Compare commits

..

25 Commits

Author SHA1 Message Date
overtrue 3340a755c1 test(e2e): verify two-pool bootstrap metadata repair 2026-09-06 16:38:10 +08:00
overtrue b4e0838b01 test(startup): observe pool repair classification and replicas 2026-09-06 16:38:09 +08:00
overtrue 6233466c5d chore: integrate current main for namespace target validation 2026-09-06 15:49:36 +08:00
overtrue 901d052c38 test(e2e): declare tempfile for fresh startup fixture 2026-09-06 15:45:26 +08:00
overtrue fdb3d9f9bb test(startup): encode observation digests with shared hex helper 2026-09-06 15:10:28 +08:00
overtrue 44333e136b test(rpc): use native part size in signed fixture 2026-09-06 13:57:07 +08:00
overtrue 9e24d23c30 test(rpc): observe bootstrap CAS during fresh startup 2026-09-06 13:01:22 +08:00
overtrue 6277287399 chore(test): merge main before startup CAS coverage 2026-09-06 12:45:50 +08:00
overtrue 358ff0f0e6 test(rpc): select fixture disks through the store topology 2026-09-06 11:42:53 +08:00
overtrue 1cea5fa1c4 test(rpc): retain bootstrap authority across delayed requests 2026-09-06 11:42:53 +08:00
overtrue b1a2235cd6 fix(ecstore): keep group fsync hooks scoped to unit tests 2026-09-06 11:25:31 +08:00
overtrue e0663c11af test(rpc): cover signed listener binding across startup 2026-09-06 11:13:09 +08:00
overtrue 11c9fd64ce fix(rpc): bind local mutations to the listener instance 2026-09-06 10:58:11 +08:00
overtrue 7f631ec378 test(rpc): mark instance regression request as v2 authenticated 2026-09-06 10:21:55 +08:00
overtrue 3205f85c2a test(rpc): preserve bootstrap metadata in instance snapshot 2026-09-06 10:21:55 +08:00
overtrue ae87ddbe2f test(rpc): reproduce same-UUID cross-instance rename 2026-09-06 10:21:55 +08:00
overtrue f09aaad2a9 Merge local namespace ownership prerequisite 2026-09-06 10:21:54 +08:00
overtrue 3ae29aab26 test(ecstore): wait for namespace owner release before asserting
The namespace owner tests decided that ownership had ended when the Weak probe stopped upgrading or when the mutation lease could be reacquired. Both signals fire before the owner guard's Drop decrements the pending counter: Arc releases its strong count before running Drop, and the lease drops its locks before its owner field. The rio-v2 lane hit that window in undo_fresh_version_keeps_physical_namespace_owner_after_timeout.

Extend every drain wait to also require namespace_commits_pending() to be false, so the assertions observe the completed release instead of racing it.
2026-09-06 10:09:44 +08:00
overtrue 3c54ac1deb Merge remote-tracking branch 'origin/main' into overtrue/fix/fsync-group-identity-capture 2026-09-06 04:49:00 +08:00
overtrue 44fe0b9950 Merge remote-tracking branch 'origin/main' into overtrue/fix/fsync-group-identity-capture 2026-09-06 04:44:25 +08:00
overtrue 58f1840630 test(ecstore): mark physical owner fixtures as inline 2026-09-06 04:25:12 +08:00
overtrue 5e33186cf6 fix(ecstore): preserve successor fsync group registration
(cherry picked from commit c7dfaad90526052e56c57dafffa4813bdcde46ca)
2026-09-06 03:54:07 +08:00
overtrue ed100103d0 fix(ecstore): capture complete fsync worker guard
(cherry picked from commit 1dc90bb836e20ea9ee45d0a629a9201e20d231c4)
2026-09-06 03:54:07 +08:00
overtrue 93c89ef132 test(ecstore): expose stale fsync group cleanup 2026-09-06 03:20:33 +08:00
overtrue ea4068b8ac fix(ecstore): retain namespace owners through local physical tails
(cherry picked from commit a2f242463316e87604feadbdac5e4148140e72c0)
2026-09-06 03:17:00 +08:00
112 changed files with 4787 additions and 14133 deletions
+93 -2
View File
@@ -582,13 +582,59 @@ jobs:
install-build-packaging-tools: 'false'
- name: Build debug binary
run: cargo build -p rustfs --bins --features e2e-test-hooks
run: |
python3 - <<'PYBUILD'
import hashlib
import json
import os
import pathlib
import subprocess
def git(*args):
return subprocess.check_output(["git", *args], text=True).strip()
def sha256(path):
digest = hashlib.sha256()
with pathlib.Path(path).open("rb") as source:
for chunk in iter(lambda: source.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
argv = ["cargo", "build", "-p", "rustfs", "--bins", "--features", "e2e-test-hooks"]
commit, tree = git("rev-parse", "HEAD"), git("rev-parse", "HEAD^{tree}")
clean_before = not git("status", "--porcelain", "--untracked-files=normal")
if not clean_before:
raise SystemExit("hooks binary requires a clean build checkout")
lock_sha256 = sha256("Cargo.lock")
lock_git_blob = git("hash-object", "Cargo.lock")
rustc = subprocess.check_output(["rustc", "-vV"], text=True)
host = next(line.removeprefix("host: ") for line in rustc.splitlines() if line.startswith("host: "))
if os.environ.get("CARGO_BUILD_TARGET") or pathlib.Path(os.environ.get("CARGO_TARGET_DIR", "target")).resolve() != pathlib.Path("target").resolve():
raise SystemExit("this artifact requires the native target/debug output")
subprocess.run(argv, check=True)
clean_after = not git("status", "--porcelain", "--untracked-files=normal")
if not clean_after or commit != git("rev-parse", "HEAD") or tree != git("rev-parse", "HEAD^{tree}") or lock_sha256 != sha256("Cargo.lock"):
raise SystemExit("hooks binary source changed while building")
manifest = {
"schema": 1, "commit": commit, "tree": tree,
"clean_before": clean_before, "clean_after": clean_after,
"lock_sha256": lock_sha256, "lock_git_blob": lock_git_blob,
"argv": argv, "profile": "debug", "target": host,
"features": ["e2e-test-hooks"],
"rustc_verbose": rustc,
"build_flags": {key: os.environ[key] for key in ("RUSTFLAGS", "CARGO_ENCODED_RUSTFLAGS", "CARGO_BUILD_TARGET", "CARGO_TARGET_DIR", "RUSTUP_TOOLCHAIN") if key in os.environ},
"binary_sha256": sha256("target/debug/rustfs"),
}
pathlib.Path("target/debug/rustfs.e2e-startup-cas-build.json").write_text(json.dumps(manifest, indent=2) + "\n")
PYBUILD
- name: Upload debug binary
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: rustfs-debug-binary
path: target/debug/rustfs
path: |
target/debug/rustfs
target/debug/rustfs.e2e-startup-cas-build.json
if-no-files-found: error
retention-days: 1
@@ -906,6 +952,36 @@ jobs:
- name: Make binary executable
run: chmod +x ./target/debug/rustfs
- name: Preserve startup CAS binary input
env:
STARTUP_CAS_INPUT: ${{ runner.temp }}/rustfs-startup-cas-input
run: |
python3 - <<'PYINPUT'
import hashlib
import json
import os
import pathlib
import shutil
import subprocess
source = pathlib.Path("target/debug/rustfs")
manifest_path = source.with_name("rustfs.e2e-startup-cas-build.json")
manifest = json.loads(manifest_path.read_text())
target = pathlib.Path(os.environ["STARTUP_CAS_INPUT"])
target.mkdir(parents=True, exist_ok=True)
binary = target / "rustfs"
shutil.copy2(source, binary)
digest = hashlib.sha256()
with binary.open("rb") as stream:
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
digest.update(chunk)
commit = subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip()
if manifest["binary_sha256"] != digest.hexdigest() or manifest["commit"] != commit:
raise SystemExit("downloaded hooks binary identity mismatch")
shutil.copy2(manifest_path, target / manifest_path.name)
binary.chmod(0o755)
PYINPUT
- name: Verify e2e full membership
env:
NEXTEST_LISTING: ${{ runner.temp }}/rustfs-e2e-full-list.json
@@ -918,6 +994,10 @@ jobs:
# extend that filter, never add ad-hoc e2e jobs here. Reuses the downloaded
# debug binary; each test spawns its own rustfs server on a random port.
- name: Run e2e full suite
env:
RUSTFS_E2E_STARTUP_CAS_BINARY: ${{ runner.temp }}/rustfs-startup-cas-input/rustfs
RUSTFS_E2E_STARTUP_CAS_BUILD_MANIFEST: ${{ runner.temp }}/rustfs-startup-cas-input/rustfs.e2e-startup-cas-build.json
RUSTFS_E2E_STARTUP_CAS_ARTIFACT_DIR: ${{ runner.temp }}/rustfs-startup-cas-evidence
run: cargo nextest run --profile e2e-full -p e2e_test
- name: Upload junit
@@ -930,6 +1010,17 @@ jobs:
${{ runner.temp }}/rustfs-e2e-full-list.json
retention-days: 7
- name: Upload startup CAS evidence
if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: fresh-startup-cas-evidence-${{ github.run_number }}
path: |
${{ runner.temp }}/rustfs-startup-cas-evidence
${{ runner.temp }}/rustfs-startup-cas-input/rustfs.e2e-startup-cas-build.json
if-no-files-found: warn
retention-days: 7
e2e-tests-rio-v2:
name: End-to-End Tests (rio-v2)
# Inherits the schedule/dispatch-only gate through needs: on every other
Generated
+35 -35
View File
@@ -2527,18 +2527,18 @@ checksum = "790eea4361631c5e7d22598ecd5723ff611904e3344ce8720784c93e3d83d40b"
[[package]]
name = "crossbeam-channel"
version = "0.5.17"
version = "0.5.16"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "98b0cc327b5bc766e7fda9c9260cc0fa81b43a8e240440422dff70788e3f9ef1"
checksum = "d85363c37faeca707aef026efa9f3b34d077bce547e48f770770625c6013679e"
dependencies = [
"crossbeam-utils",
]
[[package]]
name = "crossbeam-deque"
version = "0.8.8"
version = "0.8.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "622f3fc73690be383c7214310406f28a90e6edeadc3cea882f9d71e495b9711a"
checksum = "5181e0de7b61eb03a81e347d6dd8797bae9da5146707b51077e2d71a54ec0ceb"
dependencies = [
"crossbeam-epoch",
"crossbeam-utils",
@@ -2546,27 +2546,27 @@ dependencies = [
[[package]]
name = "crossbeam-epoch"
version = "0.9.21"
version = "0.9.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d"
checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f"
dependencies = [
"crossbeam-utils",
]
[[package]]
name = "crossbeam-queue"
version = "0.3.14"
version = "0.3.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "03e8bd762f7479489c70ed6c768ddca99d7296857de437a68dcb2a94365b3fae"
checksum = "803d13fb3b09d88be9f4dbc29062c66b19bf7170867ceb746d2a8689bf6c7a26"
dependencies = [
"crossbeam-utils",
]
[[package]]
name = "crossbeam-utils"
version = "0.8.23"
version = "0.8.22"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6"
checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17"
[[package]]
name = "crunchy"
@@ -3673,9 +3673,9 @@ dependencies = [
[[package]]
name = "der"
version = "0.8.2"
version = "0.8.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a878c850e9e421b20262e9b41f9c860e4785fa07541c266b62ff9d1ef998a80a"
checksum = "a69dedd701da44b0536442edf09c81a64b0ab97a7a4a5e3d1971f00027cbc63d"
dependencies = [
"const-oid 0.10.2",
"pem-rfc7468 1.0.0",
@@ -3946,7 +3946,7 @@ dependencies = [
"libc",
"option-ext",
"redox_users 0.5.2",
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -4057,6 +4057,7 @@ dependencies = [
"sha1 0.11.0",
"sha2 0.11.0",
"suppaftp",
"tempfile",
"time",
"tokio",
"tokio-stream",
@@ -4090,7 +4091,7 @@ version = "0.17.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c0681a4fc24c767085329728d8dfba959af91228aa4610cca4f8ce317ba46ae0"
dependencies = [
"der 0.8.2",
"der 0.8.1",
"digest 0.11.3",
"elliptic-curve 0.14.1",
"rfc6979 0.6.0",
@@ -4295,7 +4296,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb"
dependencies = [
"libc",
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -5693,7 +5694,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "20fd6de4ccfcc187e38bc21cfa543cb5a302cb86a8b114eb7f0bf0dc9f8ac00f"
dependencies = [
"io-lifetimes 3.0.1",
"windows-sys 0.59.0",
"windows-sys 0.60.2",
]
[[package]]
@@ -5734,9 +5735,9 @@ dependencies = [
[[package]]
name = "ipnet"
version = "2.12.2"
version = "2.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "791930b43c0d5973160d90a8f3894509f2b273430f5c5c73b668636d0287c5c0"
checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78"
dependencies = [
"serde",
]
@@ -5758,7 +5759,7 @@ checksum = "3640c1c38b8e4e43584d8df18be5fc6b0aa314ce6ebf51b53313d4306cca8e46"
dependencies = [
"hermit-abi",
"libc",
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -6955,7 +6956,7 @@ version = "0.50.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7957b9740744892f114936ab4a57b3f487491bbeafaf8083688b16841a4240e5"
dependencies = [
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -7933,7 +7934,7 @@ version = "0.8.0-rc.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "986d2e952779af96ea048f160fd9194e1751b4faea78bcf3ceb456efe008088e"
dependencies = [
"der 0.8.2",
"der 0.8.1",
"spki 0.8.0",
]
@@ -7976,7 +7977,7 @@ dependencies = [
"aes 0.9.3",
"aes-gcm",
"cbc 0.2.1",
"der 0.8.2",
"der 0.8.1",
"pbkdf2 0.13.0",
"rand_core 0.10.1",
"scrypt 0.12.0",
@@ -8000,7 +8001,7 @@ version = "0.11.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "451913da69c775a56034ea8d9003d27ee8948e12443eae7c038ba100a4f21cb7"
dependencies = [
"der 0.8.2",
"der 0.8.1",
"pkcs5 0.8.1",
"rand_core 0.10.1",
"spki 0.8.0",
@@ -8706,7 +8707,7 @@ dependencies = [
"once_cell",
"socket2",
"tracing",
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -8942,9 +8943,9 @@ dependencies = [
[[package]]
name = "redis"
version = "1.7.0"
version = "1.6.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2acbc41a996f7652b2ddd9dfd98cc4ff602cfd742ae35382f07f608405ab50ed"
checksum = "e37a4ca5c6ca42aa3e6df2fd32b987a65d32a4c2159a6f3fe0fd1df306a2658f"
dependencies = [
"arc-swap",
"arcstr",
@@ -9326,7 +9327,7 @@ dependencies = [
"curve25519-dalek 5.0.0",
"data-encoding",
"delegate",
"der 0.8.2",
"der 0.8.1",
"digest 0.11.3",
"ecdsa 0.17.0",
"ed25519-dalek 3.0.0",
@@ -10745,7 +10746,6 @@ dependencies = [
"rustfs-data-usage",
"rustfs-ecstore",
"rustfs-filemeta",
"rustfs-heal",
"rustfs-heal-contracts",
"rustfs-lifecycle",
"rustfs-lock",
@@ -11066,7 +11066,7 @@ dependencies = [
"errno",
"libc",
"linux-raw-sys",
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -11149,7 +11149,7 @@ dependencies = [
"security-framework",
"security-framework-sys",
"webpki-root-certs",
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -11420,7 +11420,7 @@ checksum = "d56d437c2f19203ce5f7122e507831de96f3d2d4d3be5af44a0b0a09d8a80e4d"
dependencies = [
"base16ct 1.0.0",
"ctutils",
"der 0.8.2",
"der 0.8.1",
"hybrid-array",
"subtle",
"zeroize",
@@ -11994,7 +11994,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1d9efca8738c78ee9484207732f728b1ef517bbb1833d6fc0879ca898a522f6f"
dependencies = [
"base64ct",
"der 0.8.2",
"der 0.8.1",
]
[[package]]
@@ -12406,10 +12406,10 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
dependencies = [
"fastrand",
"getrandom 0.3.4",
"getrandom 0.4.3",
"once_cell",
"rustix",
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
@@ -13528,7 +13528,7 @@ version = "0.1.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22"
dependencies = [
"windows-sys 0.59.0",
"windows-sys 0.61.2",
]
[[package]]
+5 -5
View File
@@ -256,10 +256,10 @@ clap = { version = "4.6.6" }
const-str = { version = "1.1.0" }
convert_case = "0.12.0"
criterion = { version = "0.8" }
crossbeam-queue = "0.3.14"
crossbeam-channel = "0.5.17"
crossbeam-deque = "0.8.8"
crossbeam-utils = "0.8.23"
crossbeam-queue = "0.3.13"
crossbeam-channel = "0.5.16"
crossbeam-deque = "0.8.7"
crossbeam-utils = "0.8.22"
datafusion = { default-features = false, version = "55.0.0" }
derive_builder = "0.20.2"
enumset = "1.1.14"
@@ -306,7 +306,7 @@ rustfs-erasure-codec = { version = "8.0.2" }
reed-solomon-simd = "3.1.0"
regex = { version = "1.13.1" }
rumqttc = { package = "rumqttc-next", version = "0.34.0" }
redis = { version = "1.7.0" }
redis = { version = "1.6.0" }
rustify = { version = "0.7", default-features = false }
rustix = { version = "1.1.4" }
rust-embed = { version = "8.12.0" }
+5 -8
View File
@@ -422,9 +422,9 @@ fn unix_now_ms() -> u64 {
.unwrap_or(0)
}
/// Legacy, unverified repair notice. Its identity lacks kind, set scope,
/// bucket incarnation and responsibility generation. Consumers must not use
/// it to discharge persisted repair responsibility.
/// A repair the MRF consumer landed, fanned out so retry ledgers can drop
/// entries the journal no longer tracks (backlog#1894 axis B). The payload
/// mirrors the intent identity so consumers match without re-parsing.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct MrfRepairedEvent {
pub bucket: Arc<str>,
@@ -439,8 +439,8 @@ const MRF_REPAIRED_EVENT_CAP: usize = 4096;
static MRF_REPAIRED_EVENTS: OnceLock<std::sync::Mutex<std::collections::VecDeque<MrfRepairedEvent>>> = OnceLock::new();
/// Record a legacy notification for compatibility. This is not an
/// acknowledgement of storage verification or durable repair completion.
/// Record that the MRF consumer landed a repair. Never blocks: the critical
/// section is a deque push under a std mutex.
pub fn note_mrf_repaired(bucket: &str, object: &str, version_id: Option<[u8; 16]>) {
let registry = MRF_REPAIRED_EVENTS.get_or_init(|| std::sync::Mutex::new(std::collections::VecDeque::new()));
let Ok(mut events) = registry.lock() else {
@@ -515,9 +515,6 @@ mod tests {
}
coalescer_release(&key, Some(lease));
let retry_lease = coalescer_admit(key.clone()).expect("released identity must admit a retry");
assert_ne!(lease, retry_lease);
coalescer_release(&key, Some(lease));
assert_eq!(coalescer_admit(key.clone()), Err(MrfIngressResult::Coalesced));
coalescer_release(&key, Some(retry_lease));
}
+3
View File
@@ -144,3 +144,6 @@ russh = { workspace = true, features = ["serde"] }
russh-sftp = { workspace = true }
zip.workspace = true
clap = { workspace = true, features = ["derive", "env"] }
[dev-dependencies]
tempfile.workspace = true
File diff suppressed because it is too large Load Diff
@@ -21,7 +21,7 @@
use super::common::{BoxError, OdmSourceSpec, OdmTestEnv, SeedObject};
use crate::fake_s3_target::{BucketMode, Operation};
use aws_sdk_s3::error::ProvideErrorMetadata;
use aws_sdk_s3::types::{BucketVersioningStatus, ObjectAttributes, VersioningConfiguration};
use aws_sdk_s3::types::{BucketVersioningStatus, VersioningConfiguration};
use bytes::Bytes;
use std::time::Duration;
@@ -103,19 +103,11 @@ async fn get_miss_pulls_inline_and_serves_locally_afterwards() -> TestResult {
#[tokio::test]
async fn get_large_object_streams_through_and_backfills_in_background() -> TestResult {
const PART_SIZE: usize = 5 * 1024 * 1024;
let bucket = "odm-get-large";
let env = configured_env(bucket, |spec| {
spec.policy.inline_max_bytes = 4096;
spec.policy.multipart_part_size_bytes = u64::try_from(PART_SIZE).expect("part size fits in u64");
})
.await?;
let env = configured_env(bucket, |spec| spec.policy.inline_max_bytes = 4096).await?;
let key = "large/archive.bin";
let body = payload(PART_SIZE + 4096);
let etag = env
.seed_source(SOURCE_BUCKET, &[SeedObject::new(key, body.clone())])
.remove(0);
assert_eq!(etag.len(), 32, "the source fixture has a plain MD5 ETag");
let body = payload(512 * 1024);
env.seed_source(SOURCE_BUCKET, &[SeedObject::new(key, body.clone())]);
let response = env.raw_get(bucket, key).await?;
assert_eq!(response.status, 200, "{}", String::from_utf8_lossy(&response.body));
@@ -133,68 +125,6 @@ async fn get_large_object_streams_through_and_backfills_in_background() -> TestR
vec![None, None],
"one passthrough GET plus one background pull, both unranged"
);
let source_requests = env.source.requests().len();
let second_part = env.client.get_object().bucket(bucket).key(key).part_number(2).send().await?;
assert_eq!(second_part.content_length(), Some(4096), "the completed second part is the tail");
assert_eq!(
second_part.content_range(),
Some(format!("bytes {PART_SIZE}-{}/{}", body.len() - 1, body.len()).as_str()),
"partNumber reads the stored multipart boundary"
);
assert_eq!(
second_part.body.collect().await?.into_bytes(),
body.slice(PART_SIZE..),
"the local second part contains the exact source tail"
);
let third_part = env
.client
.get_object()
.bucket(bucket)
.key(key)
.part_number(3)
.send()
.await
.expect_err("the completed object has exactly two parts");
assert_eq!(third_part.code(), Some("InvalidPart"));
let mut part_marker = None;
for (part_number, part_size) in [(1, PART_SIZE), (2, 4096)] {
let attributes = env
.client
.get_object_attributes()
.bucket(bucket)
.key(key)
.object_attributes(ObjectAttributes::ObjectParts)
.object_attributes(ObjectAttributes::Etag)
.max_parts(1)
.set_part_number_marker(part_marker.clone())
.send()
.await?;
assert_eq!(
attributes.e_tag().map(|value| value.trim_matches('"')),
Some(etag.as_str()),
"multipart write-back preserves the source MD5 ETag"
);
let parts = attributes
.object_parts()
.expect("RustFS must expose the stored multipart layout");
assert_eq!(parts.total_parts_count(), Some(2));
assert_eq!(parts.max_parts(), Some(1));
assert_eq!(parts.is_truncated(), Some(part_number == 1));
assert_eq!(parts.parts().len(), 1, "RustFS returns one stored part per requested page");
assert_eq!(parts.parts()[0].part_number(), Some(part_number));
assert_eq!(parts.parts()[0].size(), Some(i64::try_from(part_size).expect("part size fits in i64")));
part_marker = parts.next_part_number_marker().map(str::to_owned);
if part_number == 1 {
assert_eq!(part_marker.as_deref(), Some("1"), "the next request continues after the first part");
}
}
assert_eq!(
env.source.requests().len(),
source_requests,
"local part reads must not consult the source"
);
Ok(())
}
@@ -23,20 +23,16 @@
use super::common::{
ALLOW_LOOPBACK_SOURCE_ENV, AdminResponse, BackfillOp, BackfillRequest, BoxError, ODM_MODULE_SWITCH_ENV, ODM_SERVER_ENV,
OdmEnvOptions, OdmSourceSpec, OdmTestEnv, SeedObject, start_configured_env, start_configured_env_with, start_source_rustfs,
OdmEnvOptions, OdmSourceSpec, OdmTestEnv, SeedObject, start_configured_env, start_configured_env_with,
};
use crate::common::{RustFSTestEnvironment, replication_fast_env, signed_request};
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY, FakeS3Target, Operation};
use crate::object_lock::common::put_object_lock_configuration;
use crate::replication_extension_test::{
ReplicationTargetOptions, enable_bucket_versioning, set_replication_target_with_options,
};
use aws_sdk_s3::error::ProvideErrorMetadata;
use aws_sdk_s3::types::{
BucketVersioningStatus, Event, FilterRule, FilterRuleName, NotificationConfiguration, NotificationConfigurationFilter,
ObjectAttributes, ObjectLockRetentionMode, QueueConfiguration, S3KeyFilter, ServerSideEncryption,
ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging,
VersioningConfiguration,
ObjectLockRetentionMode, QueueConfiguration, S3KeyFilter, ServerSideEncryption, ServerSideEncryptionByDefault,
ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging, VersioningConfiguration,
};
use bytes::Bytes;
use local_ip_address::local_ip;
@@ -584,130 +580,6 @@ async fn test_odm_pulled_object_replicates_and_target_as_source_is_rejected() ->
"a bucket may not migrate from its own replication target: {}",
rejected.body
);
Box::pin(assert_odm_multipart_replicates_to_rustfs(&env, bucket)).await?;
Ok(())
}
async fn assert_odm_multipart_replicates_to_rustfs(env: &OdmTestEnv, bucket: &str) -> TestResult {
const PART_SIZE: usize = 5 * 1024 * 1024;
let replica = start_source_rustfs().await?;
let replica_bucket = "odm-real-replica";
replica.create_test_bucket(replica_bucket).await?;
enable_bucket_versioning(&replica, replica_bucket).await?;
let arn = set_replication_target_with_options(
&env.rustfs,
bucket,
ReplicationTargetOptions {
endpoint: &replica.address,
access_key: &replica.access_key,
secret_key: &replica.secret_key,
target_bucket: replica_bucket,
secure: false,
skip_tls_verify: false,
ca_cert_pem: None,
},
)
.await?;
put_bucket_replication(&env.rustfs, bucket, &arn).await?;
let mut spec = env.fake_source_spec(SOURCE_BUCKET);
// Below the 16 MiB inline default the pull is one tee'd PUT with a single
// part; force the passthrough + background multipart write-back instead.
spec.policy.inline_max_bytes = 4096;
spec.policy.multipart_part_size_bytes = PART_SIZE as u64;
spec.policy.preserve_etag = true;
env.configure_and_wait(bucket, &spec).await?;
let key = "replicated/preserved-md5-multipart.bin";
let body = payload(PART_SIZE + 4096);
let source_put = env
.source_client()
.put_object()
.bucket(SOURCE_BUCKET)
.key(key)
.body(aws_sdk_s3::primitives::ByteStream::from(body.clone()))
.send()
.await?;
let etag = source_put.e_tag().ok_or("source PUT omitted its ETag")?.trim_matches('"');
assert_eq!(etag.len(), 32, "the source must retain a single-PUT MD5 ETag");
assert!(etag.bytes().all(|byte| byte.is_ascii_hexdigit()));
let pulled = env.raw_get(bucket, key).await?;
assert_eq!(pulled.status, 200, "{}", String::from_utf8_lossy(&pulled.body));
assert_eq!(pulled.body, body);
assert!(env.wait_local_listed(bucket, key, SETTLE).await?, "the multipart pull must persist");
let deadline = Instant::now() + SETTLE;
let source_head = loop {
let head = env.client.head_object().bucket(bucket).key(key).send().await?;
match head.replication_status().map(|status| status.as_str()) {
Some("COMPLETED") => break head,
Some("FAILED") => return Err("the ODM multipart copy failed replication to RustFS".into()),
_ => {
assert!(Instant::now() < deadline, "the ODM multipart copy never completed replication to RustFS");
tokio::time::sleep(Duration::from_millis(200)).await;
}
}
};
let version = source_head
.version_id()
.ok_or("the versioned ODM copy omitted its version id")?;
assert_ne!(version, "null");
let replica_client = replica.create_s3_client();
for (client, object_bucket) in [(&env.client, bucket), (&replica_client, replica_bucket)] {
let attributes = client
.get_object_attributes()
.bucket(object_bucket)
.key(key)
.version_id(version)
.object_attributes(ObjectAttributes::Etag)
.object_attributes(ObjectAttributes::ObjectParts)
.send()
.await?;
assert_eq!(attributes.e_tag().map(|value| value.trim_matches('"')), Some(etag));
let parts = attributes
.object_parts()
.ok_or("the local copy and replica must both expose two parts")?;
assert_eq!(parts.total_parts_count(), Some(2));
assert_eq!(
parts
.parts()
.iter()
.map(|part| (part.part_number(), part.size()))
.collect::<Vec<_>>(),
[(Some(1), Some(PART_SIZE as i64)), (Some(2), Some(4096))]
);
}
// REPLICA status surfaces on HEAD, like the other inbound-replica checks.
let replica_head = replica_client
.head_object()
.bucket(replica_bucket)
.key(key)
.version_id(version)
.send()
.await?;
assert_eq!(replica_head.replication_status().map(|status| status.as_str()), Some("REPLICA"));
let replica_get = replica_client
.get_object()
.bucket(replica_bucket)
.key(key)
.version_id(version)
.send()
.await?;
assert_eq!(replica_get.version_id(), Some(version));
assert_eq!(replica_get.body.collect().await?.into_bytes(), body);
let boundary = replica_client
.get_object()
.bucket(replica_bucket)
.key(key)
.version_id(version)
.range(format!("bytes={}-{}", PART_SIZE - 32, PART_SIZE + 31))
.send()
.await?;
assert_eq!(boundary.body.collect().await?.into_bytes(), body.slice(PART_SIZE - 32..PART_SIZE + 32));
assert_eq!(
env.source.count_requests(Operation::GetObject, key),
2,
"one passthrough GET plus one background pull; replication and local reads must not fetch the migration source again"
);
Ok(())
}
@@ -368,23 +368,23 @@ impl Drop for SlowReplicationTargetGuard {
// Mirrors madmin-go `ResyncTargetsInfo`/`ResyncTarget` json tags — the same
// shape `mc replicate resync status` decodes.
#[derive(Debug, Clone, serde::Deserialize)]
pub(crate) struct ReplicationResetStatusResponse {
struct ReplicationResetStatusResponse {
#[serde(rename = "target", default)]
pub(crate) targets: Vec<ReplicationResetStatusTarget>,
targets: Vec<ReplicationResetStatusTarget>,
}
#[derive(Debug, Clone, serde::Deserialize)]
pub(crate) struct ReplicationResetStatusTarget {
struct ReplicationResetStatusTarget {
#[serde(rename = "arn", default)]
pub(crate) arn: String,
arn: String,
#[serde(rename = "resetid", default)]
pub(crate) reset_id: String,
reset_id: String,
#[serde(rename = "resyncStatus", default)]
pub(crate) status: String,
status: String,
#[serde(rename = "replicationCount", default)]
pub(crate) replicated_count: i64,
replicated_count: i64,
#[serde(rename = "object", default)]
pub(crate) object: String,
object: String,
}
fn extract_xml_tag(xml: &str, tag: &str) -> Option<String> {
@@ -2294,7 +2294,7 @@ async fn site_replication_state_edit(
/// return the target `(arn, reset_id)`, asserting the response carries the
/// madmin `ResyncTargetsInfo` shape (`target[0].arn` / `target[0].resetid`)
/// that `mc replicate resync start` decodes.
pub(crate) async fn start_bucket_replication_reset(
async fn start_bucket_replication_reset(
env: &RustFSTestEnvironment,
bucket: &str,
) -> Result<(String, String), Box<dyn Error + Send + Sync>> {
@@ -2314,7 +2314,7 @@ pub(crate) async fn start_bucket_replication_reset(
Ok((arn, reset_id))
}
pub(crate) async fn get_replication_reset_status(
async fn get_replication_reset_status(
env: &RustFSTestEnvironment,
bucket: &str,
arn: &str,
@@ -31,19 +31,17 @@
//! Adding a target behavior the fleet has shown: add the mode to the fake
//! target, add a row here, and record any cell that is red before the fix.
use crate::common::{init_logging, replication_fast_env};
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY};
use crate::fake_s3_target::{FakeS3Target, FaultAction as FakeTargetFault, Operation as FakeTargetOperation, RequestRecord};
use crate::on_demand_migration::common::{OdmEnvOptions, OdmTestEnv, fake_source_client};
use crate::common::{RustFSTestEnvironment, init_logging, replication_fast_env};
use crate::fake_s3_target::{FAKE_ACCESS_KEY, FAKE_SECRET_KEY};
use crate::fake_s3_target::{FakeS3Target, Operation as FakeTargetOperation, RequestRecord};
use crate::on_demand_migration::common::fake_source_client;
use crate::replication_extension_test::{
LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, enable_bucket_versioning, get_replication_reset_status,
put_bucket_replication, set_replication_target_with_options, start_bucket_replication_reset,
LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, enable_bucket_versioning, put_bucket_replication,
set_replication_target_with_options,
};
use aws_sdk_s3::Client;
use aws_sdk_s3::primitives::{ByteStream, DateTime};
use aws_sdk_s3::types::{
Checksum, CompletedMultipartUpload, CompletedPart, ObjectAttributes, ObjectLockLegalHoldStatus, ObjectLockMode,
};
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart, ObjectLockLegalHoldStatus, ObjectLockMode};
use bytes::Bytes;
use std::error::Error;
use std::time::{SystemTime, UNIX_EPOCH};
@@ -112,19 +110,16 @@ enum ObjectShape {
/// Two-part multipart upload with a GOVERNANCE retention period; the
/// lock headers travel on CreateMultipartUpload, which has no body.
LockedMultipart,
/// ODM stores two local parts while preserving a single-PUT source's MD5 ETag.
OdmPreservedMd5Multipart,
}
impl ObjectShape {
const ALL: [ObjectShape; 7] = [
const ALL: [ObjectShape; 6] = [
ObjectShape::Empty,
ObjectShape::Plain,
ObjectShape::Retention,
ObjectShape::LegalHold,
ObjectShape::Multipart,
ObjectShape::LockedMultipart,
ObjectShape::OdmPreservedMd5Multipart,
];
fn key(self) -> &'static str {
@@ -135,7 +130,6 @@ impl ObjectShape {
ObjectShape::LegalHold => "matrix/legal-hold.bin",
ObjectShape::Multipart => "matrix/multipart.bin",
ObjectShape::LockedMultipart => "matrix/locked-multipart.bin",
ObjectShape::OdmPreservedMd5Multipart => "matrix/odm-preserved-md5.bin",
}
}
@@ -145,8 +139,7 @@ impl ObjectShape {
/// Upload the shape to the source and return the bytes the target must
/// end up holding.
async fn put(self, env: &OdmTestEnv, bucket: &str) -> Result<Bytes, Box<dyn Error + Send + Sync>> {
let client = &env.client;
async fn put(self, client: &Client, bucket: &str) -> Result<Bytes, Box<dyn Error + Send + Sync>> {
let key = self.key();
match self {
ObjectShape::Empty => {
@@ -197,7 +190,6 @@ impl ObjectShape {
}
ObjectShape::Multipart => multipart_put(client, bucket, key, 0x44, false).await,
ObjectShape::LockedMultipart => multipart_put(client, bucket, key, 0x55, true).await,
ObjectShape::OdmPreservedMd5Multipart => odm_preserved_md5_multipart(env, bucket, key).await,
}
}
}
@@ -227,170 +219,6 @@ fn expectation(mode: TargetMode, shape: ObjectShape) -> Expectation {
.unwrap_or(Expectation::Completed)
}
/// rustfs/backlog#2340: a target that mints its own version ids (Wasabi,
/// AWS S3) answers 404 to a HEAD by the source uuid, which the worker used to
/// read as "replica missing" and re-drive the PUT — one more target version
/// per heal, MRF retry or resync. Two re-drive shapes, both must converge on
/// the single version the first PUT created:
/// - the first PUT lands but its response is lost, so the object is FAILED
/// and the scanner heal pass re-drives it;
/// - an existing-object resync re-drives a COMPLETED object unconditionally.
#[tokio::test]
async fn matrix_mint_own_version_ids_redrive_does_not_duplicate() -> TestResult {
init_logging();
let target = FakeS3Target::start().await?;
let target_bucket = "matrix-mint-own-redrive-dst".to_string();
target.create_bucket_with_object_lock(target_bucket.clone());
target.assign_own_version_ids(true);
let mut env_vars = replication_fast_env();
env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
env_vars.extend_from_slice(&[
("NO_PROXY", "127.0.0.1,localhost"),
("HTTP_PROXY", ""),
("HTTPS_PROXY", ""),
// The scanner heal pass is what re-drives a FAILED object.
("RUSTFS_SCANNER_CYCLE", "1"),
("RUSTFS_SCANNER_START_DELAY_SECS", "1"),
]);
let env = OdmTestEnv::start_with(OdmEnvOptions {
env: env_vars,
..OdmEnvOptions::default()
})
.await?;
let source_env = &env.rustfs;
let source_bucket = "matrix-mint-own-redrive-src";
let source_client = source_env.create_s3_client();
source_client
.create_bucket()
.bucket(source_bucket)
.object_lock_enabled_for_bucket(true)
.send()
.await?;
enable_bucket_versioning(source_env, source_bucket).await?;
let target_arn = set_replication_target_with_options(
source_env,
source_bucket,
ReplicationTargetOptions {
endpoint: &target.address(),
access_key: FAKE_ACCESS_KEY,
secret_key: FAKE_SECRET_KEY,
target_bucket: &target_bucket,
secure: false,
skip_tls_verify: false,
ca_cert_pem: None,
},
)
.await?;
put_bucket_replication(source_env, source_bucket, &target_arn).await?;
// Teach the worker the target's identity contract with one ordinary
// write, exactly as production learns it (the PUT response carries the
// minted id).
let probe_key = "redrive/identity-probe.bin";
source_client
.put_object()
.bucket(source_bucket)
.key(probe_key)
.body(ByteStream::from(payload(4 * 1024, 0x01)))
.send()
.await?;
assert_eq!(
wait_for_terminal_replication_status(&source_client, source_bucket, probe_key).await?,
"COMPLETED"
);
// Shape 1: the PUT is stored, its response never arrives, heal re-drives.
let heal_key = "redrive/heal.bin";
target.inject_for_key(FakeTargetOperation::PutObject, heal_key, FakeTargetFault::DisconnectAfterResponse, 1);
source_client
.put_object()
.bucket(source_bucket)
.key(heal_key)
.body(ByteStream::from(payload(8 * 1024, 0x02)))
.send()
.await?;
wait_for_replication_status_and_single_version(&source_client, source_bucket, &target, &target_bucket, heal_key).await?;
// Shape 2: an existing-object resync re-drives a COMPLETED object.
let resync_key = "redrive/resync.bin";
source_client
.put_object()
.bucket(source_bucket)
.key(resync_key)
.body(ByteStream::from(payload(8 * 1024, 0x03)))
.send()
.await?;
assert_eq!(
wait_for_terminal_replication_status(&source_client, source_bucket, resync_key).await?,
"COMPLETED"
);
let (reset_arn, _reset_id) = start_bucket_replication_reset(source_env, source_bucket).await?;
assert_eq!(reset_arn, target_arn);
let resync = async {
loop {
let status = get_replication_reset_status(source_env, source_bucket, &target_arn).await?;
if let Some(entry) = status.targets.iter().find(|entry| entry.arn == target_arn)
&& entry.status == "Completed"
{
return Ok::<_, Box<dyn Error + Send + Sync>>(entry.replicated_count);
}
sleep(Duration::from_millis(250)).await;
}
};
let replicated = timeout(Duration::from_secs(90), resync)
.await
.map_err(|_| "existing-object resync did not complete within 90 seconds")??;
assert!(replicated >= 3, "resync must count the located replicas as replicated, got {replicated}");
for key in [probe_key, heal_key, resync_key] {
let versions = target.stored_versions(&target_bucket, key);
assert_eq!(
versions.len(),
1,
"{key}: a re-drive against a target that mints its own version ids must not mint another one: {versions:?}"
);
}
target.shutdown().await;
Ok(())
}
/// Wait until `key` is COMPLETED on the source and, for the observation
/// window after that, the target still holds exactly one live version of it.
async fn wait_for_replication_status_and_single_version(
source_client: &Client,
source_bucket: &str,
target: &FakeS3Target,
target_bucket: &str,
key: &str,
) -> TestResult {
// The lost PUT response first settles the object FAILED; only the next
// scanner heal pass can turn that into COMPLETED, so FAILED is transient
// here and the wait is for COMPLETED alone.
let converged = async {
loop {
let head = source_client.head_object().bucket(source_bucket).key(key).send().await?;
if head.replication_status().is_some_and(|status| status.as_str() == "COMPLETED") {
return Ok::<_, Box<dyn Error + Send + Sync>>(());
}
sleep(Duration::from_millis(250)).await;
}
};
timeout(Duration::from_secs(90), converged)
.await
.map_err(|_| format!("{key}: heal re-drive did not converge to COMPLETED within 90 seconds"))??;
// The heal pass keeps visiting the key for a few scanner cycles; a
// duplicate would show up here as a second stored version.
for _ in 0..12 {
let versions = target.stored_versions(target_bucket, key);
assert_eq!(versions.len(), 1, "{key}: target minted another version on re-drive: {versions:?}");
sleep(Duration::from_millis(500)).await;
}
Ok(())
}
#[tokio::test]
async fn matrix_baseline_target() -> TestResult {
run_row(TargetMode::Baseline).await
@@ -442,15 +270,11 @@ async fn run_row(mode: TargetMode) -> TestResult {
target.create_bucket_with_object_lock(target_bucket.clone());
mode.apply(&target);
let mut source_env = RustFSTestEnvironment::new().await?;
let mut env_vars = replication_fast_env();
env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
env_vars.extend_from_slice(&[("NO_PROXY", "127.0.0.1,localhost"), ("HTTP_PROXY", ""), ("HTTPS_PROXY", "")]);
let env = OdmTestEnv::start_with(OdmEnvOptions {
env: env_vars,
..OdmEnvOptions::default()
})
.await?;
let source_env = &env.rustfs;
source_env.start_rustfs_server_with_env(vec![], &env_vars).await?;
let source_bucket = format!("matrix-{}-src", mode.slug());
let source_client = source_env.create_s3_client();
@@ -460,9 +284,9 @@ async fn run_row(mode: TargetMode) -> TestResult {
.object_lock_enabled_for_bucket(true)
.send()
.await?;
enable_bucket_versioning(source_env, &source_bucket).await?;
enable_bucket_versioning(&source_env, &source_bucket).await?;
let target_arn = set_replication_target_with_options(
source_env,
&source_env,
&source_bucket,
ReplicationTargetOptions {
endpoint: &target.address(),
@@ -475,21 +299,14 @@ async fn run_row(mode: TargetMode) -> TestResult {
},
)
.await?;
put_bucket_replication(source_env, &source_bucket, &target_arn).await?;
put_bucket_replication(&source_env, &source_bucket, &target_arn).await?;
let target_client = fake_source_client(&target);
let mut failures = Vec::new();
for shape in ObjectShape::ALL {
let cell = format!("{}/{:?}", mode.slug(), shape);
let expected_body = shape.put(&env, &source_bucket).await?;
let expected_body = shape.put(&source_client, &source_bucket).await?;
let status = wait_for_terminal_replication_status(&source_client, &source_bucket, shape.key()).await?;
if shape == ObjectShape::OdmPreservedMd5Multipart {
assert_eq!(
env.source.count_requests(FakeTargetOperation::GetObject, shape.key()),
2,
"one passthrough GET plus one background pull; replication must read the persisted local parts"
);
}
let journal = target.requests();
let outcome = match expectation(mode, shape) {
Expectation::Completed => {
@@ -562,36 +379,6 @@ async fn check_completed_cell(
if uploads.is_empty() {
return Err("no upload reached the target although the source reports COMPLETED".into());
}
if shape == ObjectShape::OdmPreservedMd5Multipart {
let key_requests: Vec<_> = journal
.iter()
.filter(|record| record.key.as_deref() == Some(shape.key()))
.collect();
for operation in [
FakeTargetOperation::CreateMultipartUpload,
FakeTargetOperation::CompleteMultipartUpload,
] {
if !key_requests.iter().any(|record| record.operation == operation) {
return Err(format!("preserved-MD5 multipart object did not use {operation:?}").into());
}
}
if key_requests
.iter()
.any(|record| record.operation == FakeTargetOperation::PutObject)
{
return Err("preserved-MD5 multipart object used a single PutObject".into());
}
let mut part_numbers: Vec<_> = key_requests
.iter()
.filter(|record| record.operation == FakeTargetOperation::UploadPart)
.map(|record| record.part_number)
.collect();
part_numbers.sort_unstable();
part_numbers.dedup();
if part_numbers != [Some(1), Some(2)] {
return Err(format!("preserved-MD5 multipart object uploaded unexpected parts: {part_numbers:?}").into());
}
}
if let Some(framed) = uploads.iter().find(|record| record.transport.aws_chunked) {
return Err(format!("{cell}: an upload went out aws-chunked (rustfs#6853 framing): {framed:?}").into());
}
@@ -668,71 +455,6 @@ async fn wait_for_terminal_replication_status(
}
}
async fn odm_preserved_md5_multipart(env: &OdmTestEnv, bucket: &str, key: &str) -> Result<Bytes, Box<dyn Error + Send + Sync>> {
const PART_SIZE: usize = 5 * 1024 * 1024;
let origin_bucket = format!("{bucket}-origin");
env.source.create_bucket_with_mode(&origin_bucket, BucketMode::Unversioned);
let mut spec = env.fake_source_spec(&origin_bucket);
// Below the 16 MiB inline default the pull is one tee'd PUT with a single
// part; force the passthrough + background multipart write-back instead.
spec.policy.inline_max_bytes = 4096;
spec.policy.multipart_part_size_bytes = PART_SIZE as u64;
spec.policy.preserve_etag = true;
env.configure_and_wait(bucket, &spec).await?;
// A normal source PUT produces the MD5 ETag; only ODM chooses the local parts.
let body = payload(PART_SIZE + 4096, 0x66);
let source_put = env
.source_client()
.put_object()
.bucket(&origin_bucket)
.key(key)
.body(ByteStream::from(body.clone()))
.send()
.await?;
let source_etag = source_put.e_tag().ok_or("source PUT omitted its ETag")?.trim_matches('"');
assert_eq!(source_etag.len(), 32, "source fixture must have a single-PUT MD5 ETag");
assert!(source_etag.bytes().all(|byte| byte.is_ascii_hexdigit()));
let pulled = env.raw_get(bucket, key).await?;
assert_eq!(pulled.status, 200, "{}", String::from_utf8_lossy(&pulled.body));
assert_eq!(pulled.body, body);
assert!(
env.wait_local_listed(bucket, key, Duration::from_secs(30)).await?,
"ODM must persist the object"
);
let attributes = env
.client
.get_object_attributes()
.bucket(bucket)
.key(key)
.object_attributes(ObjectAttributes::Etag)
.object_attributes(ObjectAttributes::ObjectParts)
.object_attributes(ObjectAttributes::Checksum)
.send()
.await?;
assert_eq!(attributes.e_tag().map(|etag| etag.trim_matches('"')), Some(source_etag));
let parts = attributes
.object_parts()
.ok_or("the ODM copy must expose its two local parts")?;
assert_eq!(parts.total_parts_count(), Some(2));
assert_eq!(
parts
.parts()
.iter()
.map(|part| (part.part_number(), part.size()))
.collect::<Vec<_>>(),
[(Some(1), Some(PART_SIZE as i64)), (Some(2), Some(4096))]
);
assert!(
attributes
.checksum()
.is_none_or(|checksum| checksum == &Checksum::builder().build()),
"multipart routing must work without an object checksum record"
);
Ok(body)
}
async fn multipart_put(
client: &Client,
bucket: &str,
+2
View File
@@ -118,6 +118,8 @@ hotpath-cpu = [
# injection, xl.meta transition assertions) via `api::tier::test_util`.
# Enable only from `[dev-dependencies]` (rustfs/backlog#1148 ilm-6).
test-util = []
# Observes real startup CAS only in the dedicated E2E binary.
e2e-test-hooks = []
[dependencies]
hotpath.workspace = true
+8 -25
View File
@@ -32,7 +32,7 @@ pub mod bucket {
pub mod bucket_target_sys {
pub use crate::bucket::bucket_target_sys::{
AdvancedPutOptions, BucketTargetError, BucketTargetSys, PutObjectOptions, RemoveObjectOptions, S3ClientError,
SsecPassthroughCapability, TargetClient, VersionIdentityCapability, append_version_id_query,
SsecPassthroughCapability, TargetClient, append_version_id_query,
};
}
@@ -76,22 +76,6 @@ pub mod bucket {
};
}
pub mod recovery_disposition {
pub use crate::bucket::lifecycle::recovery_disposition::{
CreatedIlmRecoveryDisposition, IlmRecoveryDisposition, IlmRecoveryDispositionAction, IlmRecoveryDispositionError,
IlmRecoveryDispositionIdentity, IlmRecoveryDispositionOwnerLease, IlmRecoveryDispositionReasonCode,
IlmRecoveryDispositionState, ObservedIlmRecoveryDisposition, create_recovery_disposition_if_absent,
load_recovery_disposition, recovery_disposition_id, save_recovery_disposition_if_current,
};
}
pub mod recovery_export {
pub use crate::bucket::lifecycle::recovery_export::{
IlmRecoveryExportCreated, IlmRecoveryExportObservation, create_recovery_export,
inspect_recovery_export_observation, load_recovery_export,
};
}
pub mod transition_transaction {
pub use crate::bucket::lifecycle::transition_transaction::{
TransitionOperatorDeleteResult, TransitionOperatorError, TransitionOperatorProbe, TransitionOperatorStatus,
@@ -309,7 +293,7 @@ pub mod cache {
pub mod capacity {
pub use crate::core::pools::{
DecommissionUnresolvedEntry, PoolDecommissionInfo, PoolStatus, get_total_usable_capacity, get_total_usable_capacity_free,
is_pool_activation_fleet_proof_error, path2_bucket_object, path2_bucket_object_with_base_path,
path2_bucket_object, path2_bucket_object_with_base_path,
};
pub use crate::store::utils::is_reserved_or_invalid_bucket;
}
@@ -384,6 +368,8 @@ pub mod data_usage {
pub mod disk {
pub use crate::disk::disk_store::get_object_disk_read_timeout;
pub use crate::disk::local::ScanGuard;
#[cfg(all(feature = "test-util", not(windows)))]
pub use crate::disk::os::{LocalPublicationPause, LocalPublicationStage};
pub use crate::disk::{
BATCH_READ_VERSION_MAX_ITEMS, BUCKET_META_PREFIX, BatchReadVersionItem, BatchReadVersionReq, BatchReadVersionResp,
CheckPartsResp, ConditionalFileUpdate, DeleteOptions, Disk, DiskAPI, DiskInfo, DiskInfoOptions, DiskLocation, DiskOption,
@@ -462,12 +448,9 @@ pub mod notification {
#[cfg(any(test, feature = "test-util"))]
pub use crate::services::notification_sys::rotate_cross_pool_fence_fleet_proof_for_test;
pub use crate::services::notification_sys::{
ClusterTierDailyStats, CrossPoolFenceFleetProofToken, IlmRecoveryExportFleetProofToken,
LegacyTransitionStateReconcileFleetProofToken, NotificationPeerErr, NotificationSys, ScannerPublicationLeaseGrant,
acquire_cross_pool_fence_fleet_proof, acquire_ilm_recovery_export_fleet_proof,
ClusterTierDailyStats, CrossPoolFenceFleetProofToken, LegacyTransitionStateReconcileFleetProofToken, NotificationPeerErr,
NotificationSys, ScannerPublicationLeaseGrant, acquire_cross_pool_fence_fleet_proof,
acquire_legacy_transition_state_reconcile_fleet_proof, cross_pool_fence_fleet_proof_matches, get_global_notification_sys,
ilm_recovery_export_fleet_proof_matches, ilm_recovery_export_local_process_epoch,
ilm_recovery_export_member_epochs_sha256, ilm_recovery_export_topology_generation,
legacy_transition_state_reconcile_fleet_proof_matches, new_global_notification_sys,
scanner_peer_transport_error_message_is_retryable, start_remote_version_state_fleet_probe,
};
@@ -563,8 +546,8 @@ pub mod storage {
pub use crate::core::pools::HealLifecycleExpiryContext;
pub use crate::store::HealWalkVersion;
pub use crate::store::{
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, ScannerDataMovementPauseStatus, all_local_disk, all_local_disk_path,
find_local_disk_by_ref, init_local_disks, init_local_disks_with_instance_ctx, init_lock_clients,
BootstrapLocalTarget, ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, ScannerDataMovementPauseStatus, all_local_disk,
all_local_disk_path, find_local_disk_by_ref, init_local_disks, init_local_disks_with_instance_ctx, init_lock_clients,
prewarm_local_disk_id_map, prewarm_local_disk_id_map_with_instance_ctx,
};
}
+1 -194
View File
@@ -18,7 +18,7 @@ use crate::bucket::metadata_sys::get_replication_config;
use crate::bucket::remote_s3_client::{
PathStyle, REPLICATION_TARGET_RETRY_POLICY, RemoteCredentials, RemoteS3EndpointSpec, build_remote_s3_client,
};
use crate::bucket::replication::{ObjectLockIntegrity, object_lock_put_integrity, replication_etags_match};
use crate::bucket::replication::{ObjectLockIntegrity, object_lock_put_integrity};
use crate::bucket::replication::{ReplicationStatusType, ReplicationTargetConfigBridge};
use crate::bucket::target::ARN;
use crate::bucket::target::BucketTargetType;
@@ -126,22 +126,6 @@ impl From<&BucketTarget> for RemoteS3EndpointSpec {
}
pub type HeadObjectSdkError = Box<SdkError<HeadObjectError>>;
/// Whether an edited bucket target still addresses the same remote service
/// (endpoint, bucket, path style, TLS and identity), so a verdict learned
/// about that service stays valid across the edit.
fn same_replication_service(edited: &BucketTarget, previous: &BucketTarget) -> bool {
let access_key = |target: &BucketTarget| target.credentials.as_ref().map(|credentials| credentials.access_key.clone());
edited.endpoint == previous.endpoint
&& edited.target_bucket == previous.target_bucket
&& edited.secure == previous.secure
&& edited.path == previous.path
&& access_key(edited) == access_key(previous)
}
/// Page size and page budget for [`TargetClient::find_version_by_etag`].
const FIND_VERSION_BY_ETAG_PAGE_SIZE: i32 = 1000;
const FIND_VERSION_BY_ETAG_MAX_PAGES: usize = 8;
pub type GetObjectSdkError = Box<SdkError<GetObjectError>>;
pub type GetObjectTaggingSdkError = Box<SdkError<GetObjectTaggingError>>;
pub type PutObjectTaggingSdkError = Box<SdkError<PutObjectTaggingError>>;
@@ -365,13 +349,6 @@ struct TargetClientBuildProbe {
/// their import path while the verdict vocabulary lives with the
/// replication decision logic.
pub use crate::bucket::replication::SsecPassthroughCapability;
/// Version-identity verdicts (see the enum's own docs in
/// `rustfs-replication`) are cached here per target ARN and follow the same
/// `arn_remotes_map` lifecycle. They carry no TTL: the verdict is refreshed
/// by every replication write's response, so it can only go stale on a
/// target that receives no writes — and a stale `MintsOwn` costs one extra
/// content-identity lookup before a PUT, never a lost replica.
pub use crate::bucket::replication::VersionIdentityCapability;
/// How long an audited SSE-C passthrough verdict stays authoritative.
///
@@ -398,11 +375,6 @@ pub struct BucketTargetSys {
/// SSE-C passthrough capability verdicts keyed by target ARN. See
/// [`SsecPassthroughCapability`]; reset alongside `arn_remotes_map`.
ssec_passthrough_map: Arc<RwLock<HashMap<String, SsecPassthroughRecord>>>,
/// Version-identity verdicts keyed by target ARN. See
/// [`VersionIdentityCapability`]; reset alongside `arn_remotes_map`. A std
/// lock (never held across an await) so the replication worker can record
/// a verdict from inside its synchronous PUT-response audit.
version_identity_map: Arc<std::sync::RwLock<HashMap<String, VersionIdentityCapability>>>,
pub targets_map: Arc<RwLock<HashMap<String, Vec<BucketTarget>>>>,
/// Buckets whose persisted `bucket-targets.json` exists but cannot be
/// decoded (rustfs/backlog#2282). Written under the bucket's update mutex
@@ -451,7 +423,6 @@ impl BucketTargetSys {
Self {
arn_remotes_map: Arc::new(RwLock::new(HashMap::new())),
ssec_passthrough_map: Arc::new(RwLock::new(HashMap::new())),
version_identity_map: Arc::new(std::sync::RwLock::new(HashMap::new())),
targets_map: Arc::new(RwLock::new(HashMap::new())),
unreadable_targets: Arc::new(RwLock::new(HashSet::new())),
h_mutex: Arc::new(RwLock::new(HashMap::new())),
@@ -775,40 +746,10 @@ impl BucketTargetSys {
arn_remotes_map.remove(&target.arn);
health_map.remove(&target.arn);
ssec_map.remove(&target.arn);
self.forget_version_identity_capability(&target.arn);
}
}
}
/// Cached version-identity verdict for a target ARN; `Unknown` until a
/// replication write or a replication-check VersionFidelity probe judged
/// it since the target was built.
pub fn version_identity_capability(&self, arn: &str) -> VersionIdentityCapability {
self.version_identity_map
.read()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.get(arn)
.copied()
.unwrap_or_default()
}
/// Record a version-identity verdict for a target ARN. Written by the
/// replication worker after every PutObject / CompleteMultipartUpload
/// response and by the replication-check VersionFidelity phase.
pub fn record_version_identity_capability(&self, arn: &str, capability: VersionIdentityCapability) {
self.version_identity_map
.write()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.insert(arn.to_string(), capability);
}
fn forget_version_identity_capability(&self, arn: &str) {
self.version_identity_map
.write()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.remove(arn);
}
/// Cached SSE-C passthrough capability for a target ARN, plus whether the
/// verdict is older than [`SSEC_PASSTHROUGH_CAPABILITY_TTL`]. `(Unknown,
/// false)` when no verdict has been recorded since the target was built.
@@ -1221,32 +1162,12 @@ impl BucketTargetSys {
// Remove existing targets
if let Some(existing_targets) = targets_map.remove(bucket) {
let mut ssec_map = self.ssec_passthrough_map.write().await;
let unchanged_service: HashMap<&str, &BucketTarget> = targets
.map(|new_targets| {
new_targets
.targets
.iter()
.map(|target| (target.arn.as_str(), target))
.collect()
})
.unwrap_or_default();
for target in existing_targets {
arn_remotes_map.remove(&target.arn);
health_map.remove(&target.arn);
// A rebuilt/edited target may point at a different service:
// the SSE-C passthrough verdict must be re-audited from Unknown.
ssec_map.remove(&target.arn);
// The version-identity verdict survives an edit that keeps the
// same remote service (a resync start or a bandwidth change
// rewrites the entry in place): forgetting it there would make
// the very resync that follows re-drive every object as a
// duplicate on a target that mints its own version ids.
if unchanged_service
.get(target.arn.as_str())
.is_none_or(|edited| !same_replication_service(edited, &target))
{
self.forget_version_identity_capability(&target.arn);
}
self.update_bandwidth_limit(bucket, &target.arn, 0);
}
}
@@ -1971,62 +1892,6 @@ impl TargetClient {
.map_err(Box::new)
}
/// Locate a replica by content identity on a target that mints its own
/// version ids: page `ListObjectVersions` under the exact key and return
/// the newest live version whose ETag matches `source_etag`. Delete
/// markers and prefix siblings never match. Bounded to
/// [`FIND_VERSION_BY_ETAG_MAX_PAGES`] pages so a key with a very deep
/// history cannot turn one convergence check into an unbounded scan; a
/// replica beyond that window reads as missing, which only costs a
/// re-PUT (today's behaviour), never a lost object.
pub async fn find_version_by_etag(
&self,
bucket: &str,
object: &str,
source_etag: &str,
) -> Result<Option<String>, Box<SdkError<aws_sdk_s3::operation::list_object_versions::ListObjectVersionsError>>> {
let mut key_marker: Option<String> = None;
let mut version_id_marker: Option<String> = None;
for _ in 0..FIND_VERSION_BY_ETAG_MAX_PAGES {
let page = self
.client
.list_object_versions()
.bucket(bucket)
.prefix(object)
.max_keys(FIND_VERSION_BY_ETAG_PAGE_SIZE)
.set_key_marker(key_marker.take())
.set_version_id_marker(version_id_marker.take())
.send()
.await
.map_err(Box::new)?;
if let Some(version) = page.versions().iter().find(|version| {
version.key() == Some(object)
&& version.version_id().is_some_and(|id| !id.is_empty())
&& replication_etags_match(Some(source_etag), version.e_tag())
}) {
return Ok(version.version_id().map(str::to_string));
}
// Every listed key is >= the prefix; once the listing moved past
// the exact key there is nothing left to find.
if page
.versions()
.iter()
.any(|version| version.key().is_some_and(|key| key > object))
{
return Ok(None);
}
if !page.is_truncated().unwrap_or(false) {
return Ok(None);
}
key_marker = page.next_key_marker().map(str::to_string);
version_id_marker = page.next_version_id_marker().map(str::to_string);
if key_marker.is_none() {
return Ok(None);
}
}
Ok(None)
}
/// HEAD used by the read-proxy path (GET/HEAD of an object not yet
/// replicated locally, MinIO `proxyHeadToRepTarget`).
///
@@ -3356,64 +3221,6 @@ mod tests {
assert!(message.contains("connection refused"));
}
#[test]
fn same_replication_service_ignores_resync_and_bandwidth_edits() {
let base = BucketTarget {
endpoint: "target.example:9000".to_string(),
target_bucket: "replica".to_string(),
secure: true,
path: "on".to_string(),
arn: "arn:rustfs:replication:us-east-1:bucket:same".to_string(),
credentials: Some(Credentials {
access_key: "access".to_string(),
..Default::default()
}),
..Default::default()
};
let resync_edit = BucketTarget {
reset_id: "reset-1".to_string(),
bandwidth_limit: 1024,
..base.clone()
};
assert!(same_replication_service(&resync_edit, &base));
for moved in [
BucketTarget {
endpoint: "other.example:9000".to_string(),
..base.clone()
},
BucketTarget {
target_bucket: "other".to_string(),
..base.clone()
},
BucketTarget {
secure: false,
..base.clone()
},
BucketTarget {
credentials: Some(Credentials {
access_key: "rotated".to_string(),
..Default::default()
}),
..base.clone()
},
] {
assert!(!same_replication_service(&moved, &base));
}
}
#[test]
fn version_identity_verdict_is_per_arn_and_forgotten_with_the_target() {
let sys = BucketTargetSys::default();
let arn = "arn:rustfs:replication:us-east-1:bucket:identity";
assert_eq!(sys.version_identity_capability(arn), VersionIdentityCapability::Unknown);
sys.record_version_identity_capability(arn, VersionIdentityCapability::MintsOwn);
assert_eq!(sys.version_identity_capability(arn), VersionIdentityCapability::MintsOwn);
assert_eq!(sys.version_identity_capability("other"), VersionIdentityCapability::Unknown);
// A rebuilt target may point at a different service.
sys.forget_version_identity_capability(arn);
assert_eq!(sys.version_identity_capability(arn), VersionIdentityCapability::Unknown);
}
#[test]
fn endpoint_health_key_preserves_explicit_port() {
let url = Url::parse("https://remote.example:9443").expect("url should parse");
@@ -41,21 +41,6 @@ where
com::read_config(api, file).await
}
pub(crate) async fn read_config_limited_preserve_empty<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
where
S: ObjectIO<
Error = Error,
RangeSpec = HTTPRangeSpec,
HeaderMap = HeaderMap,
ObjectOptions = ObjectOptions,
ObjectInfo = ObjectInfo,
GetObjectReader = GetObjectReader,
PutObjectReader = PutObjReader,
>,
{
com::read_config_limited_preserve_empty(api, file, max_bytes).await
}
pub(crate) async fn read_config_with_metadata<S>(api: Arc<S>, file: &str, opts: &ObjectOptions) -> Result<(Vec<u8>, ObjectInfo)>
where
S: ObjectIO<
@@ -71,26 +56,6 @@ where
com::read_config_with_metadata(api, file, opts).await
}
pub(crate) async fn read_config_limited_preserve_empty_with_metadata<S>(
api: Arc<S>,
file: &str,
opts: &ObjectOptions,
max_bytes: usize,
) -> Result<(Vec<u8>, ObjectInfo)>
where
S: ObjectIO<
Error = Error,
RangeSpec = HTTPRangeSpec,
HeaderMap = HeaderMap,
ObjectOptions = ObjectOptions,
ObjectInfo = ObjectInfo,
GetObjectReader = GetObjectReader,
PutObjectReader = PutObjReader,
>,
{
com::read_config_limited_preserve_empty_with_metadata_opts(api, file, opts, max_bytes).await
}
pub(crate) async fn save_config<S>(api: Arc<S>, file: &str, data: Vec<u8>) -> Result<()>
where
S: ObjectIO<
@@ -22,7 +22,7 @@ use super::{
bucket_lifecycle_ops::{
ManualTransitionQueueSnapshot, ManualTransitionRunReport, decode_manual_transition_continuation_token,
},
manual_transition_job, recovery_control, recovery_disposition, recovery_export, tier_delete_journal, transition_transaction,
manual_transition_job, recovery_control, tier_delete_journal, transition_transaction,
};
use crate::error::{Error, Result};
use crate::services::tier::tier_probe_intent;
@@ -42,8 +42,6 @@ pub(crate) enum DurableIlmRecordKind {
ManualTransitionTask,
ManualTransitionWorkerResult,
RecoveryControl,
RecoveryExport,
RecoveryDisposition,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
@@ -114,20 +112,8 @@ pub(crate) const RECOVERY_CONTROL_NAMESPACE: DurableIlmNamespace = DurableIlmNam
max_record_size: recovery_control::MAX_ILM_RECOVERY_CONTROL_SIZE,
kind: DurableIlmRecordKind::RecoveryControl,
};
pub(crate) const RECOVERY_EXPORT_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
name: "recovery-export",
prefix: recovery_export::ILM_RECOVERY_EXPORT_PREFIX,
max_record_size: recovery_export::MAX_ILM_RECOVERY_EXPORT_SIZE,
kind: DurableIlmRecordKind::RecoveryExport,
};
pub(crate) const RECOVERY_DISPOSITION_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
name: "recovery-disposition",
prefix: recovery_disposition::ILM_RECOVERY_DISPOSITION_PREFIX,
max_record_size: recovery_disposition::MAX_ILM_RECOVERY_DISPOSITION_SIZE,
kind: DurableIlmRecordKind::RecoveryDisposition,
};
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 12] = [
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 10] = [
TIER_DELETE_JOURNAL_NAMESPACE,
TIER_DELETE_JOURNAL_V6_NAMESPACE,
TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE,
@@ -138,8 +124,6 @@ pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 12] = [
MANUAL_TRANSITION_TASK_NAMESPACE,
MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE,
RECOVERY_CONTROL_NAMESPACE,
RECOVERY_EXPORT_NAMESPACE,
RECOVERY_DISPOSITION_NAMESPACE,
];
#[derive(Debug, Clone, PartialEq, Eq)]
@@ -277,28 +261,6 @@ pub(crate) enum DurableIlmRecordCheckpoint {
#[serde(default, skip_serializing_if = "Option::is_none")]
owner_fence_sha256: Option<String>,
},
RecoveryExport {
content_sha256: String,
source_generation_sha256: String,
topology_generation: String,
member_epochs_sha256: String,
creator_sha256: String,
retain_until_unix_nanos: i64,
},
RecoveryDisposition {
content_sha256: String,
identity_sha256: String,
copy_manifest_sha256: String,
copy_manifest_count: usize,
created_at_unix_nanos: i64,
revision: u64,
state: recovery_disposition::IlmRecoveryDispositionState,
owner_fence_sha256: Option<String>,
owner_lease_acquired_at_unix_nanos: Option<i64>,
owner_lease_expires_at_unix_nanos: Option<i64>,
confirmed_absent_sha256: Vec<String>,
retain_until_unix_nanos: i64,
},
}
impl DurableIlmRecordCheckpoint {
@@ -313,9 +275,7 @@ impl DurableIlmRecordCheckpoint {
| Self::ManualTransitionScope { content_sha256, .. }
| Self::ManualTransitionTask { content_sha256 }
| Self::ManualTransitionWorkerResult { content_sha256 }
| Self::RecoveryControl { content_sha256, .. }
| Self::RecoveryExport { content_sha256, .. }
| Self::RecoveryDisposition { content_sha256, .. } => content_sha256,
| Self::RecoveryControl { content_sha256, .. } => content_sha256,
}
}
@@ -356,9 +316,6 @@ impl DurableIlmRecordCheckpoint {
{
return Err(Error::other("durable ILM tier delete journal checkpoint is invalid"));
}
if !recovery_disposition_checkpoint_is_valid(checkpoint) {
return Err(Error::other("durable ILM recovery disposition checkpoint is invalid"));
}
}
if self == next {
if let Self::ManualTransitionJob {
@@ -637,83 +594,6 @@ impl DurableIlmRecordCheckpoint {
&& previous_attempts == next_attempts;
adjacent && (claim || source_refresh || completion)
}
(
Self::RecoveryDisposition {
identity_sha256: previous_identity,
copy_manifest_sha256: previous_manifest,
copy_manifest_count: previous_manifest_count,
created_at_unix_nanos: previous_created_at,
revision: previous_revision,
state: previous_state,
owner_fence_sha256: previous_owner,
owner_lease_acquired_at_unix_nanos: previous_owner_acquired,
owner_lease_expires_at_unix_nanos: previous_owner_expires,
confirmed_absent_sha256: previous_confirmed,
retain_until_unix_nanos: previous_retain_until,
..
},
Self::RecoveryDisposition {
identity_sha256: next_identity,
copy_manifest_sha256: next_manifest,
copy_manifest_count: next_manifest_count,
created_at_unix_nanos: next_created_at,
revision: next_revision,
state: next_state,
owner_fence_sha256: next_owner,
owner_lease_acquired_at_unix_nanos: next_owner_acquired,
owner_lease_expires_at_unix_nanos: next_owner_expires,
confirmed_absent_sha256: next_confirmed,
retain_until_unix_nanos: next_retain_until,
..
},
) => {
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
let immutable_identity_matches = previous_identity == next_identity
&& previous_manifest == next_manifest
&& previous_manifest_count == next_manifest_count
&& previous_created_at == next_created_at
&& previous_retain_until == next_retain_until;
let adjacent = previous_revision.checked_add(1) == Some(*next_revision);
let progress_is_monotonic = sorted_sha256_set_is_subset(previous_confirmed, next_confirmed);
let legal_edge = match (previous_state, next_state) {
(Prepared, Prepared) => {
let claim = previous_owner.is_none() && next_owner.is_some();
let takeover = previous_owner.is_some()
&& previous_owner != next_owner
&& previous_owner_expires
.zip(*next_owner_acquired)
.is_some_and(|(expires, acquired)| acquired >= expires);
previous_confirmed == next_confirmed && (claim || takeover)
}
(Prepared, Applying) => {
previous_confirmed == next_confirmed
&& previous_owner.is_some()
&& previous_owner == next_owner
&& previous_owner_acquired == next_owner_acquired
&& previous_owner_expires == next_owner_expires
}
(Applying, Applying) => {
let progress = previous_owner == next_owner
&& previous_owner_acquired == next_owner_acquired
&& previous_owner_expires == next_owner_expires
&& previous_confirmed.len().checked_add(1) == Some(next_confirmed.len());
let takeover = previous_owner.is_some()
&& previous_owner != next_owner
&& previous_confirmed == next_confirmed
&& previous_owner_expires
.zip(*next_owner_acquired)
.is_some_and(|(expires, acquired)| acquired >= expires);
progress || takeover
}
(Applying, Completed) => {
previous_owner.is_some() && next_owner.is_none() && previous_confirmed == next_confirmed
}
_ => false,
};
immutable_identity_matches && adjacent && progress_is_monotonic && legal_edge
}
_ => false,
};
@@ -731,11 +611,6 @@ impl DurableIlmRecordCheckpoint {
/// after the exact terminal ETag and terminal receipt were committed, to
/// purge older object versions exposed by that deletion.
pub(crate) fn is_predecessor_of_terminal(&self, terminal: &Self) -> bool {
for checkpoint in [self, terminal] {
if !recovery_disposition_checkpoint_is_valid(checkpoint) {
return false;
}
}
if let Self::TierProbeIntent { state, .. } = terminal
&& !matches!(
state,
@@ -752,11 +627,6 @@ impl DurableIlmRecordCheckpoint {
{
return false;
}
if let Self::RecoveryDisposition { state, .. } = terminal
&& state != &recovery_disposition::IlmRecoveryDispositionState::Completed
{
return false;
}
if self == terminal || self.validate_successor(terminal).is_ok() {
return true;
}
@@ -882,121 +752,11 @@ impl DurableIlmRecordCheckpoint {
&& terminal_revision > previous_revision
&& terminal_attempts >= previous_attempts
}
(
Self::RecoveryDisposition {
identity_sha256: previous_identity,
copy_manifest_sha256: previous_manifest,
copy_manifest_count: previous_manifest_count,
created_at_unix_nanos: previous_created_at,
revision: previous_revision,
state: previous_state,
owner_fence_sha256: previous_owner,
confirmed_absent_sha256: previous_confirmed,
retain_until_unix_nanos: previous_retain_until,
..
},
Self::RecoveryDisposition {
identity_sha256: terminal_identity,
copy_manifest_sha256: terminal_manifest,
copy_manifest_count: terminal_manifest_count,
created_at_unix_nanos: terminal_created_at,
revision: terminal_revision,
state: recovery_disposition::IlmRecoveryDispositionState::Completed,
confirmed_absent_sha256: terminal_confirmed,
retain_until_unix_nanos: terminal_retain_until,
..
},
) => {
matches!(
previous_state,
recovery_disposition::IlmRecoveryDispositionState::Prepared
| recovery_disposition::IlmRecoveryDispositionState::Applying
) && previous_identity == terminal_identity
&& previous_manifest == terminal_manifest
&& previous_manifest_count == terminal_manifest_count
&& previous_created_at == terminal_created_at
&& previous_retain_until == terminal_retain_until
&& terminal_revision.checked_sub(*previous_revision).is_some_and(|distance| {
let minimum_distance = match previous_state {
recovery_disposition::IlmRecoveryDispositionState::Prepared if previous_owner.is_some() => 3,
recovery_disposition::IlmRecoveryDispositionState::Prepared => 4,
recovery_disposition::IlmRecoveryDispositionState::Applying
if previous_confirmed.len() == *previous_manifest_count =>
{
1
}
recovery_disposition::IlmRecoveryDispositionState::Applying => 2,
recovery_disposition::IlmRecoveryDispositionState::Completed => u64::MAX,
};
distance >= minimum_distance
})
&& sorted_sha256_set_is_subset(previous_confirmed, terminal_confirmed)
}
_ => false,
}
}
}
fn recovery_disposition_checkpoint_is_valid(checkpoint: &DurableIlmRecordCheckpoint) -> bool {
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
let DurableIlmRecordCheckpoint::RecoveryDisposition {
content_sha256,
identity_sha256,
copy_manifest_sha256,
copy_manifest_count,
created_at_unix_nanos,
revision,
state,
owner_fence_sha256,
owner_lease_acquired_at_unix_nanos,
owner_lease_expires_at_unix_nanos,
confirmed_absent_sha256,
retain_until_unix_nanos,
} = checkpoint
else {
return true;
};
let owner_fence_sha256 = owner_fence_sha256.as_deref();
is_canonical_sha256(content_sha256)
&& is_canonical_sha256(identity_sha256)
&& is_canonical_sha256(copy_manifest_sha256)
&& *copy_manifest_count > 0
&& *created_at_unix_nanos > 0
&& *revision > 0
&& *retain_until_unix_nanos > 0
&& owner_fence_sha256.is_none_or(is_canonical_sha256)
&& match (
owner_fence_sha256,
*owner_lease_acquired_at_unix_nanos,
*owner_lease_expires_at_unix_nanos,
) {
(None, None, None) => true,
(Some(_), Some(acquired), Some(expires)) => acquired > 0 && expires > acquired,
_ => false,
}
&& confirmed_absent_sha256.len() <= *copy_manifest_count
&& confirmed_absent_sha256.iter().all(|digest| is_canonical_sha256(digest))
&& confirmed_absent_sha256.windows(2).all(|pair| pair[0] < pair[1])
&& match *state {
Prepared => confirmed_absent_sha256.is_empty(),
Applying => owner_fence_sha256.is_some(),
Completed => owner_fence_sha256.is_none() && confirmed_absent_sha256.len() == *copy_manifest_count,
}
}
fn is_canonical_sha256(value: &str) -> bool {
is_sha256_checksum(value)
&& !value
.bytes()
.any(|byte| byte.is_ascii_hexdigit() && byte.is_ascii_uppercase())
}
fn sorted_sha256_set_is_subset(subset: &[String], superset: &[String]) -> bool {
subset.iter().all(|candidate| superset.binary_search(candidate).is_ok())
}
fn tier_delete_dispatch_parent_progress_delta(
previous_sequence: u64,
previous_completed_journals: u64,
@@ -1588,55 +1348,6 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
},
)
}
DurableIlmRecordKind::RecoveryExport => {
let (protocol, export_id) = recovery_export::recovery_export_id_from_record_object_name(path)?;
let export = recovery_export::IlmRecoveryExport::decode(&export_id, data)?;
let canonical = recovery_export::recovery_export_record_object_name(protocol, &export_id)?;
if canonical != path || export.protocol != protocol {
return Err(Error::other("ILM recovery export path is not canonical"));
}
let source_generation_sha256 = checkpoint_hash(&export.source_generation)?;
(
"export_id",
export_id,
DurableIlmRecordCheckpoint::RecoveryExport {
content_sha256,
source_generation_sha256,
topology_generation: export.topology_generation,
member_epochs_sha256: export.member_epochs_sha256,
creator_sha256: export.creator_sha256,
retain_until_unix_nanos: export.retain_until_unix_nanos,
},
)
}
DurableIlmRecordKind::RecoveryDisposition => {
// The disposition module owns strict schema, checksum, canonical
// path, immutable-manifest, and state-specific validation. Keep
// this boundary limited to decommission identity/checkpoint
// projection so the two readers cannot accept different records.
let disposition = recovery_disposition::decode_recovery_disposition_checkpoint(path, data)?;
if disposition.content_sha256 != content_sha256 {
return Err(Error::other("ILM recovery disposition checkpoint content digest is invalid"));
}
(
"disposition_id",
disposition.disposition_id,
DurableIlmRecordCheckpoint::RecoveryDisposition {
content_sha256: disposition.content_sha256,
identity_sha256: disposition.identity_sha256,
copy_manifest_sha256: disposition.copy_manifest_sha256,
copy_manifest_count: disposition.copy_manifest_count,
created_at_unix_nanos: disposition.created_at_unix_nanos,
revision: disposition.revision,
state: disposition.state,
owner_fence_sha256: disposition.owner_fence_sha256,
owner_lease_acquired_at_unix_nanos: disposition.owner_lease_acquired_at_unix_nanos,
owner_lease_expires_at_unix_nanos: disposition.owner_lease_expires_at_unix_nanos,
confirmed_absent_sha256: disposition.confirmed_absent_sha256,
retain_until_unix_nanos: disposition.retain_until_unix_nanos,
},
)
}
DurableIlmRecordKind::ManualTransitionJob => {
let job_id = manual_transition_job::manual_transition_job_id_from_record_object_name(path)
.map_err(|err| Error::other(err.to_string()))?;
@@ -1792,203 +1503,6 @@ mod tests {
}
}
fn recovery_disposition_checkpoint(
revision: u64,
state: recovery_disposition::IlmRecoveryDispositionState,
owner_fence: Option<&str>,
confirmed_absent_sha256: Vec<String>,
) -> DurableIlmRecordCheckpoint {
let (owner_lease_acquired_at_unix_nanos, owner_lease_expires_at_unix_nanos) = match owner_fence {
Some("f") => (Some(10), Some(20)),
Some(_) => (Some(1), Some(10)),
None => (None, None),
};
DurableIlmRecordCheckpoint::RecoveryDisposition {
content_sha256: format!("{revision:064x}"),
identity_sha256: "a".repeat(64),
copy_manifest_sha256: "d".repeat(64),
copy_manifest_count: 2,
created_at_unix_nanos: 1_700_000_000_000_000_000,
revision,
state,
owner_fence_sha256: owner_fence.map(|digest| digest.repeat(64)),
owner_lease_acquired_at_unix_nanos,
owner_lease_expires_at_unix_nanos,
confirmed_absent_sha256,
retain_until_unix_nanos: 1_820_000_000_000_000_000,
}
}
#[test]
fn recovery_disposition_namespace_is_registered_without_shadowing_its_root() {
let disposition_id = "a".repeat(64);
let path = format!(
"{}/tier_delete_journal/{}/{}/{}.json",
recovery_disposition::ILM_RECOVERY_DISPOSITION_PREFIX,
&disposition_id[..2],
&disposition_id[2..4],
disposition_id
);
let namespace = classify_durable_ilm_record(&path)
.expect("recovery disposition path should classify")
.expect("recovery disposition should be durable");
assert_eq!(namespace, &RECOVERY_DISPOSITION_NAMESPACE);
assert!(classify_durable_ilm_record(recovery_disposition::ILM_RECOVERY_DISPOSITION_PREFIX).is_err());
}
#[test]
fn recovery_disposition_checkpoint_accepts_only_monotonic_progress() {
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
let first_copy = "b".repeat(64);
let second_copy = "c".repeat(64);
let prepared = recovery_disposition_checkpoint(1, Prepared, None, Vec::new());
let claimed = recovery_disposition_checkpoint(2, Prepared, Some("e"), Vec::new());
let applying = recovery_disposition_checkpoint(3, Applying, Some("e"), Vec::new());
let first_absent = recovery_disposition_checkpoint(4, Applying, Some("e"), vec![first_copy.clone()]);
let taken_over = recovery_disposition_checkpoint(5, Applying, Some("f"), vec![first_copy.clone()]);
let all_absent = recovery_disposition_checkpoint(6, Applying, Some("f"), vec![first_copy.clone(), second_copy.clone()]);
let completed = recovery_disposition_checkpoint(7, Completed, None, vec![first_copy.clone(), second_copy.clone()]);
prepared
.validate_successor(&claimed)
.expect("Prepared should record an owner claim without absence progress");
claimed
.validate_successor(&applying)
.expect("Prepared should advance to Applying without folding in deletion progress");
applying
.validate_successor(&first_absent)
.expect("Applying should append newly confirmed absent copies");
first_absent
.validate_successor(&taken_over)
.expect("Applying should record a fenced owner takeover without losing progress");
taken_over
.validate_successor(&all_absent)
.expect("Applying should preserve every earlier confirmation while making progress");
all_absent
.validate_successor(&completed)
.expect("a fully confirmed manifest should advance to Completed");
assert!(
prepared.validate_successor(&completed).is_err(),
"adjacent receipt updates must not skip Applying"
);
assert!(
first_absent
.validate_successor(&recovery_disposition_checkpoint(5, Applying, Some("e"), Vec::new()))
.is_err(),
"confirmed-absent progress must not move backwards"
);
assert!(
applying
.validate_successor(&recovery_disposition_checkpoint(4, Completed, None, vec![first_copy.clone()]))
.is_err(),
"Completed must cover the complete immutable copy manifest"
);
assert!(
completed
.validate_successor(&recovery_disposition_checkpoint(7, Applying, Some("e"), vec![second_copy]))
.is_err(),
"Completed is terminal"
);
assert!(
first_absent
.validate_successor(&recovery_disposition_checkpoint(5, Applying, Some("e"), vec![first_copy.clone()]))
.is_err(),
"a same-state revision bump must change the owner fence or absence progress"
);
assert!(
applying
.validate_successor(&recovery_disposition_checkpoint(4, Applying, None, vec![first_copy]))
.is_err(),
"Applying must retain a fenced owner"
);
let mut noncanonical_identity = claimed.clone();
if let DurableIlmRecordCheckpoint::RecoveryDisposition { identity_sha256, .. } = &mut noncanonical_identity {
*identity_sha256 = "A".repeat(64);
}
assert!(prepared.validate_successor(&noncanonical_identity).is_err());
let mut changed_created_at = claimed;
if let DurableIlmRecordCheckpoint::RecoveryDisposition {
created_at_unix_nanos, ..
} = &mut changed_created_at
{
*created_at_unix_nanos += 1;
}
assert!(prepared.validate_successor(&changed_created_at).is_err());
let mut early_takeover = taken_over;
if let DurableIlmRecordCheckpoint::RecoveryDisposition {
owner_lease_acquired_at_unix_nanos,
..
} = &mut early_takeover
{
*owner_lease_acquired_at_unix_nanos = Some(9);
}
assert!(first_absent.validate_successor(&early_takeover).is_err());
assert!(
applying
.validate_successor(&recovery_disposition_checkpoint(
4,
Applying,
Some("e"),
vec!["c".repeat(64), "b".repeat(64)],
))
.is_err(),
"confirmed-absent entries must be a canonical sorted set"
);
}
#[test]
fn recovery_disposition_terminal_predecessor_requires_exact_identity_and_full_manifest() {
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
let first_copy = "b".repeat(64);
let second_copy = "c".repeat(64);
let prepared = recovery_disposition_checkpoint(1, Prepared, None, Vec::new());
let applying = recovery_disposition_checkpoint(3, Applying, Some("e"), vec![first_copy.clone()]);
let completed = recovery_disposition_checkpoint(5, Completed, None, vec![first_copy.clone(), second_copy]);
assert!(prepared.is_predecessor_of_terminal(&completed));
assert!(applying.is_predecessor_of_terminal(&completed));
assert!(
!prepared.is_predecessor_of_terminal(&recovery_disposition_checkpoint(2, Applying, Some("e"), Vec::new())),
"a nonterminal disposition must not authorize terminal cleanup"
);
assert!(
!prepared.is_predecessor_of_terminal(&recovery_disposition_checkpoint(
4,
Completed,
None,
vec![first_copy.clone(), "c".repeat(64)],
)),
"terminal proof must leave enough revisions for claim, apply, progress, and completion"
);
assert!(
!applying.is_predecessor_of_terminal(&recovery_disposition_checkpoint(
4,
Completed,
None,
vec![first_copy.clone(), "c".repeat(64)],
)),
"an incomplete Applying checkpoint cannot complete without a progress generation"
);
let mut other_identity = completed;
if let DurableIlmRecordCheckpoint::RecoveryDisposition { identity_sha256, .. } = &mut other_identity {
*identity_sha256 = "e".repeat(64);
}
assert!(!prepared.is_predecessor_of_terminal(&other_identity));
let incomplete_terminal = recovery_disposition_checkpoint(4, Completed, None, vec![first_copy]);
assert!(
!prepared.is_predecessor_of_terminal(&incomplete_terminal),
"a partial confirmed-absent set must not become terminal proof"
);
}
fn tier_probe_intent_fixture() -> tier_probe_intent::TierProbeIntent {
let probe_id = Uuid::parse_str("36e2220e-9ad2-495b-b3bc-c4d2caf70a31").expect("fixture uuid should parse");
tier_probe_intent::TierProbeIntent {
@@ -25,8 +25,6 @@ mod object_handlers_common;
mod object_lock_boundary;
pub use self::core as lifecycle;
pub mod recovery_control;
pub mod recovery_disposition;
pub mod recovery_export;
mod replication_sink;
pub mod rule;
mod runtime_boundary;
@@ -168,7 +168,7 @@ impl IlmRecoverySourceGeneration {
Ok(generation)
}
pub(crate) fn validate(&self) -> Result<()> {
fn validate(&self) -> Result<()> {
if self.source_schema.trim().is_empty() {
return Err(IlmRecoveryControlError::Corrupt("source schema is empty"));
}
File diff suppressed because it is too large Load Diff
@@ -1,840 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use std::{collections::HashSet, sync::Arc};
use rustfs_utils::crypto::{hex_sha256, is_sha256_checksum};
use serde::{Deserialize, Serialize};
use super::config_boundary;
use super::recovery_control::{
IlmRecoveryClassification, IlmRecoveryControl, IlmRecoveryProtocol, IlmRecoverySourceCopy, IlmRecoverySourceGeneration,
MAX_ILM_RECOVERY_CONTROL_SIZE, ObservedIlmRecoveryControl, ObservedIlmRecoverySource, recovery_control_record_object_name,
};
use super::tier_delete_journal::{
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA, validate_legacy_tier_delete_recovery_source,
};
use crate::disk::RUSTFS_META_BUCKET;
use crate::error::{Error, Result};
use crate::object_api::{ObjectOptions, WriteCompletion};
use crate::services::notification_sys::{
acquire_ilm_recovery_export_fleet_proof, ilm_recovery_export_fleet_proof_matches, ilm_recovery_export_member_epochs_sha256,
ilm_recovery_export_topology_generation,
};
use crate::storage_api_contracts::{list::ListOperations as _, namespace::NamespaceLocking as _, object::HTTPPreconditions};
use crate::store::ECStore;
pub const ILM_RECOVERY_EXPORT_SCHEMA: &str = "rustfs-ilm-recovery-export-v1";
pub const ILM_RECOVERY_EXPORT_PREFIX: &str = "ilm/recovery-exports";
pub const MAX_ILM_RECOVERY_EXPORT_SIZE: usize = 128 * 1024;
const MAX_ILM_RECOVERY_EXPORTS: usize = 10_000;
const MAX_ILM_RECOVERY_EXPORT_BYTES: u64 = 1024 * 1024 * 1024;
const MAX_ACTOR_EXPORTS_PER_MINUTE: usize = 10;
const MAX_CLUSTER_EXPORTS_PER_MINUTE: usize = 100;
const EXPORT_RETENTION_NANOS: i64 = 90 * 24 * 60 * 60 * 1_000_000_000;
const EXPORT_ADMISSION_LOCK: &str = "ilm/recovery-admission/export.lock";
const MAX_LEGACY_TIER_DELETE_SOURCE_SIZE: usize = 64 * 1024;
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct IlmRecoveryExportObservation {
pub control_id: String,
pub protocol: IlmRecoveryProtocol,
pub control_etag: String,
pub control_revision: u64,
pub classification: IlmRecoveryClassification,
pub canonical_source_path: String,
pub source_generation: IlmRecoverySourceGeneration,
pub topology_generation: String,
pub member_epochs_sha256: String,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct IlmRecoveryExport {
pub export_id: String,
pub control_id: String,
pub protocol: IlmRecoveryProtocol,
pub control_etag: String,
pub control_revision: u64,
pub classification: IlmRecoveryClassification,
pub canonical_source_path: String,
pub source_generation: IlmRecoverySourceGeneration,
pub topology_generation: String,
pub member_epochs_sha256: String,
pub creator_sha256: String,
pub created_at_unix_nanos: i64,
pub retain_until_unix_nanos: i64,
pub source_bytes_base64: String,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
struct PersistedIlmRecoveryExport {
schema: String,
content_sha256: String,
export: IlmRecoveryExport,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
pub struct IlmRecoveryExportCreated {
pub export_id: String,
pub content_sha256: String,
pub encoded: Vec<u8>,
pub replayed: bool,
}
impl IlmRecoveryExport {
fn validate(&self) -> Result<()> {
self.source_generation.validate().map_err(Error::other)?;
validate_sha256(&self.export_id, "ILM recovery export ID is invalid")?;
validate_sha256(&self.control_id, "ILM recovery export control ID is invalid")?;
validate_sha256(&self.topology_generation, "ILM recovery export topology generation is invalid")?;
validate_sha256(&self.member_epochs_sha256, "ILM recovery export member epoch digest is invalid")?;
validate_sha256(&self.creator_sha256, "ILM recovery export creator digest is invalid")?;
if self.protocol != IlmRecoveryProtocol::TierDeleteJournal
|| self.classification != IlmRecoveryClassification::RetainedAmbiguous
|| !is_legacy_export_schema(&self.source_generation.source_schema)
{
return Err(Error::other("ILM recovery export source is not an exportable legacy journal"));
}
if self.control_etag.trim().is_empty() || self.control_revision == 0 {
return Err(Error::other("ILM recovery export control generation is invalid"));
}
if self.canonical_source_path.is_empty()
|| self.canonical_source_path.starts_with('/')
|| self.canonical_source_path.ends_with('/')
|| self.canonical_source_path.split('/').any(str::is_empty)
{
return Err(Error::other("ILM recovery export source path is invalid"));
}
if self.created_at_unix_nanos <= 0
|| self.retain_until_unix_nanos < self.created_at_unix_nanos.saturating_add(EXPORT_RETENTION_NANOS)
{
return Err(Error::other("ILM recovery export retention is invalid"));
}
let source = base64_simd::STANDARD
.decode_to_vec(self.source_bytes_base64.as_bytes())
.map_err(|_| Error::other("ILM recovery export source encoding is invalid"))?;
validate_legacy_tier_delete_recovery_source(&self.canonical_source_path, &self.source_generation.source_schema, &source)?;
let encoded_len = u64::try_from(source.len()).map_err(|_| Error::other("ILM recovery export source length overflow"))?;
if source.is_empty()
|| source.len() > MAX_LEGACY_TIER_DELETE_SOURCE_SIZE
|| hex_sha256(&source, ToOwned::to_owned) != self.source_generation.content_sha256
|| self.source_generation.copies.iter().any(|copy| {
copy.canonical_path != self.canonical_source_path
|| copy.etag != self.source_generation.source_etag
|| copy.content_sha256 != self.source_generation.content_sha256
|| copy.encoded_len != encoded_len
})
{
return Err(Error::other("ILM recovery export source bytes do not match the observed generation"));
}
if recovery_export_id(&self.control_id, &self.source_generation)? != self.export_id {
return Err(Error::other("ILM recovery export ID does not match its source generation"));
}
Ok(())
}
pub fn encode(&self) -> Result<Vec<u8>> {
self.validate()?;
let export_bytes = serde_json::to_vec(self).map_err(Error::other)?;
let persisted = PersistedIlmRecoveryExport {
schema: ILM_RECOVERY_EXPORT_SCHEMA.to_string(),
content_sha256: hex_sha256(&export_bytes, ToOwned::to_owned),
export: self.clone(),
};
let encoded = serde_json::to_vec(&persisted).map_err(Error::other)?;
if encoded.len() > MAX_ILM_RECOVERY_EXPORT_SIZE {
return Err(Error::other("encoded ILM recovery export exceeds maximum size"));
}
Ok(encoded)
}
pub fn decode(expected_export_id: &str, data: &[u8]) -> Result<Self> {
validate_sha256(expected_export_id, "ILM recovery export ID is invalid")?;
if data.len() > MAX_ILM_RECOVERY_EXPORT_SIZE {
return Err(Error::other("encoded ILM recovery export exceeds maximum size"));
}
let persisted: PersistedIlmRecoveryExport = serde_json::from_slice(data).map_err(Error::other)?;
if persisted.schema != ILM_RECOVERY_EXPORT_SCHEMA {
return Err(Error::other("ILM recovery export schema is unsupported"));
}
validate_sha256(&persisted.content_sha256, "ILM recovery export checksum is invalid")?;
let export_bytes = serde_json::to_vec(&persisted.export).map_err(Error::other)?;
if hex_sha256(&export_bytes, ToOwned::to_owned) != persisted.content_sha256 {
return Err(Error::other("ILM recovery export checksum mismatch"));
}
persisted.export.validate()?;
if persisted.export.export_id != expected_export_id {
return Err(Error::other("ILM recovery export ID does not match record key"));
}
Ok(persisted.export)
}
}
pub fn recovery_export_record_object_name(protocol: IlmRecoveryProtocol, export_id: &str) -> Result<String> {
validate_sha256(export_id, "ILM recovery export ID is invalid")?;
Ok(format!(
"{}/{}/{}/{}/{}.json",
ILM_RECOVERY_EXPORT_PREFIX,
protocol.as_str(),
&export_id[..2],
&export_id[2..4],
export_id
))
}
pub fn recovery_export_id_from_record_object_name(object: &str) -> Result<(IlmRecoveryProtocol, String)> {
let suffix = object
.strip_prefix(ILM_RECOVERY_EXPORT_PREFIX)
.and_then(|suffix| suffix.strip_prefix('/'))
.ok_or_else(|| Error::other("ILM recovery export path has wrong prefix"))?;
let mut parts = suffix.split('/');
let protocol = match parts.next() {
Some("tier_delete_journal") => IlmRecoveryProtocol::TierDeleteJournal,
_ => return Err(Error::other("ILM recovery export protocol is invalid")),
};
let shard_a = parts
.next()
.ok_or_else(|| Error::other("ILM recovery export path is incomplete"))?;
let shard_b = parts
.next()
.ok_or_else(|| Error::other("ILM recovery export path is incomplete"))?;
let export_id = parts
.next()
.and_then(|name| name.strip_suffix(".json"))
.ok_or_else(|| Error::other("ILM recovery export suffix is invalid"))?;
if parts.next().is_some() {
return Err(Error::other("ILM recovery export path is not canonical"));
}
validate_sha256(export_id, "ILM recovery export ID is invalid")?;
if shard_a != &export_id[..2] || shard_b != &export_id[2..4] {
return Err(Error::other("ILM recovery export shard does not match export ID"));
}
Ok((protocol, export_id.to_string()))
}
pub async fn inspect_recovery_export_observation(api: Arc<ECStore>, control_id: &str) -> Result<IlmRecoveryExportObservation> {
let proof = acquire_ilm_recovery_export_fleet_proof()
.await
.ok_or_else(|| Error::other("ILM recovery export fleet proof is unavailable"))?;
let observed_control = load_exportable_control(api.clone(), control_id).await?;
let observed_source = observe_export_source(
api,
&observed_control.control.identity.canonical_source_path,
&observed_control.control.observed_source_generation.source_schema,
)
.await?;
if !observed_source.is_consistent()
|| observed_source.generation != observed_control.control.observed_source_generation
|| !ilm_recovery_export_fleet_proof_matches(&proof).await
{
return Err(Error::other("ILM recovery export observation changed or is incomplete"));
}
Ok(IlmRecoveryExportObservation {
control_id: control_id.to_string(),
protocol: observed_control.control.identity.protocol,
control_etag: observed_control.etag,
control_revision: observed_control.control.revision,
classification: observed_control.control.classification,
canonical_source_path: observed_control.control.identity.canonical_source_path,
source_generation: observed_source.generation,
topology_generation: ilm_recovery_export_topology_generation(&proof),
member_epochs_sha256: ilm_recovery_export_member_epochs_sha256(&proof),
})
}
pub async fn create_recovery_export(
api: Arc<ECStore>,
observation: &IlmRecoveryExportObservation,
creator_sha256: &str,
) -> Result<IlmRecoveryExportCreated> {
validate_sha256(creator_sha256, "ILM recovery export creator digest is invalid")?;
let lock = api.new_ns_lock(RUSTFS_META_BUCKET, EXPORT_ADMISSION_LOCK).await?;
let admission_guard = lock.get_write_lock(crate::set_disk::get_lock_acquire_timeout()).await?;
let proof = acquire_ilm_recovery_export_fleet_proof()
.await
.ok_or_else(|| Error::other("ILM recovery export fleet proof is unavailable"))?;
if ilm_recovery_export_topology_generation(&proof) != observation.topology_generation
|| ilm_recovery_export_member_epochs_sha256(&proof) != observation.member_epochs_sha256
{
return Err(Error::PreconditionFailed);
}
let control_object = recovery_control_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, &observation.control_id)
.map_err(Error::other)?;
let control_lock = api.new_ns_lock(RUSTFS_META_BUCKET, &control_object).await?;
let control_guard = control_lock
.get_read_lock(crate::set_disk::get_lock_acquire_timeout())
.await?;
let source_lock = api
.new_ns_lock(RUSTFS_META_BUCKET, &observation.canonical_source_path)
.await?;
let source_guard = source_lock.get_read_lock(crate::set_disk::get_lock_acquire_timeout()).await?;
let locks_current = || !admission_guard.is_lock_lost() && !control_guard.is_lock_lost() && !source_guard.is_lock_lost();
let (current, current_source_bytes) = current_observation_under_proof_no_lock(api.clone(), observation, &proof).await?;
if &current != observation || !locks_current() {
return Err(Error::PreconditionFailed);
}
let current_source_base64 = base64_simd::STANDARD.encode_to_string(current_source_bytes);
let candidate_export_id = recovery_export_id(&current.control_id, &current.source_generation)?;
let object = recovery_export_record_object_name(current.protocol, &candidate_export_id)?;
match load_recovery_export_decoded(api.clone(), &candidate_export_id).await {
Ok((existing, export)) if export_matches_observation(&export, observation) => {
if !locks_current() || !ilm_recovery_export_fleet_proof_matches(&proof).await {
return Err(Error::PreconditionFailed);
}
api.record_durable_ilm_decommission_progress(&object, &existing.encoded)
.await?;
if !locks_current() {
return Err(Error::PreconditionFailed);
}
return Ok(existing.with_replayed());
}
Ok(_) => return Err(Error::PreconditionFailed),
Err(Error::ConfigNotFound) => {}
Err(err) => return Err(err),
}
let inventory = collect_export_inventory(api.clone()).await?;
if !locks_current() || !ilm_recovery_export_fleet_proof_matches(&proof).await {
return Err(Error::PreconditionFailed);
}
let created_at_unix_nanos = now_unix_nanos()?;
let export = build_export_from_source(&current, creator_sha256, created_at_unix_nanos, &current_source_base64)?;
let encoded = export.encode()?;
inventory.check(creator_sha256, encoded.len(), created_at_unix_nanos)?;
let mut write_options = ObjectOptions {
max_parity: true,
write_completion: WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()),
..Default::default()
}),
..Default::default()
};
write_options.add_namespace_lock_guard(&admission_guard);
write_options.add_namespace_lock_guard(&control_guard);
write_options.add_namespace_lock_guard(&source_guard);
if !locks_current() {
return Err(Error::PreconditionFailed);
}
let write_result = config_boundary::save_config_with_opts(api.clone(), &object, encoded.clone(), &write_options).await;
let stored = match load_recovery_export(api.clone(), &export.export_id).await {
Ok(stored) if stored.encoded == encoded => stored,
Ok(_) => return Err(Error::PreconditionFailed),
Err(read_err) => return Err(write_result.err().unwrap_or(read_err)),
};
if !locks_current() || !ilm_recovery_export_fleet_proof_matches(&proof).await {
return Err(Error::PreconditionFailed);
}
api.record_durable_ilm_decommission_progress(&object, &encoded).await?;
if !locks_current() {
return Err(Error::PreconditionFailed);
}
Ok(stored)
}
pub async fn load_recovery_export(api: Arc<ECStore>, export_id: &str) -> Result<IlmRecoveryExportCreated> {
let (created, _) = load_recovery_export_decoded(api, export_id).await?;
Ok(created)
}
async fn load_recovery_export_decoded(
api: Arc<ECStore>,
export_id: &str,
) -> Result<(IlmRecoveryExportCreated, IlmRecoveryExport)> {
let object = recovery_export_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, export_id)?;
let encoded = config_boundary::read_config_limited_preserve_empty(api, &object, MAX_ILM_RECOVERY_EXPORT_SIZE).await?;
let export = IlmRecoveryExport::decode(export_id, &encoded)?;
let content_sha256 = hex_sha256(&encoded, ToOwned::to_owned);
Ok((
IlmRecoveryExportCreated {
export_id: export.export_id.clone(),
content_sha256,
encoded,
replayed: false,
},
export,
))
}
impl IlmRecoveryExportCreated {
fn with_replayed(mut self) -> Self {
self.replayed = true;
self
}
}
async fn load_exportable_control(api: Arc<ECStore>, control_id: &str) -> Result<ObservedIlmRecoveryControl> {
load_exportable_control_with_options(api, control_id, &ObjectOptions::default()).await
}
async fn load_exportable_control_no_lock(api: Arc<ECStore>, control_id: &str) -> Result<ObservedIlmRecoveryControl> {
load_exportable_control_with_options(
api,
control_id,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
}
async fn load_exportable_control_with_options(
api: Arc<ECStore>,
control_id: &str,
options: &ObjectOptions,
) -> Result<ObservedIlmRecoveryControl> {
let object = recovery_control_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, control_id).map_err(Error::other)?;
let (data, metadata) =
config_boundary::read_config_limited_preserve_empty_with_metadata(api, &object, options, MAX_ILM_RECOVERY_CONTROL_SIZE)
.await?;
let etag = metadata
.etag
.filter(|etag| !etag.trim().is_empty())
.ok_or_else(|| Error::other("ILM recovery control is missing an ETag"))?;
let control = IlmRecoveryControl::decode(control_id, &data).map_err(Error::other)?;
if control.identity.protocol != IlmRecoveryProtocol::TierDeleteJournal
|| control.classification != IlmRecoveryClassification::RetainedAmbiguous
|| !is_legacy_export_schema(&control.observed_source_generation.source_schema)
{
return Err(Error::other("ILM recovery control is not exportable"));
}
Ok(ObservedIlmRecoveryControl { control, etag })
}
async fn current_observation_under_proof_no_lock(
api: Arc<ECStore>,
expected: &IlmRecoveryExportObservation,
proof: &crate::services::notification_sys::IlmRecoveryExportFleetProofToken,
) -> Result<(IlmRecoveryExportObservation, Vec<u8>)> {
let observed_control = load_exportable_control_no_lock(api.clone(), &expected.control_id).await?;
let observed_source = observe_export_source_no_lock(
api,
&observed_control.control.identity.canonical_source_path,
&observed_control.control.observed_source_generation.source_schema,
)
.await?;
let source_bytes = observed_source
.canonical_data
.clone()
.ok_or_else(|| Error::other("ILM recovery export source copies diverge"))?;
if !observed_source.is_consistent()
|| observed_source.generation != observed_control.control.observed_source_generation
|| !ilm_recovery_export_fleet_proof_matches(proof).await
{
return Err(Error::PreconditionFailed);
}
Ok((
IlmRecoveryExportObservation {
control_id: expected.control_id.clone(),
protocol: observed_control.control.identity.protocol,
control_etag: observed_control.etag,
control_revision: observed_control.control.revision,
classification: observed_control.control.classification,
canonical_source_path: observed_control.control.identity.canonical_source_path,
source_generation: observed_source.generation,
topology_generation: ilm_recovery_export_topology_generation(proof),
member_epochs_sha256: ilm_recovery_export_member_epochs_sha256(proof),
},
source_bytes,
))
}
async fn observe_export_source(
api: Arc<ECStore>,
canonical_path: &str,
source_schema: &str,
) -> Result<ObservedIlmRecoverySource> {
if canonical_path.is_empty()
|| canonical_path.starts_with('/')
|| canonical_path.ends_with('/')
|| canonical_path.split('/').any(str::is_empty)
|| !is_legacy_export_schema(source_schema)
{
return Err(Error::other("ILM recovery export source identity is invalid"));
}
let lock = api.new_ns_lock(RUSTFS_META_BUCKET, canonical_path).await?;
let _guard = lock.get_read_lock(crate::set_disk::get_lock_acquire_timeout()).await?;
observe_export_source_no_lock(api, canonical_path, source_schema).await
}
async fn observe_export_source_no_lock(
api: Arc<ECStore>,
canonical_path: &str,
source_schema: &str,
) -> Result<ObservedIlmRecoverySource> {
let mut copies = Vec::new();
let mut canonical: Option<(String, String, Vec<u8>)> = None;
let mut consistent = true;
for set in api.all_set_disks() {
let authority = format!("pool-{}/set-{}", set.pool_index, set.set_index);
let result = config_boundary::read_config_limited_preserve_empty_with_metadata(
set,
canonical_path,
&ObjectOptions {
no_lock: true,
..Default::default()
},
MAX_LEGACY_TIER_DELETE_SOURCE_SIZE,
)
.await;
match result {
Ok((data, metadata)) => {
if data.is_empty() || data.len() > MAX_LEGACY_TIER_DELETE_SOURCE_SIZE {
return Err(Error::other("ILM recovery export source exceeds its protocol size limit"));
}
validate_legacy_tier_delete_recovery_source(canonical_path, source_schema, &data)?;
let etag = metadata
.etag
.filter(|etag| !etag.trim().is_empty())
.ok_or_else(|| Error::other("ILM recovery export source copy is missing an ETag"))?;
let content_sha256 = hex_sha256(&data, ToOwned::to_owned);
let encoded_len =
u64::try_from(data.len()).map_err(|_| Error::other("ILM recovery export source length does not fit u64"))?;
copies.push(IlmRecoverySourceCopy {
authority,
canonical_path: canonical_path.to_string(),
etag: etag.clone(),
encoded_len,
content_sha256: content_sha256.clone(),
});
match canonical.as_ref() {
Some((first_etag, first_digest, first_data)) => {
consistent &= first_etag == &etag && first_digest == &content_sha256 && first_data == &data;
}
None => canonical = Some((etag, content_sha256, data)),
}
}
Err(err) if export_source_is_missing(&err) => {}
Err(err) => return Err(err),
}
}
let Some((source_etag, content_sha256, source_bytes)) = canonical else {
return Err(Error::ConfigNotFound);
};
let generation =
IlmRecoverySourceGeneration::new(source_schema, source_etag, content_sha256, copies).map_err(Error::other)?;
Ok(ObservedIlmRecoverySource {
generation,
canonical_data: consistent.then_some(source_bytes),
})
}
fn export_source_is_missing(err: &Error) -> bool {
matches!(
err,
Error::ConfigNotFound | Error::FileNotFound | Error::ObjectNotFound(_, _) | Error::VersionNotFound(_, _, _)
)
}
fn build_export_from_source(
observation: &IlmRecoveryExportObservation,
creator_sha256: &str,
created_at_unix_nanos: i64,
source_bytes_base64: &str,
) -> Result<IlmRecoveryExport> {
let retain_until_unix_nanos = created_at_unix_nanos
.checked_add(EXPORT_RETENTION_NANOS)
.ok_or_else(|| Error::other("ILM recovery export retention timestamp overflow"))?;
let export = IlmRecoveryExport {
export_id: recovery_export_id(&observation.control_id, &observation.source_generation)?,
control_id: observation.control_id.clone(),
protocol: observation.protocol,
control_etag: observation.control_etag.clone(),
control_revision: observation.control_revision,
classification: observation.classification,
canonical_source_path: observation.canonical_source_path.clone(),
source_generation: observation.source_generation.clone(),
topology_generation: observation.topology_generation.clone(),
member_epochs_sha256: observation.member_epochs_sha256.clone(),
creator_sha256: creator_sha256.to_string(),
created_at_unix_nanos,
retain_until_unix_nanos,
source_bytes_base64: source_bytes_base64.to_string(),
};
export.validate()?;
Ok(export)
}
pub(crate) fn recovery_export_id(control_id: &str, generation: &IlmRecoverySourceGeneration) -> Result<String> {
validate_sha256(control_id, "ILM recovery export control ID is invalid")?;
validate_sha256(&generation.content_sha256, "ILM recovery export source checksum is invalid")?;
validate_sha256(&generation.copy_set_sha256, "ILM recovery export copy-set checksum is invalid")?;
let mut data = Vec::new();
for part in [control_id, &generation.content_sha256, &generation.copy_set_sha256] {
data.extend_from_slice(&(part.len() as u64).to_be_bytes());
data.extend_from_slice(part.as_bytes());
}
Ok(hex_sha256(&data, ToOwned::to_owned))
}
fn export_matches_observation(export: &IlmRecoveryExport, observation: &IlmRecoveryExportObservation) -> bool {
export.control_id == observation.control_id
&& export.protocol == observation.protocol
&& export.classification == observation.classification
&& export.canonical_source_path == observation.canonical_source_path
&& export.source_generation == observation.source_generation
}
#[derive(Debug, Default)]
struct IlmRecoveryExportInventory {
count: usize,
bytes: u64,
creations: Vec<(i64, String)>,
}
impl IlmRecoveryExportInventory {
fn check(&self, creator_sha256: &str, candidate_len: usize, now: i64) -> Result<()> {
let recent_after = now.saturating_sub(60 * 1_000_000_000);
let cluster_recent = self
.creations
.iter()
.filter(|(created_at, _)| *created_at > recent_after)
.count();
let actor_recent = self
.creations
.iter()
.filter(|(created_at, creator)| *created_at > recent_after && creator == creator_sha256)
.count();
check_export_admission(self.count, self.bytes, actor_recent, cluster_recent, candidate_len)
}
}
async fn collect_export_inventory(api: Arc<ECStore>) -> Result<IlmRecoveryExportInventory> {
let mut marker = None;
let mut seen_markers = HashSet::new();
let mut inventory = IlmRecoveryExportInventory::default();
loop {
let page = api
.clone()
.list_objects_v2(
RUSTFS_META_BUCKET,
&format!("{ILM_RECOVERY_EXPORT_PREFIX}/"),
marker.clone(),
None,
1_000,
false,
None,
false,
)
.await?;
for object in page.objects {
let (_, export_id) = recovery_export_id_from_record_object_name(&object.name)?;
let (stored, export) = load_recovery_export_decoded(api.clone(), &export_id).await?;
inventory.count = inventory
.count
.checked_add(1)
.ok_or_else(|| Error::other("ILM recovery export count overflow"))?;
inventory.bytes = inventory
.bytes
.checked_add(u64::try_from(stored.encoded.len()).map_err(|_| Error::other("ILM recovery export size overflow"))?)
.ok_or_else(|| Error::other("ILM recovery export byte total overflow"))?;
inventory
.creations
.push((export.created_at_unix_nanos, export.creator_sha256));
}
if !page.is_truncated {
break;
}
let next = page
.next_continuation_token
.ok_or_else(|| Error::other("ILM recovery export inventory omitted its continuation marker"))?;
marker = Some(record_export_inventory_marker(&mut seen_markers, next)?);
}
Ok(inventory)
}
fn record_export_inventory_marker(seen_markers: &mut HashSet<String>, next: String) -> Result<String> {
if !seen_markers.insert(next.clone()) {
return Err(Error::other("ILM recovery export inventory repeated its continuation marker"));
}
Ok(next)
}
fn check_export_admission(
count: usize,
bytes: u64,
actor_recent: usize,
cluster_recent: usize,
candidate_len: usize,
) -> Result<()> {
let candidate_len = u64::try_from(candidate_len).map_err(|_| Error::other("ILM recovery export size does not fit u64"))?;
if count >= MAX_ILM_RECOVERY_EXPORTS
|| bytes
.checked_add(candidate_len)
.is_none_or(|total| total > MAX_ILM_RECOVERY_EXPORT_BYTES)
|| actor_recent >= MAX_ACTOR_EXPORTS_PER_MINUTE
|| cluster_recent >= MAX_CLUSTER_EXPORTS_PER_MINUTE
{
return Err(Error::SlowDown);
}
Ok(())
}
fn is_legacy_export_schema(schema: &str) -> bool {
matches!(schema, TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA | TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA)
}
fn validate_sha256(value: &str, message: &'static str) -> Result<()> {
if !is_sha256_checksum(value)
|| value
.bytes()
.any(|byte| byte.is_ascii_hexdigit() && byte.is_ascii_uppercase())
{
return Err(Error::other(message));
}
Ok(())
}
fn now_unix_nanos() -> Result<i64> {
i64::try_from(time::OffsetDateTime::now_utc().unix_timestamp_nanos())
.map_err(|_| Error::other("ILM recovery export timestamp does not fit i64"))
}
#[cfg(test)]
mod tests {
use super::*;
use crate::bucket::lifecycle::recovery_control::IlmRecoverySourceCopy;
const PINNED_V1_EXPORT: &[u8] = br#"{"schema":"rustfs-ilm-recovery-export-v1","content_sha256":"3dfb3ec3892256e909de1211c1a963ca7008963ff32b3a869f7161a7b9b44028","export":{"export_id":"2b78e7a825bfc2edbf7f773d0b6ed3bf93e360ff1702d73a449109c11bfaa105","control_id":"0fcd568a5cb9bdb4677b69354b11ee415af8f784519cff3da49a26f84eaee7f2","protocol":"tier_delete_journal","control_etag":"control-etag","control_revision":1,"classification":"retained_ambiguous","canonical_source_path":"ilm/tier-delete-journal/872072554f66ab326f10ce7adbae11422b7a4b0663aa7112d6061a8f6ed41b94.json","source_generation":{"source_schema":"rustfs-tier-delete-journal-v1","source_etag":"etag-a","content_sha256":"0e0b010ebdeeb7b41473fe8575e989d6bb1303c0ca551dd984e9400f0ae306bd","copy_set_sha256":"5a7406115b6c3923ffe79dcd1f43ccae7beed786e557163f019dd10ec409a653","copies":[{"authority":"pool-0/set-0","canonical_path":"ilm/tier-delete-journal/872072554f66ab326f10ce7adbae11422b7a4b0663aa7112d6061a8f6ed41b94.json","etag":"etag-a","encoded_len":81,"content_sha256":"0e0b010ebdeeb7b41473fe8575e989d6bb1303c0ca551dd984e9400f0ae306bd"}]},"topology_generation":"e6e2b826e31fca5c36125c48f130dcb6f961e698ff8a8776a1f290cf0892e8e6","member_epochs_sha256":"612dd8a861161819a4ad8f6f3e2a0567602877c043a2353ca933a13c78dc0ed4","creator_sha256":"50c9c4aeb40b5b206b6d98f516f8b8c0efd29ce2e56a76b345fb9240c225a1b7","created_at_unix_nanos":1000000000,"retain_until_unix_nanos":7776001000000000,"source_bytes_base64":"eyJ2ZXJzaW9uIjoxLCJvYmpfbmFtZSI6ImxlZ2FjeS9yZW1vdGUiLCJ2ZXJzaW9uX2lkIjoib3BhcXVlIiwidGllcl9uYW1lIjoiV0FSTSJ9"}}"#;
fn legacy_source() -> Vec<u8> {
br#"{"version":1,"obj_name":"legacy/remote","version_id":"opaque","tier_name":"WARM"}"#.to_vec()
}
fn observation() -> IlmRecoveryExportObservation {
let source = legacy_source();
let source_path = super::super::tier_delete_journal::tier_delete_journal_object_name(
&super::super::tier_delete_journal::decode_tier_delete_journal_entry(&source).expect("legacy fixture should decode"),
);
let source_sha256 = hex_sha256(&source, ToOwned::to_owned);
let generation = IlmRecoverySourceGeneration::new(
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA,
"etag-a",
source_sha256.clone(),
vec![IlmRecoverySourceCopy {
authority: "pool-0/set-0".to_string(),
canonical_path: source_path.clone(),
etag: "etag-a".to_string(),
encoded_len: source.len() as u64,
content_sha256: source_sha256,
}],
)
.expect("generation should be valid");
IlmRecoveryExportObservation {
control_id: hex_sha256(b"control", ToOwned::to_owned),
protocol: IlmRecoveryProtocol::TierDeleteJournal,
control_etag: "control-etag".to_string(),
control_revision: 1,
classification: IlmRecoveryClassification::RetainedAmbiguous,
canonical_source_path: source_path,
source_generation: generation,
topology_generation: hex_sha256(b"topology", ToOwned::to_owned),
member_epochs_sha256: hex_sha256(b"epochs", ToOwned::to_owned),
}
}
#[test]
fn recovery_export_round_trip_is_strict_and_deterministic() {
let observed = observation();
let creator = hex_sha256(b"actor", ToOwned::to_owned);
let export = build_export_from_source(
&observed,
&creator,
1_000_000_000,
&base64_simd::STANDARD.encode_to_string(legacy_source()),
)
.expect("export should be valid");
assert_eq!(
export.export_id,
recovery_export_id(&observed.control_id, &observed.source_generation).unwrap()
);
let encoded = export.encode().expect("export should encode");
assert_eq!(encoded, PINNED_V1_EXPORT, "v1 export wire format must remain pinned");
assert_eq!(IlmRecoveryExport::decode(&export.export_id, &encoded).unwrap(), export);
assert_eq!(
IlmRecoveryExport::decode("2b78e7a825bfc2edbf7f773d0b6ed3bf93e360ff1702d73a449109c11bfaa105", PINNED_V1_EXPORT)
.unwrap(),
export,
);
let path = recovery_export_record_object_name(export.protocol, &export.export_id).unwrap();
let durable = super::super::durable_namespace::validate_durable_ilm_record(&path, &encoded)
.expect("export should be registered as a durable ILM record");
assert_eq!(durable.namespace, "recovery-export");
assert_eq!(durable.id_kind, "export_id");
assert_eq!(durable.id, export.export_id);
let mut wrong_source = export.clone();
wrong_source.source_bytes_base64 = base64_simd::STANDARD.encode_to_string(b"changed");
assert!(wrong_source.encode().is_err());
let mut persisted: serde_json::Value = serde_json::from_slice(&encoded).unwrap();
persisted["unknown"] = serde_json::json!(true);
assert!(IlmRecoveryExport::decode(&export.export_id, &serde_json::to_vec(&persisted).unwrap()).is_err());
}
#[test]
fn export_inventory_rejects_non_adjacent_continuation_cycles() {
let mut seen = HashSet::new();
assert_eq!(record_export_inventory_marker(&mut seen, "a".to_string()).unwrap(), "a");
assert_eq!(record_export_inventory_marker(&mut seen, "b".to_string()).unwrap(), "b");
record_export_inventory_marker(&mut seen, "a".to_string())
.expect_err("a non-adjacent continuation marker cycle must fail closed");
}
#[test]
fn recovery_export_path_rejects_noncanonical_shards() {
let id = hex_sha256(b"export", ToOwned::to_owned);
let path = recovery_export_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, &id).unwrap();
assert_eq!(recovery_export_id_from_record_object_name(&path).unwrap().1, id);
let wrong_shard = path.replacen(&format!("/{}/", &id[..2]), "/zz/", 1);
assert!(recovery_export_id_from_record_object_name(&wrong_shard).is_err());
}
#[test]
fn canonical_replay_survives_fleet_rotation_but_not_source_change() {
let observed = observation();
let creator = hex_sha256(b"actor", ToOwned::to_owned);
let export = build_export_from_source(
&observed,
&creator,
1_000_000_000,
&base64_simd::STANDARD.encode_to_string(legacy_source()),
)
.unwrap();
let mut rotated = observed;
rotated.control_etag = "new-control-etag".to_string();
rotated.control_revision += 1;
rotated.topology_generation = hex_sha256(b"new-topology", ToOwned::to_owned);
rotated.member_epochs_sha256 = hex_sha256(b"new-members", ToOwned::to_owned);
assert!(export_matches_observation(&export, &rotated));
rotated.source_generation.content_sha256 = hex_sha256(b"changed", ToOwned::to_owned);
assert!(!export_matches_observation(&export, &rotated));
}
#[test]
fn export_admission_enforces_exact_count_byte_and_rate_boundaries() {
assert!(check_export_admission(9_999, MAX_ILM_RECOVERY_EXPORT_BYTES - 1, 9, 99, 1).is_ok());
assert!(check_export_admission(10_000, 0, 0, 0, 1).is_err());
assert!(check_export_admission(0, MAX_ILM_RECOVERY_EXPORT_BYTES, 0, 0, 1).is_err());
assert!(check_export_admission(0, 0, 10, 0, 1).is_err());
assert!(check_export_admission(0, 0, 0, 100, 1).is_err());
}
}
@@ -82,8 +82,8 @@ const TIER_DELETE_DISPATCH_MEMBER_DELETE_CONCURRENCY: usize = 32;
const TIER_DELETE_DISPATCH_PREPARE_CONCURRENCY: usize = 16;
const TIER_DELETE_DISPATCH_CAS_CONCURRENCY: usize = 32;
const TIER_DELETE_JOURNAL_VERSION: u8 = 2;
pub(crate) const TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v1";
pub(crate) const TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v2";
const TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v1";
const TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v2";
const TIER_DELETE_JOURNAL_UNKNOWN_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-unknown";
const TIER_DELETE_JOURNAL_V1_RECOVERY_CLASS: &str = "tier_delete_journal_v1";
const TIER_DELETE_JOURNAL_V2_RECOVERY_CLASS: &str = "tier_delete_journal_v2";
@@ -884,23 +884,6 @@ struct PersistedTierDeleteJournalEntry {
}
impl PersistedTierDeleteJournalEntry {
fn validate_legacy_recovery_shape(&self) -> Result<()> {
let has_later_version_fields = self.version_id_exact.is_some()
|| self.version_state.is_some()
|| self.state.is_some()
|| self.source.is_some()
|| self.dispatch.is_some();
match self.version {
1 if self.backend_identity.is_none() && !has_later_version_fields => Ok(()),
TIER_DELETE_JOURNAL_VERSION if self.backend_identity.is_some() && !has_later_version_fields => Ok(()),
1 => Err(Error::other("tier delete journal v1 entry contains fields from a later version")),
TIER_DELETE_JOURNAL_VERSION => Err(Error::other(
"tier delete journal v2 entry is missing its identity or contains fields from a later version",
)),
_ => Err(Error::other("tier delete journal is not an exportable legacy version")),
}
}
fn from_jentry(je: &Jentry) -> Result<Self> {
validate_version_state(je.version_state, &je.version_id, je.version_id_exact)?;
let legacy_unknown = je.version_state == rustfs_filemeta::TransitionVersionState::Unknown;
@@ -5548,12 +5531,6 @@ fn canonical_legacy_tier_delete_journal_identity(object_name: &str) -> Option<&s
.then_some(identity)
}
pub(crate) fn validate_legacy_tier_delete_recovery_path(object_name: &str) -> Result<()> {
canonical_legacy_tier_delete_journal_identity(object_name)
.map(|_| ())
.ok_or_else(|| Error::other("legacy tier delete journal path is not canonical"))
}
fn legacy_tier_delete_recovery_descriptor(entry: &Jentry) -> Option<(&'static str, &'static str)> {
match entry.persisted_version {
1 => Some((TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V1_RECOVERY_CLASS)),
@@ -5562,21 +5539,6 @@ fn legacy_tier_delete_recovery_descriptor(entry: &Jentry) -> Option<(&'static st
}
}
pub(crate) fn validate_legacy_tier_delete_recovery_source(object_name: &str, source_schema: &str, data: &[u8]) -> Result<()> {
validate_legacy_tier_delete_recovery_path(object_name)?;
let persisted: PersistedTierDeleteJournalEntry =
serde_json::from_slice(data).map_err(|err| Error::other_with_context("decode tier delete journal failed", err))?;
persisted.validate_legacy_recovery_shape()?;
let entry = persisted.into_jentry()?;
let Some((decoded_schema, _)) = legacy_tier_delete_recovery_descriptor(&entry) else {
return Err(Error::other("tier delete journal is not an exportable legacy version"));
};
if decoded_schema != source_schema || tier_delete_journal_object_name(&entry) != object_name {
return Err(Error::other("legacy tier delete journal identity does not match its recovery source"));
}
Ok(())
}
fn legacy_tier_delete_control_matches(
control: &IlmRecoveryControl,
identity: &IlmRecoveryControlIdentity,
@@ -6178,18 +6140,17 @@ where
#[cfg(test)]
mod tests {
use super::{
PersistedTierDeleteJournalEntry, TIER_DELETE_DISPATCH_MANIFEST_VERSION, TIER_DELETE_DISPATCH_PARENT_RECORD_TYPE,
TIER_DELETE_DISPATCH_PARENT_VERSION, TIER_DELETE_JOURNAL_EXACT_VERSION, TIER_DELETE_JOURNAL_LEGACY_PREFIX,
TIER_DELETE_JOURNAL_SOLE_OWNER_VERSION, TIER_DELETE_JOURNAL_STATE_VERSION, TIER_DELETE_JOURNAL_TRANSACTION_VERSION,
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V6_PREFIX,
TIER_DELETE_JOURNAL_VERSION, TierDeleteDispatchChunkBinding, TierDeleteDispatchManifest, TierDeleteDispatchManifestState,
TierDeleteDispatchParent, TierDeleteDispatchParentState, TierDeleteDispatchRecord, await_tier_delete_journal_recovery,
TIER_DELETE_DISPATCH_MANIFEST_VERSION, TIER_DELETE_DISPATCH_PARENT_RECORD_TYPE, TIER_DELETE_DISPATCH_PARENT_VERSION,
TIER_DELETE_JOURNAL_EXACT_VERSION, TIER_DELETE_JOURNAL_LEGACY_PREFIX, TIER_DELETE_JOURNAL_SOLE_OWNER_VERSION,
TIER_DELETE_JOURNAL_STATE_VERSION, TIER_DELETE_JOURNAL_TRANSACTION_VERSION, TIER_DELETE_JOURNAL_V6_PREFIX,
TierDeleteDispatchChunkBinding, TierDeleteDispatchManifest, TierDeleteDispatchManifestState, TierDeleteDispatchParent,
TierDeleteDispatchParentState, TierDeleteDispatchRecord, await_tier_delete_journal_recovery,
decode_tier_delete_dispatch_record, decode_tier_delete_journal_entry, encode_tier_delete_dispatch_manifest,
encode_tier_delete_dispatch_parent, encode_tier_delete_journal_entry, object_info_references_tier_delete,
record_tier_delete_journal_backend_identity, same_tier_delete_authorization_identity, same_tier_delete_journal_identity,
tier_delete_dispatch_child_matches_parent, tier_delete_dispatch_chunk_manifest_object_name,
tier_delete_dispatch_journal_set_digest, tier_delete_dispatch_manifest_object_name, tier_delete_journal_object_name,
tier_delete_source_matches_dispatch_scope, validate_legacy_tier_delete_recovery_source,
tier_delete_source_matches_dispatch_scope,
};
use crate::bucket::lifecycle::tier_sweeper::{
Jentry, TierDeleteDispatchBinding, TierDeleteJournalState, TierDeleteSourceIdentity,
@@ -6648,72 +6609,6 @@ mod tests {
}
}
#[test]
fn legacy_recovery_export_rejects_fields_from_later_journal_versions() {
let later = bound_v6_journal_entry(TierDeleteJournalState::Prepared);
let v1 = PersistedTierDeleteJournalEntry {
version: 1,
obj_name: "remote/object".to_string(),
version_id: "opaque".to_string(),
tier_name: "WARM".to_string(),
backend_identity: None,
version_id_exact: None,
version_state: None,
state: None,
source: None,
dispatch: None,
};
let mut v2 = v1.clone();
v2.version = TIER_DELETE_JOURNAL_VERSION;
v2.backend_identity = Some([7; 32]);
let assert_rejected = |persisted: PersistedTierDeleteJournalEntry, schema: &str| {
let normalized = persisted
.clone()
.into_jentry()
.expect("the generic compatibility decoder should demonstrate the discarded field");
let object_name = tier_delete_journal_object_name(&normalized);
let encoded = serde_json::to_vec(&persisted).expect("mixed-version journal fixture should encode");
let err = validate_legacy_tier_delete_recovery_source(&object_name, schema, &encoded)
.expect_err("legacy recovery export must reject fields from later versions");
assert!(err.to_string().contains("later version"));
};
let mut invalid_v1 = Vec::new();
let mut with_backend = v1.clone();
with_backend.backend_identity = Some([7; 32]);
invalid_v1.push(with_backend);
for persisted in [&v1, &v2] {
let schema = if persisted.version == 1 {
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA
} else {
TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA
};
let mut invalid = Vec::new();
let mut with_exact = persisted.clone();
with_exact.version_id_exact = Some(false);
invalid.push(with_exact);
let mut with_version_state = persisted.clone();
with_version_state.version_state = Some(rustfs_filemeta::TransitionVersionState::Unknown);
invalid.push(with_version_state);
let mut with_state = persisted.clone();
with_state.state = Some(TierDeleteJournalState::Committed);
invalid.push(with_state);
let mut with_source = persisted.clone();
with_source.source = later.source.clone();
invalid.push(with_source);
let mut with_dispatch = persisted.clone();
with_dispatch.dispatch = later.dispatch.clone();
invalid.push(with_dispatch);
for record in invalid {
assert_rejected(record, schema);
}
}
for record in invalid_v1 {
assert_rejected(record, TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA);
}
}
#[test]
fn tier_delete_journal_path_is_stable_and_sanitized() {
let je = journal_entry();
@@ -66,7 +66,6 @@ pub(crate) use replication_lifecycle_bridge::ReplicationLifecycleBridge;
pub(crate) use replication_migration_bridge::ReplicationMigrationBridge;
pub use replication_object_bridge::ReplicationObjectBridge;
pub use replication_object_config::{DeleteReplicationConfigSnapshot, ReplicationConfig};
pub(crate) use replication_object_decision_boundary::replication_etags_match;
pub use replication_object_decision_boundary::{
MustReplicateOptions, ReplicationDeleteScheduleInput, ReplicationDeleteStateSource, delete_replication_state_from_config,
delete_replication_version_id, should_schedule_delete_replication, should_use_existing_delete_replication_info,
@@ -89,6 +88,5 @@ pub use replication_state::{ReplicationStats, RuntimeReplicationTargetBacklog};
pub use replication_stats_boundary::{BucketReplicationStat, BucketReplicationStats, BucketStats, InQueueMetric, XferStats};
pub use replication_storage_boundary::{ReplicationObjectIO, ReplicationStorage};
pub use replication_target_boundary::SsecPassthroughCapability;
pub use replication_target_boundary::VersionIdentityCapability;
pub use replication_target_boundary::{ObjectLockIntegrity, object_lock_put_integrity};
pub(crate) use replication_target_config_bridge::ReplicationTargetConfigBridge;
@@ -12,8 +12,6 @@
// See the License for the specific language governing permissions and
// limitations under the License.
#[cfg(test)]
pub(crate) use rustfs_filemeta::ObjectPartInfo;
pub use rustfs_replication::{MrfOpKind, MrfReplicateEntry};
pub(crate) use rustfs_replication::{
REPLICATE_EXISTING, REPLICATE_HEAL_DELETE, ReplicateTargetDecision, ReplicatedInfos, ReplicatedTargetInfo, ReplicationAction,
@@ -12,8 +12,6 @@
// See the License for the specific language governing permissions and
// limitations under the License.
#[cfg(test)]
pub(crate) use rustfs_replication::ReplicationMultipartPlanError;
pub use rustfs_replication::{
MustReplicateOptions, ReplicationDeleteScheduleInput, ReplicationDeleteStateSource, delete_replication_state_from_config,
delete_replication_version_id, should_schedule_delete_replication, should_use_existing_delete_replication_info,
@@ -56,8 +56,6 @@ use super::replication_storage_boundary::{
};
#[cfg(test)]
use super::replication_storage_boundary::{NamespaceLockFence, NamespaceLockSignalTestFence, ReplicationDeletedObject};
#[cfg(test)]
use super::replication_target_boundary::VersionIdentityCapability;
use super::replication_target_boundary::{
ERR_REPLICATION_SSEC_PASSTHROUGH_UNSUPPORTED, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions,
RemotePutObjectResponse, ReplicationTargetStore, S3ClientError, SsecPassthroughCapability, SsecPassthroughGate, TargetClient,
@@ -65,7 +63,7 @@ use super::replication_target_boundary::{
replication_delete_marker_purge_remove_options, replication_delete_remove_options, replication_force_delete_remove_options,
replication_object_is_ssec_encrypted, replication_put_object_header_size, replication_put_object_options,
replication_target_head_is_newer_null_version, resolve_read_api_version_id, ssec_passthrough_evidence_present,
ssec_passthrough_gate, version_identity_capability_from_put, version_identity_drifted,
ssec_passthrough_gate, version_identity_drifted,
};
use super::replication_versioning_boundary::ReplicationVersioningStore;
use super::runtime_boundary as runtime_sources;
@@ -125,7 +123,6 @@ const EVENT_DELETE_MARKER_PURGE_FAILED: &str = "replication_delete_marker_purge_
const EVENT_DELETE_MARKER_PURGE_MRF: &str = "replication_delete_marker_purge_mrf";
const METRIC_DELETE_MARKER_PURGE_TOTAL: &str = "rustfs_replication_delete_marker_purge_total";
const EVENT_REPLICATION_VERSION_IDENTITY_DRIFT: &str = "replication_version_identity_drift";
const EVENT_REPLICATION_DRIFTED_REPLICA_LOCATED: &str = "replication_drifted_replica_located";
const EVENT_REPLICATION_OBJECT_FAILED: &str = "replication_object_failed";
const EVENT_REPLICATION_PURGE_OBJECT_LOCK_DENIED: &str = "replication_purge_object_lock_denied";
@@ -335,12 +332,6 @@ fn verify_single_part_replica(
const REPLICA_ETAG_MISMATCH_ERROR: &str = "replica etag mismatch: the target persisted different bytes than were sent";
fn audit_target_version_identity(tgt_client: &TargetClient, source_version_id: &str, assigned_version_id: Option<&str>) {
// Every write refreshes the cached verdict, so the convergence fallback
// below (`replica_head_fallback`) knows whether a 404 on a
// version-addressed HEAD can mean "replica missing" on this target.
if let Some(capability) = version_identity_capability_from_put(source_version_id, assigned_version_id) {
ReplicationTargetStore::record_version_identity_capability(&tgt_client.arn, capability);
}
if !version_identity_drifted(source_version_id, assigned_version_id) {
return;
}
@@ -413,70 +404,11 @@ async fn head_object_fallback(
) -> std::result::Result<Option<HeadObjectOutput>, HeadObjectSdkError> {
match head_object_for_worker(tgt_client, &tgt_client.bucket, object, None).await {
Ok(oi) => Ok(Some(oi)),
Err(e) if head_object_not_found(&e) => Ok(None),
Err(e) if e.as_service_error().is_some_and(|se| se.is_not_found()) || has_raw_status(&e, 404) => Ok(None),
Err(e) => Err(e),
}
}
fn head_object_not_found(err: &HeadObjectSdkError) -> bool {
err.as_service_error().is_some_and(|se| se.is_not_found()) || has_raw_status(err, 404)
}
/// Second look at a replica whose version-addressed HEAD failed, for the two
/// target shapes where that failure is not a verdict on the replica:
///
/// - AWS-style 400/403 (the RustFS uuid is rejected as malformed): HEAD the
/// current version without a version id; callers compare ETags.
/// - 404 on a target known to mint its own version ids (the Wasabi shape,
/// rustfs/backlog#2340): the source id never existed there, so locate the
/// replica by exact key and ETag through ListObjectVersions and HEAD the id
/// the target assigned. Without this, every heal, MRF retry and
/// existing-object resync re-drive PUTs the object again and mints one
/// more target version.
///
/// `None` when the error stands as-is: a real miss on an adopting target, or
/// a target whose identity contract is still unknown. A failed lookup is
/// returned as a HEAD-shaped error so callers keep their "target operation
/// failed" handling (retry later) instead of re-driving the PUT.
async fn replica_head_fallback(
tgt_client: &TargetClient,
object: &str,
source_etag: Option<&str>,
err: &HeadObjectSdkError,
) -> Option<std::result::Result<Option<HeadObjectOutput>, HeadObjectSdkError>> {
if is_version_id_format_mismatch(err) {
return Some(head_object_fallback(tgt_client, object).await);
}
if !head_object_not_found(err)
|| !ReplicationTargetStore::version_identity_capability(&tgt_client.arn).version_addressing_unreliable()
{
return None;
}
let etag = source_etag.filter(|etag| !etag.trim().is_empty())?;
Some(match tgt_client.find_version_by_etag(&tgt_client.bucket, object, etag).await {
Ok(Some(assigned_version_id)) => {
debug!(
event = EVENT_REPLICATION_DRIFTED_REPLICA_LOCATED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
bucket = %tgt_client.bucket,
object = %object,
arn = %tgt_client.arn,
assigned_version_id = %assigned_version_id,
"Located replica by content identity on a target that mints its own version ids"
);
match head_object_for_worker(tgt_client, &tgt_client.bucket, object, Some(assigned_version_id)).await {
Ok(oi) => Ok(Some(oi)),
// The located version disappeared between LIST and HEAD.
Err(e) if head_object_not_found(&e) => Ok(None),
Err(e) => Err(e),
}
}
Ok(None) => Ok(None),
Err(list_err) => Err(Box::new(SdkError::construction_failure(*list_err))),
})
}
/// Resolve the N2 fail-closed gate for an SSE-C passthrough attempt against
/// this target. Returns `Some(audit_required)` when replication may proceed;
/// on a freshly-flagged header-dropping target it settles `rinfo` as FAILED
@@ -1468,26 +1400,31 @@ async fn verify_resync_head_result(
(0, None)
}
}
Err(err) => {
// A version-addressed HEAD is not the last word on every target:
// re-verify through the fallback before counting a well-replicated
// object as failed (see `replica_head_fallback`).
match replica_head_fallback(target_client.as_ref(), &roi.name, roi.etag.as_deref(), &err).await {
Some(Ok(Some(_))) => {
Err(err) if is_version_id_format_mismatch(&err) => {
// AWS-style target rejects the RustFS UUID versionId
// (400). Re-verify without the versionId before
// concluding the object failed to replicate, instead
// of counting a well-replicated object as failed.
match head_object_fallback(target_client.as_ref(), &roi.name).await {
Ok(Some(_)) => {
st.replicated_count += 1;
st.replicated_size += roi.size;
(roi.size, None)
}
Some(Ok(None)) | None => {
Ok(None) => {
st.failed_count += 1;
(0, Some(err))
}
Some(Err(e2)) => {
Err(e2) => {
st.failed_count += 1;
(0, Some(e2))
}
}
}
Err(err) => {
st.failed_count += 1;
(0, Some(err))
}
}
}
@@ -3660,8 +3597,11 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo {
}
}
Err(e) => {
if let Some(fallback) = replica_head_fallback(&tgt_client, &object, object_info.etag.as_deref(), &e).await {
match fallback {
if e.as_service_error().is_some_and(|se| se.is_not_found()) || has_raw_status(&e, 404) {
// Object not on target yet → fall through to PUT.
} else if is_version_id_format_mismatch(&e) {
// Version-ID format mismatch: retry without versionId and compare ETags.
match head_object_fallback(&tgt_client, &object).await {
Ok(Some(oi)) if replication_etags_match(object_info.etag.as_deref(), oi.e_tag.as_deref()) => {
if ssec_audit_required
&& !settle_ssec_passthrough_evidence(&oi, &tgt_client, &bucket, &object, &mut rinfo).await
@@ -3691,8 +3631,6 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo {
return rinfo;
}
}
} else if head_object_not_found(&e) {
// Object not on target yet → fall through to PUT.
} else {
rinfo.error = Some(e.to_string());
warn!(
@@ -4292,8 +4230,9 @@ async fn resolve_replicate_all_action(
}
}
Err(e) => {
if let Some(fallback) = replica_head_fallback(tgt_client, object, object_info.etag.as_deref(), &e).await {
match fallback {
if is_version_id_format_mismatch(&e) {
// Version-ID format mismatch: retry without versionId and compare ETags.
match head_object_fallback(tgt_client, object).await {
Ok(Some(oi)) => {
let etags_match = replication_etags_match(object_info.etag.as_deref(), oi.e_tag.as_deref());
if require_existing_target && !etags_match {
@@ -4345,7 +4284,7 @@ async fn resolve_replicate_all_action(
return None;
}
}
} else if head_object_not_found(&e) {
} else if e.as_service_error().is_some_and(|se| se.is_not_found()) || has_raw_status(&e, 404) {
if require_existing_target {
rinfo.error = Some("replica metadata target does not contain this object version".to_string());
rinfo.duration = (OffsetDateTime::now_utc() - start_time).unsigned_abs();
@@ -4717,58 +4656,6 @@ where
result
}
#[derive(Debug)]
struct MultipartReplicationReadPlan {
part_number: i32,
part_size: i64,
range: Option<HTTPRangeSpec>,
next_offset: i64,
}
fn multipart_replication_read_plan(
object_info: &ObjectInfo,
obj_opts: &ObjectOptions,
mut input: ReplicationMultipartPartInput,
stored_size: usize,
is_last: bool,
) -> std::io::Result<MultipartReplicationReadPlan> {
let empty_last_part = is_last && input.part_size == 0 && stored_size == 0;
// Raw reads address stored bytes. Only untransformed legacy parts may
// substitute their stored size for a missing logical size.
if obj_opts.raw_data_movement_read || (input.part_size == 0 && !object_info.is_compressed() && !object_info.is_encrypted()) {
input.part_size = i64::try_from(stored_size).map_err(|_| {
std::io::Error::new(std::io::ErrorKind::InvalidData, "multipart replication stored part size exceeds i64")
})?;
}
if empty_last_part {
if input.offset < 0 {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"empty multipart replication part has a negative offset",
));
}
let part_number = i32::try_from(input.part_number)
.map_err(|_| std::io::Error::new(std::io::ErrorKind::InvalidData, "multipart replication part number exceeds i32"))?;
return Ok(MultipartReplicationReadPlan {
part_number,
part_size: 0,
range: None,
next_offset: input.offset,
});
}
let plan = replication_multipart_part_plan(input).map_err(std::io::Error::other)?;
Ok(MultipartReplicationReadPlan {
part_number: plan.part_number,
part_size: plan.part_size,
range: Some(HTTPRangeSpec {
is_suffix_length: false,
start: plan.range.start,
end: plan.range.end,
}),
next_offset: plan.next_offset,
})
}
async fn replicate_multipart_parts_and_complete<S: ReplicationObjectIO>(
ctx: MultipartReplicationContext<'_, S>,
upload_id: &str,
@@ -4789,31 +4676,35 @@ async fn replicate_multipart_parts_and_complete<S: ReplicationObjectIO>(
let mut header_size = replication_put_object_header_size(&put_opts);
let mut offset: i64 = 0;
for (index, part_info) in object_info.parts.iter().enumerate() {
let part_plan = multipart_replication_read_plan(
object_info,
obj_opts,
ReplicationMultipartPartInput {
offset,
part_number: part_info.number,
part_size: part_info.actual_size,
},
part_info.size,
index + 1 == object_info.parts.len(),
)?;
for part_info in object_info.parts.iter() {
// Ciphertext passthrough (raw read) ranges over the stored part
// bytes; decrypted reads range over the logical plaintext parts.
let part_size = if obj_opts.raw_data_movement_read {
part_info.size as i64
} else {
part_info.actual_size
};
let part_plan = replication_multipart_part_plan(ReplicationMultipartPartInput {
offset,
part_number: part_info.number,
part_size,
})
.map_err(|err| std::io::Error::other(err.to_string()))?;
let range_spec = HTTPRangeSpec {
is_suffix_length: false,
start: part_plan.range.start,
end: part_plan.range.end,
};
offset = part_plan.next_offset;
let byte_stream = if let Some(range_spec) = part_plan.range {
let part_reader = storage
.get_object_reader(src_bucket, object, Some(range_spec), HeaderMap::new(), obj_opts)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let part_stream = wrap_with_bandwidth_monitor_with_header(part_reader.stream, src_bucket, arn, header_size);
async_read_to_bytestream(part_stream)
} else {
ByteStream::from_static(b"")
};
let part_reader = storage
.get_object_reader(src_bucket, object, Some(range_spec), HeaderMap::new(), obj_opts)
.await
.map_err(|e| std::io::Error::other(e.to_string()))?;
let part_stream = wrap_with_bandwidth_monitor_with_header(part_reader.stream, src_bucket, arn, header_size);
header_size = 0;
let byte_stream = async_read_to_bytestream(part_stream);
let object_part = cli
.put_object_part(
@@ -4869,173 +4760,6 @@ async fn replicate_multipart_parts_and_complete<S: ReplicationObjectIO>(
#[cfg(test)]
mod tests {
use super::super::replication_filemeta_boundary::ReplicateTargetDecision;
use super::super::replication_object_decision_boundary::ReplicationMultipartPlanError;
#[test]
fn multipart_read_plan_preserves_legacy_plain_part_ranges() {
const MIB: usize = 1024 * 1024;
let object_info = ObjectInfo {
etag: Some("0123456789abcdef0123456789abcdef".to_string()),
size: 6 * 1024 * 1024,
..Default::default()
};
let mut offset = 0;
for (part_number, stored_size, start, end) in [
(1, 5 * MIB, 0, 5 * 1024 * 1024 - 1),
(2, MIB, 5 * 1024 * 1024, 6 * 1024 * 1024 - 1),
] {
let plan = multipart_replication_read_plan(
&object_info,
&ObjectOptions::default(),
ReplicationMultipartPartInput {
offset,
part_number,
part_size: 0,
},
stored_size,
part_number == 2,
)
.expect("legacy plain parts must use their stored sizes");
assert_eq!(plan.part_number, i32::try_from(part_number).expect("part number fits"));
assert_eq!(plan.part_size, i64::try_from(stored_size).expect("stored size fits"));
let range = plan.range.expect("a nonempty part must read a range");
assert!(!range.is_suffix_length);
assert_eq!((range.start, range.end), (start, end));
assert_eq!(plan.next_offset, end + 1);
offset = plan.next_offset;
}
assert_eq!(offset, object_info.size);
}
#[test]
fn multipart_read_plan_distinguishes_transformed_and_raw_sizes() {
for metadata in [
HashMap::from([("x-rustfs-internal-compression".to_string(), "klauspost/compress/s2".to_string())]),
HashMap::from([("x-amz-server-side-encryption".to_string(), "AES256".to_string())]),
] {
let object_info = ObjectInfo {
user_defined: Arc::new(metadata),
..Default::default()
};
assert!(object_info.is_compressed() || object_info.is_encrypted());
for raw in [false, true] {
for actual_size in [-1, 0, 5] {
let result = multipart_replication_read_plan(
&object_info,
&ObjectOptions {
raw_data_movement_read: raw,
..Default::default()
},
ReplicationMultipartPartInput {
offset: 7,
part_number: 2,
part_size: actual_size,
},
9,
true,
);
if !raw && actual_size <= 0 {
let err = result.expect_err("transformed reads cannot substitute physical bytes for unknown plaintext");
assert!(matches!(
err.get_ref().and_then(|err| err.downcast_ref::<ReplicationMultipartPlanError>()),
Some(ReplicationMultipartPlanError::InvalidPartSize { part_size })
if *part_size == actual_size
));
} else {
let plan = result.expect("the selected representation has a known positive size");
let expected_size = if raw { 9 } else { 5 };
assert_eq!(plan.part_number, 2);
assert_eq!(plan.part_size, expected_size);
let range = plan.range.expect("a nonempty part must read a range");
assert_eq!((range.start, range.end), (7, 7 + expected_size - 1));
assert_eq!(plan.next_offset, 7 + expected_size);
}
}
}
}
}
#[test]
fn multipart_read_plan_retains_an_empty_last_part_without_advancing() {
for offset in [5 * 1024 * 1024, i64::MAX] {
for raw in [false, true] {
let plan = multipart_replication_read_plan(
&ObjectInfo::default(),
&ObjectOptions {
raw_data_movement_read: raw,
..Default::default()
},
ReplicationMultipartPartInput {
offset,
part_number: 2,
part_size: 0,
},
0,
true,
)
.expect("an empty final part needs no range read");
assert_eq!(plan.part_number, 2);
assert_eq!(plan.part_size, 0);
assert!(plan.range.is_none());
assert_eq!(plan.next_offset, offset);
}
}
}
#[test]
fn multipart_read_plan_rejects_invalid_empty_parts_and_ranges() {
for (offset, part_number, actual_size, stored_size, is_last) in [
(0, 1, 0, 0, false),
(0, 2, -1, 0, true),
(0, 2, -1, 9, true),
(-1, 2, 0, 0, true),
(0, usize::try_from(i32::MAX).expect("i32 fits usize") + 1, 0, 0, true),
(i64::MAX, 2, 1, 1, true),
(i64::MAX, 2, 2, 2, true),
] {
let err = multipart_replication_read_plan(
&ObjectInfo::default(),
&ObjectOptions::default(),
ReplicationMultipartPartInput {
offset,
part_number,
part_size: actual_size,
},
stored_size,
is_last,
)
.expect_err("invalid part metadata must not become a successful transport plan");
assert!(
err.kind() == std::io::ErrorKind::InvalidData
|| err.get_ref().is_some_and(|err| { err.is::<ReplicationMultipartPlanError>() }),
"the failure must preserve a typed metadata or planner error: {err}"
);
}
}
#[cfg(target_pointer_width = "64")]
#[test]
fn multipart_read_plan_rejects_physical_size_overflow() {
for raw in [false, true] {
let err = multipart_replication_read_plan(
&ObjectInfo::default(),
&ObjectOptions {
raw_data_movement_read: raw,
..Default::default()
},
ReplicationMultipartPartInput {
offset: 0,
part_number: 1,
part_size: 0,
},
usize::MAX,
true,
)
.expect_err("a physical size outside the range API must be rejected before casting");
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert_eq!(err.to_string(), "multipart replication stored part size exceeds i64");
}
}
#[test]
fn same_state_terminal_retry_uses_validate_only() {
@@ -5444,178 +5168,6 @@ mod tests {
ReplicationTargetStore::register_test_target(target).await;
}
const DRIFTED_ASSIGNED_VERSION_ID: &str = "001788697733811332140-fR6j6uXKV-";
const DRIFTED_ETAG: &str = "9a0364b9e99bb480dd25e1f0284c8555";
/// The Wasabi shape (rustfs/backlog#2340): a version-addressed HEAD with
/// the source uuid answers 404 (not the AWS 400), ListObjectVersions shows
/// the id the target minted, and a HEAD by that id succeeds. Serves exactly
/// `requests` connections and returns the request lines it saw.
fn spawn_drifted_target_server(requests: usize) -> (String, std::thread::JoinHandle<Vec<String>>) {
use std::io::{Read, Write};
let listener = std::net::TcpListener::bind(("127.0.0.1", 0)).expect("test HTTP listener should bind");
let endpoint = format!("http://{}", listener.local_addr().expect("test HTTP listener should have an address"));
let handle = std::thread::spawn(move || {
let mut seen = Vec::new();
for _ in 0..requests {
let (mut stream, _) = listener.accept().expect("test HTTP client should connect");
let mut request = [0_u8; 8192];
let bytes_read = stream.read(&mut request).expect("test HTTP request should be read");
let text = String::from_utf8_lossy(&request[..bytes_read]).to_string();
let request_line = text.lines().next().unwrap_or_default().to_string();
let response = if request_line.starts_with("HEAD ") {
if request_line.contains(&format!("versionId={DRIFTED_ASSIGNED_VERSION_ID}")) {
format!(
"HTTP/1.1 200 OK\r\nETag: \"{DRIFTED_ETAG}\"\r\nContent-Length: 4\r\nLast-Modified: Sun, 06 Sep 2026 10:00:00 GMT\r\nConnection: close\r\n\r\n"
)
} else {
"HTTP/1.1 404 Not Found\r\nContent-Length: 0\r\nConnection: close\r\n\r\n".to_string()
}
} else if request_line.starts_with("GET ") && request_line.contains("versions") {
let body = format!(
"<?xml version=\"1.0\" encoding=\"UTF-8\"?><ListVersionsResult xmlns=\"http://s3.amazonaws.com/doc/2006-03-01/\"><Name>target-bucket</Name><Prefix>object</Prefix><MaxKeys>1000</MaxKeys><IsTruncated>false</IsTruncated><Version><Key>object</Key><VersionId>{DRIFTED_ASSIGNED_VERSION_ID}</VersionId><IsLatest>true</IsLatest><LastModified>2026-09-06T10:00:00.000Z</LastModified><ETag>&quot;{DRIFTED_ETAG}&quot;</ETag><Size>4</Size><StorageClass>STANDARD</StorageClass></Version></ListVersionsResult>"
);
format!(
"HTTP/1.1 200 OK\r\nContent-Type: application/xml\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}",
body.len()
)
} else {
"HTTP/1.1 500 Unexpected\r\nContent-Length: 0\r\nConnection: close\r\n\r\n".to_string()
};
stream
.write_all(response.as_bytes())
.expect("test HTTP response should be written");
seen.push(request_line);
}
seen
});
(endpoint, handle)
}
fn drifted_roi_and_object() -> (ReplicateObjectInfo, ObjectInfo) {
let roi = ReplicateObjectInfo {
bucket: "source".to_string(),
name: "object".to_string(),
version_id: Some(Uuid::new_v4()),
op_type: ReplicationType::Heal,
replication_status: ReplicationStatusType::Pending,
etag: Some(DRIFTED_ETAG.to_string()),
size: 4,
..Default::default()
};
let object_info = ObjectInfo {
bucket: roi.bucket.clone(),
name: roi.name.clone(),
version_id: roi.version_id,
etag: Some(DRIFTED_ETAG.to_string()),
size: 4,
..Default::default()
};
(roi, object_info)
}
#[tokio::test]
async fn heal_redrive_locates_replica_by_etag_on_target_that_mints_own_version_ids() {
let (endpoint, server) = spawn_drifted_target_server(3);
let target = test_target_client(endpoint);
ReplicationTargetStore::record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn);
let (roi, object_info) = drifted_roi_and_object();
let mut rinfo = replicate_all_target_info(&roi, &target);
let action = resolve_replicate_all_action(
ReplicateAllActionContext {
roi: &roi,
tgt_client: &target,
bucket: &roi.bucket,
object: &roi.name,
start_time: OffsetDateTime::now_utc(),
ssec_audit_required: false,
},
object_info,
&mut rinfo,
)
.await;
assert!(
matches!(action, Some((ReplicationAction::None, _))),
"a replica located by content identity must not be re-driven: {action:?}"
);
assert!(rinfo.error.is_none(), "{:?}", rinfo.error);
let seen = server.join().expect("test HTTP server should finish");
assert_eq!(seen.len(), 3, "HEAD by source id, ListObjectVersions, HEAD by assigned id: {seen:?}");
assert!(seen[0].starts_with("HEAD ") && seen[0].contains(&roi.version_id.unwrap().to_string()));
assert!(seen[1].starts_with("GET ") && seen[1].contains("prefix=object"), "{}", seen[1]);
assert!(seen[2].starts_with("HEAD ") && seen[2].contains(DRIFTED_ASSIGNED_VERSION_ID));
}
#[tokio::test]
async fn head_not_found_still_replicates_when_identity_contract_is_unknown() {
// Same 404, but the target never revealed whether it adopts version
// ids: a 404 keeps meaning "replica missing" (adopting targets, e.g.
// RustFS/MinIO peers, must not skip a genuinely missing version).
let (endpoint, server) = spawn_head_status_server(404);
let target = test_target_client(endpoint);
let (roi, object_info) = drifted_roi_and_object();
let mut rinfo = replicate_all_target_info(&roi, &target);
let action = resolve_replicate_all_action(
ReplicateAllActionContext {
roi: &roi,
tgt_client: &target,
bucket: &roi.bucket,
object: &roi.name,
start_time: OffsetDateTime::now_utc(),
ssec_audit_required: false,
},
object_info,
&mut rinfo,
)
.await;
assert!(matches!(action, Some((ReplicationAction::All, _))));
server.join().expect("test HTTP server should finish");
}
#[tokio::test]
async fn resync_verification_counts_drifted_replica_as_replicated() {
let (endpoint, server) = spawn_drifted_target_server(3);
let target = test_target_client(endpoint);
ReplicationTargetStore::record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn);
let (roi, _) = drifted_roi_and_object();
let mut st = TargetReplicationResyncStatus::default();
let head_result =
head_object_for_worker(target.as_ref(), &target.bucket, &roi.name, roi.version_id.map(|v| v.to_string())).await;
let (size, err) = verify_resync_head_result(head_result, &roi, &mut st, &target).await;
assert!(err.is_none(), "{err:?}");
assert_eq!((size, st.replicated_count, st.failed_count), (4, 1, 0));
server.join().expect("test HTTP server should finish");
}
#[test]
fn put_response_audit_records_identity_verdict() {
let target = test_target_client("http://127.0.0.1:1".to_string());
let source = Uuid::new_v4().to_string();
audit_target_version_identity(&target, &source, Some(DRIFTED_ASSIGNED_VERSION_ID));
assert_eq!(
ReplicationTargetStore::version_identity_capability(&target.arn),
VersionIdentityCapability::MintsOwn
);
audit_target_version_identity(&target, &source, Some(&source));
assert_eq!(
ReplicationTargetStore::version_identity_capability(&target.arn),
VersionIdentityCapability::Adopts
);
// An unversioned write carries no contract and must not overwrite it.
audit_target_version_identity(&target, "null", None);
assert_eq!(
ReplicationTargetStore::version_identity_capability(&target.arn),
VersionIdentityCapability::Adopts
);
}
#[test]
fn resync_admission_configuration_is_bounded() {
assert_eq!(ENV_REPL_RESYNC_MAX_JOBS, "RUSTFS_REPL_RESYNC_MAX_JOBS");
@@ -6787,326 +6339,4 @@ mod tests {
"one target's report must not silence another's"
);
}
mod multipart_transport_tests {
use super::super::super::replication_filemeta_boundary::ObjectPartInfo;
use super::super::super::replication_storage_boundary::ObjectIO as _;
use super::*;
use bytes::Bytes;
use http_body_util::{BodyExt, Full};
use std::convert::Infallible;
#[derive(Debug)]
struct Source {
body: Bytes,
info: ObjectInfo,
ranges: StdMutex<Vec<(i64, i64)>>,
full_reads: std::sync::atomic::AtomicUsize,
}
#[async_trait::async_trait]
impl super::super::super::replication_storage_boundary::ObjectIO for Source {
type Error = Error;
type RangeSpec = HTTPRangeSpec;
type HeaderMap = HeaderMap;
type ObjectOptions = ObjectOptions;
type ObjectInfo = ObjectInfo;
type GetObjectReader = GetObjectReader;
type PutObjectReader = super::super::super::replication_storage_boundary::PutObjReader;
async fn get_object_reader(
&self,
_bucket: &str,
_object: &str,
range: Option<HTTPRangeSpec>,
_headers: HeaderMap,
opts: &ObjectOptions,
) -> Result<GetObjectReader> {
assert_eq!(
opts.version_id,
self.info.version_id.map(|id| id.to_string()),
"every read retains the selected source version"
);
if range.is_none() {
self.full_reads.fetch_add(1, Ordering::Relaxed);
return Ok(GetObjectReader {
stream: Box::new(std::io::Cursor::new(self.body.clone())),
object_info: self.info.clone(),
buffered_body: None,
body_source: Default::default(),
});
}
let range = range.expect("multipart transport must request an explicit nonempty range");
assert!(!range.is_suffix_length);
assert!(range.start <= range.end, "empty parts must not issue an inverted range");
self.ranges.lock().expect("range journal lock").push((range.start, range.end));
let start = usize::try_from(range.start).expect("nonnegative start");
let end = usize::try_from(range.end).expect("nonnegative end");
let body = self.body.slice(start..=end);
Ok(GetObjectReader {
stream: Box::new(std::io::Cursor::new(body)),
object_info: self.info.clone(),
buffered_body: None,
body_source: Default::default(),
})
}
async fn put_object(
&self,
_bucket: &str,
_object: &str,
_data: &mut Self::PutObjectReader,
_opts: &ObjectOptions,
) -> Result<ObjectInfo> {
panic!("replication must not overwrite its source")
}
}
#[derive(Debug)]
struct RequestRecord {
method: http::Method,
query: HashMap<String, String>,
headers: HeaderMap,
body: Bytes,
}
#[tokio::test]
async fn multipart_transport_preserves_legacy_zero_actual_sizes() {
run_transport(4096, None).await;
}
#[tokio::test]
async fn multipart_transport_uploads_an_empty_last_part_without_reading_a_range() {
run_transport(0, None).await;
}
#[tokio::test]
async fn multipart_transport_preserves_transformed_unknown_nonempty_parts() {
for unknown_part in [(0, 0), (1, 0), (0, -1), (1, -1)] {
run_transport(4096, Some(unknown_part)).await;
}
}
#[tokio::test]
async fn multipart_transport_preserves_transformed_empty_tail() {
run_transport(0, Some((1, 0))).await;
}
async fn run_transport(tail_size: usize, unknown_part: Option<(usize, i64)>) {
const FIRST_SIZE: usize = 5 * 1024 * 1024;
let body = Bytes::from([vec![0x35; FIRST_SIZE], vec![0xa7; tail_size]].concat());
let etag = faster_hex::hex_string(rustfs_utils::hash::HashAlgorithm::Md5.hash_encode(&body).as_ref());
let source = Arc::new(Source {
info: ObjectInfo {
size: i64::try_from(body.len() + if unknown_part.is_some() { 16 } else { 0 }).expect("stored size"),
actual_size: i64::try_from(body.len()).expect("body size"),
etag: Some(etag.clone()),
version_id: Some(Uuid::new_v4()),
user_defined: Arc::new(if unknown_part.is_some() {
HashMap::from([("x-amz-server-side-encryption".to_string(), "AES256".to_string())])
} else {
HashMap::new()
}),
parts: Arc::new(vec![
ObjectPartInfo {
number: 1,
size: FIRST_SIZE + if unknown_part.is_some() { 8 } else { 0 },
actual_size: if let Some((0, size)) = unknown_part {
size
} else if unknown_part.is_some() || tail_size == 0 {
i64::try_from(FIRST_SIZE).expect("first part size")
} else {
0
},
..Default::default()
},
ObjectPartInfo {
number: 2,
size: tail_size + if unknown_part.is_some() { 8 } else { 0 },
actual_size: if let Some((1, size)) = unknown_part {
size
} else if unknown_part.is_some() {
i64::try_from(tail_size).expect("tail logical size")
} else {
0
},
..Default::default()
},
]),
..Default::default()
},
body: body.clone(),
ranges: StdMutex::new(Vec::new()),
full_reads: std::sync::atomic::AtomicUsize::new(0),
});
let journal = Arc::new(StdMutex::new(Vec::<RequestRecord>::new()));
let listener = tokio::net::TcpListener::bind(("127.0.0.1", 0))
.await
.expect("bind multipart target");
let endpoint = format!("http://{}", listener.local_addr().expect("multipart target address"));
let server_journal = journal.clone();
let server = tokio::spawn(async move {
let mut connections = JoinSet::new();
loop {
let (stream, _) = listener.accept().await.expect("accept multipart request");
let journal = server_journal.clone();
connections.spawn(async move {
let service = hyper::service::service_fn(move |request: hyper::Request<hyper::body::Incoming>| {
let journal = journal.clone();
async move {
let (request, body) = request.into_parts();
let query: HashMap<String, String> = url::form_urlencoded::parse(
request.uri.query().unwrap_or_default().as_bytes(),
).into_owned().collect();
let body = body.collect().await.expect("read complete multipart request body").to_bytes();
let response = if request.method == http::Method::POST && query.contains_key("uploads") {
"<InitiateMultipartUploadResult><Bucket>target-bucket</Bucket><Key>object</Key><UploadId>upload-1</UploadId></InitiateMultipartUploadResult>"
} else if request.method == http::Method::PUT {
""
} else if request.method == http::Method::POST && query.contains_key("uploadId") {
"<CompleteMultipartUploadResult><Location>http://localhost/object</Location><Bucket>target-bucket</Bucket><Key>object</Key><ETag>&quot;target-2&quot;</ETag></CompleteMultipartUploadResult>"
} else if request.method == http::Method::DELETE && query.contains_key("uploadId") {
""
} else {
panic!("unexpected multipart request: {} {}", request.method, request.uri)
};
let response_etag = if request.method == http::Method::PUT && !query.contains_key("partNumber") {
format!("\"{}\"", faster_hex::hex_string(rustfs_utils::hash::HashAlgorithm::Md5.hash_encode(&body).as_ref()))
} else {
"\"uploaded-part\"".to_string()
};
journal.lock().expect("request journal lock").push(RequestRecord {
method: request.method, query, headers: request.headers, body,
});
Ok::<_, Infallible>(hyper::Response::builder()
.header("content-type", "application/xml")
.header("etag", response_etag)
.body(Full::new(Bytes::from_static(response.as_bytes())))
.expect("multipart response"))
}
});
hyper::server::conn::http1::Builder::new()
.serve_connection(hyper_util::rt::TokioIo::new(stream), service)
.await.expect("serve multipart connection");
});
}
});
let mut target = test_target_client(endpoint);
let config = target
.client
.config()
.to_builder()
.request_checksum_calculation(aws_sdk_s3::config::RequestChecksumCalculation::WhenRequired)
.force_path_style(true)
.build();
Arc::get_mut(&mut target).expect("unshared test target").client = Arc::new(aws_sdk_s3::Client::from_conf(config));
let (put_opts, is_multipart) = replication_put_object_options("STANDARD", &source.info).expect("replication options");
let opts = ObjectOptions {
version_id: source.info.version_id.map(|id| id.to_string()),
..Default::default()
};
let reader = source
.get_object_reader("source", "object", None, HeaderMap::new(), &opts)
.await
.expect("open the existing full-object stream");
let result = tokio::time::timeout(
std::time::Duration::from_secs(30),
replicate_all_payload_to_target(
ReplicateAllPayloadContext {
storage: &source,
tgt_client: &target,
bucket: "source",
object: "object",
object_info: &source.info,
obj_opts: &opts,
arn: &target.arn,
transfer_size: i64::try_from(body.len()).expect("plaintext size"),
is_multipart,
put_opts,
},
reader,
),
)
.await;
server.abort();
assert!(server.await.expect_err("fixture server is stopped").is_cancelled());
if let Some(error) = result.expect("replication must finish") {
panic!("legacy parts must replicate successfully: {error}");
}
assert_eq!(
source.full_reads.load(Ordering::Relaxed),
1,
"reuse the initial full stream without an extra read"
);
if unknown_part.is_some() {
let requests = journal.lock().expect("request journal lock");
assert_eq!(requests.len(), 1, "unknown transformed boundaries retain one streaming PUT");
let request = &requests[0];
assert_eq!(request.method, http::Method::PUT);
let source_version = source.info.version_id.map(|id| id.to_string()).expect("versioned fixture");
assert_eq!(
request.query,
HashMap::from([
("x-id".to_string(), "PutObject".to_string()),
("versionId".to_string(), source_version.clone()),
]),
"single PUT carries only the SDK operation query and the source versionId the target must reuse"
);
assert_eq!(request.body, body, "single PUT includes every byte of both source parts");
assert_eq!(
request.headers.get("content-length").expect("body length"),
body.len().to_string().as_str()
);
assert_eq!(
rustfs_utils::http::get_header(&request.headers, rustfs_utils::http::SUFFIX_SOURCE_ETAG).as_deref(),
Some(etag.as_str())
);
assert_eq!(
rustfs_utils::http::get_header(&request.headers, rustfs_utils::http::SUFFIX_SOURCE_VERSION_ID)
.map(|value| value.into_owned()),
Some(source_version),
"single PUT preserves the selected source version"
);
assert!(
source.ranges.lock().expect("range journal lock").is_empty(),
"unknown logical boundaries must not issue guessed ranges"
);
return;
}
let requests = journal.lock().expect("request journal lock");
assert_eq!(requests.len(), 4, "initiate, two upload parts, and complete without retries");
assert!(requests[0].query.contains_key("uploads"));
for (index, expected) in [(1, body.slice(..FIRST_SIZE)), (2, body.slice(FIRST_SIZE..))] {
assert_eq!(requests[index].method, http::Method::PUT);
assert_eq!(requests[index].query.get("partNumber"), Some(&index.to_string()));
assert_eq!(requests[index].body, expected, "upload part contains the exact source range");
assert_eq!(
requests[index].headers.get("content-length").expect("part content length"),
expected.len().to_string().as_str()
);
}
let complete = &requests[3];
assert_eq!(complete.method, http::Method::POST);
assert_eq!(
rustfs_utils::http::get_header(&complete.headers, rustfs_utils::http::SUFFIX_SOURCE_ETAG).as_deref(),
Some(etag.as_str())
);
let complete_xml = std::str::from_utf8(&complete.body).expect("complete XML");
assert_eq!(
complete_xml.matches("<Part>").count(),
2,
"the empty final part must remain in the completion list"
);
assert!(complete_xml.contains("<PartNumber>1</PartNumber>"));
assert!(complete_xml.contains("<PartNumber>2</PartNumber>"));
let mut expected_ranges = vec![(0, i64::try_from(FIRST_SIZE - 1).expect("first end"))];
if tail_size > 0 {
expected_ranges.push((
i64::try_from(FIRST_SIZE).expect("tail start"),
i64::try_from(body.len() - 1).expect("tail end"),
));
}
assert_eq!(*source.ranges.lock().expect("range journal lock"), expected_ranges);
}
}
}
@@ -48,7 +48,6 @@ pub use rustfs_replication::{ObjectLockIntegrity, object_lock_put_integrity};
pub(crate) use rustfs_replication::{
SsecPassthroughGate, is_replication_target_offline_error, ssec_passthrough_gate, version_identity_drifted,
};
pub use rustfs_replication::{VersionIdentityCapability, version_identity_capability_from_put};
use super::replication_config_store::ReplicationConfigStore;
use super::replication_error_boundary::{Error, Result};
@@ -193,14 +192,6 @@ impl ReplicationTargetStore {
.await
}
pub(crate) fn version_identity_capability(arn: &str) -> VersionIdentityCapability {
BucketTargetSys::get().version_identity_capability(arn)
}
pub(crate) fn record_version_identity_capability(arn: &str, capability: VersionIdentityCapability) {
BucketTargetSys::get().record_version_identity_capability(arn, capability)
}
#[cfg(test)]
pub(crate) async fn register_test_target(target_client: &Arc<TargetClient>) {
BucketTargetSys::get().arn_remotes_map.write().await.insert(
@@ -257,16 +248,7 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
meta.insert(AMZ_SERVER_SIDE_ENCRYPTION.to_string(), "aws:kms".to_string());
}
// Older transformed objects can have physical parts without logical part
// lengths. Keep their existing whole-object transport: physical sizes are
// not plaintext boundaries for a multipart replication read.
let legacy_single_put = object_info.etag.as_deref().is_none_or(|etag| etag.len() == 32);
let base_is_multipart = object_info.is_multipart()
&& !(legacy_single_put
&& object_info.parts.len() > 1
&& (object_info.is_compressed() || object_info.is_encrypted())
&& object_info.parts.iter().any(|part| part.actual_size <= 0));
let mut is_multipart = base_is_multipart;
let mut is_multipart = object_info.is_multipart();
if let Some(checksum_data) = &object_info.checksum
&& !checksum_data.is_empty()
@@ -277,8 +259,8 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
} else if object_info.is_encrypted() {
// Encrypted checksums cannot be exposed as plaintext headers, and
// decrypt_checksums reports is_multipart=false for them (a value
// the response path relies on). Keep the transport selected from
// the object's layout and readable part boundaries.
// the response path relies on). Keep the object's own multipart
// flag so encrypted objects stay on the multipart route.
} else {
let (checksum_meta, checksum_record_is_multipart) = object_info.decrypt_checksums(0, &HeaderMap::new())?;
// The checksum record describes how the *checksum* is composed,
@@ -286,9 +268,9 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
// MULTIPART flag even on a multipart upload, so trusting it here
// routed a 768-part object through a single PutObject and the
// target rejected the 6 GiB body with EntityTooLarge
// (rustfs#6825). The usable part layout is the authority: the
// (rustfs#6825). The object's own shape is the authority: the
// record may only add multipart-ness, never take it away.
is_multipart = base_is_multipart || checksum_record_is_multipart;
is_multipart = object_info.is_multipart() || checksum_record_is_multipart;
for (key, value) in checksum_meta.iter() {
if key != AMZ_CHECKSUM_TYPE {
@@ -296,7 +278,7 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
}
}
if !base_is_multipart
if !object_info.is_multipart()
&& checksum_meta
.get(AMZ_CHECKSUM_TYPE)
.is_some_and(|value| value == AMZ_CHECKSUM_TYPE_FULL_OBJECT)
@@ -534,7 +516,6 @@ fn is_standard_header(key: &str) -> bool {
#[cfg(test)]
mod tests {
use super::super::replication_filemeta_boundary::ObjectPartInfo;
use super::*;
use aws_smithy_types::DateTime;
use rustfs_replication::content_matches_by_etag;
@@ -569,109 +550,6 @@ mod tests {
checksum.to_bytes(&combined)
}
fn replication_route_metadata() -> [(&'static str, Arc<HashMap<String, String>>); 4] {
let mut compressed = HashMap::new();
rustfs_utils::http::insert_str(&mut compressed, rustfs_utils::http::SUFFIX_COMPRESSION, "zstd".to_string());
[
("plain", Arc::new(HashMap::new())),
("compressed", Arc::new(compressed)),
(
"encrypted",
Arc::new(HashMap::from([(AMZ_SERVER_SIDE_ENCRYPTION.to_string(), "AES256".to_string())])),
),
(
"ssec",
Arc::new(HashMap::from([(SSEC_ALGORITHM_HEADER.to_string(), "AES256".to_string())])),
),
]
}
fn replication_route_object(
etag: Option<&str>,
actual_sizes: [i64; 3],
metadata: Arc<HashMap<String, String>>,
) -> ObjectInfo {
ObjectInfo {
etag: etag.map(str::to_string),
size: 48,
actual_size: 12,
user_defined: metadata,
parts: Arc::new(
actual_sizes
.into_iter()
.enumerate()
.map(|(index, actual_size)| ObjectPartInfo {
number: index + 1,
size: 16,
actual_size,
..Default::default()
})
.collect(),
),
..Default::default()
}
}
#[test]
fn legacy_transformed_single_put_parts_keep_the_previous_replication_route() {
let [_, (_, compressed), (_, encrypted), (_, ssec)] = replication_route_metadata();
let cases = [
(
"compressed middle zero",
compressed.clone(),
Some("0123456789abcdef0123456789abcdef"),
[4, 0, 4],
),
("compressed tail unknown", compressed, None, [4, 4, -1]),
(
"encrypted middle unknown",
encrypted,
Some("gggggggggggggggggggggggggggggggg"),
[4, -1, 4],
),
("ssec tail zero", ssec.clone(), None, [4, 4, 0]),
("ssec middle unknown", ssec, Some("gggggggggggggggggggggggggggggggg"), [4, -1, 4]),
];
for (name, metadata, etag, actual_sizes) in cases {
for checksum in [None, Some(full_object_multipart_checksum_record())] {
let mut object_info = replication_route_object(etag, actual_sizes, metadata.clone());
object_info.checksum = checksum;
assert!(object_info.is_multipart(), "{name}: physical parts remain visible to metadata APIs");
assert!(object_info.is_compressed() || object_info.is_encrypted());
let (options, is_multipart) =
replication_put_object_options("STANDARD", &object_info).expect("legacy transformed put options");
assert!(
!is_multipart,
"{name}: unknown logical part sizes must preserve the old whole-object route"
);
assert_eq!(options.internal.source_etag, etag.unwrap_or_default());
if metadata.contains_key(SSEC_ALGORITHM_HEADER) {
assert_eq!(
get_header_map(&options.user_metadata, SUFFIX_REPLICATION_SSEC_CRC).is_some(),
object_info.checksum.is_some(),
"SSE-C checksums retain their raw passthrough transport"
);
}
}
}
}
#[test]
fn positive_part_sizes_and_legacy_multipart_etags_keep_the_replication_route() {
for (name, metadata) in replication_route_metadata() {
for (etag, actual_sizes) in [
("0123456789abcdef0123456789abcdef", [4, 4, 4]),
("0123456789abcdef0123456789abcdef-3", [4, 0, -1]),
] {
let mut object_info = replication_route_object(Some(etag), actual_sizes, metadata.clone());
object_info.checksum = Some(full_object_multipart_checksum_record());
let (_, is_multipart) = replication_put_object_options("STANDARD", &object_info).expect("multipart put options");
assert!(is_multipart, "{name}/{etag}: usable sizes and old multipart ETags must retain MPU");
}
}
}
#[test]
fn multipart_object_with_full_object_checksum_keeps_the_multipart_route() {
// rustfs#6825: a 768-part upload was replicated with a single
@@ -704,36 +582,6 @@ mod tests {
);
}
#[test]
fn stored_multipart_parts_keep_the_replication_route_without_a_multipart_etag() {
for etag in [Some("0123456789abcdef0123456789abcdef"), None] {
for checksum in [None, Some(full_object_multipart_checksum_record())] {
let object_info = ObjectInfo {
etag: etag.map(str::to_string),
checksum,
parts: Arc::new(
(1..=2)
.map(|number| ObjectPartInfo {
number,
..Default::default()
})
.collect(),
),
..Default::default()
};
let (options, is_multipart) =
replication_put_object_options("STANDARD", &object_info).expect("build put options");
assert!(
is_multipart,
"stored parts must retain multipart routing: etag={etag:?}, checksum={:?}",
object_info.checksum
);
assert_eq!(options.internal.source_etag, etag.unwrap_or_default());
}
}
}
#[test]
fn checksum_record_never_changes_the_transport_a_single_part_object_needs() {
// The mirror of the rustfs#6825 guard: an object stored as one PUT
@@ -744,10 +592,6 @@ mod tests {
let object_info = ObjectInfo {
etag: Some("0123456789abcdef0123456789abcdef".to_string()),
checksum: Some(checksum.to_bytes(&[])),
parts: Arc::new(vec![ObjectPartInfo {
number: 1,
..Default::default()
}]),
..Default::default()
};
@@ -784,19 +628,6 @@ mod tests {
let (_, is_multipart) = replication_put_object_options("STANDARD", &object_info).expect("build put options");
assert!(is_multipart, "a composite-checksum multipart object must stay on the multipart transport");
for (name, metadata) in replication_route_metadata() {
let mut legacy = replication_route_object(Some("0123456789abcdef0123456789abcdef"), [4, 0, 4], metadata);
legacy.checksum = Some(checksum.to_bytes(&combined));
let (_, record_is_multipart) = legacy.decrypt_checksums(0, &HeaderMap::new()).expect("decode checksum");
let (_, is_multipart) = replication_put_object_options("STANDARD", &legacy).expect("legacy checksum put options");
if legacy.is_encrypted() {
assert!(!is_multipart, "{name}: encrypted checksum records must not change the old transport");
} else {
assert!(record_is_multipart, "the composite checksum must carry its own multipart signal");
assert!(is_multipart, "{name}: a composite record can still promote the legacy route to MPU");
}
}
}
#[test]
@@ -1688,15 +1688,6 @@ impl PeerRestClient {
Ok((self.topology_member.clone(), supported_version, epoch))
}
pub async fn probe_ilm_recovery_export(&self, topology_fingerprint: String) -> Result<(String, Uuid)> {
let probe = rustfs_protos::ilm_recovery_export_capability_probe(Uuid::new_v4().as_bytes());
let result = self
.heal_control(rustfs_protos::HEAL_CONTROL_PROTOCOL_VERSION, topology_fingerprint, probe)
.await?;
let epoch = decode_remote_version_state_capability(&self.topology_member, &result)?;
Ok((self.topology_member.clone(), epoch))
}
pub async fn load_bucket_metadata(&self, bucket: &str, scanner_maintenance_change: bool) -> Result<()> {
let result = tokio::time::timeout(BUCKET_METADATA_RELOAD_TIMEOUT, async {
let result = self.load_bucket_metadata_once(bucket, scanner_maintenance_change).await;
+8 -20
View File
@@ -449,14 +449,6 @@ where
Ok(data)
}
pub(crate) async fn read_config_limited_preserve_empty<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
where
S: EcstoreObjectIO,
{
let (data, _obj) = read_config_limited_preserve_empty_with_metadata(api, file, max_bytes).await?;
Ok(data)
}
pub(crate) async fn read_config_limited<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
where
S: EcstoreObjectIO,
@@ -465,6 +457,14 @@ where
Ok(data)
}
pub(crate) async fn read_config_limited_preserve_empty<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
where
S: EcstoreObjectIO,
{
let (data, _obj) = read_config_limited_preserve_empty_with_metadata(api, file, max_bytes).await?;
Ok(data)
}
pub(crate) async fn read_config_limited_preserve_empty_with_metadata<S>(
api: Arc<S>,
file: &str,
@@ -476,18 +476,6 @@ where
read_config_with_metadata_inner(api, file, &ObjectOptions::default(), true, Some(max_bytes)).await
}
pub(crate) async fn read_config_limited_preserve_empty_with_metadata_opts<S>(
api: Arc<S>,
file: &str,
opts: &ObjectOptions,
max_bytes: usize,
) -> Result<(Vec<u8>, ObjectInfo)>
where
S: EcstoreObjectIO,
{
read_config_with_metadata_inner(api, file, opts, true, Some(max_bytes)).await
}
/// Read an existing config object without treating an empty payload as absent.
/// Callers that validate their own payload format need to distinguish corruption
/// from `ConfigNotFound`.
+138 -5
View File
@@ -3385,7 +3385,7 @@ pub(crate) async fn acquire_pool_activation_fleet_proof(
.ok_or_else(|| Error::other(POOL_ACTIVATION_FLEET_PROOF_REQUIRED))
}
pub fn is_pool_activation_fleet_proof_error(err: &Error) -> bool {
pub(crate) fn is_pool_activation_fleet_proof_error(err: &Error) -> bool {
// Save-stage helpers add context by formatting the original error, so the
// marker may be nested in the display string. Restrict matching to the
// `Error::other` I/O shape used by this activation path.
@@ -5108,7 +5108,49 @@ async fn read_pool_meta_replicas<S>(pools: Vec<Arc<S>>, no_lock: bool) -> Vec<Po
where
S: EcstoreObjectIO,
{
join_all(pools.into_iter().map(|pool| read_pool_meta_replica(pool, no_lock))).await
let reads = join_all(pools.into_iter().map(|pool| read_pool_meta_replica(pool, no_lock))).await;
#[cfg(feature = "e2e-test-hooks")]
if STARTUP_CAS_OBSERVATION.try_with(|_| ()).is_ok() {
let batch = uuid::Uuid::new_v4();
for (pool, read) in reads.iter().enumerate() {
let mut observation = serde_json::json!({
"kind": "replica-read", "object": POOL_META_NAME, "batch": batch, "pool": pool,
"cas": match &read.cas {
PoolMetaCasToken::Missing => "missing",
PoolMetaCasToken::Existing(_) => "existing",
PoolMetaCasToken::Unsafe => "unsafe",
},
"etag": match &read.cas { PoolMetaCasToken::Existing(etag) => Some(etag), _ => None },
});
match &read.replica {
PoolMetaReplica::Valid {
raw,
canonical,
meta,
revision,
committed,
..
} => {
observation["state"] = serde_json::json!("valid");
observation["committed"] = serde_json::json!(committed);
observation["version"] = serde_json::json!(revision.version);
observation["cluster_id"] = serde_json::json!(revision.cluster_id);
observation["epoch"] = serde_json::json!(revision.epoch);
observation["generation"] = serde_json::json!(revision.generation);
observation["transaction_id"] = serde_json::json!(revision.transaction_id);
observation["pool_count"] = serde_json::json!(meta.pools.len());
observation["payload_sha256"] = serde_json::json!(rustfs_utils::crypto::hex(Sha256::digest(canonical)));
observation["raw_sha256"] = serde_json::json!(rustfs_utils::crypto::hex(Sha256::digest(raw)));
}
PoolMetaReplica::Missing => observation["state"] = serde_json::json!("missing"),
PoolMetaReplica::Corrupt(_) => observation["state"] = serde_json::json!("corrupt"),
PoolMetaReplica::Incompatible(_) => observation["state"] = serde_json::json!("incompatible"),
PoolMetaReplica::Unreadable(_) => observation["state"] = serde_json::json!("unreadable"),
}
startup_cas_test_observe(observation);
}
}
reads
}
fn select_pool_meta_replicas_observing<R>(write_state: &mut PoolMetaWriteState, replicas: Vec<R>) -> Result<PoolMetaSelection>
@@ -5480,6 +5522,60 @@ fn pool_meta_cas_preconditions(token: &PoolMetaCasToken, object: &str) -> Result
}
}
#[cfg(feature = "e2e-test-hooks")]
struct StartupCasObservation {
attempt: uuid::Uuid,
phase: &'static str,
pools: Vec<usize>,
}
#[cfg(feature = "e2e-test-hooks")]
tokio::task_local! {
static STARTUP_CAS_OBSERVATION: StartupCasObservation;
}
// This scope follows only the directly polled startup future. Spawned work
// does not inherit it; receiver evidence retains its existing RPC tuple.
#[cfg(feature = "e2e-test-hooks")]
pub(crate) async fn startup_cas_test_scope<S, F: std::future::Future>(
attempt: uuid::Uuid,
phase: &'static str,
pools: &[Arc<S>],
future: F,
) -> F::Output {
STARTUP_CAS_OBSERVATION
.scope(
StartupCasObservation {
attempt,
phase,
// These identities are never dereferenced or logged. The
// caller and operation keep the same pool Arcs alive.
pools: pools.iter().map(|pool| Arc::as_ptr(pool) as usize).collect(),
},
future,
)
.await
}
// Direct JSON diagnostics are independent of the startup tracing subscriber.
#[cfg(feature = "e2e-test-hooks")]
pub(crate) fn startup_cas_test_observe(mut observation: serde_json::Value) {
let Some(nonce) = std::env::var("RUSTFS_E2E_STARTUP_CAS_NONCE")
.ok()
.and_then(|value| uuid::Uuid::parse_str(&value).ok())
else {
return;
};
observation["nonce"] = serde_json::json!(nonce);
observation["pid"] = serde_json::json!(std::process::id());
let _ = STARTUP_CAS_OBSERVATION.try_with(|scope| {
observation["attempt"] = serde_json::json!(scope.attempt);
observation["startup_phase"] = serde_json::json!(scope.phase);
});
let line = format!("RUSTFS_E2E_STARTUP_CAS {observation}\n");
let _ = std::io::Write::write_all(&mut std::io::stderr().lock(), line.as_bytes());
}
async fn save_pool_meta_object_cas<S>(
pool: Arc<S>,
object: &str,
@@ -5500,13 +5596,43 @@ where
..Default::default()
};
fence.add_to_options(&mut opts);
#[cfg(feature = "e2e-test-hooks")]
let observation = std::env::var_os("RUSTFS_E2E_STARTUP_CAS_NONCE").map(|_| {
serde_json::json!({
"kind": "cas", "object": object, "phase": phase,
"pool": STARTUP_CAS_OBSERVATION.try_with(|scope| {
scope.pools.iter().position(|identity| *identity == Arc::as_ptr(&pool) as usize)
}).ok().flatten(),
"payload_sha256": rustfs_utils::crypto::hex(Sha256::digest(&data)),
"if_match": opts.http_preconditions.as_ref().and_then(|p| p.if_match.as_deref()),
"if_none_match": opts.http_preconditions.as_ref().and_then(|p| p.if_none_match.as_deref()),
"tail_drained": opts.write_completion == crate::object_api::WriteCompletion::TailDrained,
"no_lock": opts.no_lock,
})
});
let result = save_config_with_opts_and_metadata(pool, object, data, &opts).await;
if matches!(&result, Err(Error::PreconditionFailed)) {
record_pool_meta_stale_write_rejection(phase);
}
let object_info = result?;
fence.ensure_held()?;
Ok(object_info)
let result = result.and_then(|object_info| {
fence.ensure_held()?;
Ok(object_info)
});
#[cfg(feature = "e2e-test-hooks")]
if let Some(mut observation) = observation {
observation["ok"] = serde_json::json!(result.is_ok());
observation["etag"] = serde_json::json!(result.as_ref().ok().and_then(|info| info.etag.as_deref()));
observation["mod_time"] = serde_json::json!(
result
.as_ref()
.ok()
.and_then(|info| info.mod_time)
.map(|time| time.unix_timestamp_nanos().to_string())
);
observation["error"] = serde_json::json!(result.as_ref().err().map(ToString::to_string));
startup_cas_test_observe(observation);
}
result
}
async fn persist_pool_meta_identity<S>(
@@ -6805,6 +6931,13 @@ impl PoolMeta {
};
if confirmed.revision == revision && confirmed.canonical.as_ref() == Some(&durable) {
persist_pool_meta_identity(pools, write_state, true, fence).await?;
#[cfg(feature = "e2e-test-hooks")]
startup_cas_test_observe(serde_json::json!({
"kind": "confirmed", "object": POOL_META_NAME,
"payload_sha256": rustfs_utils::crypto::hex(Sha256::digest(&durable)),
"generation": confirmed.revision.generation,
"transaction_id": confirmed.revision.transaction_id,
}));
return Ok(confirmed.meta);
}
if !commit_succeeded {
+2 -2
View File
@@ -2673,7 +2673,7 @@ mod tests {
]),
..Default::default()
};
assert!(object_info.is_multipart());
assert!(!object_info.is_multipart());
assert!(should_use_multipart_data_movement(&object_info, false));
let single_nonstandard_part = ObjectInfo {
@@ -3050,7 +3050,7 @@ mod tests {
..Default::default()
};
assert!(object_info.is_multipart());
assert!(!object_info.is_multipart());
assert!(object_info.parts.iter().any(|part| part.checksums.is_some()));
let opts = data_movement_put_object_opts(&object_info, 0);
assert!(!rustfs_utils::http::contains_key_str(&opts.user_defined, SUFFIX_PART_CHECKSUMS));
-40
View File
@@ -324,46 +324,6 @@ impl DiskStoreRenameDataExt for LocalDiskWrapper {
}
impl LocalDiskWrapper {
pub(in crate::disk) async fn delete_version_with_namespace_owner(
&self,
volume: &str,
path: &str,
fi: FileInfo,
force_del_marker: bool,
opts: DeleteOptions,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
self.track_disk_health_mutation(
"delete_version",
DiskMetricMutation::Delete,
|| async {
Box::pin(
self.disk
.delete_version_with_namespace_owner(volume, path, fi, force_del_marker, opts, namespace_owner),
)
.await
},
get_max_timeout_duration(),
)
.await
}
pub(in crate::disk) async fn delete_with_namespace_owner(
&self,
volume: &str,
path: &str,
opts: DeleteOptions,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
self.track_disk_health_mutation(
"delete",
DiskMetricMutation::Delete,
|| async { Box::pin(self.disk.delete_with_namespace_owner(volume, path, opts, namespace_owner)).await },
get_max_timeout_duration(),
)
.await
}
pub(in crate::disk) async fn undo_write_with_namespace_owner(
&self,
volume: &str,
+109 -459
View File
@@ -191,33 +191,11 @@ fn restore_part_transaction_file(current: &Path, backup: &Path, absent: &Path, r
}
async fn write_metadata_rollback_backup(object_dir: &Path, rollback_dir: Uuid, data: &[u8]) -> Result<()> {
write_delete_rollback_file(object_dir, rollback_dir, STORAGE_FORMAT_FILE_BACKUP, data, None).await
}
async fn write_delete_rollback_file(
object_dir: &Path,
rollback_dir: Uuid,
name: &str,
data: &[u8],
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
let backup_dir = object_dir.join(rollback_dir.to_string());
let path = backup_dir.join(name);
if namespace_owner.is_none() {
fs::create_dir_all(&backup_dir).await.map_err(to_file_error)?;
fs::write(path, data).await.map_err(to_file_error)?;
return Ok(());
}
let lease = os::acquire_namespace_mutation_lease_with_owner(&path, namespace_owner).await;
let data = data.to_vec();
os::run_blocking_namespace_operation(lease, move || {
std::fs::create_dir_all(&backup_dir)?;
#[cfg(test)]
run_owned_file_write_before_open(&path);
std::fs::write(path, data)
})
.await
.map_err(to_file_error)?;
fs::create_dir_all(&backup_dir).await.map_err(to_file_error)?;
fs::write(backup_dir.join(STORAGE_FORMAT_FILE_BACKUP), data)
.await
.map_err(to_file_error)?;
Ok(())
}
@@ -364,7 +342,6 @@ struct DeleteVersionMutation {
struct DeleteRollbackFailure {
stage: &'static str,
error: DiskError,
namespace_owner: Option<Arc<dyn Send + Sync>>,
}
async fn restore_delete_rollback_after_error(
@@ -376,18 +353,12 @@ async fn restore_delete_rollback_after_error(
failure: DeleteRollbackFailure,
publication_root: &os::PublicationRoot,
) -> DiskError {
let DeleteRollbackFailure {
stage,
error,
namespace_owner,
} = failure;
let DeleteRollbackFailure { stage, error } = failure;
let Some(rollback_dir) = rollback_dir else {
return error;
};
if let Err(restore_err) =
restore_delete_rollback_with_namespace_owner(object_dir, xl_path, rollback_dir, publication_root, namespace_owner).await
{
if let Err(restore_err) = restore_delete_rollback(object_dir, xl_path, rollback_dir, publication_root).await {
warn!(
volume,
path,
@@ -5683,123 +5654,6 @@ impl LocalDisk {
// })
// }
#[tracing::instrument(name = "delete_version", level = "trace", skip_all)]
pub(in crate::disk) async fn delete_version_with_namespace_owner(
&self,
volume: &str,
path: &str,
fi: FileInfo,
force_del_marker: bool,
opts: DeleteOptions,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
self.delete_version_inner(
volume,
path,
fi,
DeleteVersionMutation {
force_del_marker,
opts,
namespace_owner,
},
)
.await
}
#[tracing::instrument(name = "write_metadata", level = "trace", skip_all)]
async fn write_metadata_with_namespace_owner(
&self,
volume: &str,
path: &str,
fi: FileInfo,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
crate::hp_guard!("LocalDisk::write_metadata");
fi.validate_for_metadata_read()?;
let p = self.io_get_object_path(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str())?;
let mut meta = FileMeta::new();
if !fi.fresh {
let (buf, _) = read_file_exists(&p).await?;
if !buf.is_empty() {
let _ = meta.unmarshal_msg(&buf).map_err(|_| {
meta = FileMeta::new();
});
}
}
meta.add_version(fi)?;
let fm_data = meta.marshal_msg()?;
// Atomic temp+rename: this path also rewrites live xl.meta (delete markers,
// decommission), where an in-place truncate would expose torn metadata.
self.write_all_meta_with_namespace_owner(
volume,
format!("{path}/{STORAGE_FORMAT_FILE}").as_str(),
&fm_data,
true,
namespace_owner,
)
.await?;
Ok(())
}
async fn delete_data_dir_with_namespace_owner(
&self,
volume: &str,
path: &str,
opts: DeleteOptions,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<DataDirDeleteStatus> {
let key = SnapshotLeaseKey {
volume: volume.to_string(),
path: path.to_string(),
};
{
let mut registry = self.snapshot_leases.lock().await;
if let Some(entry) = registry.entries.get_mut(&key) {
if !entry.tokens.is_empty() {
entry.pending_delete.get_or_insert_with(|| opts.clone());
return Ok(DataDirDeleteStatus::Deferred);
}
if entry.deleting {
entry.pending_delete.get_or_insert_with(|| opts.clone());
return Ok(DataDirDeleteStatus::Deferred);
}
entry.deleting = true;
entry.pending_delete.get_or_insert_with(|| opts.clone());
} else {
registry.entries.insert(
key.clone(),
SnapshotLeaseEntry {
pending_delete: Some(opts.clone()),
deleting: true,
..Default::default()
},
);
}
}
let result = self
.delete_unleased_with_namespace_owner(volume, path, &opts, namespace_owner)
.await;
let mut registry = self.snapshot_leases.lock().await;
match result {
Ok(()) => {
registry.entries.remove(&key);
Ok(DataDirDeleteStatus::Deleted)
}
Err(err) => {
if let Some(entry) = registry.entries.get_mut(&key) {
entry.deleting = false;
}
Err(err)
}
}
}
async fn delete_version_inner(&self, volume: &str, path: &str, fi: FileInfo, mutation: DeleteVersionMutation) -> Result<()> {
let DeleteVersionMutation {
force_del_marker,
@@ -5842,7 +5696,7 @@ impl LocalDisk {
if fi.deleted && force_del_marker {
return self
.write_missing_delete_marker(volume, path, fi, file_path.as_path(), rollback_dir, namespace_owner.clone())
.write_missing_delete_marker(volume, path, fi, file_path.as_path(), &xl_path, rollback_dir)
.await;
}
@@ -5858,14 +5712,7 @@ impl LocalDisk {
let old_dir = meta.delete_version(&fi)?;
let mut reserved_version_delete = false;
if let Some(rollback_dir) = rollback_dir {
write_delete_rollback_file(
file_path.as_path(),
rollback_dir,
STORAGE_FORMAT_FILE_BACKUP,
&buf,
namespace_owner.clone(),
)
.await?;
write_metadata_rollback_backup(file_path.as_path(), rollback_dir, &buf).await?;
}
if let Some(uuid) = old_dir {
@@ -5881,7 +5728,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_metadata_update",
error: err,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
@@ -5899,7 +5745,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_data_path",
error: err,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
@@ -5908,7 +5753,7 @@ impl LocalDisk {
if let Some(rollback_dir) = rollback_dir {
let rollback_path = file_path.join(rollback_dir.to_string());
if let Err(err) = os::create_dir_all_with_namespace_owner(&rollback_path, namespace_owner.clone()).await {
if let Err(err) = fs::create_dir_all(&rollback_path).await {
let err: DiskError = to_file_error(err).into();
return Err(restore_delete_rollback_after_error(
file_path.as_path(),
@@ -5919,16 +5764,12 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_rollback_dir",
error: err,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
.await);
}
reserved_version_delete = match self
.reserve_version_delete_with_namespace_owner(volume, path, uuid, rollback_dir, namespace_owner.clone())
.await
{
reserved_version_delete = match self.reserve_version_delete(volume, path, uuid, rollback_dir).await {
Ok(reserved) => reserved,
Err(err) => {
return Err(restore_delete_rollback_after_error(
@@ -5940,7 +5781,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_reserve_data",
error: err,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
@@ -5949,14 +5789,9 @@ impl LocalDisk {
};
let rollback_data_path = rollback_path.join(uuid.to_string());
if !reserved_version_delete
&& let Err(err) = os::rename_all_ignore_missing_source_with_owner(
&old_path,
&rollback_data_path,
&rollback_path,
&self.publication_root,
namespace_owner.clone(),
)
.await
&& let Err(err) =
rename_all_ignore_missing_source(&old_path, &rollback_data_path, &rollback_path, &self.publication_root)
.await
{
return Err(restore_delete_rollback_after_error(
file_path.as_path(),
@@ -5967,7 +5802,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_stage_data",
error: err,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
@@ -5976,16 +5810,13 @@ impl LocalDisk {
if should_fail_after_delete_data_staged(path) {
if reserved_version_delete {
return Err(self
.abort_reserved_version_delete_with_failure(
.abort_reserved_version_delete(
file_path.as_path(),
rollback_dir,
volume,
path,
DeleteRollbackFailure {
stage: "delete_version_test_after_stage",
error: DiskError::Unexpected,
namespace_owner: namespace_owner.clone(),
},
"delete_version_test_after_stage",
DiskError::Unexpected,
)
.await);
}
@@ -5998,7 +5829,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_test_after_stage",
error: DiskError::Unexpected,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
@@ -6028,16 +5858,13 @@ impl LocalDisk {
let err: DiskError = err.into();
if reserved_version_delete && let Some(rollback_dir) = rollback_dir {
return Err(self
.abort_reserved_version_delete_with_failure(
.abort_reserved_version_delete(
file_path.as_path(),
rollback_dir,
volume,
path,
DeleteRollbackFailure {
stage: "delete_version_metadata_encode",
error: err,
namespace_owner: namespace_owner.clone(),
},
"delete_version_metadata_encode",
err,
)
.await);
}
@@ -6050,7 +5877,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_metadata_encode",
error: err,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
@@ -6073,17 +5899,7 @@ impl LocalDisk {
if let Err(err) = commit_result {
if reserved_version_delete && let Some(rollback_dir) = rollback_dir {
return Err(self
.abort_reserved_version_delete_with_failure(
file_path.as_path(),
rollback_dir,
volume,
path,
DeleteRollbackFailure {
stage: "delete_version_commit",
error: err,
namespace_owner: namespace_owner.clone(),
},
)
.abort_reserved_version_delete(file_path.as_path(), rollback_dir, volume, path, "delete_version_commit", err)
.await);
}
return Err(restore_delete_rollback_after_error(
@@ -6095,7 +5911,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_version_commit",
error: err,
namespace_owner: namespace_owner.clone(),
},
&self.publication_root,
)
@@ -6104,21 +5919,16 @@ impl LocalDisk {
if reserved_version_delete
&& let Some(rollback_dir) = rollback_dir
&& let Err(err) = self
.commit_reserved_version_delete_with_namespace_owner(volume, path, rollback_dir, namespace_owner.clone())
.await
&& let Err(err) = self.commit_reserved_version_delete(volume, path, rollback_dir).await
{
return Err(self
.abort_reserved_version_delete_with_failure(
.abort_reserved_version_delete(
file_path.as_path(),
rollback_dir,
volume,
path,
DeleteRollbackFailure {
stage: "delete_version_commit_intent",
error: err,
namespace_owner: namespace_owner.clone(),
},
"delete_version_commit_intent",
err,
)
.await);
}
@@ -6288,7 +6098,7 @@ impl LocalDisk {
}
#[tracing::instrument(name = "delete", level = "trace", skip_all)]
pub(in crate::disk) async fn delete_with_namespace_owner(
async fn delete_with_namespace_owner(
&self,
volume: &str,
path: &str,
@@ -6301,8 +6111,7 @@ impl LocalDisk {
&& let Some((object, transaction_id)) = path.rsplit_once('/')
&& let Ok(transaction_id) = Uuid::parse_str(transaction_id)
{
self.finish_version_delete(volume, object, transaction_id, namespace_owner.clone())
.await?
self.finish_version_delete(volume, object, transaction_id).await?
} else {
false
};
@@ -6644,27 +6453,19 @@ impl LocalDisk {
path: &str,
fi: FileInfo,
object_dir: &Path,
xl_path: &Path,
rollback_dir: Option<Uuid>,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
let xl_path = object_dir.join(STORAGE_FORMAT_FILE);
if let Some(rollback_dir) = rollback_dir {
write_delete_rollback_file(object_dir, rollback_dir, DELETE_MARKER_ROLLBACK_FILE, &[], namespace_owner.clone())
.await?;
}
if let Err(err) = self
.write_metadata_with_namespace_owner(volume, path, fi, namespace_owner.clone())
.await
{
if let Some(rollback_dir) = rollback_dir
&& let Err(restore_err) = restore_delete_rollback_with_namespace_owner(
object_dir,
&xl_path,
rollback_dir,
&self.publication_root,
namespace_owner,
)
let rollback_path = object_dir.join(rollback_dir.to_string());
fs::create_dir_all(&rollback_path).await.map_err(to_file_error)?;
fs::write(rollback_path.join(DELETE_MARKER_ROLLBACK_FILE), [])
.await
.map_err(to_file_error)?;
}
if let Err(err) = self.write_metadata("", volume, path, fi).await {
if let Some(rollback_dir) = rollback_dir
&& let Err(restore_err) = restore_delete_rollback(object_dir, xl_path, rollback_dir, &self.publication_root).await
{
warn!(
event = EVENT_DISK_LOCAL_DELETE_ROLLBACK_FAILED,
@@ -6709,7 +6510,7 @@ impl LocalDisk {
return Err(DiskError::FileNotFound);
};
return self
.write_missing_delete_marker(volume, path, delete_marker, object_dir, opts.old_data_dir, None)
.write_missing_delete_marker(volume, path, delete_marker, object_dir, &xlpath, opts.old_data_dir)
.await;
}
Err(err) => return Err(err),
@@ -6758,7 +6559,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_metadata_update",
error: err,
namespace_owner: None,
},
&self.publication_root,
)
@@ -6794,7 +6594,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_data_path",
error: err,
namespace_owner: None,
},
&self.publication_root,
)
@@ -6826,7 +6625,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_rollback_dir",
error: err,
namespace_owner: None,
},
&self.publication_root,
)
@@ -6867,7 +6665,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_stage_data",
error: err,
namespace_owner: None,
},
&self.publication_root,
)
@@ -6895,7 +6692,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_test_after_stage",
error: DiskError::Unexpected,
namespace_owner: None,
},
&self.publication_root,
)
@@ -6938,7 +6734,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_commit_delete",
error: err,
namespace_owner: None,
},
&self.publication_root,
)
@@ -6985,7 +6780,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_metadata_encode",
error: err,
namespace_owner: None,
},
&self.publication_root,
)
@@ -7011,7 +6805,6 @@ impl LocalDisk {
DeleteRollbackFailure {
stage: "delete_versions_commit_write",
error: err,
namespace_owner: None,
},
&self.publication_root,
)
@@ -8222,18 +8015,6 @@ impl LocalDisk {
}
async fn reserve_version_delete(&self, volume: &str, object: &str, data_dir: Uuid, rollback_dir: Uuid) -> Result<bool> {
self.reserve_version_delete_with_namespace_owner(volume, object, data_dir, rollback_dir, None)
.await
}
async fn reserve_version_delete_with_namespace_owner(
&self,
volume: &str,
object: &str,
data_dir: Uuid,
rollback_dir: Uuid,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<bool> {
let path = format!("{object}/{data_dir}");
let data_path = self.io_get_object_path(volume, &path)?;
match fs::metadata(&data_path).await {
@@ -8243,28 +8024,6 @@ impl LocalDisk {
Err(err) => return Err(to_file_error(err).into()),
}
let marker_path = data_path.join(format!("{RESERVED_DELETE_DATA_DIR_MARKER_PREFIX}{rollback_dir}"));
if namespace_owner.is_some() {
let lease = os::acquire_namespace_mutation_lease_with_owner(&marker_path, namespace_owner.clone()).await;
let volume = volume.to_string();
let sync = os::run_blocking_namespace_operation(lease, move || {
#[cfg(test)]
run_owned_file_write_before_open(&marker_path);
let marker = std::fs::File::create(marker_path)?;
let sync = effective_durability(&volume).syncs_commit_metadata();
if sync {
marker.sync_all()?;
}
Ok(sync)
})
.await
.map_err(to_file_error)?;
if sync {
os::fsync_dir_with_owner(&data_path, namespace_owner)
.await
.map_err(to_file_error)?;
}
return Ok(true);
}
let marker = File::create(marker_path).await.map_err(to_file_error)?;
if effective_durability(volume).syncs_commit_metadata() {
marker.sync_all().await.map_err(to_file_error)?;
@@ -8274,17 +8033,6 @@ impl LocalDisk {
}
async fn commit_reserved_version_delete(&self, volume: &str, object: &str, rollback_dir: Uuid) -> Result<()> {
self.commit_reserved_version_delete_with_namespace_owner(volume, object, rollback_dir, None)
.await
}
async fn commit_reserved_version_delete_with_namespace_owner(
&self,
volume: &str,
object: &str,
rollback_dir: Uuid,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
let object_path = self.io_get_object_path(volume, object)?;
let mut entries = match fs::read_dir(object_path).await {
Ok(entries) => entries,
@@ -8300,14 +8048,10 @@ impl LocalDisk {
continue;
}
let reserved_path = entry.path().join(&reserved_name);
match os::rename_with_namespace_owner(&reserved_path, &entry.path().join(&committed_name), namespace_owner.clone())
.await
{
match fs::rename(&reserved_path, entry.path().join(&committed_name)).await {
Ok(()) => {
if effective_durability(volume).syncs_commit_metadata() {
os::fsync_dir_with_owner(&entry.path(), namespace_owner.clone())
.await
.map_err(to_file_error)?;
os::fsync_dir(&entry.path()).await.map_err(to_file_error)?;
}
}
Err(err) if err.kind() == ErrorKind::NotFound => {}
@@ -8317,13 +8061,7 @@ impl LocalDisk {
Ok(())
}
async fn finish_version_delete(
&self,
volume: &str,
object: &str,
rollback_dir: Uuid,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<bool> {
async fn finish_version_delete(&self, volume: &str, object: &str, rollback_dir: Uuid) -> Result<bool> {
let object_path = self.io_get_object_path(volume, object)?;
let mut entries = match fs::read_dir(object_path).await {
Ok(entries) => entries,
@@ -8344,14 +8082,13 @@ impl LocalDisk {
Err(err) => return Err(to_file_error(err).into()),
}
if let Err(err) = self
.delete_data_dir_with_namespace_owner(
.delete_data_dir(
volume,
&format!("{object}/{data_dir}"),
DeleteOptions {
recursive: true,
..Default::default()
},
namespace_owner.clone(),
)
.await
&& first_err.is_none()
@@ -8372,28 +8109,6 @@ impl LocalDisk {
object: &str,
stage: &'static str,
err: DiskError,
) -> DiskError {
self.abort_reserved_version_delete_with_failure(
object_dir,
rollback_dir,
volume,
object,
DeleteRollbackFailure {
stage,
error: err,
namespace_owner: None,
},
)
.await
}
async fn abort_reserved_version_delete_with_failure(
&self,
object_dir: &Path,
rollback_dir: Uuid,
volume: &str,
object: &str,
failure: DeleteRollbackFailure,
) -> DiskError {
let xl_path = object_dir.join(STORAGE_FORMAT_FILE);
restore_delete_rollback_after_error(
@@ -8402,7 +8117,7 @@ impl LocalDisk {
Some(rollback_dir),
volume,
object,
failure,
DeleteRollbackFailure { stage, error: err },
&self.publication_root,
)
.await
@@ -9893,7 +9608,49 @@ impl DiskAPI for LocalDisk {
}
async fn delete_data_dir(&self, volume: &str, path: &str, opts: DeleteOptions) -> Result<DataDirDeleteStatus> {
self.delete_data_dir_with_namespace_owner(volume, path, opts, None).await
let key = SnapshotLeaseKey {
volume: volume.to_string(),
path: path.to_string(),
};
{
let mut registry = self.snapshot_leases.lock().await;
if let Some(entry) = registry.entries.get_mut(&key) {
if !entry.tokens.is_empty() {
entry.pending_delete.get_or_insert_with(|| opts.clone());
return Ok(DataDirDeleteStatus::Deferred);
}
if entry.deleting {
entry.pending_delete.get_or_insert_with(|| opts.clone());
return Ok(DataDirDeleteStatus::Deferred);
}
entry.deleting = true;
entry.pending_delete.get_or_insert_with(|| opts.clone());
} else {
registry.entries.insert(
key.clone(),
SnapshotLeaseEntry {
pending_delete: Some(opts.clone()),
deleting: true,
..Default::default()
},
);
}
}
let result = self.delete_unleased(volume, path, &opts).await;
let mut registry = self.snapshot_leases.lock().await;
match result {
Ok(()) => {
registry.entries.remove(&key);
Ok(DataDirDeleteStatus::Deleted)
}
Err(err) => {
if let Some(entry) = registry.entries.get_mut(&key) {
entry.deleting = false;
}
Err(err)
}
}
}
#[tracing::instrument(level = "trace", skip_all)]
@@ -9932,8 +9689,32 @@ impl DiskAPI for LocalDisk {
Err(Error::other("Invalid Argument"))
}
#[tracing::instrument(level = "trace", skip_all)]
async fn write_metadata(&self, _org_volume: &str, volume: &str, path: &str, fi: FileInfo) -> Result<()> {
self.write_metadata_with_namespace_owner(volume, path, fi, None).await
crate::hp_guard!("LocalDisk::write_metadata");
fi.validate_for_metadata_read()?;
let p = self.io_get_object_path(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str())?;
let mut meta = FileMeta::new();
if !fi.fresh {
let (buf, _) = read_file_exists(&p).await?;
if !buf.is_empty() {
let _ = meta.unmarshal_msg(&buf).map_err(|_| {
meta = FileMeta::new();
});
}
}
meta.add_version(fi)?;
let fm_data = meta.marshal_msg()?;
// Atomic temp+rename: this path also rewrites live xl.meta (delete markers,
// decommission), where an in-place truncate would expose torn metadata.
self.write_all_meta(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str(), &fm_data, true)
.await?;
Ok(())
}
#[tracing::instrument(level = "trace", skip_all)]
@@ -22396,135 +22177,4 @@ mod test {
assert_eq!(mount_id_from_mountinfo_contents(mountinfo, Path::new("/mnt/replacement disk")), Some(202));
assert_eq!(mount_id_from_mountinfo_contents(mountinfo, Path::new("/mnt/replacement")), None);
}
#[cfg(not(windows))]
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn single_delete_internal_restore_keeps_owner_after_cancellation() {
use crate::disk::os::prepared_publication_test_hooks as hooks;
use futures::FutureExt;
let dir = tempfile::tempdir().expect("fixture directory");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("UTF-8 fixture path")).expect("endpoint");
let disk = Arc::new(LocalDisk::new(&endpoint, false).await.expect("local disk"));
let bucket = "single-delete-internal-restore";
let object = format!("object-{}", Uuid::new_v4());
ensure_test_volume(&disk, bucket).await;
let version = Uuid::new_v4();
let data_dir = Uuid::new_v4();
let rollback_dir = Uuid::new_v4();
let fi = test_file_info(&object, version, Some(data_dir), None);
let original = test_meta(fi.clone());
let object_dir = disk.io_get_object_path(bucket, &object).expect("object IO path");
let part = object_dir.join(data_dir.to_string()).join("part.1");
let metadata = object_dir.join(STORAGE_FORMAT_FILE);
let backup = object_dir.join(rollback_dir.to_string()).join(STORAGE_FORMAT_FILE_BACKUP);
fs::create_dir_all(part.parent().expect("data parent"))
.await
.expect("data directory");
fs::write(&part, b"x").await.expect("real shard");
fs::write(&metadata, &original).await.expect("real version metadata");
set_delete_version_fail_after_data_staged(&object);
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
let (release, release_rx) = std::sync::mpsc::channel::<()>();
let hook = hooks::install_at(hooks::Stage::Rename, &metadata, move || {
let _ = entered_tx.send(());
let _ = release_rx.recv();
});
let ctx = Arc::new(crate::runtime::instance::InstanceContext::new());
let before = ctx.namespace_commit_generation();
let owner = ctx.begin_namespace_commit();
let deleting_disk = Arc::clone(&disk);
let deleting_object = object.clone();
let mut delete = tokio::spawn(async move {
deleting_disk
.delete_version_inner(
bucket,
&deleting_object,
fi,
DeleteVersionMutation {
force_del_marker: false,
opts: DeleteOptions {
old_data_dir: Some(rollback_dir),
..Default::default()
},
namespace_owner: Some(owner),
},
)
.await
});
let mut joined = false;
let mut entered = false;
let mut counts = None;
let observations = std::panic::AssertUnwindSafe(async {
tokio::time::timeout(Duration::from_secs(10), async {
tokio::select! {
result = entered_rx => {
result.expect("actual internal restore entry");
entered = true;
}
result = &mut delete => {
joined = true;
panic!("delete returned before internal physical restore: {result:?}");
}
}
})
.await
.expect("internal restore must reach the physical rename");
assert_eq!(std::fs::read(&backup).expect("real undo backup"), original);
assert_eq!(std::fs::read(&part).expect("reserved shard"), b"x");
delete.abort();
let result = tokio::time::timeout(Duration::from_secs(5), &mut delete).await;
joined = result.is_ok();
assert!(
result
.expect("cancelled caller joins")
.expect_err("cancelled caller")
.is_cancelled()
);
counts = Some((ctx.namespace_commits_pending(), ctx.namespace_commit_generation()));
assert!(hooks::drain_namespace_key(&metadata).now_or_never().is_none());
})
.catch_unwind()
.await;
drop(release);
drop(hook);
let coordinator_drained = joined || tokio::time::timeout(Duration::from_secs(10), &mut delete).await.is_ok();
if !coordinator_drained {
delete.abort();
let _ = tokio::time::timeout(Duration::from_secs(5), &mut delete).await;
}
let physical_drained = tokio::time::timeout(Duration::from_secs(5), hooks::drain_namespace_key(&metadata))
.await
.is_ok();
let owner_drained = tokio::time::timeout(Duration::from_secs(5), async {
while ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
})
.await
.is_ok();
if !entered || !coordinator_drained || !physical_drained || !owner_drained {
eprintln!("internal restore cleanup incomplete; retained root: {:?}", dir.keep());
if let Err(panic) = observations {
std::panic::resume_unwind(panic);
}
panic!("internal restore cleanup must drain before removing its root");
}
if let Err(panic) = observations {
std::panic::resume_unwind(panic);
}
assert_eq!(std::fs::read(&metadata).expect("late restored metadata"), original);
assert!(!backup.exists(), "the actual backup rename must have completed");
assert_eq!(std::fs::read(&part).expect("old shard survives"), b"x");
disk.read_version("", bucket, &object, &version.to_string(), &ReadOptions::default())
.await
.expect("restored version");
let (pending, generation) = counts.expect("observations completed");
assert!(pending, "internal error recovery lost the physical namespace owner");
assert_eq!(generation, before + 1);
assert_eq!(ctx.namespace_commit_generation(), before + 2);
assert!(!ctx.namespace_commits_pending());
}
}
-40
View File
@@ -732,46 +732,6 @@ impl Disk {
}
}
pub(crate) async fn delete_version_with_namespace_owner(
&self,
volume: &str,
path: &str,
fi: FileInfo,
force_del_marker: bool,
opts: DeleteOptions,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
match self {
Self::Local(disk) => {
disk.delete_version_with_namespace_owner(volume, path, fi, force_del_marker, opts, namespace_owner)
.await
}
Self::Remote(disk) => {
let result = disk.delete_version(volume, path, fi, force_del_marker, opts).await;
// This is sender lifetime only, not proof of a remote physical drain.
drop(namespace_owner);
result
}
}
}
pub(crate) async fn delete_with_namespace_owner(
&self,
volume: &str,
path: &str,
opts: DeleteOptions,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Result<()> {
match self {
Self::Local(disk) => disk.delete_with_namespace_owner(volume, path, opts, namespace_owner).await,
Self::Remote(disk) => {
let result = disk.delete(volume, path, opts).await;
drop(namespace_owner);
result
}
}
}
/// Keep local undo publication owned independently of the wrapper deadline.
/// Remote undo retains its existing RPC contract; this is not a remote drain proof.
pub(crate) async fn undo_write_with_namespace_owner(
+47 -77
View File
@@ -247,7 +247,7 @@ pub(crate) mod fsync_dir_recorder {
}
/// Pause a real namespace mutation inside its physical executor.
#[cfg(all(test, not(windows)))]
#[cfg(all(any(test, feature = "test-util"), not(windows)))]
pub(crate) mod prepared_publication_test_hooks {
use super::*;
@@ -256,7 +256,9 @@ pub(crate) mod prepared_publication_test_hooks {
PreparedRename,
Rename,
Remove,
#[cfg(test)]
Rollback,
#[cfg(test)]
DirFsync,
}
@@ -272,6 +274,7 @@ pub(crate) mod prepared_publication_test_hooks {
}
}
#[cfg(test)]
pub(crate) fn install(path: &Path, hook: impl FnOnce() + Send + 'static) -> Guard {
install_at(Stage::PreparedRename, path, hook)
}
@@ -288,45 +291,50 @@ pub(crate) mod prepared_publication_test_hooks {
hook();
}
}
}
#[cfg(test)]
type RenameDestinationHook = Box<dyn FnOnce(&Path) + Send>;
#[cfg(test)]
static RENAME_DESTINATIONS: LazyLock<Mutex<HashMap<PathBuf, RenameDestinationHook>>> =
LazyLock::new(|| Mutex::new(HashMap::new()));
/// Controlled application-test pause at an existing physical executor boundary.
#[cfg(all(feature = "test-util", not(windows)))]
pub struct LocalPublicationPause {
_hook: prepared_publication_test_hooks::Guard,
entered: oneshot::Receiver<()>,
_release: std::sync::mpsc::Sender<()>,
}
#[cfg(test)]
pub(crate) struct RenameDestinationGuard(PathBuf);
#[cfg(all(feature = "test-util", not(windows)))]
#[derive(Clone, Copy)]
pub enum LocalPublicationStage {
PreparedRename,
Rename,
Remove,
}
#[cfg(test)]
impl Drop for RenameDestinationGuard {
fn drop(&mut self) {
RENAME_DESTINATIONS.lock().remove(&self.0);
}
#[cfg(all(feature = "test-util", not(windows)))]
impl LocalPublicationPause {
pub fn install(disk: &crate::disk::Disk, volume: &str, path: &str, stage: LocalPublicationStage) -> Result<Self> {
let path = disk
.get_object_path_for_io_if_local(volume, path)
.ok_or(DiskError::DiskNotFound)??;
let stage = match stage {
LocalPublicationStage::PreparedRename => prepared_publication_test_hooks::Stage::PreparedRename,
LocalPublicationStage::Rename => prepared_publication_test_hooks::Stage::Rename,
LocalPublicationStage::Remove => prepared_publication_test_hooks::Stage::Remove,
};
let (entered_tx, entered) = oneshot::channel();
let (release, release_rx) = std::sync::mpsc::channel::<()>();
let hook = prepared_publication_test_hooks::install_at(stage, &path, move || {
let _ = entered_tx.send(());
let _ = release_rx.recv();
});
Ok(Self {
_hook: hook,
entered,
_release: release,
})
}
#[cfg(test)]
pub(crate) fn observe_rename_destination(source: &Path, hook: impl FnOnce(&Path) + Send + 'static) -> RenameDestinationGuard {
assert!(
RENAME_DESTINATIONS
.lock()
.insert(source.to_path_buf(), Box::new(hook))
.is_none()
);
RenameDestinationGuard(source.to_path_buf())
}
#[cfg(test)]
pub(crate) async fn drain_namespace_key(path: &Path) {
drop(super::acquire_namespace_mutation_lease(path).await);
}
#[cfg(test)]
pub(super) fn run_rename_destination(source: &Path, destination: &Path) {
let hook = RENAME_DESTINATIONS.lock().remove(source);
if let Some(hook) = hook {
hook(destination);
}
pub async fn entered(&mut self) -> std::result::Result<(), oneshot::error::RecvError> {
(&mut self.entered).await
}
}
@@ -1401,7 +1409,7 @@ async fn acquire_namespace_mutation_lease(path: &Path) -> Arc<NamespaceMutationL
acquire_namespace_mutation_lease_with_owner(path, None).await
}
pub(in crate::disk) async fn acquire_namespace_mutation_lease_with_owner(
async fn acquire_namespace_mutation_lease_with_owner(
path: &Path,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> Arc<NamespaceMutationLease> {
@@ -1956,7 +1964,7 @@ pub(crate) async fn remove_file_with_owner(
let path = path.as_ref().to_path_buf();
let lease = acquire_namespace_mutation_lease_with_owner(&path, namespace_owner).await;
run_blocking_namespace_operation(lease, move || {
#[cfg(all(test, not(windows)))]
#[cfg(all(any(test, feature = "test-util"), not(windows)))]
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Remove, &path);
std::fs::remove_file(path)
})
@@ -1976,42 +1984,6 @@ pub(crate) async fn remove_dir_with_owner(
run_blocking_namespace_operation(lease, move || std::fs::remove_dir(path)).await
}
/// Preserve raw rename semantics while retaining a counted owner in the syscall.
/// Unlike reliable rename, this never creates parents or retries a missing source.
pub(in crate::disk) async fn rename_with_namespace_owner(
src: &Path,
dst: &Path,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> io::Result<()> {
if namespace_owner.is_none() {
return tokio::fs::rename(src, dst).await;
}
let src = src.to_path_buf();
let dst = dst.to_path_buf();
let lease = acquire_namespace_mutation_lease_with_owner(&dst, namespace_owner).await;
run_blocking_namespace_operation(lease, move || {
#[cfg(all(test, not(windows)))]
{
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Rename, &src);
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Rename, &dst);
}
std::fs::rename(src, dst)
})
.await
}
pub(in crate::disk) async fn create_dir_all_with_namespace_owner(
path: &Path,
namespace_owner: Option<Arc<dyn Send + Sync>>,
) -> io::Result<()> {
if namespace_owner.is_none() {
return tokio::fs::create_dir_all(path).await;
}
let path = path.to_path_buf();
let lease = acquire_namespace_mutation_lease_with_owner(&path, namespace_owner).await;
run_blocking_namespace_operation(lease, move || std::fs::create_dir_all(path)).await
}
#[tracing::instrument(name = "rename_all", level = "debug", skip_all)]
pub(crate) async fn rename_all_with_owner(
src_file_path: impl AsRef<Path>,
@@ -2216,7 +2188,7 @@ pub(crate) async fn rename_all_with_prepared_source(
move || {
validate_prepared_rename_source(&prepared_source, &src_file_path)?;
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
#[cfg(test)]
#[cfg(any(test, feature = "test-util"))]
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::PreparedRename, &dst_file_path);
rename_prepared(&src_file_path, &dst_file_path, &preparation)
}
@@ -2347,9 +2319,7 @@ async fn reliable_rename_inner_with_lease(
let base_dir = base_dir.clone();
move || {
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
#[cfg(all(test, not(windows)))]
prepared_publication_test_hooks::run_rename_destination(&src_file_path, &dst_file_path);
#[cfg(all(test, not(windows)))]
#[cfg(all(any(test, feature = "test-util"), not(windows)))]
{
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Rename, &src_file_path);
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Rename, &dst_file_path);
-170
View File
@@ -2278,121 +2278,6 @@ mod tests {
assert_eq!(read, fixture.plaintext, "SSE-C + compression full GET must reassemble all parts");
}
#[tokio::test]
async fn multipart_empty_tail_full_reads_preserve_plaintext() {
let key = [0x6Eu8; 32];
let part_sizes = [5 * 1024 * 1024, 0];
let encrypted = build_legacy_ssec_multipart_fixture(key, &part_sizes).await;
for (kind, mut fixture, headers) in [
(
"encrypted",
CompressedMultipartFixture {
object_info: encrypted.object_info,
stored: encrypted.ciphertext,
plaintext: encrypted.plaintext,
},
ssec_headers_from_key(key),
),
("compressed", compressed_multipart_fixture(&part_sizes).await, HeaderMap::new()),
(
"compressed and encrypted",
compressed_encrypted_multipart_fixture(key, &part_sizes).await,
ssec_headers_from_key(key),
),
] {
fixture.object_info.etag = Some(faster_hex::hex_string(Md5::digest(&fixture.plaintext).as_ref()));
assert_eq!(fixture.object_info.etag.as_ref().expect("source ETag").len(), 32);
assert_eq!(fixture.object_info.parts.len(), 2);
let tail = &fixture.object_info.parts[1];
assert_eq!(tail.actual_size, 0, "{kind}: final part has no plaintext");
if kind == "compressed" {
assert_eq!(tail.size, 0, "unpadded compression emits no bytes for an empty part");
} else {
assert!(tail.size > 0, "{kind}: the empty part still has a stored frame");
}
let stored_size = i64::try_from(fixture.stored.len()).expect("fixture size fits i64");
let (mut reader, offset, length) = GetObjectReader::new(
Box::new(Cursor::new(fixture.stored)),
None,
&fixture.object_info,
&ObjectOptions::default(),
&headers,
)
.await
.expect("full transformed read must include the empty tail");
assert_eq!((offset, length), (0, stored_size), "{kind}: full read includes all stored parts");
let mut body = Vec::new();
reader
.stream
.read_to_end(&mut body)
.await
.expect("read through the complete decoder EOF");
assert_eq!(body, fixture.plaintext, "{kind}: no plaintext is added or lost by the empty tail");
}
}
#[tokio::test]
async fn multipart_empty_tail_full_read_authenticates_v2_final_frame() {
let key = [0x6Eu8; 32];
let plaintext = legacy_fixture_part_plaintext(1, 5 * 1024 * 1024);
let mut ciphertext = Vec::new();
let mut parts = Vec::new();
for (number, body) in [(1, plaintext.as_slice()), (2, b"".as_slice())] {
let start = ciphertext.len();
rustfs_rio::EncryptReader::new_multipart_v2(Cursor::new(body), key, LEGACY_FIXTURE_BASE_NONCE, number)
.read_to_end(&mut ciphertext)
.await
.expect("encrypt a v2 fixture part with an authenticated final frame");
parts.push(ObjectPartInfo {
number,
size: ciphertext.len() - start,
actual_size: i64::try_from(body.len()).expect("fixture plaintext size fits"),
..Default::default()
});
}
let tail_start = parts[0].size;
assert_eq!(parts[1].actual_size, 0);
assert!(parts[1].size > 8, "the empty final frame carries more than an END marker");
let object_info = ObjectInfo {
bucket: "bucket".to_string(),
name: "v2-empty-tail".to_string(),
size: i64::try_from(ciphertext.len()).expect("fixture ciphertext size fits"),
etag: Some(faster_hex::hex_string(Md5::digest(&plaintext).as_ref())),
parts: Arc::new(parts),
user_defined: Arc::new(legacy_ssec_multipart_metadata(key, plaintext.len())),
..Default::default()
};
for corrupt_tail in [false, true] {
let mut stored = ciphertext.clone();
if corrupt_tail {
// The v2 header is authenticated associated data, including
// the header of a final frame containing zero plaintext.
stored[tail_start + 5] ^= 1;
}
let (mut reader, offset, length) = GetObjectReader::new(
Box::new(Cursor::new(stored)),
None,
&object_info,
&ObjectOptions::default(),
&ssec_headers_from_key(key),
)
.await
.expect("construct the full reader before consuming the final frame");
assert_eq!((offset, length), (0, object_info.size));
let result = tokio::io::copy(&mut reader.stream, &mut tokio::io::sink()).await;
if corrupt_tail {
let err = result.expect_err("EOF must authenticate the empty final frame after all plaintext is returned");
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert_eq!(err.to_string(), "v2 encrypted frame failed authentication");
} else {
assert_eq!(
result.expect("valid empty final frame must reach EOF"),
u64::try_from(plaintext.len()).expect("plaintext length fits")
);
}
}
}
#[tokio::test]
async fn compressed_encrypted_multipart_range_crosses_part_boundary() {
let key_bytes = [0x6Eu8; 32];
@@ -3771,61 +3656,6 @@ mod tests {
.await;
}
#[tokio::test]
async fn multipart_full_read_preserves_legacy_zero_and_negative_part_sizes() {
let key = [0x77; 32];
let part_sizes = [5 * 1024 * 1024, 1024 * 1024];
let encrypted = build_legacy_ssec_multipart_fixture(key, &part_sizes).await;
// The encrypted case supplies the fixture key explicitly. This covers
// full decrypted reads, not managed-key acquisition.
for (kind, fixture, headers) in [
("compressed", compressed_multipart_fixture(&part_sizes).await, HeaderMap::new()),
(
"encrypted with supplied key",
CompressedMultipartFixture {
object_info: encrypted.object_info,
stored: encrypted.ciphertext,
plaintext: encrypted.plaintext,
},
ssec_headers_from_key(key),
),
] {
let source_etag = faster_hex::hex_string(Md5::digest(&fixture.plaintext).as_ref());
assert_eq!(source_etag.len(), 32);
assert_eq!(fixture.plaintext.len(), 6 * 1024 * 1024);
for part_index in 0..part_sizes.len() {
assert!(fixture.object_info.parts[part_index].actual_size > 0, "the selected part is nonempty");
for actual_size in [0, -1] {
let mut object_info = fixture.object_info.clone();
object_info.etag = Some(source_etag.clone());
Arc::make_mut(&mut object_info.parts)[part_index].actual_size = actual_size;
let (mut reader, offset, length) = GetObjectReader::new(
Box::new(Cursor::new(fixture.stored.clone())),
None,
&object_info,
&ObjectOptions::default(),
&headers,
)
.await
.expect("the authoritative total size must keep full legacy reads available");
assert_eq!(offset, 0);
assert_eq!(length, i64::try_from(fixture.stored.len()).expect("stored size fits"));
let mut body = Vec::new();
reader
.stream
.read_to_end(&mut body)
.await
.expect("full read must reach EOF despite an unspecified per-part logical size");
assert_eq!(
body, fixture.plaintext,
"{kind}: part {part_index} with actual_size={actual_size} must not lose readable data"
);
assert_eq!(reader.object_info.etag.as_deref(), Some(source_etag.as_str()));
}
}
}
}
/// The physical part sizes must add up to `oi.size` for a seek to be safe;
/// inconsistent metadata must fall back to the previous full-object read
/// instead of scheduling an erasure read past the object end.
+1 -30
View File
@@ -1597,7 +1597,7 @@ impl ObjectInfo {
}
pub fn is_multipart(&self) -> bool {
self.parts.len() > 1 || self.etag.as_ref().is_some_and(|v| v.len() != 32)
self.etag.as_ref().is_some_and(|v| v.len() != 32)
}
pub fn is_encrypted(&self) -> bool {
@@ -2235,35 +2235,6 @@ mod tests {
}
use rustfs_filemeta::{FileInfo, FileMeta, MetaCacheEntry, TRANSITION_COMPLETE};
#[test]
fn multipart_identity_uses_stored_parts_and_preserves_the_etag_fallback() {
let plain_etag = "0123456789abcdef0123456789abcdef";
let multipart_etag = "0123456789abcdef0123456789abcdef-1";
for (case, part_count, etag, expected) in [
("preserved source ETag", 2, Some(plain_etag), true),
("missing ETag", 2, None, true),
("ordinary PUT", 1, Some(plain_etag), false),
("ordinary PUT without ETag", 1, None, false),
("single-part MPU", 1, Some(multipart_etag), true),
("legacy MPU without parts", 0, Some(multipart_etag), true),
] {
let object = ObjectInfo {
etag: etag.map(str::to_string),
parts: Arc::new(
(1..=part_count)
.map(|number| ObjectPartInfo {
number,
..Default::default()
})
.collect(),
),
..Default::default()
};
assert_eq!(object.is_multipart(), expected, "{case}");
}
}
fn inline_fast_path_object(size: i64, versioned: bool) -> ObjectInfo {
ObjectInfo {
size,
+19 -305
View File
@@ -33,11 +33,11 @@ use rustfs_madmin::net::NetInfo;
use rustfs_madmin::{ItemState, ServerProperties, StorageInfo};
use rustfs_utils::XHost;
use sha2::{Digest, Sha256};
use std::collections::{BTreeMap, BTreeSet, HashMap, hash_map::DefaultHasher};
use std::collections::{BTreeMap, HashMap, hash_map::DefaultHasher};
use std::future::Future;
use std::hash::{Hash, Hasher};
use std::sync::{
Arc, LazyLock, Mutex, OnceLock,
Arc, Mutex, OnceLock,
atomic::{AtomicBool, AtomicUsize, Ordering},
};
use std::time::{Duration, Instant, SystemTime};
@@ -311,20 +311,12 @@ pub struct LegacyTransitionStateReconcileFleetProofToken {
_permit: FleetCapabilityProofPermit,
}
/// Effect-window authority for one immutable ILM recovery export.
pub struct IlmRecoveryExportFleetProofToken {
token: FleetCapabilityProofToken,
_permit: FleetCapabilityProofPermit,
}
static REMOTE_VERSION_STATE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static CROSS_POOL_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static TIER_DELETE_JOURNAL_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static DECOMMISSION_TARGET_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static LEGACY_TRANSITION_STATE_RECONCILE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static ILM_RECOVERY_EXPORT_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static REMOTE_VERSION_STATE_PROBE_TOPOLOGY: OnceLock<String> = OnceLock::new();
static ILM_RECOVERY_EXPORT_LOCAL_PROCESS_EPOCH: LazyLock<Uuid> = LazyLock::new(Uuid::new_v4);
fn cross_pool_fence_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
CROSS_POOL_FENCE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
@@ -346,10 +338,6 @@ fn legacy_transition_state_reconcile_fleet_proof_slot() -> &'static std::sync::R
LEGACY_TRANSITION_STATE_RECONCILE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
}
fn ilm_recovery_export_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
ILM_RECOVERY_EXPORT_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
}
fn revoke_fleet_capability_proof_state(state: &mut FleetCapabilityProofState) {
if let Some(proof) = state.proof.take() {
proof.generation.revoke();
@@ -585,117 +573,6 @@ pub async fn legacy_transition_state_reconcile_fleet_proof_matches(
.await
}
pub async fn acquire_ilm_recovery_export_fleet_proof() -> Option<IlmRecoveryExportFleetProofToken> {
let expected_topology = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get()?;
let proof = {
let state = ilm_recovery_export_fleet_proof_slot()
.read()
.unwrap_or_else(std::sync::PoisonError::into_inner);
acquire_ilm_recovery_export_fleet_proof_from(&state, expected_topology, Instant::now())?
};
let observed = observe_ilm_recovery_export_fleet(expected_topology).await?;
let state = ilm_recovery_export_fleet_proof_slot()
.read()
.unwrap_or_else(std::sync::PoisonError::into_inner);
ilm_recovery_export_fleet_proof_matches_observation_at(&state, &proof, expected_topology, &observed, Instant::now())
.then_some(proof)
}
fn acquire_ilm_recovery_export_fleet_proof_from(
state: &FleetCapabilityProofState,
expected_topology: &str,
now: Instant,
) -> Option<IlmRecoveryExportFleetProofToken> {
let token = acquire_fleet_capability_proof_from(state, expected_topology, now)?;
let permit = state.proof.as_ref()?.generation.try_acquire()?;
Some(IlmRecoveryExportFleetProofToken { token, _permit: permit })
}
pub async fn ilm_recovery_export_fleet_proof_matches(proof: &IlmRecoveryExportFleetProofToken) -> bool {
let Some(expected_topology) = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get() else {
return false;
};
{
let state = ilm_recovery_export_fleet_proof_slot()
.read()
.unwrap_or_else(std::sync::PoisonError::into_inner);
if !ilm_recovery_export_fleet_proof_matches_at(&state, proof, expected_topology, Instant::now()) {
return false;
}
}
let Some(observed) = observe_ilm_recovery_export_fleet(expected_topology).await else {
return false;
};
let state = ilm_recovery_export_fleet_proof_slot()
.read()
.unwrap_or_else(std::sync::PoisonError::into_inner);
ilm_recovery_export_fleet_proof_matches_observation_at(&state, proof, expected_topology, &observed, Instant::now())
}
pub fn ilm_recovery_export_topology_generation(proof: &IlmRecoveryExportFleetProofToken) -> String {
let mut hasher = Sha256::new();
hasher.update(b"rustfs-ilm-recovery-export-topology-v1\0");
hasher.update(proof.token.topology_fingerprint.as_bytes());
rustfs_utils::crypto::hex(hasher.finalize().as_slice())
}
pub fn ilm_recovery_export_member_epochs_sha256(proof: &IlmRecoveryExportFleetProofToken) -> String {
let encoded = serde_json::to_vec(proof.token.peer_epochs.as_ref()).expect("member epoch map is JSON encodable");
let mut hasher = Sha256::new();
hasher.update(b"rustfs-ilm-recovery-export-members-v1\0");
hasher.update(encoded);
rustfs_utils::crypto::hex(hasher.finalize().as_slice())
}
pub fn ilm_recovery_export_local_process_epoch() -> Uuid {
*ILM_RECOVERY_EXPORT_LOCAL_PROCESS_EPOCH
}
fn ilm_recovery_export_fleet_proof_matches_at(
state: &FleetCapabilityProofState,
proof: &IlmRecoveryExportFleetProofToken,
expected_topology: &str,
now: Instant,
) -> bool {
proof._permit.generation.is_accepting()
&& fleet_capability_proof_matches_at(state, &proof.token, expected_topology, now)
&& state
.proof
.as_ref()
.is_some_and(|current| Arc::ptr_eq(&current.generation, &proof._permit.generation))
}
fn ilm_recovery_export_fleet_proof_matches_observation_at(
state: &FleetCapabilityProofState,
proof: &IlmRecoveryExportFleetProofToken,
expected_topology: &str,
observed: &BTreeMap<String, Uuid>,
now: Instant,
) -> bool {
ilm_recovery_export_fleet_proof_matches_at(state, proof, expected_topology, now)
&& proof.token.peer_epochs.as_ref() == observed
}
async fn observe_ilm_recovery_export_fleet(expected_topology: &str) -> Option<BTreeMap<String, Uuid>> {
#[cfg(test)]
{
let state = ilm_recovery_export_fleet_proof_slot()
.read()
.unwrap_or_else(std::sync::PoisonError::into_inner);
if fleet_capability_proof_valid_at(state.proof.as_ref(), expected_topology, Instant::now()) {
return state.proof.as_ref().map(|proof| proof.peer_epochs.as_ref().clone());
}
}
let notification_sys = get_global_notification_sys()?;
timeout(
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
notification_sys.probe_ilm_recovery_export_fleet(expected_topology),
)
.await
.ok()?
.ok()
}
async fn legacy_transition_state_reconcile_fleet_proof_matches_with_observer<F, Fut>(
slot: &std::sync::RwLock<FleetCapabilityProofState>,
proof: &LegacyTransitionStateReconcileFleetProofToken,
@@ -784,7 +661,7 @@ pub(crate) fn install_cross_pool_fence_fleet_proof_for_test() {
state.proof.clone()
} else {
Some(FleetCapabilityProof::new(
topology.clone(),
topology,
Arc::new(BTreeMap::new()),
now + Duration::from_secs(60 * 60),
))
@@ -817,21 +694,6 @@ pub(crate) fn install_cross_pool_fence_fleet_proof_for_test() {
decommission_state.topology_conflict = false;
decommission_state.draining_generation = None;
decommission_state.proof = proof.as_ref().map(FleetCapabilityProof::with_fresh_generation);
drop(decommission_state);
let mut export_state = ilm_recovery_export_fleet_proof_slot()
.write()
.unwrap_or_else(std::sync::PoisonError::into_inner);
if !fleet_capability_proof_valid_at(export_state.proof.as_ref(), &topology, now) {
debug_assert!(
export_state
.proof
.as_ref()
.is_none_or(|current| current.generation.is_drained())
);
export_state.topology_conflict = false;
export_state.draining_generation = None;
export_state.proof = proof.as_ref().map(FleetCapabilityProof::with_fresh_generation);
}
}
#[cfg(test)]
@@ -1088,7 +950,6 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
tier_delete_journal_fleet_proof_slot(),
decommission_target_fence_fleet_proof_slot(),
legacy_transition_state_reconcile_fleet_proof_slot(),
ilm_recovery_export_fleet_proof_slot(),
] {
mark_fleet_capability_topology_conflict(slot);
}
@@ -1098,42 +959,29 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
tokio::spawn(async move {
loop {
let notification_sys = get_global_notification_sys();
let remote_version_state_probe = async {
match notification_sys.as_ref() {
Some(notification_sys) => timeout(
let result = match get_global_notification_sys() {
Some(notification_sys) => {
match timeout(
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
notification_sys.probe_remote_version_state_fleet(&topology_fingerprint),
)
.await
.unwrap_or_else(|_| Err(Error::other("remote version state fleet capability probe timed out"))),
None => Err(Error::other("remote version state fleet capability notification system is unavailable")),
{
Ok(result) => result,
Err(_) => Err(Error::other("remote version state fleet capability probe timed out")),
}
}
None => Err(Error::other("remote version state fleet capability notification system is unavailable")),
};
let cross_pool_fence_probe = async {
match notification_sys.as_ref() {
Some(notification_sys) => timeout(
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
notification_sys.probe_cross_pool_fence_fleet(&topology_fingerprint),
)
.await
.unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))),
None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")),
}
let fence_probe = match get_global_notification_sys() {
Some(notification_sys) => timeout(
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
notification_sys.probe_cross_pool_fence_fleet(&topology_fingerprint),
)
.await
.unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))),
None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")),
};
let recovery_export_probe = async {
match notification_sys.as_ref() {
Some(notification_sys) => timeout(
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
notification_sys.probe_ilm_recovery_export_fleet(&topology_fingerprint),
)
.await
.unwrap_or_else(|_| Err(Error::other("ILM recovery export fleet capability probe timed out"))),
None => Err(Error::other("ILM recovery export fleet capability notification system is unavailable")),
}
};
let (result, fence_probe, recovery_export_result) =
tokio::join!(remote_version_state_probe, cross_pool_fence_probe, recovery_export_probe);
let (fence_result, journal_result, decommission_target_fence_result, reconcile_result) = match fence_probe {
Ok((peer_epochs, minimum_version)) => cross_pool_fence_policy_results(peer_epochs, minimum_version),
Err(err) => {
@@ -1156,7 +1004,6 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
revoke_fleet_capability_proof(tier_delete_journal_fleet_proof_slot());
revoke_fleet_capability_proof(decommission_target_fence_fleet_proof_slot());
revoke_fleet_capability_proof(legacy_transition_state_reconcile_fleet_proof_slot());
revoke_fleet_capability_proof(ilm_recovery_export_fleet_proof_slot());
} else if let Some(err) = publish_fleet_capability_probe_result(
remote_version_state_fleet_proof_slot(),
&topology_fingerprint,
@@ -1183,24 +1030,6 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
"notification capability probe"
);
}
if !topology_conflict
&& let Some(err) = publish_fleet_capability_probe_result(
ilm_recovery_export_fleet_proof_slot(),
&topology_fingerprint,
recovery_export_result,
Instant::now(),
)
{
debug!(
event = EVENT_NOTIFICATION_CAPABILITY_PROBE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_NOTIFICATION,
capability = "ilm_recovery_export_v1",
state = "failed_closed",
error = %err,
"notification capability probe"
);
}
if !topology_conflict
&& let Some(err) = publish_fleet_capability_probe_result(
tier_delete_journal_fleet_proof_slot(),
@@ -1345,46 +1174,6 @@ impl NotificationSys {
}
Ok((peer_epochs, minimum_version))
}
async fn probe_ilm_recovery_export_fleet(&self, topology_fingerprint: &str) -> Result<BTreeMap<String, Uuid>> {
if self.peer_clients.len() != self.peer_topology_hosts.len() {
return Err(Error::other("ILM recovery export capability fleet membership is incomplete"));
}
let local_member = runtime_sources::local_node_name().await;
if local_member.trim().is_empty() {
return Err(Error::other("ILM recovery export local member identity is unavailable"));
}
let mut peer_epochs = BTreeMap::new();
insert_remote_version_state_peer(&mut peer_epochs, local_member.clone(), ilm_recovery_export_local_process_epoch())?;
let probes = self.peer_clients.iter().map(|client| async {
let client = client
.as_ref()
.ok_or_else(|| Error::other("ILM recovery export capability peer is unreachable"))?;
client.probe_ilm_recovery_export(topology_fingerprint.to_string()).await
});
for result in join_all(probes).await {
let (peer, epoch) = result?;
insert_remote_version_state_peer(&mut peer_epochs, peer, epoch)?;
}
validate_ilm_recovery_export_members(&self.peer_topology_hosts, &local_member, &peer_epochs)?;
Ok(peer_epochs)
}
}
fn validate_ilm_recovery_export_members(
expected_remote_members: &[String],
local_member: &str,
observed: &BTreeMap<String, Uuid>,
) -> Result<()> {
let expected = expected_remote_members
.iter()
.cloned()
.chain(std::iter::once(local_member.to_string()))
.collect::<BTreeSet<_>>();
if expected.len() != expected_remote_members.len().saturating_add(1) || observed.keys().ne(expected.iter()) {
return Err(Error::other("ILM recovery export capability fleet membership does not match topology"));
}
Ok(())
}
/// Rolling tier activity summed over every cluster member that answered, with
@@ -3718,81 +3507,6 @@ mod tests {
assert!(captured != restarted.token());
}
#[test]
fn ilm_recovery_export_member_digest_is_order_independent_and_epoch_bound() {
let now = Instant::now();
let local_epoch = ilm_recovery_export_local_process_epoch();
assert!(!local_epoch.is_nil());
assert_eq!(local_epoch, ilm_recovery_export_local_process_epoch());
let remote_epoch = Uuid::new_v4();
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let peers = BTreeMap::from([("node-b".to_string(), remote_epoch), ("node-a".to_string(), local_epoch)]);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(peers), now).is_none());
let proof = {
let state = slot.read().expect("export proof slot should not poison");
acquire_ilm_recovery_export_fleet_proof_from(&state, "topology-a", now).expect("complete fleet should admit export")
};
let digest = ilm_recovery_export_member_epochs_sha256(&proof);
let changed_slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let changed = BTreeMap::from([("node-a".to_string(), local_epoch), ("node-b".to_string(), Uuid::new_v4())]);
assert!(publish_fleet_capability_probe_result(&changed_slot, "topology-a", Ok(changed), now).is_none());
let changed_proof = {
let state = changed_slot.read().expect("export proof slot should not poison");
acquire_ilm_recovery_export_fleet_proof_from(&state, "topology-a", now).expect("complete fleet should admit export")
};
assert_ne!(digest, ilm_recovery_export_member_epochs_sha256(&changed_proof));
}
#[test]
fn ilm_recovery_export_members_must_match_the_exact_topology() {
let expected_remote = vec!["node-b".to_string()];
let local = "node-a";
let complete = BTreeMap::from([
(local.to_string(), Uuid::new_v4()),
(expected_remote[0].clone(), Uuid::new_v4()),
]);
assert!(validate_ilm_recovery_export_members(&expected_remote, local, &complete).is_ok());
let unexpected = BTreeMap::from([(local.to_string(), Uuid::new_v4()), ("node-c".to_string(), Uuid::new_v4())]);
assert!(validate_ilm_recovery_export_members(&expected_remote, local, &unexpected).is_err());
assert!(
validate_ilm_recovery_export_members(&[local.to_string()], local, &complete).is_err(),
"the configured remote set cannot repeat the local member"
);
}
#[test]
fn ilm_recovery_export_restart_revokes_authority_until_permit_drains() {
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let now = Instant::now();
let original = BTreeMap::from([("node-a".to_string(), Uuid::new_v4())]);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(original), now).is_none());
let admitted = {
let state = slot.read().expect("export proof slot should not poison");
acquire_ilm_recovery_export_fleet_proof_from(&state, "topology-a", now).expect("fresh fleet should admit export")
};
let restarted = BTreeMap::from([("node-a".to_string(), Uuid::new_v4())]);
let draining = publish_fleet_capability_probe_result(&slot, "topology-a", Ok(restarted.clone()), now)
.expect("restart must wait for the admitted export effect window");
assert!(draining.to_string().contains("previous generation to drain"));
{
let state = slot.read().expect("export proof slot should not poison");
assert!(!ilm_recovery_export_fleet_proof_matches_at(&state, &admitted, "topology-a", now));
assert!(
acquire_ilm_recovery_export_fleet_proof_from(&state, "topology-a", now).is_none(),
"successor authority must wait for the old effect window to drain"
);
}
drop(admitted);
assert!(
publish_fleet_capability_probe_result(&slot, "topology-a", Ok(restarted), now + Duration::from_millis(1)).is_none()
);
let state = slot.read().expect("export proof slot should not poison");
assert!(acquire_ilm_recovery_export_fleet_proof_from(&state, "topology-a", now).is_some());
}
#[test]
fn tier_delete_journal_generation_is_stable_across_members_and_process_restarts() {
let topology = "topology-a";
+14 -142
View File
@@ -5458,7 +5458,7 @@ impl TierConfigMgr {
let manager = handle.read().await;
let published_digest = if intents
.iter()
.any(|recovered| recovered.intent.state == TierMutationIntentState::Committed)
.any(|recovered| recovered.is_peer_only_terminal() && recovered.intent.state == TierMutationIntentState::Committed)
{
Some(tier_config_candidate_digest(&manager).map_err(|err| {
let mut admin_err = ERR_TIER_INVALID_CONFIG.clone();
@@ -5468,29 +5468,18 @@ impl TierConfigMgr {
} else {
None
};
let locally_published_committed_mutations = intents
.iter()
.filter(|recovered| {
recovered.intent.state == TierMutationIntentState::Committed
&& published_digest == Some(recovered.intent.candidate_digest)
})
.map(|recovered| recovered.intent.mutation_id)
.collect::<HashSet<_>>();
let mut prepared_mutation_blocks = HashMap::new();
let mut committed_mutation_blocks: HashMap<String, HashSet<uuid::Uuid>> = HashMap::new();
for recovered in intents {
// A matching in-memory manager has already crossed the local
// publication boundary. Keep replaying and durably cleaning the
// record, but do not re-fence object operations while that
// terminal work finishes.
let skip_runtime_fence = locally_published_committed_mutations.contains(&recovered.intent.mutation_id)
|| (recovered.is_peer_only_terminal()
&& match recovered.intent.state {
TierMutationIntentState::Aborted => true,
TierMutationIntentState::Committed => !retain_missing_mutation_blocks,
TierMutationIntentState::Prepared => false,
});
if skip_runtime_fence {
let settled_tombstone = recovered.is_peer_only_terminal()
&& match recovered.intent.state {
TierMutationIntentState::Aborted => true,
TierMutationIntentState::Committed => {
!retain_missing_mutation_blocks || published_digest == Some(recovered.intent.candidate_digest)
}
TierMutationIntentState::Prepared => false,
};
if settled_tombstone {
continue;
}
Self::collect_prepared_mutation_intent_block(&mut prepared_mutation_blocks, &recovered.intent)?;
@@ -5503,9 +5492,6 @@ impl TierConfigMgr {
}
if retain_missing_mutation_blocks {
for (tier_name, mutation_id) in &runtime.prepared_mutation_blocks {
if locally_published_committed_mutations.contains(mutation_id) {
continue;
}
match prepared_mutation_blocks.entry(tier_name.clone()) {
Entry::Vacant(entry) => {
entry.insert(*mutation_id);
@@ -5519,15 +5505,10 @@ impl TierConfigMgr {
}
}
for (tier_name, mutation_ids) in &runtime.committed_mutation_blocks {
for mutation_id in mutation_ids {
if locally_published_committed_mutations.contains(mutation_id) {
continue;
}
committed_mutation_blocks
.entry(tier_name.clone())
.or_default()
.insert(*mutation_id);
}
committed_mutation_blocks
.entry(tier_name.clone())
.or_default()
.extend(mutation_ids);
}
}
let changed = runtime.prepared_mutation_blocks != prepared_mutation_blocks
@@ -11317,115 +11298,6 @@ mod tests {
);
}
#[tokio::test]
async fn published_dual_terminal_intent_does_not_restore_local_runtime_fence_during_replay() {
use crate::services::tier::tier_mutation_intent::save_tier_mutation_intent_record;
let store = Arc::new(CasConfigStore::default());
let mut persisted = empty_mgr();
persisted.tiers.insert("COLD-A".to_string(), build_rustfs_tier("COLD-A"));
persisted
.save_tiering_config_if_current(store.clone(), None)
.await
.expect("published tier config fixture should persist");
let (_, current_etag) = load_tier_config_for_update(store.clone())
.await
.expect("published tier config fixture should load with metadata");
let current_etag = current_etag.expect("published tier config fixture should have an ETag");
let affected_targets = build_tier_mutation_affected_targets(
TierMutationIntentKind::Add,
HashSet::from(["COLD-A".to_string()]),
&empty_mgr(),
&persisted,
)
.expect("published AddTier targets should build");
let mut intent = build_coordinator_tier_mutation_intent(TierMutationIntentKind::Add, None, &persisted, affected_targets)
.expect("published AddTier intent should build")
.expect("published AddTier should require a durable intent");
intent
.advance(TierMutationIntentState::Committed, Some(current_etag))
.expect("published AddTier intent should commit");
save_tier_coordinator_mutation_intent_record_if_absent(store.clone(), &intent)
.await
.expect("published coordinator intent should persist");
save_tier_mutation_intent_record(store.clone(), &intent)
.await
.expect("published peer intent should persist");
let manager = TierConfigMgr::new();
{
let mut guard = manager.write().await;
install_lease_backend(&mut guard, "COLD-A", LeaseTestBackend::ready("published"));
}
{
let guard = manager.read().await;
assert_eq!(
tier_config_candidate_digest(&guard).expect("published manager digest should build"),
intent.candidate_digest
);
}
TierConfigMgr::apply_committed_mutation_intent_block(&manager, &intent)
.await
.expect("pre-existing committed runtime fence should install");
assert!(
TierConfigMgr::acquire_operation_lease(&manager, "COLD-A").await.is_err(),
"fixture must begin with the committed runtime fence installed"
);
let started = Arc::new(Notify::new());
let release = Arc::new(tokio::sync::Semaphore::new(0));
TIER_MUTATION_TEST_PEERS
.scope(
vec![Arc::new(BlockingCommitTierMutationPeer {
started: started.clone(),
release: release.clone(),
})],
async {
let reload = TierConfigMgr::reload_handle_with(&manager, store.clone());
tokio::pin!(reload);
tokio::time::timeout(Duration::from_secs(5), async {
tokio::select! {
result = &mut reload => panic!("reload finished before terminal replay was released: {result:?}"),
_ = started.notified() => {}
}
})
.await
.expect("terminal replay should reach the blocking peer");
let lease = TierConfigMgr::acquire_operation_lease(&manager, "COLD-A")
.await
.expect("terminal cleanup must not re-fence an already-published tier");
drop(lease);
release.add_permits(1);
tokio::time::timeout(Duration::from_secs(5), &mut reload)
.await
.expect("terminal replay should finish after the peer responds")
.expect("terminal replay should succeed after the peer responds");
},
)
.await;
assert!(manager.read().await.tiers.contains_key("COLD-A"));
assert!(
TierConfigMgr::load_coordinator_mutation_intents(store.clone())
.await
.expect("coordinator cleanup should be readable")
.is_empty()
);
assert_eq!(
TierConfigMgr::load_tier_mutation_intents(store)
.await
.expect("retained peer tombstone should be readable"),
vec![intent]
);
let guard = manager.read().await;
let runtime = registered_tier_driver_runtime(&guard).expect("runtime should remain registered");
assert!(
lock_unpoisoned(&runtime).committed_mutation_blocks.is_empty(),
"retained terminal evidence must not restore the published runtime fence"
);
}
#[tokio::test]
async fn peer_terminal_tombstone_gc_uses_etag_and_retains_racing_replacement() {
use crate::services::tier::tier_mutation_intent::{
+2 -302
View File
@@ -3301,18 +3301,13 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
// (rustfs/backlog#1009): CompleteMultipartUpload keeps its pre-commit
// `get_object_info` lookup, so the backfill has no consumer here yet.
Self::assign_rename_data_indexes(&mut parts_metadatas);
// Disk deadlines can expire before physical publication or failure undo drains.
let mut rename_result = SetDisks::rename_data_owned_with_fence(
let mut rename_result = SetDisks::rename_data_owned(
&commit_disks,
(RUSTFS_META_MULTIPART_BUCKET, &commit_upload_id_path),
parts_metadatas,
(&commit_bucket, &commit_object),
write_quorum,
commit_allows_early_ack,
crate::set_disk::core::io_primitives::RenameDataFenceOptions::new(write_quorum, None)
.with_namespace_commit_guard(
(!crate::bucket::utils::is_meta_bucketname(&commit_bucket))
.then(|| commit_set.ctx.begin_namespace_commit()),
),
)
.await;
if let Ok(rename_commit) = rename_result.as_mut() {
@@ -6763,301 +6758,6 @@ mod tests {
.await;
}
#[tokio::test]
#[serial(capacity_dirty_scope)]
async fn complete_multipart_advances_namespace_generation_after_commit() {
let (dirs, disks, set_disks) = hermetic_set_disks(4).await;
let bucket = "multipart-namespace-commit";
let object = "completed-object";
let body = vec![0x65; 4096];
make_bucket_on_all(&disks, bucket).await;
let before = set_disks.ctx.namespace_commit_generation();
let (upload_id, parts) =
stage_upload_with_create_opts(&set_disks, bucket, object, &body, &ObjectOptions::default()).await;
assert_eq!(set_disks.ctx.namespace_commit_generation(), before, "staging is not publication");
assert!(!set_disks.ctx.namespace_commits_pending());
tokio::time::timeout(
Duration::from_secs(10),
set_disks
.clone()
.complete_multipart_upload(bucket, object, &upload_id, parts, &ObjectOptions::default()),
)
.await
.expect("completion must finish")
.expect("the four real shards must commit");
let mut reader = tokio::time::timeout(
Duration::from_secs(5),
set_disks.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default()),
)
.await
.expect("GET after completion must finish")
.expect("a successful completion must be immediately readable");
let mut actual = Vec::new();
tokio::time::timeout(Duration::from_secs(5), reader.stream.read_to_end(&mut actual))
.await
.expect("the completed body stream must finish")
.expect("read completed body");
assert_eq!(actual, body);
let upload_path = SetDisks::get_upload_id_dir(bucket, object, &upload_id);
for dir in &dirs {
assert!(!dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_path).exists());
}
assert!(matches!(
set_disks.check_upload_id_exists(bucket, object, &upload_id, false).await,
Err(StorageError::InvalidUploadID(..))
));
assert!(!set_disks.ctx.namespace_commits_pending());
assert_eq!(
set_disks.ctx.namespace_commit_generation(),
before + 2,
"one completed MPU must invalidate snapshots at namespace admission and physical retirement"
);
}
#[cfg(not(windows))]
async fn assert_complete_multipart_physical_namespace_owner(undo: bool) {
use crate::disk::os::{self, prepared_publication_test_hooks as hooks};
use crate::set_disk::core::io_primitives::rename_fault_injection;
use futures::FutureExt;
temp_env::async_with_vars([(rustfs_config::ENV_DRIVE_MAX_TIMEOUT_DURATION, Some("60"))], async {
let (dirs, disks, set_disks) = hermetic_set_disks(4).await;
let bucket = "multipart-physical-namespace";
let object = if undo { "undo-tail" } else { "publication-tail" };
let old_body = vec![0x41; 1024];
let new_body = vec![0x62; 4096];
make_bucket_on_all(&disks, bucket).await;
let old = set_disks
.put_object(
bucket,
object,
&mut PutObjReader::from_vec(old_body.clone()),
&ObjectOptions {
write_completion: crate::object_api::WriteCompletion::TailDrained,
..Default::default()
},
)
.await
.expect("seed a real readable old version");
let old_etag = old.etag.expect("the committed old object must have an ETag");
let before = set_disks.ctx.namespace_commit_generation();
assert!(!set_disks.ctx.namespace_commits_pending());
let (upload_id, parts) =
stage_upload_with_create_opts(&set_disks, bucket, object, &new_body, &ObjectOptions::default()).await;
let new_etag = get_complete_multipart_md5(&parts);
let upload_path = SetDisks::get_upload_id_dir(bucket, object, &upload_id);
for dir in &dirs {
assert!(dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_path).exists());
}
assert_eq!(set_disks.ctx.namespace_commit_generation(), before);
let _fault = undo.then(|| rename_fault_injection::fail_rename_on(object, &[2, 3]));
let stage = if undo {
hooks::Stage::Rename
} else {
hooks::Stage::PreparedRename
};
let (entered_tx, mut entered_rx) = tokio::sync::mpsc::unbounded_channel();
let mut hooks = Vec::new();
let mut releases = Vec::new();
let mut mutation_paths = Vec::new();
for (index, disk) in disks.iter().enumerate() {
let disk::Disk::Local(local) = disk.as_ref() else {
panic!("physical MPU fixture requires local disks");
};
let destination = local
.get_disk()
.get_object_path_for_io(bucket, object)
.expect("leased IO path");
let entered_tx = entered_tx.clone();
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
hooks.push(hooks::install_at(stage, &destination.join(STORAGE_FORMAT_FILE), move || {
let _ = entered_tx.send(index);
// Dropping senders releases every syscall on assertion failure, too.
let _ = release_rx.recv();
}));
releases.push(release_tx);
// Canonical rename serializes the object directory; backup restore
// serializes its xl.meta destination. Drain the actual executor key.
mutation_paths.push(if undo {
destination.join(STORAGE_FORMAT_FILE)
} else {
destination
});
}
drop(entered_tx);
let complete_set = set_disks.clone();
let complete_upload = upload_id.clone();
let mut complete = tokio::spawn(async move {
complete_set
.complete_multipart_upload(bucket, object, &complete_upload, parts, &ObjectOptions::default())
.await
});
let mut complete_joined = false;
let mut observed_counts = None;
let observations = std::panic::AssertUnwindSafe(async {
let expected_publishers = if undo { 2 } else { 4 };
let entered = tokio::time::timeout(Duration::from_secs(10), async {
let mut entered = HashSet::new();
while entered.len() < expected_publishers {
tokio::select! {
index = entered_rx.recv() => {
assert!(entered.insert(index.expect("physical publisher must signal entry")));
}
result = &mut complete => {
complete_joined = true;
panic!("completion returned before physical entry: {result:?}");
}
}
}
entered
})
.await
.expect("all expected physical metadata operations must enter");
let pending_at_entry = set_disks.ctx.namespace_commits_pending();
let generation_at_entry = set_disks.ctx.namespace_commit_generation();
for &index in &entered {
let metadata = tokio::time::timeout(
Duration::from_secs(5),
disks[index].read_version("", bucket, object, "", &ReadOptions::default()),
)
.await
.expect("metadata observation must finish while publication is paused")
.expect("metadata before the paused physical action must be readable");
assert_eq!(
metadata.metadata.get("etag"),
Some(if undo { &new_etag } else { &old_etag }),
"undo must follow actual publication; prepared rename must precede publication"
);
}
// The entry signals run inside the real blocking closures, after each
// wrapper installed its normal deadline. No quota/external guard disables it.
tokio::time::pause();
tokio::time::advance(Duration::from_secs(61)).await;
tokio::time::resume();
let joined = tokio::time::timeout(Duration::from_secs(5), &mut complete).await;
complete_joined = joined.is_ok();
let result = joined
.expect("ordinary MPU disk/undo deadlines must still return before physical drain")
.expect("completion task must not panic");
assert!(result.is_err(), "a timed-out or two-shard commit cannot acknowledge success");
let pending_after_timeout = set_disks.ctx.namespace_commits_pending();
let generation_after_timeout = set_disks.ctx.namespace_commit_generation();
observed_counts = Some((pending_at_entry, generation_at_entry, pending_after_timeout, generation_after_timeout));
for &index in &entered {
assert!(
os::acquire_rename_data_mutation_lease(&disks[index].path(), bucket, &mutation_paths[index])
.now_or_never()
.is_none(),
"timed-out physical work must still own its object serialization"
);
}
for dir in &dirs {
assert!(
dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_path).exists(),
"failed completion must not clean the upload staging"
);
}
})
.catch_unwind()
.await;
// Release even after a failed observation, then finish dispatch before
// draining every physical key. No per-disk assertion may skip a later drain.
drop(releases);
drop(hooks);
let coordinator_drained = complete_joined
|| tokio::time::timeout(Duration::from_secs(10), &mut complete).await.is_ok();
if !coordinator_drained {
complete.abort();
let _ = tokio::time::timeout(Duration::from_secs(5), &mut complete).await;
}
let drains = futures::future::join_all(disks.iter().zip(&mutation_paths).map(|(disk, destination)| async move {
tokio::time::timeout(
Duration::from_secs(5),
os::acquire_rename_data_mutation_lease(&disk.path(), bucket, destination),
)
.await
.map(drop)
}))
.await;
let owner_drained = tokio::time::timeout(Duration::from_secs(5), async {
while set_disks.ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
})
.await;
let physical_drained = drains.iter().all(|drain| drain.is_ok());
if !coordinator_drained || !physical_drained || owner_drained.is_err() {
// A bounded cleanup failure cannot justify deleting roots that a
// detached executor might still use. Keep them for diagnosis.
let retained = dirs.into_iter().map(TempDir::keep).collect::<Vec<_>>();
eprintln!("MPU cleanup incomplete: coordinator={coordinator_drained}, physical={physical_drained}, retained={retained:?}");
if let Err(panic) = observations {
std::panic::resume_unwind(panic);
}
panic!("MPU cleanup did not drain: coordinator={coordinator_drained}, physical={physical_drained}, retained={retained:?}");
}
if let Err(panic) = observations {
std::panic::resume_unwind(panic);
}
// Preserve the original drain checks after collecting every result.
for drained in drains {
drained.expect("released physical MPU work must drain");
}
owner_drained.expect("physical retirement must finish its namespace counter decrement");
for disk in &disks {
let metadata = disk
.read_version("", bucket, object, "", &ReadOptions::default())
.await
.expect("all disks must expose the expected final metadata");
assert_eq!(metadata.metadata.get("etag"), Some(if undo { &old_etag } else { &new_etag }));
}
let (pending_at_entry, generation_at_entry, pending_after_timeout, generation_after_timeout) =
observed_counts.expect("successful observations must record namespace counters");
let mut reader = tokio::time::timeout(
Duration::from_secs(5),
set_disks.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default()),
)
.await
.expect("GET after physical drain must finish")
.expect("the real final object must be readable");
let mut actual = Vec::new();
tokio::time::timeout(Duration::from_secs(5), reader.stream.read_to_end(&mut actual))
.await
.expect("the final object stream must finish")
.expect("read final object bytes");
assert_eq!(actual, if undo { old_body } else { new_body });
assert!(
pending_at_entry && pending_after_timeout,
"physical MPU work outlived namespace accounting: undo={undo}"
);
assert_eq!(generation_at_entry, before + 1);
assert_eq!(
generation_after_timeout, generation_at_entry,
"the blocked physical owner cannot retire early"
);
assert_eq!(set_disks.ctx.namespace_commit_generation(), before + 2);
assert!(!set_disks.ctx.namespace_commits_pending());
})
.await;
}
#[cfg(not(windows))]
#[tokio::test]
#[serial(capacity_dirty_scope)]
async fn complete_multipart_timeout_keeps_namespace_owner_until_physical_publication() {
assert_complete_multipart_physical_namespace_owner(false).await;
}
#[cfg(not(windows))]
#[tokio::test]
#[serial(capacity_dirty_scope)]
async fn complete_multipart_failed_quorum_keeps_namespace_owner_until_physical_undo() {
assert_complete_multipart_physical_namespace_owner(true).await;
}
#[tokio::test(flavor = "multi_thread")]
#[serial]
async fn complete_multipart_releases_disk_snapshot_before_cleanup() {
+4 -441
View File
@@ -7497,7 +7497,6 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
let transported = delete_file_info_with_replication_transport_metadata(fi);
let fi = &transported;
let disks = self.disk_inventory().await;
let namespace_owner = (!is_meta_bucketname(bucket)).then(|| self.ctx.begin_namespace_commit());
let write_quorum = disks.len() / 2 + 1;
let rollback_dir = Uuid::new_v4();
@@ -7505,11 +7504,10 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
let mut errs = Vec::with_capacity(disks.len());
for disk in disks.iter() {
let disk_namespace_owner = namespace_owner.clone().map(|owner| owner as Arc<dyn Send + Sync>);
futures.push(async move {
if let Some(disk) = disk {
match disk
.delete_version_with_namespace_owner(
.delete_version(
bucket,
object,
fi.clone(),
@@ -7518,7 +7516,6 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
old_data_dir: Some(rollback_dir),
..Default::default()
},
disk_namespace_owner,
)
.await
{
@@ -7566,11 +7563,10 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
let bucket = bucket.to_string();
let object = object.to_string();
let fi = fi.clone();
let disk_namespace_owner = namespace_owner.clone().map(|owner| owner as Arc<dyn Send + Sync>);
rollback_futures.push(async move {
if should_rollback {
if let Err(err) = disk
.delete_version_with_namespace_owner(
.delete_version(
&bucket,
&object,
fi,
@@ -7581,7 +7577,6 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
old_data_dir: Some(rollback_dir),
..Default::default()
},
disk_namespace_owner,
)
.await
{
@@ -7596,7 +7591,7 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
} else {
let rollback_path = format!("{object}/{rollback_dir}");
if let Err(err) = disk
.delete_with_namespace_owner(
.delete(
&bucket,
&rollback_path,
DeleteOptions {
@@ -7604,7 +7599,6 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
immediate: true,
..Default::default()
},
disk_namespace_owner,
)
.await
&& err != DiskError::FileNotFound
@@ -7623,7 +7617,6 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
}
join_all(rollback_futures).await;
drop(namespace_owner);
quorum_result
}
@@ -16291,23 +16284,7 @@ mod transition_upload_integrity_tests {
let bucket = format!("transition-real-bitrot-{}", position.label());
let object = format!("{}-corrupt.bin", position.label());
let payload = vec![0x41; 2 * 1024 * 1024];
// Corruption edits physical shards, so every healthy rename must finish first.
for disk in &disk_stores {
disk.make_volume(&bucket).await.expect("bucket volume should be created");
}
let mut reader = PutObjReader::from_vec(payload.to_vec());
let original = set_disks
.put_object(
&bucket,
&object,
&mut reader,
&ObjectOptions {
write_completion: WriteCompletion::TailDrained,
..Default::default()
},
)
.await
.expect("source object should be written");
let original = write_source(&set_disks, &disk_stores, &bucket, &object, &payload).await;
let source = set_disks
.get_object_fileinfo(
&bucket,
@@ -21067,417 +21044,3 @@ mod body_cache_hook_e2e_tests {
);
}
}
#[cfg(test)]
mod single_delete_namespace_owner_tests {
use super::hermetic_set_disks_support::hermetic_set_disks_isolated;
use super::*;
use crate::disk::ReadOptions;
#[cfg(not(windows))]
use crate::disk::STORAGE_FORMAT_FILE;
use crate::object_api::WriteCompletion;
#[cfg(not(windows))]
use tokio::io::AsyncReadExt;
async fn seed_version(set: &Arc<SetDisks>, bucket: &str, object: &str, version: Uuid, body: &[u8]) {
set.put_object(
bucket,
object,
&mut PutObjReader::from_vec(body.to_vec()),
&ObjectOptions {
versioned: true,
version_id: Some(version.to_string()),
write_completion: WriteCompletion::TailDrained,
..Default::default()
},
)
.await
.expect("seed a complete real object version");
}
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn single_delete_advances_namespace_generation_through_cleanup() {
let (dirs, disks, set) = hermetic_set_disks_isolated(4).await;
let bucket = "single-delete-namespace";
let object = "last-version";
for disk in &disks {
disk.make_volume(bucket).await.expect("fixture bucket");
}
let version = Uuid::new_v4();
seed_version(&set, bucket, object, version, &vec![0x41; 256 * 1024]).await;
let before = set.ctx.namespace_commit_generation();
assert!(!set.ctx.namespace_commits_pending());
let request = FileInfo {
name: object.to_string(),
version_id: Some(version),
mod_time: Some(OffsetDateTime::now_utc()),
..Default::default()
};
let result =
tokio::time::timeout(Duration::from_secs(10), set.delete_object_version(bucket, object, &request, false)).await;
if !matches!(result, Ok(Ok(()))) {
let retained = dirs.into_iter().map(tempfile::TempDir::keep).collect::<Vec<_>>();
panic!("single delete and cleanup did not finish: {result:?}; retained roots: {retained:?}");
}
for (disk, dir) in disks.iter().zip(&dirs) {
let result = disk
.read_version("", bucket, object, &version.to_string(), &ReadOptions::default())
.await;
assert!(matches!(result, Err(DiskError::FileNotFound | DiskError::FileVersionNotFound)));
assert!(
!dir.path().join(bucket).join(object).exists(),
"immediate cleanup must remove the rollback object tree"
);
}
assert!(!set.ctx.namespace_commits_pending());
assert_eq!(
set.ctx.namespace_commit_generation(),
before + 2,
"single delete must count one complete root lifetime"
);
}
#[cfg(not(windows))]
async fn assert_single_delete_physical_owner(case: &'static str) {
use crate::disk::os::prepared_publication_test_hooks as hooks;
use futures::FutureExt;
temp_env::async_with_vars([(rustfs_config::ENV_DRIVE_MAX_TIMEOUT_DURATION, Some("60"))], async {
let (dirs, disks, set) = hermetic_set_disks_isolated(4).await;
let bucket = "single-delete-physical-owner";
let first = Uuid::new_v4();
let second = Uuid::new_v4();
let first_body = vec![0x51; 256 * 1024];
let second_body = vec![0x62; 4096];
let missing = case == "missing-marker";
let rollback = case == "rollback";
let last = case == "last-version";
let cleanup = case == "immediate-cleanup";
for disk in &disks {
disk.make_volume(bucket).await.expect("fixture bucket");
}
if !missing {
seed_version(&set, bucket, case, first, &first_body).await;
if !last && !cleanup {
seed_version(&set, bucket, case, second, &second_body).await;
}
}
let request = FileInfo {
name: case.to_string(),
version_id: Some(first),
deleted: missing,
mark_deleted: missing,
mod_time: Some(OffsetDateTime::now_utc()),
..Default::default()
};
let before = set.ctx.namespace_commit_generation();
assert!(!set.ctx.namespace_commits_pending());
let mut metadata_paths = Vec::new();
let mut originals = Vec::new();
let mut cleanup_sources = Vec::new();
let mut cleanup_parts = Vec::new();
for disk in &disks {
let crate::disk::Disk::Local(local) = disk.as_ref() else {
panic!("local fixture required");
};
let path = local
.get_disk()
.get_object_path_for_io(bucket, case)
.expect("leased IO path")
.join(STORAGE_FORMAT_FILE);
originals.push(if missing {
None
} else {
Some(std::fs::read(&path).expect("seeded raw metadata"))
});
if cleanup {
let fi = disk.read_version("", bucket, case, &first.to_string(), &ReadOptions::default())
.await.expect("real non-inline data directory");
assert!(!fi.inline_data(), "cleanup fixture must have real shard files");
let data = path.parent().expect("object parent").join(fi.data_dir.expect("data directory").to_string());
cleanup_parts.push(std::fs::read(data.join("part.1")).expect("real pre-delete shard"));
cleanup_sources.push(data);
}
metadata_paths.push(path);
}
if rollback {
// Two disks apply the real deletion and then error. Undo must
// restore all four disks, including these post-apply failures.
for disk in disks.iter().take(2) {
crate::disk::local::set_delete_version_fail_after_commit(disk.path().as_path(), case);
}
}
let (entered_tx, mut entered_rx) = tokio::sync::mpsc::unbounded_channel();
let mut guards = Vec::new();
let mut destination_guards = Vec::new();
let later_guards = Arc::new(std::sync::Mutex::new(Vec::new()));
let mut releases = Vec::new();
for (index, path) in metadata_paths.iter().enumerate() {
let tx = entered_tx.clone();
let (release, rx) = std::sync::mpsc::channel::<()>();
if last || cleanup {
let source = if cleanup { &cleanup_sources[index] } else { path };
destination_guards.push(hooks::observe_rename_destination(source, move |destination| {
let _ = tx.send((index, destination.to_path_buf()));
let _ = rx.recv();
}));
} else {
let hook_path = path.clone();
let path = path.clone();
let pause_path = path.clone();
let later_guards = Arc::clone(&later_guards);
guards.push(hooks::install_at(hooks::Stage::Rename, &hook_path, move || {
let pause = move || {
let _ = tx.send((index, pause_path));
let _ = rx.recv();
};
if rollback {
// This first callback precedes forward metadata publication.
// Arm only the subsequent real backup-restore rename.
let next = hooks::install_at(hooks::Stage::Rename, &path, pause);
later_guards.lock().expect("fixture hook guards").push(next);
} else {
pause();
}
}));
}
releases.push(release);
}
drop(entered_tx);
let deleting_set = Arc::clone(&set);
let mut delete =
tokio::spawn(async move { deleting_set.delete_object_version(bucket, case, &request, missing).await });
let mut joined = false;
let mut counts = None;
let mut physical_keys = std::collections::BTreeMap::new();
let observations = std::panic::AssertUnwindSafe(async {
tokio::time::timeout(Duration::from_secs(10), async {
while physical_keys.len() < 4 {
tokio::select! {
entry = entered_rx.recv() => {
let (index, key) = entry.expect("actual physical delete entry");
assert!(physical_keys.insert(index, key).is_none());
}
result = &mut delete => {
joined = true;
panic!("delete returned before physical entry: {result:?}");
}
}
}
})
.await
.expect("all four physical mutations must enter");
let pending_at_entry = set.ctx.namespace_commits_pending();
let generation_at_entry = set.ctx.namespace_commit_generation();
for (path, original) in metadata_paths.iter().zip(&originals) {
if missing {
assert!(!path.exists(), "missing marker must still be unpublished at entry");
} else if cleanup {
assert!(!path.exists(), "cleanup must follow the actual last-version deletion");
} else {
let bytes = std::fs::read(path).expect("paused metadata is readable");
let metadata = rustfs_filemeta::FileMeta::load(&bytes).expect("real metadata must parse");
if !last && !cleanup {
assert!(metadata.find_version(Some(second)).is_ok());
}
assert_eq!(
metadata.find_version(Some(first)).is_err(),
rollback,
"undo entry must follow actual deletion"
);
if !rollback {
assert_eq!(Some(&bytes), original.as_ref());
}
}
}
if rollback {
tokio::time::pause();
tokio::time::advance(Duration::from_secs(61)).await;
tokio::time::resume();
let result = tokio::time::timeout(Duration::from_secs(5), &mut delete).await;
joined = result.is_ok();
let result = result
.expect("ordinary undo deadlines must return")
.expect("delete coordinator must not panic");
assert!(
matches!(&result, Err(StorageError::InsufficientWriteQuorum(error_bucket, error_object)) if error_bucket == bucket && error_object == case),
"keep the original failed delete quorum: {result:?}"
);
} else {
delete.abort();
let result = tokio::time::timeout(Duration::from_secs(5), &mut delete).await;
joined = result.is_ok();
assert!(
result
.expect("cancelled caller must join")
.expect_err("the caller must be cancelled")
.is_cancelled()
);
}
counts = Some((
pending_at_entry,
generation_at_entry,
set.ctx.namespace_commits_pending(),
set.ctx.namespace_commit_generation(),
));
for path in physical_keys.values() {
assert!(
hooks::drain_namespace_key(path)
.now_or_never()
.is_none(),
"the physical metadata executor must still own its exact key"
);
}
})
.catch_unwind()
.await;
drop(releases);
drop(guards);
drop(destination_guards);
let coordinator_drained = joined || tokio::time::timeout(Duration::from_secs(10), &mut delete).await.is_ok();
if !coordinator_drained {
delete.abort();
let _ = tokio::time::timeout(Duration::from_secs(5), &mut delete).await;
}
later_guards.lock().unwrap_or_else(std::sync::PoisonError::into_inner).clear();
let drains = futures::future::join_all(physical_keys.values().map(|path| {
tokio::time::timeout(Duration::from_secs(5), hooks::drain_namespace_key(path))
})).await;
let owner_drained = tokio::time::timeout(Duration::from_secs(5), async {
while set.ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
})
.await
.is_ok();
if physical_keys.len() != 4 || !coordinator_drained || !owner_drained || drains.iter().any(|result| result.is_err()) {
let retained = dirs.into_iter().map(tempfile::TempDir::keep).collect::<Vec<_>>();
eprintln!("single delete cleanup incomplete; retained roots: {retained:?}");
if let Err(panic) = observations {
std::panic::resume_unwind(panic);
}
panic!("single delete physical cleanup did not drain");
}
if let Err(panic) = observations {
std::panic::resume_unwind(panic);
}
for (index, (disk, (path, original))) in disks.iter().zip(metadata_paths.iter().zip(&originals)).enumerate() {
if cleanup {
assert!(!path.exists(), "metadata must remain deleted after cleanup cancellation");
assert!(!cleanup_sources[index].exists(), "physical cleanup must remove the shard directory");
assert_eq!(
std::fs::read(physical_keys[&index].join("part.1")).expect("actual trashed shard"),
cleanup_parts[index],
"trash must contain the exact original shard"
);
continue;
}
if last {
assert!(!path.exists(), "late trash rename must remove the last metadata");
assert_eq!(
Some(std::fs::read(&physical_keys[&index]).expect("actual trash destination")),
*original,
"last-version trash must contain the exact old metadata"
);
continue;
}
let bytes = std::fs::read(path).expect("late metadata publication must finish");
let metadata = rustfs_filemeta::FileMeta::load(&bytes).expect("final metadata must parse");
if missing {
assert!(
metadata
.find_version(Some(first))
.expect("the marker must be published")
.1
.delete_marker
.is_some()
);
} else {
assert!(metadata.find_version(Some(second)).is_ok());
assert_eq!(metadata.find_version(Some(first)).is_ok(), rollback);
if rollback {
assert_eq!(Some(&bytes), original.as_ref(), "physical undo must restore exact old metadata");
}
let fi = disk
.read_version("", bucket, case, &second.to_string(), &ReadOptions { read_data: true, ..Default::default() })
.await
.expect("remaining version");
if fi.inline_data() {
assert!(fi.data.as_ref().is_some_and(|data| !data.is_empty()), "remaining inline shard must survive");
} else {
let parts = disk.check_parts(bucket, case, &fi).await.expect("remaining shard check");
assert_eq!(parts.results, vec![crate::disk::CHECK_PART_SUCCESS; fi.parts.len()]);
}
}
}
if !missing && !last && !cleanup {
let mut actual = Vec::new();
let read_opts = ObjectOptions {
version_id: Some(if rollback { first } else { second }.to_string()),
versioned: true,
..Default::default()
};
let mut reader = tokio::time::timeout(
Duration::from_secs(5),
set.get_object_reader(bucket, case, None, HeaderMap::new(), &read_opts),
)
.await
.expect("final GET must finish")
.expect("the surviving version must be readable");
tokio::time::timeout(Duration::from_secs(5), reader.stream.read_to_end(&mut actual))
.await
.expect("body must drain")
.expect("read surviving body");
assert_eq!(actual, if rollback { first_body } else { second_body });
}
let (pending_at_entry, generation_at_entry, pending_after_return, generation_after_return) =
counts.expect("complete observations");
assert!(
pending_at_entry && pending_after_return,
"physical single delete outlived namespace accounting: {case}"
);
assert_eq!(generation_at_entry, before + 1);
assert_eq!(generation_after_return, generation_at_entry);
assert_eq!(set.ctx.namespace_commit_generation(), before + 2);
assert!(!set.ctx.namespace_commits_pending());
})
.await;
}
#[cfg(not(windows))]
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn single_delete_cancel_keeps_owner_until_immediate_data_cleanup() {
assert_single_delete_physical_owner("immediate-cleanup").await;
}
#[cfg(not(windows))]
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn single_delete_cancel_keeps_owner_until_last_version_trash() {
assert_single_delete_physical_owner("last-version").await;
}
#[cfg(not(windows))]
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn single_delete_cancel_keeps_owner_until_metadata_rewrite() {
assert_single_delete_physical_owner("remaining-version").await;
}
#[cfg(not(windows))]
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn single_delete_cancel_keeps_owner_until_missing_marker_publication() {
assert_single_delete_physical_owner("missing-marker").await;
}
#[cfg(not(windows))]
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn single_delete_failed_quorum_keeps_owner_until_physical_undo() {
assert_single_delete_physical_owner("rollback").await;
}
}
+20 -46
View File
@@ -630,14 +630,27 @@ impl ECStore {
.pools
.first()
.is_some_and(|pool| pool_first_endpoint_is_local(&pool.endpoints));
#[cfg(feature = "e2e-test-hooks")]
let startup_attempt = uuid::Uuid::new_v4();
let (meta, pool_meta_replica_state) = {
let mut write_state = self.pool_meta_save_gate.lock().await;
establish_pool_meta_bootstrap_identity_if_proven(self.pools.clone(), &mut write_state, should_persist_pool_meta)
.await
.map_err(|err| Error::other(format!("store init failed during establish_pool_meta_bootstrap_identity: {err}")))?;
load_pool_meta_for_startup(self.pools.clone(), &mut write_state).await?
let load = load_pool_meta_for_startup(self.pools.clone(), &mut write_state);
#[cfg(feature = "e2e-test-hooks")]
let load = crate::core::pools::startup_cas_test_scope(startup_attempt, "load", &self.pools, load);
load.await?
};
let update = meta.validate(self.pools.clone())?;
#[cfg(feature = "e2e-test-hooks")]
crate::core::pools::startup_cas_test_observe(serde_json::json!({
"kind": "startup-classifier", "attempt": startup_attempt,
"elected_writer": should_persist_pool_meta,
"needs_repair": pool_meta_replica_state.needs_repair,
"repair_write_safe": pool_meta_replica_state.repair_write_safe,
"topology_update": update,
}));
let endpoints = runtime_sources::endpoint_pools_or_default();
let mut installed_pool_meta = if update {
@@ -649,15 +662,17 @@ impl ECStore {
// distributed startup can race on the same lock and replay the prior init bug.
{
let mut write_state = self.pool_meta_save_gate.lock().await;
installed_pool_meta = persist_pool_meta_for_startup_if_safe(
let persist = persist_pool_meta_for_startup_if_safe(
&installed_pool_meta,
self.pools.clone(),
pool_meta_replica_state,
&mut write_state,
update,
should_persist_pool_meta,
)
.await?;
);
#[cfg(feature = "e2e-test-hooks")]
let persist = crate::core::pools::startup_cas_test_scope(startup_attempt, "persist", &self.pools, persist);
installed_pool_meta = persist.await?;
}
{
@@ -830,10 +845,6 @@ mod tests {
IlmRecoveryProtocol, MAX_RECOVERY_ATTEMPTS, list_recovery_controls, load_recovery_control,
observe_recovery_source, save_recovery_control_if_absent,
},
recovery_export::{
create_recovery_export, inspect_recovery_export_observation, load_recovery_export,
recovery_export_record_object_name,
},
tier_delete_journal::{
DecommissionCheckpointTargetFailureHook, TIER_DELETE_DISPATCH_MANIFEST_PREFIX, TIER_DELETE_JOURNAL_PREFIX,
TierDeleteChunkTestBarrier, TierDeleteChunkTestStage, TierDeleteDispatchBatchLimitGuard,
@@ -4373,7 +4384,7 @@ mod tests {
}
retry_source_info.parts = Arc::new(retry_source_parts);
assert_eq!(retry_source_info.etag.as_deref(), Some(retry_object_etag.as_str()));
assert!(retry_source_info.is_multipart());
assert!(!retry_source_info.is_multipart());
assert!(retry_source_info.parts.iter().all(|part| part.checksums.is_some()));
assert_eq!(retry_source_info.checksum.as_deref(), Some(retry_object_checksum_bytes.as_ref()));
assert!(
@@ -16874,43 +16885,6 @@ mod tests {
}
}
let creator_sha256 = rustfs_utils::crypto::hex_sha256(b"legacy-export-actor", ToOwned::to_owned);
let mut created_exports = Vec::new();
for exportable in first_controls
.iter()
.filter(|control| control.classification == IlmRecoveryClassification::RetainedAmbiguous)
{
let observation = inspect_recovery_export_observation(store.clone(), &exportable.control_id)
.await
.expect("fresh legacy recovery observation should be exportable");
let created = create_recovery_export(store.clone(), &observation, &creator_sha256)
.await
.expect("legacy recovery export should be created exactly once");
assert!(!created.replayed);
let loaded = load_recovery_export(store.clone(), &created.export_id)
.await
.expect("created legacy recovery export should load");
assert_eq!(loaded.encoded, created.encoded, "export readback must preserve the exact committed bytes");
let replayed = create_recovery_export(store.clone(), &observation, &creator_sha256)
.await
.expect("the same observed generation should replay its immutable export");
assert!(replayed.replayed);
assert_eq!(replayed.encoded, created.encoded);
created_exports.push(created);
}
assert_eq!(created_exports.len(), 2, "both v1 and v2 legacy journals must have an export path");
let corrupt_export_id = &created_exports[0].export_id;
let export_path = recovery_export_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, corrupt_export_id)
.expect("export path should build");
com::save_config(store.clone(), &export_path, Vec::new())
.await
.expect("zero-byte corruption fixture should persist");
let corrupt_export = load_recovery_export(store.clone(), corrupt_export_id)
.await
.expect_err("an existing zero-byte export must fail closed");
assert!(!matches!(corrupt_export, Error::ConfigNotFound));
com::save_config(
store.clone(),
&journal_paths[0],
+2 -2
View File
@@ -442,7 +442,7 @@ pub(crate) mod utils;
use peer::init_local_peer;
pub use peer::{
all_local_disk, all_local_disk_path, find_local_disk_by_ref, get_disk_infos, init_local_disks,
BootstrapLocalTarget, all_local_disk, all_local_disk_path, find_local_disk_by_ref, get_disk_infos, init_local_disks,
init_local_disks_with_instance_ctx, init_lock_clients, prewarm_local_disk_id_map,
prewarm_local_disk_id_map_with_instance_ctx,
};
@@ -1787,7 +1787,7 @@ mod tests {
// Build a minimal ECStore carrying an explicit instance context. Empty
// pools/disks are sufficient: the Phase 5 accessors read only `self.ctx`.
fn build_store_with_ctx(ctx: Arc<InstanceContext>) -> Arc<ECStore> {
pub(super) fn build_store_with_ctx(ctx: Arc<InstanceContext>) -> Arc<ECStore> {
let endpoint_pools = EndpointServerPools::default();
Arc::new(ECStore {
id: uuid::Uuid::new_v4(),
+717 -1
View File
@@ -13,7 +13,10 @@
// limitations under the License.
use super::*;
use crate::runtime::instance::InstanceContext;
use crate::bucket::utils::has_bad_path_component;
use crate::disk::error::{DiskError, Result as DiskResult};
use crate::disk::{DeleteOptions, Disk, RenameDataGuards, RenameDataResp};
use crate::runtime::instance::{InstanceContext, NamespaceCommitGuard};
use crate::runtime::sources as runtime_sources;
use tracing::{debug, error};
@@ -22,6 +25,203 @@ const LOG_SUBSYSTEM_DISK_STARTUP: &str = "disk_startup";
const EVENT_LOCAL_DISK_ID_PREWARM_SKIPPED: &str = "local_disk_id_prewarm_skipped";
const EVENT_LOCK_CLIENT_INITIALIZATION_FAILED: &str = "lock_client_initialization_failed";
/// An instance-bound capability for internal writes before ECStore/IAM startup.
/// Its private context and volume checks cannot be replaced by a caller guard.
#[derive(Clone)]
pub struct BootstrapLocalTarget {
ctx: Arc<InstanceContext>,
}
impl BootstrapLocalTarget {
pub fn new(ctx: Arc<InstanceContext>) -> Self {
Self { ctx }
}
pub fn is_for_store(&self, store: &ECStore) -> bool {
Arc::ptr_eq(&self.ctx, &store.ctx)
}
pub async fn rename_local_data(
&self,
disk_ref: &str,
source: (&str, &str),
fi: &FileInfo,
destination: (&str, &str),
scanner_token: Option<Uuid>,
) -> DiskResult<RenameDataResp> {
if scanner_token.is_some() {
return Err(DiskError::other("bootstrap rename cannot use a scanner publication lease"));
}
validate_bootstrap_volume(source.0)?;
validate_bootstrap_volume(destination.0)?;
rename_local_data_with_ctx(&self.ctx, disk_ref, source, fi, destination, RenameDataGuards::default()).await
}
pub async fn undo_local_write(
&self,
disk_ref: &str,
volume: &str,
path: &str,
fi: FileInfo,
opts: DeleteOptions,
) -> DiskResult<()> {
validate_bootstrap_volume(volume)?;
undo_local_write_with_ctx(&self.ctx, disk_ref, volume, path, fi, opts).await
}
}
fn validate_bootstrap_volume(volume: &str) -> DiskResult<()> {
// Prefix membership alone permits aliases such as .rustfs.sys/../bucket.
// Validate both raw rename volumes before any disk lookup or admission.
if has_bad_path_component(volume) || !is_meta_bucketname(volume) {
return Err(DiskError::FileAccessDenied);
}
Ok(())
}
impl ECStore {
/// Execute on this instance's active local disk through the physical owner.
pub async fn rename_local_data(
&self,
disk_ref: &str,
source: (&str, &str),
fi: &FileInfo,
destination: (&str, &str),
scanner_token: Option<Uuid>,
) -> DiskResult<RenameDataResp> {
let external_guard: Option<Arc<dyn Send + Sync>> = if let Some(token) = scanner_token {
Some(Arc::new(
self.acquire_scanner_publication_lease_guard(token)
.await
.map_err(|err| DiskError::other(err.to_string()))?,
))
} else {
None
};
rename_local_data_with_ctx(
&self.ctx,
disk_ref,
source,
fi,
destination,
RenameDataGuards {
scanner_publication_lease_token: scanner_token,
external_guard,
namespace_owner: None,
},
)
.await
}
pub async fn undo_local_write(
&self,
disk_ref: &str,
volume: &str,
path: &str,
fi: FileInfo,
opts: DeleteOptions,
) -> DiskResult<()> {
undo_local_write_with_ctx(&self.ctx, disk_ref, volume, path, fi, opts).await
}
}
// The optional ID is a cold lookup to cache only after final admission.
async fn local_disk_candidate(ctx: &Arc<InstanceContext>, disk_ref: &str) -> DiskResult<(DiskStore, Option<Uuid>)> {
let map = ctx.local_disk_map();
if let Some(disk) = map.read().await.get(disk_ref).and_then(Option::as_ref).cloned() {
return Ok((disk, None));
}
let disk_id = Uuid::parse_str(disk_ref).map_err(|_| DiskError::DiskNotFound)?;
let cached_path = ctx.local_disk_id_map().read().await.get(&disk_id).cloned();
if let Some(path) = cached_path {
let cached_disk = map.read().await.get(&path).and_then(Option::as_ref).cloned();
if let Some(disk) = cached_disk
&& matches!(disk.as_ref(), Disk::Local(_))
&& disk.get_disk_id().await? == Some(disk_id)
{
return Ok((disk, None));
}
}
let disks: Vec<_> = map.read().await.values().filter_map(Clone::clone).collect();
// Disk identity may perform format I/O. No registry guard spans this await.
for disk in disks {
if matches!(disk.as_ref(), Disk::Local(_)) && disk.get_disk_id().await.ok().flatten() == Some(disk_id) {
return Ok((disk, Some(disk_id)));
}
}
Err(DiskError::DiskNotFound)
}
async fn admit_local_disk(
ctx: &Arc<InstanceContext>,
disk: &DiskStore,
disk_id: Option<Uuid>,
volume: &str,
) -> DiskResult<Option<Arc<NamespaceCommitGuard>>> {
if !matches!(disk.as_ref(), Disk::Local(_)) {
return Err(DiskError::DiskNotFound);
}
let map = ctx.local_disk_map();
let active = map.read().await;
if !active
.get(&disk.endpoint().to_string())
.and_then(Option::as_ref)
.is_some_and(|current| Arc::ptr_eq(current, disk))
{
return Err(DiskError::DiskNotFound);
}
// Preserve registry -> ID-cache lock order; no filesystem I/O under either.
if let Some(disk_id) = disk_id {
ctx.local_disk_id_map()
.write()
.await
.insert(disk_id, disk.endpoint().to_string());
}
// Admission linearizes under the registry read: replacement/quarantine
// before this point rejects; later changes do not revoke physical I/O.
Ok((!is_meta_bucketname(volume)).then(|| ctx.begin_namespace_commit()))
}
async fn rename_local_data_with_ctx(
ctx: &Arc<InstanceContext>,
disk_ref: &str,
source: (&str, &str),
fi: &FileInfo,
destination: (&str, &str),
mut guards: RenameDataGuards,
) -> DiskResult<RenameDataResp> {
let (disk, disk_id) = local_disk_candidate(ctx, disk_ref).await?;
let owner = admit_local_disk(ctx, &disk, disk_id, destination.0).await?;
guards.namespace_owner = owner.as_ref().map(|owner| owner.clone() as Arc<dyn Send + Sync>);
let result = disk
.rename_data_borrowed_with_fence_observed(source.0, source.1, fi, destination.0, destination.1, guards)
.await
.result;
drop(owner);
result
}
async fn undo_local_write_with_ctx(
ctx: &Arc<InstanceContext>,
disk_ref: &str,
volume: &str,
path: &str,
fi: FileInfo,
opts: DeleteOptions,
) -> DiskResult<()> {
if !opts.undo_write {
return Err(DiskError::other("target undo requires undo_write"));
}
let (disk, disk_id) = local_disk_candidate(ctx, disk_ref).await?;
let owner = admit_local_disk(ctx, &disk, disk_id, volume).await?;
let physical_owner = owner.as_ref().map(|owner| owner.clone() as Arc<dyn Send + Sync>);
let result = disk
.undo_write_with_namespace_owner(volume, path, fi, opts, physical_owner)
.await;
drop(owner);
result
}
async fn remember_local_disk_id(disk: &DiskStore) -> Option<Uuid> {
remember_local_disk_id_with_instance_ctx(&crate::runtime::global::current_ctx(), disk).await
}
@@ -265,6 +465,522 @@ mod tests {
}])
}
async fn target_disk(ctx: &Arc<InstanceContext>, root: &std::path::Path, id: Uuid) -> DiskStore {
let mut format = crate::layout::format::FormatV3::new(1, 1);
format.erasure.this = id;
format.erasure.sets[0][0] = id;
let meta = root.join(crate::disk::RUSTFS_META_BUCKET);
tokio::fs::create_dir_all(&meta).await.expect("create format volume");
tokio::fs::write(
meta.join(crate::disk::FORMAT_CONFIG_FILE),
serde_json::to_vec(&format).expect("encode format"),
)
.await
.expect("write real disk identity");
let mut endpoint = Endpoint::try_from(root.to_str().expect("UTF-8 root")).expect("endpoint");
endpoint.set_pool_index(0);
endpoint.set_set_index(0);
endpoint.set_disk_index(0);
let disk = new_disk(
&endpoint,
&DiskOption {
cleanup: false,
health_check: false,
},
)
.await
.expect("open real local disk");
assert_eq!(disk.get_disk_id().await.expect("read disk format identity"), Some(id));
ctx.local_disk_map()
.write()
.await
.insert(disk.endpoint().to_string(), Some(disk.clone()));
disk
}
fn target_file_info(object: &str, version: Uuid, body: &'static [u8]) -> FileInfo {
let mut fi = FileInfo::new(object, 1, 0);
fi.erasure.index = 1;
fi.version_id = Some(version);
fi.mod_time = Some(OffsetDateTime::now_utc());
fi.size = i64::try_from(body.len()).expect("fixture length");
fi.parts = vec![rustfs_filemeta::ObjectPartInfo {
number: 1,
size: body.len(),
actual_size: fi.size,
..Default::default()
}];
fi.data = Some(bytes::Bytes::from_static(body));
fi.set_inline_data();
fi
}
async fn seed_target(disk: &DiskStore, volume: &str, object: &str, fi: FileInfo) -> Vec<u8> {
let dir = disk.path().join(volume);
tokio::fs::create_dir_all(&dir).await.expect("real fixture volume");
disk.write_metadata(volume, volume, object, fi.clone())
.await
.expect("seed real metadata");
let read = disk
.read_version(
volume,
volume,
object,
&fi.version_id.expect("fixture version").to_string(),
&crate::disk::ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("read fixture before mutation");
assert_eq!(read.data, fi.data, "fixture must contain readable inline bytes");
tokio::fs::read(dir.join(object).join(crate::disk::STORAGE_FORMAT_FILE))
.await
.expect("seeded metadata bytes")
}
#[tokio::test]
async fn target_uuid_lookup_binds_real_disk_and_owner_to_one_instance() {
for warm in [false, true] {
let ctx_a = Arc::new(InstanceContext::new());
let ctx_b = Arc::new(InstanceContext::new());
let a = tempfile::tempdir().expect("A root");
let b = tempfile::tempdir().expect("B root");
let id = Uuid::new_v4();
let disk_a = target_disk(&ctx_a, a.path(), id).await;
let disk_b = target_disk(&ctx_b, b.path(), id).await;
if warm {
assert!(record_local_disk_id_if_active(&ctx_a, &disk_a, id).await);
assert!(record_local_disk_id_if_active(&ctx_b, &disk_b, id).await);
}
let version = Uuid::new_v4();
let fi = target_file_info("destination", version, b"new-A");
for disk in [&disk_a, &disk_b] {
seed_target(disk, "target-bucket", "staged", fi.clone()).await;
}
let b_before = seed_target(
&disk_b,
"target-bucket",
"destination",
target_file_info("destination", version, b"old-B"),
)
.await;
let store = super::super::tests::build_store_with_ctx(ctx_a.clone());
store
.rename_local_data(&id.to_string(), ("target-bucket", "staged"), &fi, ("target-bucket", "destination"), None)
.await
.expect("rename on A");
let read = disk_a
.read_version(
"target-bucket",
"target-bucket",
"destination",
&version.to_string(),
&crate::disk::ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("read committed A");
assert_eq!(read.data, fi.data, "warm={warm}");
assert_eq!(
tokio::fs::read(b.path().join("target-bucket/destination/xl.meta"))
.await
.expect("B metadata"),
b_before
);
assert!(b.path().join("target-bucket/staged/xl.meta").exists());
assert!(ctx_a.namespace_commit_generation() > 0);
assert_eq!(ctx_b.namespace_commit_generation(), 0);
assert!(!ctx_a.namespace_commits_pending());
assert!(!ctx_b.namespace_commits_pending());
assert_eq!(ctx_a.local_disk_id_map().read().await.get(&id), Some(&disk_a.endpoint().to_string()));
}
}
#[tokio::test]
async fn target_admission_rejects_removed_quarantined_and_replaced_arcs() {
let ctx = Arc::new(InstanceContext::new());
let root = tempfile::tempdir().expect("root");
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
let endpoint = disk.endpoint().to_string();
for state in ["removed", "quarantined", "replaced"] {
let replacement = new_disk(
&disk.endpoint(),
&DiskOption {
cleanup: false,
health_check: false,
},
)
.await
.expect("separate active Arc");
let map = ctx.local_disk_map();
let mut entries = map.write().await;
match state {
"removed" => {
entries.remove(&endpoint);
}
"quarantined" => {
entries.insert(endpoint.clone(), None);
}
_ => {
entries.insert(endpoint.clone(), Some(replacement));
}
}
drop(entries);
assert!(
matches!(admit_local_disk(&ctx, &disk, None, "target-bucket").await, Err(DiskError::DiskNotFound)),
"{state}"
);
assert!(!ctx.namespace_commits_pending());
assert_eq!(ctx.namespace_commit_generation(), 0);
}
}
#[tokio::test]
async fn target_uuid_cache_cannot_admit_a_different_format_at_the_same_path() {
let ctx = Arc::new(InstanceContext::new());
let root = tempfile::tempdir().expect("root");
let old_id = Uuid::new_v4();
let old = target_disk(&ctx, root.path(), old_id).await;
assert!(record_local_disk_id_if_active(&ctx, &old, old_id).await);
let replacement_id = Uuid::new_v4();
let replacement = target_disk(&ctx, root.path(), replacement_id).await;
assert!(!Arc::ptr_eq(&old, &replacement));
assert!(matches!(
local_disk_candidate(&ctx, &old_id.to_string()).await,
Err(DiskError::DiskNotFound)
));
let (candidate, verified) = local_disk_candidate(&ctx, &replacement_id.to_string())
.await
.expect("replacement UUID");
assert!(Arc::ptr_eq(&candidate, &replacement));
assert_eq!(verified, Some(replacement_id));
assert!(!ctx.namespace_commits_pending());
}
#[tokio::test]
async fn bootstrap_rejects_user_volumes_aliases_and_scanner_tokens_without_mutation() {
let ctx = Arc::new(InstanceContext::new());
let root = tempfile::tempdir().expect("root");
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
let target = BootstrapLocalTarget::new(ctx.clone());
let fi = target_file_info("destination", Uuid::new_v4(), b"body");
let user_before = seed_target(&disk, "victim", "staged", fi.clone()).await;
let meta_before = seed_target(&disk, ".rustfs.sys/tmp", "staged", fi.clone()).await;
for invalid in [
"victim",
".rustfs.sys/../victim",
".rustfs.sys/./tmp",
".rustfs.sys/ .. /victim",
".rustfs.sys\\..\\victim",
".minio.sys/../victim",
] {
for (src, dst) in [(invalid, ".rustfs.sys/tmp"), (".rustfs.sys/tmp", invalid)] {
assert!(
target
.rename_local_data(&disk.endpoint().to_string(), (src, "staged"), &fi, (dst, "destination"), None)
.await
.is_err(),
"src={src}, dst={dst}"
);
}
assert!(
target
.undo_local_write(
&disk.endpoint().to_string(),
invalid,
"staged",
fi.clone(),
DeleteOptions {
undo_write: true,
..Default::default()
}
)
.await
.is_err(),
"{invalid}"
);
}
assert!(
target
.rename_local_data(
&disk.endpoint().to_string(),
(".rustfs.sys/tmp", "staged"),
&fi,
(".rustfs.sys/tmp", "destination"),
Some(Uuid::new_v4())
)
.await
.is_err()
);
assert_eq!(
tokio::fs::read(root.path().join("victim/staged/xl.meta"))
.await
.expect("user source"),
user_before
);
assert_eq!(
tokio::fs::read(root.path().join(".rustfs.sys/tmp/staged/xl.meta"))
.await
.expect("metadata source"),
meta_before
);
assert!(!root.path().join("victim/destination").exists());
assert!(!root.path().join(".rustfs.sys/tmp/destination").exists());
assert_eq!(ctx.namespace_commit_generation(), 0);
assert!(!ctx.namespace_commits_pending());
}
#[tokio::test]
async fn bootstrap_allows_internal_multisegment_rename_without_namespace_owner() {
for volume in [".rustfs.sys/tmp", ".rustfs.sys/multipart", ".minio.sys/config"] {
let ctx = Arc::new(InstanceContext::new());
let root = tempfile::tempdir().expect("root");
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
let fi = target_file_info("destination", Uuid::new_v4(), b"internal-CAS-body");
seed_target(&disk, volume, "staged", fi.clone()).await;
BootstrapLocalTarget::new(ctx.clone())
.rename_local_data(&disk.endpoint().to_string(), (volume, "staged"), &fi, (volume, "destination"), None)
.await
.expect("legitimate bootstrap metadata write");
let read = disk
.read_version(
volume,
volume,
"destination",
&fi.version_id.expect("version").to_string(),
&crate::disk::ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("read bootstrap result");
assert_eq!(read.data, fi.data);
assert_eq!(ctx.namespace_commit_generation(), 0);
assert!(!ctx.namespace_commits_pending());
}
}
#[cfg(not(windows))]
#[tokio::test]
async fn target_rename_cancellation_retains_real_namespace_and_scanner_owners() {
use crate::disk::os::prepared_publication_test_hooks as hooks;
let ctx = Arc::new(InstanceContext::new());
let sibling = Arc::new(InstanceContext::new());
let root = tempfile::tempdir().expect("root");
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
let store = super::super::tests::build_store_with_ctx(ctx.clone());
let fi = target_file_info("destination", Uuid::new_v4(), b"physically-owned");
seed_target(&disk, "target-bucket", "staged", fi.clone()).await;
let (token, _) = store
.acquire_scanner_publication_lease(0, crate::runtime::instance::SCANNER_PUBLICATION_LEASE_TTL)
.await
.expect("real scanner token in A");
let destination = disk
.get_object_path_for_io_if_local("target-bucket", "destination/xl.meta")
.expect("local disk")
.expect("destination IO path");
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
let _hook = hooks::install(&destination, move || {
let _ = entered_tx.send(());
let _ = release_rx.recv();
});
let disk_ref = disk.endpoint().to_string();
let mut rename = Box::pin(store.rename_local_data(
&disk_ref,
("target-bucket", "staged"),
&fi,
("target-bucket", "destination"),
Some(token),
));
tokio::time::timeout(std::time::Duration::from_secs(10), async {
tokio::select! {
result = &mut rename => panic!("rename completed before physical pause: {result:?}"),
entered = entered_rx => entered.expect("physical rename entered"),
}
})
.await
.expect("bounded physical entry");
drop(rename);
assert!(store.scanner_data_usage_publication_blocked().await);
assert!(ctx.namespace_commits_pending());
assert!(!sibling.namespace_commits_pending());
assert!(
store
.rename_local_data(&disk_ref, ("target-bucket", "staged"), &fi, ("target-bucket", "another"), Some(token))
.await
.is_err(),
"real pending rename blocks another scanner publication"
);
assert!(store.release_scanner_publication_lease(token).await, "remove registered token");
let gate = ctx.data_movement_operation_gate();
assert!(
gate.clone().try_write_owned().is_err(),
"physical operation still owns the scanner read guard"
);
drop(release_tx);
let _drained = tokio::time::timeout(std::time::Duration::from_secs(10), gate.write_owned())
.await
.expect("physical tail must release scanner guard");
tokio::time::timeout(std::time::Duration::from_secs(10), async {
while ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
})
.await
.expect("namespace owner drains");
let read = disk
.read_version(
"target-bucket",
"target-bucket",
"destination",
&fi.version_id.expect("version").to_string(),
&crate::disk::ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("read actual late commit");
assert_eq!(read.data, fi.data);
assert!(ctx.namespace_commit_generation() >= 2);
assert_eq!(sibling.namespace_commit_generation(), 0);
}
#[tokio::test]
async fn target_ready_rejects_unknown_foreign_released_and_expired_scanner_tokens() {
let ctx = Arc::new(InstanceContext::new());
let other = Arc::new(InstanceContext::new());
let store = super::super::tests::build_store_with_ctx(ctx.clone());
let other_store = super::super::tests::build_store_with_ctx(other);
let root = tempfile::tempdir().expect("root");
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
let fi = target_file_info("destination", Uuid::new_v4(), b"unchanged");
let before = seed_target(&disk, "target-bucket", "staged", fi.clone()).await;
let ttl = crate::runtime::instance::SCANNER_PUBLICATION_LEASE_TTL;
let (foreign, _) = other_store.acquire_scanner_publication_lease(0, ttl).await.expect("B token");
let (released, _) = store.acquire_scanner_publication_lease(0, ttl).await.expect("A token");
assert!(store.release_scanner_publication_lease(released).await);
let (valid, _) = store.acquire_scanner_publication_lease(0, ttl).await.expect("new A token");
for token in [Uuid::new_v4(), foreign, released] {
assert!(
store
.rename_local_data(
&disk.endpoint().to_string(),
("target-bucket", "staged"),
&fi,
("target-bucket", "destination"),
Some(token)
)
.await
.is_err()
);
}
tokio::time::pause();
tokio::time::advance(ttl + std::time::Duration::from_secs(1)).await;
tokio::time::resume();
assert!(
store
.rename_local_data(
&disk.endpoint().to_string(),
("target-bucket", "staged"),
&fi,
("target-bucket", "destination"),
Some(valid)
)
.await
.is_err(),
"expired real token"
);
let _ = other_store.release_scanner_publication_lease(foreign).await;
assert_eq!(
tokio::fs::read(root.path().join("target-bucket/staged/xl.meta"))
.await
.expect("source bytes"),
before
);
assert!(!root.path().join("target-bucket/destination").exists());
assert!(!ctx.namespace_commits_pending());
}
#[cfg(not(windows))]
#[tokio::test]
#[serial_test::serial]
async fn target_ordinary_timeout_keeps_its_physical_namespace_owner() {
use crate::disk::os::prepared_publication_test_hooks as hooks;
temp_env::async_with_vars([(rustfs_config::ENV_DRIVE_MAX_TIMEOUT_DURATION, Some("1"))], async {
let ctx = Arc::new(InstanceContext::new());
let store = super::super::tests::build_store_with_ctx(ctx.clone());
let root = tempfile::tempdir().expect("root");
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
let fi = target_file_info("destination", Uuid::new_v4(), b"timed-out-physical-commit");
seed_target(&disk, "target-bucket", "staged", fi.clone()).await;
let path = disk
.get_object_path_for_io_if_local("target-bucket", "destination/xl.meta")
.expect("local")
.expect("destination IO path");
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
let _hook = hooks::install(&path, move || {
let _ = entered_tx.send(());
let _ = release_rx.recv();
});
let disk_ref = disk.endpoint().to_string();
let mut rename = Box::pin(store.rename_local_data(
&disk_ref,
("target-bucket", "staged"),
&fi,
("target-bucket", "destination"),
None,
));
tokio::time::timeout(std::time::Duration::from_secs(10), async {
tokio::select! {
result = &mut rename => panic!("completed before physical pause: {result:?}"),
entered = entered_rx => entered.expect("physical entry"),
}
})
.await
.expect("bounded entry");
tokio::time::pause();
tokio::time::advance(std::time::Duration::from_secs(2)).await;
tokio::time::resume();
let result = tokio::time::timeout(std::time::Duration::from_secs(5), &mut rename)
.await
.expect("ordinary deadline remains enabled");
assert!(matches!(result, Err(DiskError::Timeout)), "{result:?}");
drop(rename);
assert!(ctx.namespace_commits_pending(), "timeout is not a physical drain");
drop(release_tx);
tokio::time::timeout(std::time::Duration::from_secs(10), async {
while ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
})
.await
.expect("late physical owner drains");
let read = disk
.read_version(
"target-bucket",
"target-bucket",
"destination",
&fi.version_id.expect("version").to_string(),
&crate::disk::ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("read actual timeout tail");
assert_eq!(read.data, fi.data);
})
.await;
}
#[test]
fn endpoint_rpc_authority_preserves_port_and_ipv6_brackets() {
let endpoint = Endpoint::try_from("https://127.0.0.1:9001/d1").expect("URL endpoint");
+19 -155
View File
@@ -76,8 +76,6 @@ struct HealTaskStatusPayload<'a> {
min_seq: u64,
#[serde(skip_serializing_if = "Option::is_none")]
progress: Option<&'a HealProgress>,
#[serde(skip_serializing_if = "Option::is_none")]
outcome: Option<&'a super::outcome::HealTaskOutcome>,
}
fn u64_is_zero(value: &u64) -> bool {
@@ -89,18 +87,17 @@ fn encode_heal_task_status_payload(
mut items: Vec<HealResultItem>,
progress: Option<&HealProgress>,
mut truncated: bool,
sequence: (u64, u64),
outcome: Option<&super::outcome::HealTaskOutcome>,
next_seq: u64,
min_seq: u64,
) -> Result<(Vec<u8>, bool)> {
loop {
let data = serde_json::to_vec(&HealTaskStatusPayload {
summary,
items: &items,
truncated,
next_seq: sequence.0,
min_seq: sequence.1,
next_seq,
min_seq,
progress,
outcome,
})
.map_err(|e| Error::Serialization(format!("failed to serialize heal task status: {e}")))?;
if data.len() <= MAX_HEAL_STATUS_PAYLOAD_SIZE {
@@ -114,21 +111,25 @@ fn encode_heal_task_status_payload(
}
}
fn heal_status_detail(detail: Option<String>, truncated: bool) -> Option<String> {
if !truncated {
return detail;
}
let truncation = "heal result items were truncated";
Some(detail.map_or_else(|| truncation.to_string(), |detail| format!("{detail}; {truncation}")))
}
fn encode_heal_status_response(
summary: &str,
items: Vec<HealResultItem>,
progress: Option<&HealProgress>,
detail: Option<String>,
truncated: bool,
sequence: (u64, u64),
outcome: Option<&super::outcome::HealTaskOutcome>,
next_seq: u64,
min_seq: u64,
) -> Result<(Vec<u8>, Option<String>)> {
let (summary, detail) = match outcome {
Some(outcome) => outcome.legacy_status(summary, detail),
None => (summary, detail),
};
let (data, truncated) = encode_heal_task_status_payload(summary, items, progress, truncated, sequence, outcome)?;
Ok((data, super::outcome::heal_status_detail(detail, truncated)))
let (data, truncated) = encode_heal_task_status_payload(summary, items, progress, truncated, next_seq, min_seq)?;
Ok((data, heal_status_detail(detail, truncated)))
}
impl HealChannelProcessor {
@@ -438,7 +439,6 @@ impl HealChannelProcessor {
.await
};
let outcome = report.as_ref().ok().and_then(|report| report.outcome.clone());
let (summary, detail, items, truncated, progress, next_seq, min_seq) = match report {
Ok(HealTaskReport {
status: HealTaskStatus::Pending | HealTaskStatus::Running,
@@ -576,15 +576,8 @@ impl HealChannelProcessor {
}
};
let (data, detail) = encode_heal_status_response(
&summary,
items,
progress.as_ref(),
detail,
truncated,
(next_seq, min_seq),
outcome.as_deref(),
)?;
let (data, detail) =
encode_heal_status_response(&summary, items, progress.as_ref(), detail, truncated, next_seq, min_seq)?;
let response = HealChannelResponse {
request_id: client_token,
@@ -873,7 +866,7 @@ mod tests {
..Default::default()
}];
let (data, detail) = encode_heal_status_response("running", items, None, None, false, (0, 0), None).unwrap();
let (data, detail) = encode_heal_status_response("running", items, None, None, false, 0, 0).unwrap();
assert!(data.len() <= MAX_HEAL_STATUS_PAYLOAD_SIZE);
let payload: serde_json::Value = serde_json::from_slice(&data).unwrap();
@@ -882,135 +875,6 @@ mod tests {
assert_eq!(detail.as_deref(), Some("heal result items were truncated"));
}
#[test]
fn outcome_v3_fixture_matches_canonical_owner_and_preserves_legacy_terminals() {
use crate::heal::outcome::*;
let cases: serde_json::Value = serde_json::from_str(include_str!("../../../madmin/tests/fixtures/heal-outcome-v3.json"))
.expect("shared client fixtures");
for case in cases.as_array().expect("fixture cases") {
if case.get("remoteResponse").is_some() {
continue;
}
let mut outcome = HealTaskOutcome::default();
let name = case["name"].as_str().expect("case name");
if matches!(name, "unknown" | "completed_with_errors") {
let disposition = if name == "unknown" {
HealObjectDisposition::Unknown
} else {
outcome.attempt_failed();
HealObjectDisposition::Failed(HealFailureClass::RetryExhausted)
};
outcome.record(HealObjectOutcome {
identity: HealObjectIdentity {
kind: HealObjectKind::Object,
bucket: "bucket".into(),
object: "object".into(),
version_id: None,
bucket_incarnation_id: None,
pool_index: None,
set_index: None,
},
disposition,
detail: None,
});
}
let abort = match name {
"cancelled" => Some(HealAbortReason::Cancelled),
"deadline" => Some(HealAbortReason::Deadline),
"untraversable" => Some(HealAbortReason::Untraversable),
_ => None,
};
outcome.finish(abort);
let expected = &case["response"];
let initial_detail = abort.map(|reason| {
match reason {
HealAbortReason::Cancelled => "heal task cancelled",
HealAbortReason::Deadline => "heal task timed out",
HealAbortReason::Untraversable => "heal listing is untraversable",
}
.to_string()
});
let (bytes, detail) = encode_heal_status_response(
if abort.is_some() { "stopped" } else { "finished" },
Vec::new(),
None,
initial_detail,
true,
(9, 4),
Some(&outcome),
)
.expect("canonical owner encoding");
let decoded: serde_json::Value = serde_json::from_slice(&bytes).expect("wire payload");
assert_eq!(decoded["summary"], expected["summary"], "{name}");
assert_eq!(detail.unwrap_or_default(), expected["detail"].as_str().expect("detail"), "{name}");
assert_eq!(decoded["outcome"], expected["outcome"], "{name}");
assert_eq!((decoded["next_seq"].as_u64(), decoded["min_seq"].as_u64()), (Some(9), Some(4)));
assert!(decoded["outcome"].get("retainedObjectBytes").is_none());
assert!(decoded["outcome"].get("untraversable").is_none());
}
}
#[test]
fn outcome_v3_abort_cannot_be_hidden_by_a_finished_status() {
use crate::heal::outcome::{HealAbortReason, HealTaskOutcome};
for reason in [
HealAbortReason::Cancelled,
HealAbortReason::Deadline,
HealAbortReason::Untraversable,
] {
let mut outcome = HealTaskOutcome::default();
outcome.finish(Some(reason));
let (data, detail) = encode_heal_status_response("finished", Vec::new(), None, None, false, (0, 0), Some(&outcome))
.expect("canonical abort adapter");
let json: serde_json::Value = serde_json::from_slice(&data).expect("public state");
assert_eq!(json["summary"], "stopped");
assert_eq!(json["outcome"]["execution"]["state"], "aborted");
assert!(detail.is_some());
}
}
#[test]
fn outcome_v3_payload_bound_keeps_cumulative_outcome_and_cursors() {
use crate::heal::outcome::*;
let mut outcome = HealTaskOutcome::default();
outcome.start();
for index in 0..256 {
outcome.record(HealObjectOutcome {
identity: HealObjectIdentity {
kind: HealObjectKind::Object,
bucket: "bucket".into(),
object: format!("object-{index}"),
version_id: None,
bucket_incarnation_id: None,
pool_index: None,
set_index: None,
},
disposition: HealObjectDisposition::Unknown,
detail: Some("\"".repeat(1024)),
});
}
let retained = outcome.objects.len();
let items = vec![
HealResultItem::default(),
HealResultItem {
detail: "x".repeat(MAX_HEAL_STATUS_PAYLOAD_SIZE + 1),
..Default::default()
},
];
let (bytes, detail) = encode_heal_status_response("running", items, None, None, false, (9, 4), Some(&outcome))
.expect("bounded status with cumulative outcome");
assert!(bytes.len() <= MAX_HEAL_STATUS_PAYLOAD_SIZE);
let wire: serde_json::Value = serde_json::from_slice(&bytes).expect("bounded payload");
assert_eq!(wire["items"].as_array().expect("items").len(), 1);
assert_eq!(wire["truncated"], true);
assert_eq!((wire["next_seq"].as_u64(), wire["min_seq"].as_u64()), (Some(9), Some(4)));
assert_eq!(wire["outcome"]["counters"]["processed"], 256);
assert_eq!(wire["outcome"]["counters"]["healed"], 0);
assert_eq!(wire["outcome"]["objects"].as_array().expect("outcome window").len(), retained);
assert!(retained < 256 && outcome.objects_truncated);
assert_eq!(detail.as_deref(), Some("heal result items were truncated"));
}
#[test]
fn admission_response_preserves_all_admission_outcomes() {
let cases = [
+25 -7
View File
@@ -313,6 +313,7 @@ impl HealManager {
completed_status_entry.outcome = Some(Arc::new(task.get_outcome().await));
}
let terminal_completion = !matches!(completed_status, HealTaskStatus::Retrying { .. });
let successful_completion = matches!(completed_status, HealTaskStatus::Completed);
// Keep retry ownership continuous: status snapshots acquire
// these locks in the same active -> retrying order.
let mut retrying_heals_guard = if let (Some((request, _, error)), Some(cancel_token)) =
@@ -374,11 +375,11 @@ impl HealManager {
drop(stats);
if terminal_completion {
let notice_targets = take_mrf_repair_notice_targets(&mrf_repair_notice_targets_clone, &task_id);
// Neither task status nor the diagnostic outcome
// window supplies a storage-owned repair receipt.
// Release only the ingress lease for rediscovery;
// preserve the producer's existing retry hints.
release_mrf_repair_notice_targets(notice_targets);
if successful_completion {
emit_mrf_repaired_events(notice_targets);
} else {
release_mrf_repair_notice_targets(notice_targets);
}
}
}
@@ -705,6 +706,20 @@ fn move_mrf_repair_notice_targets(
}
}
fn emit_mrf_repaired_events(targets: Vec<MrfRepairNoticeTarget>) {
for target in targets {
rustfs_common::mrf_channel::note_mrf_repaired(&target.bucket, &target.object, target.version_id);
rustfs_common::mrf_channel::release_mrf_identity(
target.kind,
&target.bucket,
&target.object,
target.version_id,
target.scope,
target.lease,
);
}
}
fn release_mrf_repair_notice_targets(targets: Vec<MrfRepairNoticeTarget>) {
for target in targets {
rustfs_common::mrf_channel::release_mrf_identity(
@@ -731,8 +746,11 @@ pub(super) fn prune_completed_heal_statuses(completed_heals: &mut HashMap<String
}
pub(super) fn prune_completed_heal_statuses_at(completed_heals: &mut HashMap<String, Arc<CompletedHealStatus>>, now: SystemTime) {
completed_heals
.retain(|_, completed| now.duration_since(completed.completed_at).unwrap_or_default() <= KEEP_HEAL_TASK_STATUS_DURATION);
completed_heals.retain(|_, completed| {
now.duration_since(completed.completed_at)
.map(|age| age <= KEEP_HEAL_TASK_STATUS_DURATION)
.unwrap_or(false)
});
let entry_bytes = |key: &String, value: &Arc<CompletedHealStatus>| {
key.capacity()
.saturating_add(size_of::<(String, Arc<CompletedHealStatus>)>())
+34 -161
View File
@@ -189,104 +189,37 @@ fn completed_retention_count_ttl_and_alias_eviction_are_bounded() {
);
entries.insert("future".to_string(), Arc::new(completed_retention_fixture(now + Duration::from_nanos(1))));
prune_completed_heal_statuses_at(&mut entries, now);
assert_eq!(entries.len(), 2);
assert!(entries.contains_key("ttl-boundary"));
assert!(entries.contains_key("future"), "clock rollback must not expire a new completion");
prune_completed_heal_statuses_at(&mut entries, now + Duration::from_nanos(1));
assert_eq!(entries.len(), 1);
assert!(entries.contains_key("future"));
prune_completed_heal_statuses_at(&mut entries, now + Duration::from_nanos(1) + KEEP_HEAL_TASK_STATUS_DURATION);
assert!(entries.contains_key("future"), "the exact TTL boundary remains retained");
prune_completed_heal_statuses_at(&mut entries, now + Duration::from_nanos(2) + KEEP_HEAL_TASK_STATUS_DURATION);
assert!(entries.contains_key("ttl-boundary"));
prune_completed_heal_statuses_at(&mut entries, now + Duration::from_nanos(1));
assert!(entries.is_empty());
}
#[tokio::test]
async fn completed_retention_clock_rollback_preserves_terminal_alias_queries() {
let completed_at = SystemTime::now() + Duration::from_secs(3600);
for status in [
HealTaskStatus::Completed,
HealTaskStatus::Failed {
error: "fixture failure".to_string(),
},
HealTaskStatus::Cancelled,
] {
let manager = HealManager::new(Arc::new(MockStorage), None);
let mut snapshot = completed_retention_fixture(completed_at);
snapshot.status = status.clone();
let expected_progress = snapshot.progress.clone();
let snapshot = Arc::new(snapshot);
{
let mut completed = manager.completed_heals.lock().await;
completed.insert("canonical".to_string(), Arc::clone(&snapshot));
completed.insert("alias".to_string(), Arc::clone(&snapshot));
}
for token in ["canonical", "alias"] {
let report = manager
.get_task_report_since(token, Some(3))
.await
.expect("a clock rollback must retain terminal queries");
assert_eq!(report.status, status);
assert_eq!(report.progress, expected_progress);
assert_eq!(report.result_items.len(), 1);
assert_eq!((report.min_seq, report.next_seq), (3, 5));
assert!(!report.result_items_truncated);
}
let mut completed = manager.completed_heals.lock().await;
prune_completed_heal_statuses_at(&mut completed, completed_at + KEEP_HEAL_TASK_STATUS_DURATION);
assert_eq!(completed.len(), 2, "both tokens remain at the exact TTL boundary");
prune_completed_heal_statuses_at(&mut completed, completed_at + KEEP_HEAL_TASK_STATUS_DURATION + Duration::from_nanos(1));
assert!(completed.is_empty(), "both tokens expire after the TTL");
}
}
#[test]
fn completed_retention_clock_rollback_keeps_count_and_alias_eviction_bounded() {
let now = SystemTime::UNIX_EPOCH + Duration::from_secs(3600);
let oldest = Arc::new(completed_retention_fixture(now + Duration::from_secs(1)));
let mut entries = HashMap::from([
("oldest".to_string(), Arc::clone(&oldest)),
("oldest-alias".to_string(), oldest),
]);
for index in 2..=MAX_COMPLETED_HEAL_TOKENS {
entries.insert(
format!("task-{index}"),
Arc::new(completed_retention_fixture(now + Duration::from_secs(2))),
);
}
prune_completed_heal_statuses_at(&mut entries, now);
assert_eq!(entries.len(), MAX_COMPLETED_HEAL_TOKENS - 1);
assert!(!entries.contains_key("oldest"));
assert!(!entries.contains_key("oldest-alias"));
}
#[test]
fn completed_retention_total_byte_cap_and_cap_plus_one() {
let now = SystemTime::now();
for completed_at in [now, now + Duration::from_secs(1)] {
let key = "large".to_string();
let mut entry = completed_retention_fixture(completed_at);
let base_bytes = entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>();
entry.retained_bytes.take();
entry.status = HealTaskStatus::Failed {
error: "x".repeat(MAX_COMPLETED_HEAL_BYTES - base_bytes),
};
assert_eq!(
entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>(),
MAX_COMPLETED_HEAL_BYTES
);
let mut entries = HashMap::from([(key, Arc::new(entry))]);
prune_completed_heal_statuses_at(&mut entries, now);
assert_eq!(entries.len(), 1, "exact byte cap remains retained");
let mut over = Arc::try_unwrap(entries.remove("large").expect("entry retained")).expect("entry not shared");
over.retained_bytes.take();
if let HealTaskStatus::Failed { error } = &mut over.status {
*error = "x".repeat(error.len() + 1);
}
entries.insert("large".to_string(), Arc::new(over));
prune_completed_heal_statuses_at(&mut entries, now);
assert!(entries.is_empty(), "oversized metadata cannot escape total byte bound");
let key = "large".to_string();
let mut entry = completed_retention_fixture(now);
let base_bytes = entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>();
entry.retained_bytes.take();
entry.status = HealTaskStatus::Failed {
error: "x".repeat(MAX_COMPLETED_HEAL_BYTES - base_bytes),
};
assert_eq!(
entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>(),
MAX_COMPLETED_HEAL_BYTES
);
let mut entries = HashMap::from([(key, Arc::new(entry))]);
prune_completed_heal_statuses_at(&mut entries, now);
assert_eq!(entries.len(), 1, "exact byte cap remains retained");
let mut over = Arc::try_unwrap(entries.remove("large").expect("entry retained")).expect("entry not shared");
over.retained_bytes.take();
if let HealTaskStatus::Failed { error } = &mut over.status {
*error = "x".repeat(error.len() + 1);
}
entries.insert("large".to_string(), Arc::new(over));
prune_completed_heal_statuses_at(&mut entries, now);
assert!(entries.is_empty(), "oversized metadata cannot escape total byte bound");
}
#[tokio::test]
@@ -3163,7 +3096,7 @@ async fn test_cancel_task_removes_queued_request() {
}
#[tokio::test]
async fn mrf_ownership_unverified_completion_does_not_emit_repaired() {
async fn test_mrf_repaired_notice_waits_for_successful_completion() {
let bucket = "mrf-completion-success";
let object = "object";
let version_id = Some([9u8; 16]);
@@ -3193,81 +3126,21 @@ async fn mrf_ownership_unverified_completion_does_not_emit_repaired() {
);
process_manager_queue_once(&manager).await;
tokio::time::timeout(Duration::from_secs(5), async {
loop {
let stats = manager.get_statistics().await;
if stats.successful_tasks + stats.failed_tasks > 0 {
break;
}
tokio::task::yield_now().await;
for _ in 0..100 {
let events = rustfs_common::mrf_channel::take_mrf_repaired_events_for(bucket);
if !events.is_empty() {
assert_eq!(events.len(), 1);
assert_eq!(events[0].object.as_ref(), object);
assert_eq!(events[0].version_id, version_id);
return;
}
})
.await
.expect("scheduler completes the task");
assert!(rustfs_common::mrf_channel::take_mrf_repaired_events_for(bucket).is_empty());
assert!(!lock_mrf_repair_notice_targets(&manager.mrf_repair_notice_targets).contains_key(&receipt.task_id));
}
#[tokio::test]
async fn mrf_ownership_dry_run_and_empty_window_do_not_emit_repaired() {
for empty_window in [false, true] {
let bucket = if empty_window {
"mrf-empty-outcome"
} else {
"mrf-dry-run-outcome"
};
let manager = HealManager::new(Arc::new(MockStorage), None);
let request = HealRequest::new(
if empty_window {
HealType::Cluster
} else {
HealType::Object {
bucket: bucket.to_string(),
object: "object".to_string(),
version_id: None,
}
},
HealOptions {
recursive: true,
dry_run: !empty_window,
recreate_missing: true,
..Default::default()
},
HealPriority::Normal,
);
let receipt = manager
.submit_mrf_heal_request_with_receipt(request, Arc::from(bucket), Arc::from("object"), None)
.await
.expect("notice target registered");
process_manager_queue_once(&manager).await;
tokio::time::timeout(Duration::from_secs(5), async {
loop {
let stats = manager.get_statistics().await;
if stats.successful_tasks + stats.failed_tasks > 0 {
break;
}
tokio::task::yield_now().await;
}
})
.await
.expect("scheduler completed");
let report = manager.get_task_report(&receipt.task_id).await.expect("completed report");
assert_eq!(report.status, HealTaskStatus::Completed);
let outcome = report.outcome.expect("canonical outcome");
if empty_window {
assert!(outcome.objects.is_empty());
} else {
assert_eq!(
outcome.objects[0].disposition,
crate::heal::outcome::HealObjectDisposition::DryRunObserved
);
}
assert!(rustfs_common::mrf_channel::take_mrf_repaired_events_for(bucket).is_empty());
tokio::time::sleep(Duration::from_millis(10)).await;
}
panic!("successful MRF-owned heal should emit one repaired event");
}
#[tokio::test]
async fn mrf_ownership_queued_cancel_does_not_emit_repaired() {
async fn test_mrf_repaired_notice_removed_on_queued_cancel_without_event() {
let bucket = "mrf-completion-cancel";
let object = "object";
let _ = rustfs_common::mrf_channel::take_mrf_repaired_events_for(bucket);
+8 -9
View File
@@ -25,13 +25,12 @@
//! set, rewritten on a group-commit cadence (every flush interval or flush
//! threshold new intents). A rewrite is atomic at the record level only — a
//! torn tail simply truncates during replay because every record carries its
//! own CRC32. Neither ingress nor manager admission is a durable ownership
//! receipt. The last flush window can be lost. Read-repair can rediscover a
//! failed read; the scanner retains bounded, expiring retry hints. Partial
//! writes also use a best-effort in-memory fast path, not a durable successor.
//! These mechanisms must not be reported as verified repair completion.
//! The partial-write caller's restart-survival requirement remains unmet by
//! admission alone; a verified durable handoff is still required.
//! own CRC32. Losing the last flush window (≤500 ms) is acceptable because
//! every producer keeps its own safety net: read-repair re-detects on the
//! next failing read, and the scanner's corrupt-metadata branch leaves a
//! pending-ledger entry behind even when its MRF intent is accepted
//! (backlog#1894 axis A), so a lost intent is retried by the ledger rather
//! than waiting for the failed-object TTL to re-scan the path.
use super::{DiskStore, HealDiskExt as _, local_disk_map_read};
use crate::heal::manager::{HealManager, MrfRepairNoticeTarget};
@@ -586,8 +585,8 @@ impl MrfRuntime {
self.dirty = true;
match submit_mrf_heal_request(manager, &intent).await {
// Accepted intents leave the pending set; the next flush persists the
// smaller snapshot. This is not a durable successor receipt and
// does not discharge the producer's existing retry hints.
// smaller snapshot. The scanner ledger is cleared later, when the
// canonical heal task reaches a successful terminal completion.
Ok(HealAdmissionResult::Accepted) | Ok(HealAdmissionResult::Merged) => {}
Ok(HealAdmissionResult::Full) | Ok(HealAdmissionResult::Dropped(HealAdmissionDropReason::QueueFull)) => {
intent.attempts = intent.attempts.saturating_add(1);
+11 -179
View File
@@ -15,7 +15,6 @@
//! Execution results are separate from repair responsibility. A legacy
//! successful storage call supplies no authoritative repair receipt.
use serde::{Deserialize, Serialize};
use std::{collections::VecDeque, time::SystemTime};
use uuid::Uuid;
@@ -23,16 +22,14 @@ const MAX_OUTCOME_ITEMS: usize = 128;
const MAX_OUTCOME_BYTES: usize = 64 * 1024;
const MAX_OUTCOME_DETAIL_BYTES: usize = 1024;
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "snake_case")]
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum HealObjectKind {
Object,
Metadata,
Decode,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
#[serde(rename_all = "camelCase")]
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct HealObjectIdentity {
pub kind: HealObjectKind,
pub bucket: String,
@@ -44,8 +41,7 @@ pub struct HealObjectIdentity {
pub set_index: Option<usize>,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "snake_case")]
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum HealDeferredReason {
DanglingDeleteGrace,
TransientUsageCache,
@@ -53,21 +49,14 @@ pub enum HealDeferredReason {
Deadline,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)]
#[serde(rename_all = "snake_case")]
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum HealFailureClass {
Recoverable,
RetryExhausted,
Permanent,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
#[serde(
tag = "state",
content = "details",
rename_all = "snake_case",
rename_all_fields = "camelCase"
)]
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum HealObjectDisposition {
/// The legacy storage response does not prove the requested check or commit.
Unknown,
@@ -83,8 +72,7 @@ pub enum HealObjectDisposition {
DryRunObserved,
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
#[serde(rename_all = "camelCase")]
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct HealObjectOutcome {
pub identity: HealObjectIdentity,
pub disposition: HealObjectDisposition,
@@ -101,8 +89,7 @@ impl HealObjectOutcome {
}
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize)]
#[serde(rename_all = "snake_case")]
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub enum HealTraversalCoverage {
#[default]
Unknown,
@@ -110,16 +97,14 @@ pub enum HealTraversalCoverage {
Complete,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum HealAbortReason {
Cancelled,
Deadline,
Untraversable,
}
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
#[serde(tag = "state", content = "reason", rename_all = "snake_case")]
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub enum HealExecutionOutcome {
#[default]
Pending,
@@ -129,8 +114,7 @@ pub enum HealExecutionOutcome {
Aborted(HealAbortReason),
}
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "camelCase")]
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct HealOutcomeCounters {
pub processed: u64,
pub healed: u64,
@@ -143,8 +127,7 @@ pub struct HealOutcomeCounters {
pub overflowed: bool,
}
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize)]
#[serde(rename_all = "camelCase")]
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct HealTaskOutcome {
pub execution: HealExecutionOutcome,
pub coverage: HealTraversalCoverage,
@@ -152,99 +135,11 @@ pub struct HealTaskOutcome {
/// A bounded diagnostic window, not a complete responsibility ledger.
pub objects: VecDeque<HealObjectOutcome>,
pub objects_truncated: bool,
#[serde(skip)]
retained_object_bytes: usize,
#[serde(skip)]
untraversable: bool,
}
#[derive(Debug, thiserror::Error)]
pub enum HealOutcomeWireError {
#[error("heal outcome is missing execution or counters")]
MissingFields,
#[error("heal outcome has invalid or unsupported execution fields")]
InvalidFields(#[from] serde_json::Error),
#[error("finished heal summary contradicts its canonical outcome")]
ContradictoryCompletion,
}
/// Reconcile a peer's successful legacy summary using the canonical owner types.
/// A running retry may legitimately retain the preceding attempt's outcome.
pub fn legacy_wire_status<'a>(
summary: &'a str,
wire: &serde_json::Value,
truncated: bool,
) -> Result<(&'a str, Option<String>), HealOutcomeWireError> {
if summary != "finished" {
return Ok((summary, None));
}
let execution = HealExecutionOutcome::deserialize(wire.get("execution").ok_or(HealOutcomeWireError::MissingFields)?)?;
let counters = HealOutcomeCounters::deserialize(wire.get("counters").ok_or(HealOutcomeWireError::MissingFields)?)?;
if matches!(execution, HealExecutionOutcome::Pending | HealExecutionOutcome::Running)
|| (execution == HealExecutionOutcome::Completed && counters.failed > 0)
{
return Err(HealOutcomeWireError::ContradictoryCompletion);
}
let (adapted, detail) = legacy_execution_status(summary, None, execution, &counters);
Ok((
adapted,
if adapted != summary {
heal_status_detail(detail, truncated)
} else {
None
},
))
}
pub(crate) fn heal_status_detail(detail: Option<String>, truncated: bool) -> Option<String> {
if !truncated {
return detail;
}
let truncation = "heal result items were truncated";
Some(detail.map_or_else(|| truncation.to_string(), |detail| format!("{detail}; {truncation}")))
}
fn legacy_execution_status<'a>(
summary: &'a str,
detail: Option<String>,
execution: HealExecutionOutcome,
counters: &HealOutcomeCounters,
) -> (&'a str, Option<String>) {
if summary != "finished" {
return (summary, detail);
}
match execution {
HealExecutionOutcome::CompletedWithErrors => (
"stopped",
Some(format!("heal traversal completed with errors: {} failed objects", counters.failed)),
),
HealExecutionOutcome::Aborted(reason) => {
let reason = match reason {
HealAbortReason::Cancelled => "cancelled",
HealAbortReason::Deadline => "timed out",
HealAbortReason::Untraversable => "untraversable",
};
("stopped", Some(format!("heal task {reason}")))
}
HealExecutionOutcome::Completed if counters.unknown > 0 => (
summary,
Some(format!(
"heal traversal completed; authoritative storage proof is unavailable for {} objects",
counters.unknown
)),
),
HealExecutionOutcome::Pending | HealExecutionOutcome::Running => {
("running", Some("heal execution has not reached a terminal outcome".to_string()))
}
HealExecutionOutcome::Completed => (summary, detail),
}
}
impl HealTaskOutcome {
pub(crate) fn legacy_status<'a>(&self, summary: &'a str, detail: Option<String>) -> (&'a str, Option<String>) {
legacy_execution_status(summary, detail, self.execution, &self.counters)
}
pub(crate) fn start(&mut self) {
if self.execution != HealExecutionOutcome::Aborted(HealAbortReason::Cancelled) {
self.execution = HealExecutionOutcome::Running;
@@ -397,69 +292,6 @@ mod canonical_outcome_tests {
assert!(outcome.objects.len() < MAX_OUTCOME_ITEMS);
}
#[test]
fn outcome_v3_serialization_keeps_unverified_dispositions_and_window_bounds() {
let mut outcome = HealTaskOutcome::default();
for disposition in [
HealObjectDisposition::Unknown,
HealObjectDisposition::Deferred {
reason: HealDeferredReason::DanglingDeleteGrace,
retry_not_before: None,
},
HealObjectDisposition::DryRunObserved,
] {
outcome.record(item(disposition));
}
outcome.finish(None);
let wire = serde_json::to_value(&outcome).expect("canonical wire view");
assert_eq!(wire["execution"]["state"], "completed");
assert_eq!(wire["counters"]["healed"], 0);
assert_eq!(wire["counters"]["skipped"], 3);
assert_eq!(wire["objects"][1]["disposition"]["details"]["reason"], "dangling_delete_grace");
assert!(wire["objects"][1]["identity"]["bucketIncarnationId"].is_null());
for _ in 0..MAX_OUTCOME_ITEMS + 1 {
let mut result = item(HealObjectDisposition::Unknown);
result.detail = Some("\"".repeat(MAX_OUTCOME_DETAIL_BYTES));
outcome.record(result);
}
let bytes = serde_json::to_vec(&outcome).expect("bounded canonical samples");
assert!(
bytes.len() < 8 * MAX_OUTCOME_BYTES,
"JSON escaping remains bounded independently of object count"
);
assert!(outcome.objects_truncated);
}
#[test]
fn outcome_v3_wire_consistency_rejects_unknown_success_without_rejecting_extensions() {
let mut outcome = HealTaskOutcome::default();
outcome.finish(None);
let mut wire = serde_json::to_value(&outcome).expect("canonical snapshot");
wire["execution"]["futureField"] = serde_json::json!({"new": true});
wire["counters"]["futureCounter"] = serde_json::json!(42);
assert_eq!(
legacy_wire_status("finished", &wire, false).expect("unknown extension fields"),
("finished", None)
);
wire["execution"]["state"] = serde_json::json!("future_execution");
assert!(legacy_wire_status("finished", &wire, false).is_err());
assert_eq!(
legacy_wire_status("running", &wire, false).expect("unknown nonterminal outcome"),
("running", None)
);
wire["execution"] = serde_json::json!({"state":"completed"});
wire["counters"]["failed"] = serde_json::json!(1);
assert!(matches!(
legacy_wire_status("finished", &wire, false),
Err(HealOutcomeWireError::ContradictoryCompletion)
));
wire.as_object_mut().expect("object").remove("execution");
assert!(matches!(
legacy_wire_status("finished", &wire, false),
Err(HealOutcomeWireError::MissingFields)
));
}
#[test]
fn canonical_outcome_counter_overflow_cannot_claim_complete_coverage() {
let mut outcome = HealTaskOutcome::default();
+192 -366
View File
@@ -14,107 +14,8 @@
/// bucket/cluster/prefix heal: the recursive bucket-objects sweep and the erasure-set usage baseline
use super::*;
use crate::heal::progress::{add_bytes, increment_counter, stable_generation};
use crate::heal::storage::HealListItem;
use crate::heal::utils::format_set_disk_id;
const MAX_DEFERRED_OBJECTS: usize = 256;
const MAX_DEFERRED_BYTES: usize = 256 * 1024;
const MAX_DEFERRED_FORWARD_PAGES: u64 = 2;
const MAX_DEFERRED_AGE: Duration = Duration::from_secs(30);
struct DeferredObject {
name: String,
version_id: Option<String>,
attempt: u32,
page: u64,
first_failure: Option<tokio::time::Instant>,
due: tokio::time::Instant,
}
impl DeferredObject {
fn new(item: HealListItem, page: u64) -> Self {
Self {
name: item.name,
version_id: item.version_id,
attempt: 0,
page,
first_failure: None,
due: tokio::time::Instant::now(),
}
}
fn payload_bytes(&self) -> usize {
self.name
.capacity()
.saturating_add(self.version_id.as_ref().map_or(0, String::capacity))
}
fn expired(&self) -> bool {
self.first_failure.is_some_and(|first| first.elapsed() >= MAX_DEFERRED_AGE)
}
fn defer(&mut self, delay: Duration) {
let now = tokio::time::Instant::now();
let first = *self.first_failure.get_or_insert(now);
self.attempt += 1;
self.due = (now + delay).min(first + MAX_DEFERRED_AGE);
}
}
// Only failed identities are retained. The current listing page remains owned
// by the caller; capacity pressure stops fetching, never discards that page.
struct DeferredWindow {
objects: VecDeque<DeferredObject>,
bytes: usize,
}
impl Default for DeferredWindow {
fn default() -> Self {
Self {
objects: VecDeque::new(),
// Charge every possible slot up front, including spare capacity.
bytes: MAX_DEFERRED_OBJECTS * size_of::<DeferredObject>(),
}
}
}
impl DeferredWindow {
fn push(&mut self, item: DeferredObject) -> std::result::Result<(), DeferredObject> {
let bytes = item.payload_bytes();
if self.objects.len() >= MAX_DEFERRED_OBJECTS || bytes > MAX_DEFERRED_BYTES.saturating_sub(self.bytes) {
return Err(item);
}
self.bytes += bytes;
self.objects.push_back(item);
Ok(())
}
fn pop_due(&mut self) -> Option<DeferredObject> {
let now = tokio::time::Instant::now();
let index = self.objects.iter().position(|item| item.due <= now)?;
let item = self.objects.remove(index)?;
self.bytes -= item.payload_bytes();
Some(item)
}
fn next_due(&self) -> Option<tokio::time::Instant> {
self.objects.iter().map(|item| item.due).min()
}
fn can_advance(&self, page: u64) -> bool {
self.objects.len() < MAX_DEFERRED_OBJECTS
&& self.bytes < MAX_DEFERRED_BYTES
&& self
.objects
.iter()
.all(|item| page.saturating_sub(item.page) < MAX_DEFERRED_FORWARD_PAGES)
}
}
#[cfg(test)]
#[path = "tests/deferred_retry_window.rs"]
mod deferred_retry_window;
fn unavailable_recreate_error(result: &HealResultItem, opts: &HealOpts) -> Option<Error> {
if opts.dry_run || !opts.recreate {
return None;
@@ -404,122 +305,83 @@ impl HealTask {
for (set_disk_id, heal_opts) in listing_scopes {
let mut continuation_token: Option<String> = None;
let mut deferred = DeferredWindow::default();
let mut inline_retry: Option<DeferredObject> = None;
let mut page_number = 0_u64;
let mut aborted_progress_unknown = false;
let mut pending = Vec::<HealListItem>::new().into_iter();
let mut listing_finished = false;
let mut listing_attempt = 0;
let mut listing_due = tokio::time::Instant::now();
let scope_result: Result<()> = async {
loop {
self.check_control_flags().await?;
if listing_finished && pending.as_slice().is_empty() && deferred.objects.is_empty() && inline_retry.is_none()
{
break;
}
loop {
self.check_control_flags().await?;
let mut listing_attempt = 0;
let (objects, next_token, is_truncated) = loop {
self.pace_mainline().await?;
// Listing and object retries share this safe boundary. A
// failed listing never hides an already-due object retry.
let item = deferred.pop_due().or_else(|| {
if inline_retry
.as_ref()
.is_some_and(|item| item.due <= tokio::time::Instant::now())
{
inline_retry.take()
} else if inline_retry.is_none() {
pending.next().map(|item| DeferredObject::new(item, page_number))
} else {
None
}
});
let Some(mut item) = item else {
let can_list = !listing_finished && inline_retry.is_none() && deferred.can_advance(page_number);
if can_list && listing_due <= tokio::time::Instant::now() {
let page = if let Some(set_disk_id) = set_disk_id.as_deref() {
self.await_with_control(self.storage.list_versions_for_heal_page_disk_walk(
set_disk_id,
bucket,
prefix,
continuation_token.as_deref(),
false,
))
.await
} else {
self.await_with_control(self.storage.list_objects_for_heal_page(
bucket,
prefix,
continuation_token.as_deref(),
false,
))
.await
};
match page {
Ok((objects, next_token, is_truncated)) => {
page_number = page_number.saturating_add(1);
continuation_token = next_heal_listing_token(bucket, prefix, next_token, is_truncated)?;
listing_finished = continuation_token.is_none();
listing_attempt = 0;
listing_due = tokio::time::Instant::now();
pending = objects.into_iter();
}
Err(error @ (Error::TaskCancelled | Error::TaskTimeout)) => return Err(error),
Err(error) => {
self.outcome.write().await.attempt_failed();
if error.is_recoverable_heal() && listing_attempt < MAX_BUCKET_OBJECT_HEAL_RETRIES {
listing_attempt += 1;
listing_due =
tokio::time::Instant::now() + self.bucket_object_retry_delay(listing_attempt);
continue;
}
self.outcome.write().await.mark_untraversable();
return Err(Error::HealListingFailed {
bucket: bucket.to_string(),
source: Box::new(error),
});
}
}
continue;
} else {
let due = deferred
.next_due()
.into_iter()
.chain(inline_retry.as_ref().map(|item| item.due))
.chain(can_list.then_some(listing_due))
.min();
if let Some(due) = due {
let page = if let Some(set_disk_id) = set_disk_id.as_deref() {
self.await_with_control(self.storage.list_versions_for_heal_page_disk_walk(
set_disk_id,
bucket,
prefix,
continuation_token.as_deref(),
false,
))
.await
} else {
self.await_with_control(self.storage.list_objects_for_heal_page(
bucket,
prefix,
continuation_token.as_deref(),
false,
))
.await
};
match page {
Ok(page) => break page,
Err(error @ (Error::TaskCancelled | Error::TaskTimeout)) => return Err(error),
Err(error) => {
self.outcome.write().await.attempt_failed();
if error.is_recoverable_heal() && listing_attempt < MAX_BUCKET_OBJECT_HEAL_RETRIES {
listing_attempt += 1;
self.await_with_control(async {
tokio::time::sleep_until(due).await;
tokio::time::sleep(self.bucket_object_retry_delay(listing_attempt)).await;
Ok(())
})
.await?;
continue;
}
self.outcome.write().await.mark_untraversable();
return Err(Error::HealListingFailed {
bucket: bucket.to_string(),
source: Box::new(error),
});
}
continue;
};
let retry_attempt = item.attempt;
let mut telemetry_unknown = false;
let object = item.name.as_str();
let identity =
self.outcome_identity(bucket, object, item.version_id.as_deref(), heal_opts.pool, heal_opts.set);
let mut disposition = if heal_opts.dry_run {
HealObjectDisposition::DryRunObserved
} else {
HealObjectDisposition::Unknown
};
let mut detail = None;
{
let mut progress = self.progress.write().await;
progress.set_current_object(Some(format!("{bucket}/{object}")));
}
};
let mut terminal_outcome = true;
let age_exhausted = item.expired();
let error = if age_exhausted {
Some(Error::other("heal object retry age exhausted"))
} else {
match self
let mut pending = objects;
let mut retry_attempt = 0_u32;
while !pending.is_empty() {
if retry_attempt > 0 {
self.await_with_control(async {
tokio::time::sleep(self.bucket_object_retry_delay(retry_attempt)).await;
Ok(())
})
.await?;
}
let mut retry = Vec::with_capacity(pending.len());
for item in pending {
self.check_control_flags().await?;
self.pace_mainline().await?;
let mut telemetry_unknown = false;
let object = item.name.as_str();
let identity =
self.outcome_identity(bucket, object, item.version_id.as_deref(), heal_opts.pool, heal_opts.set);
let mut disposition = if heal_opts.dry_run {
HealObjectDisposition::DryRunObserved
} else {
HealObjectDisposition::Unknown
};
let mut detail = None;
{
let mut progress = self.progress.write().await;
progress.set_current_object(Some(format!("{bucket}/{object}")));
}
let mut terminal_outcome = true;
let error = match self
.await_with_control(
self.storage
.heal_object(bucket, object, item.version_id.as_deref(), &heal_opts),
@@ -552,103 +414,35 @@ impl HealTask {
None
}
Ok((_, Some(err))) | Err(err) => Some(err),
}
};
};
if let Some(err) = error {
match err {
Error::TaskCancelled | Error::TaskTimeout => {
let disposition = if matches!(err, Error::TaskCancelled) {
HealObjectDisposition::Cancelled
} else {
HealObjectDisposition::Deferred {
reason: HealDeferredReason::Deadline,
retry_not_before: None,
}
if let Some(err) = error {
match err {
Error::TaskCancelled | Error::TaskTimeout => {
let disposition = if matches!(err, Error::TaskCancelled) {
HealObjectDisposition::Cancelled
} else {
HealObjectDisposition::Deferred {
reason: HealDeferredReason::Deadline,
retry_not_before: None,
}
};
self.outcome.write().await.record(HealObjectOutcome {
identity,
disposition,
detail: None,
});
return Err(err);
}
_ => self.outcome.write().await.attempt_failed(),
}
detail = Some(err.to_string());
if Self::is_dangling_delete_grace_error(&err) {
disposition = HealObjectDisposition::Deferred {
reason: HealDeferredReason::DanglingDeleteGrace,
retry_not_before: None,
};
self.outcome.write().await.record(HealObjectOutcome {
identity,
disposition,
detail: None,
});
aborted_progress_unknown |= !increment_counter(&mut scanned);
aborted_progress_unknown |= !increment_counter(&mut skipped);
return Err(err);
}
_ if !age_exhausted => self.outcome.write().await.attempt_failed(),
_ => {}
}
detail = Some(err.to_string());
if Self::is_dangling_delete_grace_error(&err) {
disposition = HealObjectDisposition::Deferred {
reason: HealDeferredReason::DanglingDeleteGrace,
retry_not_before: None,
};
telemetry_unknown |= !increment_counter(&mut skipped);
warn!(
target: "rustfs::heal::task",
event = EVENT_HEAL_BUCKET_RESULT,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
bucket,
object,
result = "dangling_delete_grace_skip",
error = %err,
"Heal bucket object dangling cleanup deferred by grace window"
);
} else if Self::should_skip_data_usage_cache_heal_error(bucket, object, &err) {
disposition = HealObjectDisposition::Deferred {
reason: HealDeferredReason::TransientUsageCache,
retry_not_before: None,
};
telemetry_unknown |= !increment_counter(&mut skipped);
warn!(
target: "rustfs::heal::task",
event = EVENT_HEAL_BUCKET_RESULT,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
bucket,
object,
result = "transient_skip",
error = %err,
"Heal bucket object repair skipped due to transient metadata error"
);
} else if !age_exhausted && err.is_recoverable_heal() && retry_attempt < MAX_BUCKET_OBJECT_HEAL_RETRIES {
terminal_outcome = false;
debug!(
target: "rustfs::heal::task",
event = EVENT_HEAL_BUCKET_RESULT,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
bucket,
object,
retry_attempt = retry_attempt.saturating_add(1),
error = %err,
result = "object_retry_scheduled",
"Heal bucket object retry scheduled"
);
item.defer(self.bucket_object_retry_delay(retry_attempt + 1));
if let Err(item) = deferred.push(item) {
inline_retry = Some(item);
}
} else {
disposition = HealObjectDisposition::Failed(if age_exhausted || err.is_recoverable_heal() {
HealFailureClass::RetryExhausted
} else {
HealFailureClass::Permanent
});
telemetry_unknown |= !increment_counter(&mut failed);
if age_exhausted || err.is_recoverable_heal() {
retryable_failed = retryable_failed.saturating_add(1);
} else {
permanent_failed = permanent_failed.saturating_add(1);
}
first_failed_object.get_or_insert_with(|| object.to_string());
first_error.get_or_insert_with(|| err.to_string());
if take_failure_log_sample(&mut failure_samples_logged) {
telemetry_unknown |= !increment_counter(&mut skipped);
warn!(
target: "rustfs::heal::task",
event = EVENT_HEAL_BUCKET_RESULT,
@@ -657,83 +451,115 @@ impl HealTask {
task_id = %self.id,
bucket,
object,
retry_attempt,
result = "dangling_delete_grace_skip",
error = %err,
result = "object_failed",
"Heal bucket object repair failed"
"Heal bucket object dangling cleanup deferred by grace window"
);
} else if Self::should_skip_data_usage_cache_heal_error(bucket, object, &err) {
disposition = HealObjectDisposition::Deferred {
reason: HealDeferredReason::TransientUsageCache,
retry_not_before: None,
};
telemetry_unknown |= !increment_counter(&mut skipped);
warn!(
target: "rustfs::heal::task",
event = EVENT_HEAL_BUCKET_RESULT,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
bucket,
object,
result = "transient_skip",
error = %err,
"Heal bucket object repair skipped due to transient metadata error"
);
} else if err.is_recoverable_heal() && retry_attempt < MAX_BUCKET_OBJECT_HEAL_RETRIES {
terminal_outcome = false;
debug!(
target: "rustfs::heal::task",
event = EVENT_HEAL_BUCKET_RESULT,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
bucket,
object,
retry_attempt = retry_attempt.saturating_add(1),
error = %err,
result = "object_retry_scheduled",
"Heal bucket object retry scheduled"
);
retry.push(item);
} else {
disposition = HealObjectDisposition::Failed(if err.is_recoverable_heal() {
HealFailureClass::RetryExhausted
} else {
HealFailureClass::Permanent
});
telemetry_unknown |= !increment_counter(&mut failed);
if err.is_recoverable_heal() {
retryable_failed = retryable_failed.saturating_add(1);
} else {
permanent_failed = permanent_failed.saturating_add(1);
}
first_failed_object.get_or_insert_with(|| object.to_string());
first_error.get_or_insert_with(|| err.to_string());
if take_failure_log_sample(&mut failure_samples_logged) {
warn!(
target: "rustfs::heal::task",
event = EVENT_HEAL_BUCKET_RESULT,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
bucket,
object,
retry_attempt,
error = %err,
result = "object_failed",
"Heal bucket object repair failed"
);
}
}
}
}
if terminal_outcome {
telemetry_unknown |= !increment_counter(&mut scanned);
}
if terminal_outcome {
telemetry_unknown |= !increment_counter(&mut scanned);
}
if !terminal_outcome {
continue;
}
if !terminal_outcome {
continue;
}
self.outcome.write().await.record(HealObjectOutcome {
identity,
disposition,
detail,
});
self.outcome.write().await.record(HealObjectOutcome {
identity,
disposition,
detail,
});
let mut progress = self.progress.write().await;
progress.update_object_progress(
previous_progress.objects_scanned.saturating_add(scanned),
previous_progress.objects_healed.saturating_add(healed),
previous_progress.objects_failed.saturating_add(failed),
previous_progress.skipped_objects.saturating_add(skipped),
previous_progress.bytes_processed.saturating_add(bytes),
);
if telemetry_unknown {
progress.mark_unknown();
let mut progress = self.progress.write().await;
progress.update_object_progress(
previous_progress.objects_scanned.saturating_add(scanned),
previous_progress.objects_healed.saturating_add(healed),
previous_progress.objects_failed.saturating_add(failed),
previous_progress.skipped_objects.saturating_add(skipped),
previous_progress.bytes_processed.saturating_add(bytes),
);
if telemetry_unknown {
progress.mark_unknown();
}
}
pending = retry;
retry_attempt = retry_attempt.saturating_add(1);
}
Ok(())
}
.await;
if let Err(error) = scope_result {
let disposition = match error {
Error::TaskCancelled => HealObjectDisposition::Cancelled,
Error::TaskTimeout => HealObjectDisposition::Deferred {
reason: HealDeferredReason::Deadline,
retry_not_before: None,
},
_ => HealObjectDisposition::Unknown,
};
// Only attempted identities have terminal outcomes. Unstarted
// page tails remain unprocessed under the task's partial coverage.
// No detached sleepers survive abort.
for item in deferred.objects.into_iter().chain(inline_retry) {
self.outcome.write().await.record(HealObjectOutcome {
identity: self.outcome_identity(
bucket,
&item.name,
item.version_id.as_deref(),
heal_opts.pool,
heal_opts.set,
),
disposition: disposition.clone(),
detail: None,
});
aborted_progress_unknown |= !increment_counter(&mut scanned);
aborted_progress_unknown |= !increment_counter(&mut skipped);
if !is_truncated {
break;
}
let mut progress = self.progress.write().await;
progress.update_object_progress(
previous_progress.objects_scanned.saturating_add(scanned),
previous_progress.objects_healed.saturating_add(healed),
previous_progress.objects_failed.saturating_add(failed),
previous_progress.skipped_objects.saturating_add(skipped),
previous_progress.bytes_processed.saturating_add(bytes),
);
if aborted_progress_unknown {
progress.mark_unknown();
continuation_token = next_heal_listing_token(bucket, prefix, next_token, is_truncated)?;
if continuation_token.is_none() {
// Truncated without a continuation token is a compatibility EOF.
break;
}
return Err(error);
}
}
-38
View File
@@ -15,8 +15,6 @@
use super::super::{DiskOption, DiskStore, Endpoint, new_disk};
use super::*;
mod deferred_retry;
mod canonical_outcome {
use super::*;
use crate::heal::outcome::{HealExecutionOutcome, HealTraversalCoverage};
@@ -941,10 +939,6 @@ async fn verified_recovery_keeps_state_when_marker_clear_fails() {
#[derive(Default)]
struct MockStorage {
retry_test_pages: Option<Vec<Vec<HealListItem>>>,
retry_test_delays: HashMap<String, Duration>,
retry_test_listing_delays: Mutex<VecDeque<Duration>>,
retry_test_events: Mutex<Vec<String>>,
listed: Mutex<bool>,
list_each_bucket: bool,
fail_second_listing_page: bool,
@@ -1115,7 +1109,6 @@ fn replacement_identity(
}
enum MockHealObjectOutcome {
RetryableLock,
OkWithOtherError(&'static str),
ErrOther(&'static str),
DanglingGraceDeferred,
@@ -1220,10 +1213,6 @@ impl HealStorageAPI for MockStorage {
opts: &HealOpts,
) -> Result<(HealResultItem, Option<Error>)> {
self.heal_object_calls.lock().unwrap().push(object.to_string());
self.retry_test_events.lock().expect("events").push(format!("heal:{object}"));
if let Some(delay) = self.retry_test_delays.get(object) {
tokio::time::sleep(*delay).await;
}
self.heal_object_version_ids
.lock()
.unwrap()
@@ -1252,13 +1241,6 @@ impl HealStorageAPI for MockStorage {
bucket.to_string(),
object.to_string(),
))),
MockHealObjectOutcome::RetryableLock => Ok((
HealResultItem::default(),
Some(Error::Storage(EcstoreError::Lock(rustfs_lock::LockError::AlreadyLocked {
resource: object.to_string(),
owner: "competing-writer".to_string(),
}))),
)),
MockHealObjectOutcome::RetryableSlowDown => {
Ok((HealResultItem::default(), Some(Error::Storage(EcstoreError::SlowDown))))
}
@@ -1284,13 +1266,6 @@ impl HealStorageAPI for MockStorage {
bucket.to_string(),
object.to_string(),
))),
MockHealObjectOutcome::RetryableLock => Ok((
HealResultItem::default(),
Some(Error::Storage(EcstoreError::Lock(rustfs_lock::LockError::AlreadyLocked {
resource: object.to_string(),
owner: "competing-writer".to_string(),
}))),
)),
MockHealObjectOutcome::RetryableSlowDown => {
Ok((HealResultItem::default(), Some(Error::Storage(EcstoreError::SlowDown))))
}
@@ -1386,19 +1361,6 @@ impl HealStorageAPI for MockStorage {
.lock()
.expect("listing tokens")
.push(continuation_token.map(ToOwned::to_owned));
self.retry_test_events
.lock()
.expect("events")
.push(format!("list:{}", continuation_token.unwrap_or("first")));
let delay = self.retry_test_listing_delays.lock().expect("listing delays").pop_front();
if let Some(delay) = delay {
tokio::time::sleep(delay).await;
}
if let Some(pages) = &self.retry_test_pages {
let page = continuation_token.map_or(0, |token| token.parse::<usize>().expect("test page token"));
let next = (page + 1 < pages.len()).then(|| (page + 1).to_string());
return Ok((pages[page].clone(), next.clone(), next.is_some()));
}
if let Some(remaining) = self
.recoverable_second_page_failures
.lock()
@@ -1,487 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use super::*;
fn bucket_task(storage: Arc<MockStorage>) -> HealTask {
HealTask::from_request(
HealRequest::new(
HealType::Bucket {
bucket: "bucket-a".to_string(),
},
HealOptions {
recursive: true,
timeout: None,
..Default::default()
},
HealPriority::Normal,
),
storage,
)
}
fn pages_storage(pages: &[&[&str]]) -> MockStorage {
MockStorage {
retry_test_pages: Some(
pages
.iter()
.map(|page| page.iter().map(|name| heal_item(name)).collect())
.collect(),
),
..Default::default()
}
}
fn fail_once(storage: &MockStorage, name: &str) {
storage
.heal_object_outcomes
.lock()
.expect("outcomes")
.insert(name.to_string(), VecDeque::from([MockHealObjectOutcome::RetryableLock]));
}
#[tokio::test(start_paused = true)]
async fn slow_listing_retry_services_due_object_then_age_before_next_listing() {
let storage = Arc::new(MockStorage {
recoverable_second_page_failures: Mutex::new(Some(1)),
retry_test_listing_delays: Mutex::new(VecDeque::from([Duration::ZERO, Duration::from_secs(29)])),
..Default::default()
});
storage.heal_object_outcomes.lock().expect("outcomes").insert(
"object-a".to_string(),
VecDeque::from([
MockHealObjectOutcome::RetryableSlowDown,
MockHealObjectOutcome::RetryableSlowDown,
]),
);
let task = bucket_task(storage.clone());
let execution = task.execute();
tokio::pin!(execution);
assert!(
tokio::time::timeout(Duration::from_millis(30_500), &mut execution)
.await
.is_err()
);
assert_eq!(
storage.retry_test_events.lock().expect("events").as_slice(),
["list:first", "heal:object-a", "list:second", "heal:object-a"]
);
let outcome = task.get_outcome().await;
assert_eq!(outcome.counters.processed, 1);
assert_eq!(
outcome.objects[0].disposition,
HealObjectDisposition::Failed(HealFailureClass::RetryExhausted)
);
execution.await.expect_err("age exhausted object must remain a batch failure");
assert_eq!(
storage.retry_test_events.lock().expect("events").as_slice(),
[
"list:first",
"heal:object-a",
"list:second",
"heal:object-a",
"list:second",
"heal:object-b"
]
);
let outcome = task.get_outcome().await;
assert_eq!((outcome.counters.processed, outcome.counters.attempt_failures), (2, 3));
}
#[tokio::test(start_paused = true)]
async fn listing_return_after_age_expires_does_not_start_another_heal_attempt() {
let storage = Arc::new(MockStorage {
recoverable_second_page_failures: Mutex::new(Some(1)),
retry_test_listing_delays: Mutex::new(VecDeque::from([Duration::ZERO, Duration::from_secs(31)])),
..Default::default()
});
fail_once(&storage, "object-a");
let task = bucket_task(storage.clone());
let execution = task.execute();
tokio::pin!(execution);
assert!(
tokio::time::timeout(Duration::from_millis(31_500), &mut execution)
.await
.is_err()
);
assert_eq!(
storage.retry_test_events.lock().expect("events").as_slice(),
["list:first", "heal:object-a", "list:second"]
);
assert_eq!(task.get_outcome().await.counters.failed, 1);
execution.await.expect_err("age exhaustion remains a failure");
assert_eq!(
storage.retry_test_events.lock().expect("events").as_slice(),
["list:first", "heal:object-a", "list:second", "list:second", "heal:object-b"]
);
}
#[tokio::test(start_paused = true)]
async fn full_window_abort_accounts_inline_once_and_leaves_unstarted_tail_unprocessed() {
for cancel in [true, false] {
let names: Vec<String> = (0..258).map(|index| format!("blocked-{index}")).collect();
let mut page: Vec<HealListItem> = names.iter().map(|name| heal_item(name)).collect();
page[256].version_id = Some("inline-version".to_string());
let storage = Arc::new(MockStorage {
retry_test_pages: Some(vec![page, vec![heal_item("healthy")]]),
..Default::default()
});
for name in &names {
fail_once(&storage, name);
}
let mut task = bucket_task(storage.clone());
if !cancel {
task.options.timeout = Some(Duration::from_secs(1));
}
let execution = task.execute();
tokio::pin!(execution);
assert!(
tokio::time::timeout(Duration::from_millis(500), &mut execution)
.await
.is_err()
);
assert_eq!(storage.heal_object_calls.lock().expect("calls").len(), 257);
assert_eq!(storage.listing_tokens.lock().expect("tokens").len(), 1);
if cancel {
task.cancel().await.expect("cancel");
}
let result = execution.await;
assert!(matches!(
(&result, cancel),
(Err(Error::TaskCancelled), true) | (Err(Error::TaskTimeout), false)
));
let outcome = task.get_outcome().await;
assert_eq!(
(
outcome.counters.processed,
outcome.counters.skipped,
outcome.counters.failed,
outcome.counters.healed
),
(257, 257, 0, 0)
);
assert_eq!(outcome.coverage, crate::heal::outcome::HealTraversalCoverage::Partial);
assert_eq!(
outcome.execution,
crate::heal::outcome::HealExecutionOutcome::Aborted(if cancel {
HealAbortReason::Cancelled
} else {
HealAbortReason::Deadline
})
);
let inline: Vec<_> = outcome
.objects
.iter()
.filter(|item| item.identity.object == "blocked-256")
.collect();
assert_eq!(inline.len(), 1);
assert_eq!(inline[0].identity.version_id.as_deref(), Some("inline-version"));
assert_eq!(
inline[0].disposition,
if cancel {
HealObjectDisposition::Cancelled
} else {
HealObjectDisposition::Deferred {
reason: HealDeferredReason::Deadline,
retry_not_before: None,
}
}
);
let progress = task.get_progress().await;
assert_eq!(
(
progress.objects_scanned,
progress.skipped_objects,
progress.objects_failed,
progress.objects_healed
),
(257, 257, 0, 0)
);
assert!(
!outcome
.objects
.iter()
.any(|item| item.identity.object == "blocked-257" || item.identity.object == "healthy")
);
tokio::time::advance(Duration::from_secs(60)).await;
assert_eq!(storage.heal_object_calls.lock().expect("calls").len(), 257);
assert_eq!(storage.listing_tokens.lock().expect("tokens").len(), 1);
}
}
#[tokio::test(start_paused = true)]
async fn repeated_slowdown_keeps_attempts_and_forward_pages_bounded() {
let storage = Arc::new(pages_storage(&[&["a"], &["b"], &["c"], &["d"]]));
for name in ["a", "b", "c", "d"] {
storage
.heal_object_outcomes
.lock()
.expect("outcomes")
.insert(name.to_string(), (0..4).map(|_| MockHealObjectOutcome::RetryableSlowDown).collect());
}
let task = bucket_task(storage.clone());
let execution = task.execute();
tokio::pin!(execution);
assert!(tokio::time::timeout(Duration::from_secs(1), &mut execution).await.is_err());
assert_eq!(storage.listing_tokens.lock().expect("tokens").len(), 3);
execution.await.expect_err("all four objects exhaust retries");
let outcome = task.get_outcome().await;
assert_eq!(
(outcome.counters.processed, outcome.counters.failed, outcome.counters.attempt_failures),
(4, 4, 16)
);
assert_eq!(storage.heal_object_calls.lock().expect("calls").len(), 16);
for name in ["a", "b", "c", "d"] {
assert_eq!(outcome.objects.iter().filter(|item| item.identity.object == name).count(), 1);
}
}
#[tokio::test(start_paused = true)]
async fn listing_retry_keeps_cursor_and_does_not_replay_successful_objects() {
let storage = Arc::new(MockStorage {
recoverable_second_page_failures: Mutex::new(Some(1)),
..Default::default()
});
fail_once(&storage, "object-a");
let task = bucket_task(storage.clone());
task.execute().await.expect("both retries complete");
assert_eq!(
storage.listing_tokens.lock().expect("tokens").as_slice(),
[None, Some("second".to_string()), Some("second".to_string())]
);
assert_eq!(
storage.heal_object_calls.lock().expect("calls").as_slice(),
["object-a", "object-a", "object-b"]
);
let outcome = task.get_outcome().await;
assert_eq!((outcome.counters.processed, outcome.counters.attempt_failures), (2, 2));
}
#[tokio::test(start_paused = true)]
async fn typed_lock_contention_allows_only_two_forward_pages() {
let storage = Arc::new(pages_storage(&[&["a"], &["b"], &["c"], &["d"]]));
fail_once(&storage, "a");
let task = bucket_task(storage.clone());
let execution = task.execute();
tokio::pin!(execution);
assert!(tokio::time::timeout(Duration::from_secs(1), &mut execution).await.is_err());
assert_eq!(storage.heal_object_calls.lock().expect("calls").as_slice(), ["a", "b", "c"]);
assert_eq!(storage.listing_tokens.lock().expect("tokens").len(), 3);
execution.await.expect("all objects complete");
assert_eq!(storage.heal_object_calls.lock().expect("calls").as_slice(), ["a", "b", "c", "a", "d"]);
assert_eq!(task.get_outcome().await.counters.processed, 4);
}
#[tokio::test(start_paused = true)]
async fn due_retry_runs_before_next_object_in_a_slow_healthy_page() {
let mut storage = pages_storage(&[&["a"], &["b", "c"]]);
storage.retry_test_delays.insert("b".to_string(), Duration::from_secs(3));
fail_once(&storage, "a");
let storage = Arc::new(storage);
bucket_task(storage.clone()).execute().await.expect("all objects complete");
assert_eq!(storage.heal_object_calls.lock().expect("calls").as_slice(), ["a", "b", "a", "c"]);
}
#[tokio::test(start_paused = true)]
async fn expired_retry_is_terminal_without_an_extra_storage_attempt() {
let mut storage = pages_storage(&[&["a"], &["b"]]);
storage.retry_test_delays.insert("b".to_string(), Duration::from_secs(31));
fail_once(&storage, "a");
let storage = Arc::new(storage);
let task = bucket_task(storage.clone());
task.execute().await.expect_err("aged pending responsibility is not success");
assert_eq!(storage.heal_object_calls.lock().expect("calls").as_slice(), ["a", "b"]);
let outcome = task.get_outcome().await;
assert_eq!(outcome.counters.processed, 2);
assert_eq!(outcome.counters.attempt_failures, 1);
assert_eq!(outcome.counters.failed, 1);
assert_eq!(
outcome
.objects
.iter()
.find(|item| item.identity.object == "a")
.expect("a outcome")
.disposition,
HealObjectDisposition::Failed(HealFailureClass::RetryExhausted)
);
}
#[tokio::test(start_paused = true)]
async fn cancellation_drains_owned_retries_once() {
let storage = Arc::new(pages_storage(&[&["a"], &["b"]]));
fail_once(&storage, "a");
let task = bucket_task(storage.clone());
let execution = task.execute();
tokio::pin!(execution);
assert!(tokio::time::timeout(Duration::from_secs(1), &mut execution).await.is_err());
task.cancel().await.expect("cancel");
assert!(matches!(execution.await, Err(Error::TaskCancelled)));
tokio::time::advance(Duration::from_secs(60)).await;
assert_eq!(storage.heal_object_calls.lock().expect("calls").as_slice(), ["a", "b"]);
let outcome = task.get_outcome().await;
assert_eq!(outcome.counters.processed, 2);
assert_eq!(outcome.objects.iter().filter(|item| item.identity.object == "a").count(), 1);
assert_eq!(
outcome
.objects
.iter()
.find(|item| item.identity.object == "a")
.expect("a")
.disposition,
HealObjectDisposition::Cancelled
);
}
#[tokio::test(start_paused = true)]
async fn deadline_drains_owned_retries_without_false_completion() {
let storage = Arc::new(pages_storage(&[&["a"], &["b"]]));
fail_once(&storage, "a");
let mut task = bucket_task(storage.clone());
task.options.timeout = Some(Duration::from_secs(1));
assert!(matches!(task.execute().await, Err(Error::TaskTimeout)));
let outcome = task.get_outcome().await;
assert_eq!(outcome.counters.processed, 2);
assert_eq!(
outcome
.objects
.iter()
.find(|item| item.identity.object == "a")
.expect("a")
.disposition,
HealObjectDisposition::Deferred {
reason: HealDeferredReason::Deadline,
retry_not_before: None
}
);
tokio::time::advance(Duration::from_secs(60)).await;
assert_eq!(storage.heal_object_calls.lock().expect("calls").as_slice(), ["a", "b"]);
}
#[tokio::test(start_paused = true)]
async fn full_window_backpressures_without_losing_the_current_page_tail() {
let names: Vec<String> = (0..258).map(|index| format!("blocked-{index}")).collect();
let mut storage = MockStorage {
retry_test_pages: Some(vec![names.iter().map(|name| heal_item(name)).collect(), vec![heal_item("healthy")]]),
..Default::default()
};
for name in &names {
fail_once(&storage, name);
}
// The last item has a version, proving the current-page tail is not rebuilt
// from names alone when the window fills.
storage.retry_test_pages.as_mut().expect("pages")[0][257].version_id = Some("version-tail".to_string());
let storage = Arc::new(storage);
let task = bucket_task(storage.clone());
let execution = task.execute();
tokio::pin!(execution);
assert!(tokio::time::timeout(Duration::from_secs(1), &mut execution).await.is_err());
assert_eq!(storage.heal_object_calls.lock().expect("calls").len(), 257);
assert_eq!(storage.listing_tokens.lock().expect("tokens").len(), 1);
execution.await.expect("every owned item eventually completes");
assert_eq!(task.get_outcome().await.counters.processed, 259);
let calls = storage.heal_object_calls.lock().expect("calls");
for name in &names {
assert_eq!(calls.iter().filter(|called| *called == name).count(), 2);
}
let versions = storage.heal_object_version_ids.lock().expect("versions");
for (name, version) in calls.iter().zip(versions.iter()) {
if name == "blocked-257" {
assert_eq!(version.as_deref(), Some("version-tail"));
}
}
}
#[tokio::test(start_paused = true)]
async fn oversized_identity_stays_inline_without_losing_version() {
let name = "k".repeat(256 * 1024);
let mut item = heal_item(&name);
item.version_id = Some("v".repeat(1024));
let storage = Arc::new(MockStorage {
retry_test_pages: Some(vec![vec![item], vec![heal_item("healthy")]]),
..Default::default()
});
fail_once(&storage, &name);
let task = bucket_task(storage.clone());
let execution = task.execute();
tokio::pin!(execution);
assert!(tokio::time::timeout(Duration::from_secs(1), &mut execution).await.is_err());
assert_eq!(storage.listing_tokens.lock().expect("tokens").len(), 1);
execution.await.expect("oversized identity retries inline");
assert_eq!(task.get_outcome().await.counters.processed, 2);
let versions = storage.heal_object_version_ids.lock().expect("versions");
assert_eq!(versions[0], versions[1]);
assert_eq!(versions[0].as_ref().expect("version").len(), 1024);
}
#[tokio::test(start_paused = true)]
async fn terminal_listing_failure_keeps_deferred_identity_unknown() {
let storage = Arc::new(MockStorage {
fail_second_listing_page: true,
..Default::default()
});
fail_once(&storage, "object-a");
let task = bucket_task(storage.clone());
task.execute().await.expect_err("listing cannot continue");
let outcome = task.get_outcome().await;
assert_eq!(outcome.counters.processed, 1);
assert_eq!(outcome.objects[0].disposition, HealObjectDisposition::Unknown);
assert_eq!(outcome.coverage, crate::heal::outcome::HealTraversalCoverage::Partial);
assert_eq!(storage.heal_object_calls.lock().expect("calls").as_slice(), ["object-a"]);
}
#[tokio::test(start_paused = true)]
async fn healthy_second_page_advances_before_first_retry_is_due() {
let storage = Arc::new(MockStorage {
recoverable_second_page_failures: Mutex::new(Some(0)),
..Default::default()
});
storage
.heal_object_outcomes
.lock()
.expect("outcomes")
.insert("object-a".to_string(), VecDeque::from([MockHealObjectOutcome::RetryableSlowDown]));
let task = HealTask::from_request(
HealRequest::new(
HealType::Bucket {
bucket: "bucket-a".to_string(),
},
HealOptions {
recursive: true,
timeout: None,
..Default::default()
},
HealPriority::Normal,
),
storage.clone(),
);
let execution = task.execute();
tokio::pin!(execution);
assert!(
tokio::time::timeout(Duration::from_secs(1), &mut execution).await.is_err(),
"the deferred first object must remain pending before its retry is due"
);
assert_eq!(
storage.heal_object_calls.lock().expect("calls").as_slice(),
["object-a", "object-b"],
"a retryable page head must not hold the healthy second page behind its backoff"
);
execution.await.expect("retry eventually succeeds");
let outcome = task.get_outcome().await;
assert_eq!(outcome.counters.processed, 2);
assert_eq!(outcome.counters.attempt_failures, 1);
assert_eq!(
storage.heal_object_calls.lock().expect("calls").as_slice(),
["object-a", "object-b", "object-a"]
);
}
@@ -1,76 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use super::*;
fn item(name: String, version_id: Option<String>) -> DeferredObject {
DeferredObject::new(
HealListItem {
name,
version_id,
mod_time_unix_nanos: None,
lifecycle_object_info: None,
is_delete_marker: false,
},
1,
)
}
#[tokio::test(start_paused = true)]
async fn count_cap_and_next_item_preserve_ownership() {
let mut window = DeferredWindow::default();
for _ in 0..MAX_DEFERRED_OBJECTS {
assert!(window.push(item("key".to_string(), None)).is_ok());
}
let rejected = window
.push(item("next".to_string(), Some("version".to_string())))
.expect_err("count cap");
assert_eq!(rejected.name, "next");
assert_eq!(rejected.version_id.as_deref(), Some("version"));
assert_eq!(window.objects.len(), MAX_DEFERRED_OBJECTS);
assert!(window.bytes <= MAX_DEFERRED_BYTES);
assert!(!window.can_advance(1));
assert!(window.pop_due().is_some());
assert!(window.push(rejected).is_ok());
}
#[tokio::test(start_paused = true)]
async fn byte_cap_counts_key_version_and_reserved_slots() {
let mut window = DeferredWindow::default();
let available = MAX_DEFERRED_BYTES - window.bytes;
let key = "k".repeat(available / 2);
let version = "v".repeat(available - key.capacity());
assert_eq!(key.capacity() + version.capacity(), available);
assert!(window.push(item(key, Some(version))).is_ok());
assert_eq!(window.bytes, MAX_DEFERRED_BYTES);
assert!(!window.can_advance(1));
assert!(window.push(item("x".to_string(), None)).is_err());
assert!(window.pop_due().is_some());
assert_eq!(window.bytes, MAX_DEFERRED_OBJECTS * size_of::<DeferredObject>());
assert!(window.push(item("x".to_string(), None)).is_ok());
}
#[tokio::test(start_paused = true)]
async fn retry_age_caps_due_time_and_is_not_reset_by_rescheduling() {
let mut entry = item("a".to_string(), None);
entry.defer(Duration::from_secs(2));
let first = entry.first_failure.expect("first failure");
tokio::time::advance(Duration::from_secs(29)).await;
entry.defer(Duration::from_secs(8));
assert_eq!(entry.first_failure, Some(first));
assert_eq!(entry.due, first + MAX_DEFERRED_AGE);
assert!(!entry.expired());
tokio::time::advance(Duration::from_secs(1)).await;
assert!(entry.expired());
}
+2 -1
View File
@@ -66,8 +66,9 @@ const ERR_LIFECYCLE_EXPIRED_OBJECT_DELETE_MARKER_WITH_TAGS: &str =
const ERR_LIFECYCLE_RULE_MUST_HAVE_ACTION: &str = "Rule must have at least one of Expiration, Transition, NoncurrentVersionExpiration, NoncurrentVersionTransition, or DelMarkerExpiration";
const ERR_LIFECYCLE_PREFIX_FILTER_CONFLICT: &str = "Legacy Prefix and Filter cannot both be present in a lifecycle rule. Use Filter.Prefix instead of the top-level Prefix element.";
const ERR_LIFECYCLE_INVALID_NEWER_NONCURRENT_VERSIONS: &str = "'NewerNoncurrentVersions' must be a non-negative integer";
const ERR_LIFECYCLE_FILTER_TOO_MANY_PREDICATES: &str =
"Filter must have at most one of Prefix, Tag, ObjectSizeGreaterThan, ObjectSizeLessThan or And; combine predicates with And";
const ERR_LIFECYCLE_FILTER_AND_TOO_FEW_PREDICATES: &str = "Filter And must contain at least two predicates";
const ERR_LIFECYCLE_FILTER_TOO_MANY_PREDICATES: &str = "Filter has too many predicates";
const ERR_LIFECYCLE_FILTER_DUPLICATE_TAG_KEY: &str = "Filter must not repeat a tag key";
const ERR_LIFECYCLE_FILTER_INVALID_TAG: &str = "Tag key must be 1-128 characters and tag value must be at most 256 characters";
const ERR_LIFECYCLE_FILTER_NEGATIVE_SIZE: &str = "ObjectSizeGreaterThan and ObjectSizeLessThan must not be negative";
+4 -82
View File
@@ -167,13 +167,6 @@ pub struct HealTaskStatus {
/// Live progress snapshot; the exact shape is owned by the heal runtime.
#[serde(default)]
pub progress: Option<serde_json::Value>,
/// Canonical heal-owner result. Missing or future states are not repair proof.
#[serde(default)]
pub outcome: Option<serde_json::Value>,
#[serde(default, alias = "next_seq")]
pub next_seq: Option<u64>,
#[serde(default, alias = "min_seq")]
pub min_seq: Option<u64>,
}
/// `POST /v3/background-heal/status` response. Known top-level fields are
@@ -365,22 +358,8 @@ impl AdminClient {
prefix: Option<&str>,
client_token: &str,
) -> Result<HealTaskStatus, AdminClientError> {
self.heal_status_since(bucket, prefix, client_token, None).await
}
/// Query a retained result window. Missing cursors and outcome remain unknown.
pub async fn heal_status_since(
&self,
bucket: Option<&str>,
prefix: Option<&str>,
client_token: &str,
since_seq: Option<u64>,
) -> Result<HealTaskStatus, AdminClientError> {
let mut query = vec![("clientToken", client_token.to_string())];
if let Some(since_seq) = since_seq {
query.push(("sinceSeq", since_seq.to_string()));
}
self.post_json(&heal_path(bucket, prefix), &query, Vec::new()).await
self.post_json(&heal_path(bucket, prefix), &[("clientToken", client_token.to_string())], Vec::new())
.await
}
/// Stop a heal: with a `client_token` only that task is cancelled and its
@@ -399,7 +378,7 @@ impl AdminClient {
match client_token {
Some(_) => {
let status: HealTaskStatus = self.post_json(&heal_path(bucket, prefix), &query, Vec::new()).await?;
Ok(HealStopOutcome::Stopped(Box::new(status)))
Ok(HealStopOutcome::Stopped(status))
}
None => {
let success: HealStartSuccess = self.post_json(&heal_path(bucket, prefix), &query, Vec::new()).await?;
@@ -554,7 +533,7 @@ impl AdminClient {
/// start-success-shaped receipt.
#[derive(Debug, Clone)]
pub enum HealStopOutcome {
Stopped(Box<HealTaskStatus>),
Stopped(HealTaskStatus),
PathStopped(HealStartSuccess),
}
@@ -656,24 +635,6 @@ mod tests {
assert!(status.progress.is_none());
}
#[test]
fn outcome_v3_decoder_preserves_canonical_unknown_and_future_fields() {
let cases: serde_json::Value =
serde_json::from_str(include_str!("../tests/fixtures/heal-outcome-v3.json")).expect("shared fixtures");
for case in cases.as_array().expect("cases") {
let status: HealTaskStatus = serde_json::from_value(case["response"].clone()).expect("optional outcome response");
assert_eq!(status.outcome.as_ref(), Some(&case["response"]["outcome"]));
assert_eq!((status.next_seq, status.min_seq), (Some(9), Some(4)));
assert!(status.truncated);
}
let old: HealTaskStatus = serde_json::from_value(json!({"summary":"finished"})).expect("legacy response");
assert!(old.outcome.is_none() && old.next_seq.is_none() && old.min_seq.is_none());
let future = json!({"execution":{"state":"future_state"},"newField":7});
let status: HealTaskStatus =
serde_json::from_value(json!({"summary":"running","outcome":future})).expect("future outcome remains opaque");
assert_eq!(status.outcome, Some(future));
}
#[test]
fn background_heal_status_types_known_fields_and_passes_the_rest_through() {
let raw = json!({
@@ -789,26 +750,6 @@ mod tests {
assert!(!request.query.contains("forceStop"));
}
#[tokio::test]
async fn outcome_v3_since_query_preserves_cursor_and_never_sends_force_start() {
let server = TestServer::spawn(
r#"{"summary":"running","nextSeq":9,"minSeq":4,"truncated":true,"outcome":{"execution":{"state":"future_state"}}}"#,
200,
)
.await;
let client = AdminClient::new(&format!("http://{}", server.addr), "ak", "sk").expect("test client");
let status = client
.heal_status_since(Some("bucket"), None, "token-1", Some(3))
.await
.expect("window response");
assert_eq!((status.next_seq, status.min_seq), (Some(9), Some(4)));
assert!(status.truncated);
assert_eq!(status.outcome.expect("future state is preserved")["execution"]["state"], "future_state");
let request = server.recorded();
assert!(request.query.contains("sinceSeq=3") && request.query.contains("clientToken=token-1"));
assert!(!request.query.contains("forceStart") && !request.query.contains("forceStop"));
}
#[tokio::test]
async fn stop_without_token_takes_the_path_cancel_branch() {
let server = TestServer::spawn(r#"{"clientToken":"path","clientAddress":"c","startTime":"t"}"#, 200).await;
@@ -821,25 +762,6 @@ mod tests {
assert!(!request.query.contains("clientToken"));
}
#[tokio::test]
async fn stop_with_token_decodes_boxed_task_status() {
let body = r#"{"summary":"stopped","detail":"","settings":{"recursive":false},"items":[],"truncated":false}"#;
let server = TestServer::spawn(body, 200).await;
let client = AdminClient::new(&format!("http://{}", server.addr), "ak", "sk").unwrap();
let outcome = client
.heal_stop(Some("bucket"), None, Some("token-1"))
.await
.expect("token stop decodes");
let super::HealStopOutcome::Stopped(status) = outcome else {
panic!("token stop should return task status");
};
assert_eq!(status.summary, "stopped");
let request = server.recorded();
assert!(request.query.contains("forceStop=true"));
assert!(request.query.contains("clientToken=token-1"));
}
#[tokio::test]
async fn background_heal_status_posts_to_the_registered_route() {
let body = r#"{"state":"idle","healQueueLength":0,"healActiveTasks":0,"clusterStatusComplete":true}"#;
-472
View File
@@ -1,472 +0,0 @@
[
{
"name": "completed",
"cliExit": 0,
"response": {
"summary": "finished",
"detail": "heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "completed"
},
"coverage": "complete",
"counters": {
"processed": 0,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 0,
"unknown": 0,
"attemptFailures": 0,
"overflowed": false
},
"objects": [],
"objectsTruncated": false
}
}
},
{
"name": "unknown",
"cliExit": 0,
"response": {
"summary": "finished",
"detail": "heal traversal completed; authoritative storage proof is unavailable for 1 objects; heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "completed"
},
"coverage": "complete",
"counters": {
"processed": 1,
"healed": 0,
"unchanged": 0,
"skipped": 1,
"failed": 0,
"unknown": 1,
"attemptFailures": 0,
"overflowed": false
},
"objects": [
{
"identity": {
"kind": "object",
"bucket": "bucket",
"object": "object",
"versionId": null,
"bucketIncarnationId": null,
"poolIndex": null,
"setIndex": null
},
"disposition": {
"state": "unknown"
},
"detail": null
}
],
"objectsTruncated": false
}
}
},
{
"name": "completed_with_errors",
"cliExit": 1,
"response": {
"summary": "stopped",
"detail": "heal traversal completed with errors: 1 failed objects; heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "completed_with_errors"
},
"coverage": "complete",
"counters": {
"processed": 1,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 1,
"unknown": 0,
"attemptFailures": 1,
"overflowed": false
},
"objects": [
{
"identity": {
"kind": "object",
"bucket": "bucket",
"object": "object",
"versionId": null,
"bucketIncarnationId": null,
"poolIndex": null,
"setIndex": null
},
"disposition": {
"state": "failed",
"details": "retry_exhausted"
},
"detail": null
}
],
"objectsTruncated": false
}
}
},
{
"name": "cancelled",
"cliExit": 1,
"response": {
"summary": "stopped",
"detail": "heal task cancelled; heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "aborted",
"reason": "cancelled"
},
"coverage": "partial",
"counters": {
"processed": 0,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 0,
"unknown": 0,
"attemptFailures": 0,
"overflowed": false
},
"objects": [],
"objectsTruncated": false
}
}
},
{
"name": "deadline",
"cliExit": 1,
"response": {
"summary": "stopped",
"detail": "heal task timed out; heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "aborted",
"reason": "deadline"
},
"coverage": "partial",
"counters": {
"processed": 0,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 0,
"unknown": 0,
"attemptFailures": 0,
"overflowed": false
},
"objects": [],
"objectsTruncated": false
}
}
},
{
"name": "untraversable",
"cliExit": 1,
"response": {
"summary": "stopped",
"detail": "heal listing is untraversable; heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "aborted",
"reason": "untraversable"
},
"coverage": "partial",
"counters": {
"processed": 0,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 0,
"unknown": 0,
"attemptFailures": 0,
"overflowed": false
},
"objects": [],
"objectsTruncated": false
}
}
},
{
"name": "remote_completed_with_errors",
"cliExit": 1,
"response": {
"summary": "stopped",
"detail": "heal traversal completed with errors: 1 failed objects; heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "completed_with_errors",
"futureExtension": {
"value": 7
}
},
"coverage": "complete",
"counters": {
"processed": 1,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 1,
"unknown": 0,
"attemptFailures": 1,
"overflowed": false,
"futureCounter": 11
},
"objects": [
{
"identity": {
"kind": "object",
"bucket": "bucket",
"object": "object",
"versionId": null,
"bucketIncarnationId": null,
"poolIndex": null,
"setIndex": null
},
"disposition": {
"state": "failed",
"details": "retry_exhausted"
},
"detail": null
}
],
"objectsTruncated": false
}
},
"remoteResponse": {
"summary": "finished",
"detail": "",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "completed_with_errors",
"futureExtension": {
"value": 7
}
},
"coverage": "complete",
"counters": {
"processed": 1,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 1,
"unknown": 0,
"attemptFailures": 1,
"overflowed": false,
"futureCounter": 11
},
"objects": [
{
"identity": {
"kind": "object",
"bucket": "bucket",
"object": "object",
"versionId": null,
"bucketIncarnationId": null,
"poolIndex": null,
"setIndex": null
},
"disposition": {
"state": "failed",
"details": "retry_exhausted"
},
"detail": null
}
],
"objectsTruncated": false
}
}
},
{
"name": "remote_cancelled",
"cliExit": 1,
"response": {
"summary": "stopped",
"detail": "heal task cancelled; heal result items were truncated",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "aborted",
"reason": "cancelled",
"futureExtension": {
"value": 7
}
},
"coverage": "partial",
"counters": {
"processed": 0,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 0,
"unknown": 0,
"attemptFailures": 0,
"overflowed": false,
"futureCounter": 11
},
"objects": [],
"objectsTruncated": false
}
},
"remoteResponse": {
"summary": "finished",
"detail": "",
"startTime": "2026-01-01T00:00:00Z",
"settings": {
"recursive": true,
"scanMode": 1
},
"items": [],
"truncated": true,
"nextSeq": 9,
"minSeq": 4,
"progress": {
"objectsScanned": 11,
"objectsHealed": 7
},
"outcome": {
"execution": {
"state": "aborted",
"reason": "cancelled",
"futureExtension": {
"value": 7
}
},
"coverage": "partial",
"counters": {
"processed": 0,
"healed": 0,
"unchanged": 0,
"skipped": 0,
"failed": 0,
"unknown": 0,
"attemptFailures": 0,
"overflowed": false,
"futureCounter": 11
},
"objects": [],
"objectsTruncated": false
}
}
}
]
+6 -33
View File
@@ -175,7 +175,6 @@ pub const BACKGROUND_HEAL_STATUS_PROTOCOL_VERSION: u32 = 2;
pub const HEAL_CONTROL_CAPABILITY_PROBE_PREFIX: &[u8] = b"rustfs-heal-control-capability-v3\0";
pub const REMOTE_VERSION_STATE_CAPABILITY_PROBE_PREFIX: &[u8] = b"rustfs-tier-remote-version-state-capability-v1\0";
pub const CROSS_POOL_FENCE_CAPABILITY_PROBE_PREFIX: &[u8] = b"rustfs-cross-pool-fence-capability-v1\0";
pub const ILM_RECOVERY_EXPORT_CAPABILITY_PROBE_PREFIX: &[u8] = b"rustfs-ilm-recovery-export-capability-v1\0";
pub const TIER_MUTATION_RPC_MAX_PREPARE_PAYLOAD_SIZE: usize = 64 * 1024;
pub const TIER_MUTATION_RPC_MAX_COMMIT_PAYLOAD_SIZE: usize = 1024;
pub const TIER_MUTATION_RPC_MAX_ABORT_PAYLOAD_SIZE: usize = TIER_MUTATION_RPC_MAX_PREPARE_PAYLOAD_SIZE;
@@ -220,18 +219,6 @@ pub fn is_cross_pool_fence_capability_probe(command: &[u8]) -> bool {
&& command.starts_with(CROSS_POOL_FENCE_CAPABILITY_PROBE_PREFIX)
}
pub fn ilm_recovery_export_capability_probe(nonce: &[u8; 16]) -> Vec<u8> {
let mut probe = Vec::with_capacity(ILM_RECOVERY_EXPORT_CAPABILITY_PROBE_PREFIX.len() + nonce.len());
probe.extend_from_slice(ILM_RECOVERY_EXPORT_CAPABILITY_PROBE_PREFIX);
probe.extend_from_slice(nonce);
probe
}
pub fn is_ilm_recovery_export_capability_probe(command: &[u8]) -> bool {
command.len() == ILM_RECOVERY_EXPORT_CAPABILITY_PROBE_PREFIX.len() + 16
&& command.starts_with(ILM_RECOVERY_EXPORT_CAPABILITY_PROBE_PREFIX)
}
pub fn encode_remote_version_state_capability(
topology_member: &str,
process_epoch: &[u8; 16],
@@ -2140,13 +2127,12 @@ mod scanner_activity_tests {
mod heal_control_tests {
use super::{
CROSS_POOL_FENCE_CAPABILITY_PROBE_PREFIX, HEAL_CONTROL_CAPABILITY_PROBE_PREFIX, HEAL_CONTROL_PROTOCOL_VERSION,
ILM_RECOVERY_EXPORT_CAPABILITY_PROBE_PREFIX, REMOTE_VERSION_STATE_CAPABILITY_PROBE_PREFIX,
canonical_heal_control_capability_ack, canonical_heal_control_request_body, canonical_heal_control_response_body,
decode_remote_version_state_capability, encode_cross_pool_fence_capability, encode_remote_version_state_capability,
heal_control_capability_probe, heal_control_coordinator_epoch, heal_control_execution_timeout,
heal_control_execution_timeout_for, ilm_recovery_export_capability_probe, internode_rpc_timeout,
is_cross_pool_fence_capability_probe, is_heal_control_capability_probe, is_ilm_recovery_export_capability_probe,
is_remote_version_state_capability_probe, normalize_internode_rpc_timeout, remote_version_state_capability_probe,
REMOTE_VERSION_STATE_CAPABILITY_PROBE_PREFIX, canonical_heal_control_capability_ack, canonical_heal_control_request_body,
canonical_heal_control_response_body, decode_remote_version_state_capability, encode_cross_pool_fence_capability,
encode_remote_version_state_capability, heal_control_capability_probe, heal_control_coordinator_epoch,
heal_control_execution_timeout, heal_control_execution_timeout_for, internode_rpc_timeout,
is_cross_pool_fence_capability_probe, is_heal_control_capability_probe, is_remote_version_state_capability_probe,
normalize_internode_rpc_timeout, remote_version_state_capability_probe,
};
use crate::heal_control;
use std::time::Duration;
@@ -2210,19 +2196,6 @@ mod heal_control_tests {
assert!(!is_remote_version_state_capability_probe(REMOTE_VERSION_STATE_CAPABILITY_PROBE_PREFIX));
}
#[test]
fn ilm_recovery_export_capability_probe_requires_exact_prefix_and_nonce() {
let probe = ilm_recovery_export_capability_probe(&[7; 16]);
assert!(is_ilm_recovery_export_capability_probe(&probe));
assert!(!is_ilm_recovery_export_capability_probe(ILM_RECOVERY_EXPORT_CAPABILITY_PROBE_PREFIX));
let mut wrong_prefix = probe.clone();
wrong_prefix[0] ^= 1;
assert!(!is_ilm_recovery_export_capability_probe(&wrong_prefix));
let mut extra = probe;
extra.push(0);
assert!(!is_ilm_recovery_export_capability_probe(&extra));
}
#[test]
fn remote_version_state_capability_binds_member_and_process_epoch() {
let encoded =
+3 -4
View File
@@ -65,10 +65,9 @@ pub use multipart::{
};
pub use object::{
ObjectLockIntegrity, ReplicationSourceObject, ReplicationTargetObject, SsecPassthroughCapability, SsecPassthroughGate,
VersionIdentityCapability, content_matches_by_etag, is_replication_target_offline_error, object_lock_put_integrity,
replication_action_for_target, replication_etags_match, single_part_replica_etag_mismatch, ssec_passthrough_evidence_present,
ssec_passthrough_gate, target_is_newer_than_source_null_version, version_identity_capability_from_put,
version_identity_drifted,
content_matches_by_etag, is_replication_target_offline_error, object_lock_put_integrity, replication_action_for_target,
replication_etags_match, single_part_replica_etag_mismatch, ssec_passthrough_evidence_present, ssec_passthrough_gate,
target_is_newer_than_source_null_version, version_identity_drifted,
};
pub use operation::{
MustReplicateOptions, ReplicationDeleteScheduleInput, ReplicationDeleteSource, ReplicationDeleteStateSource,
+8 -78
View File
@@ -234,62 +234,18 @@ fn comparable_metadata(metadata: Option<&HashMap<String, String>>) -> HashMap<St
/// real (non-nil) version uuid, and drift means the target answered with
/// anything else — including nothing at all.
pub fn version_identity_drifted(source_version_id: &str, assigned_version_id: Option<&str>) -> bool {
version_identity_capability_from_put(source_version_id, assigned_version_id) == Some(VersionIdentityCapability::MintsOwn)
}
/// Whether a replication target adopts the source version id it is handed on
/// PutObject / CompleteMultipartUpload, or mints its own.
///
/// A target that mints its own ids (AWS S3, Wasabi, Impossible Cloud) still
/// stores the bytes, but every later version-addressed request from the
/// source names an id the target never had. Its HEAD then answers 404 —
/// indistinguishable from a replica that is really missing — so a heal, MRF
/// retry or existing-object resync re-drive would PUT the object again and
/// mint yet another target version (rustfs/backlog#2340). The replication
/// worker learns the verdict from each PUT response (and replication-check's
/// VersionFidelity phase) and, once `MintsOwn` is known, locates a replica by
/// exact key and ETag before concluding that it is missing. The verdict cache
/// is owned by the runtime's bucket target system; this crate owns only the
/// vocabulary and the judgment.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
pub enum VersionIdentityCapability {
#[default]
Unknown,
Adopts,
MintsOwn,
}
impl VersionIdentityCapability {
/// True when a 404 from a version-addressed HEAD on this target cannot be
/// read as "replica missing": the source-side id was never the target's.
pub fn version_addressing_unreliable(self) -> bool {
self == VersionIdentityCapability::MintsOwn
}
}
/// Judge the identity contract from one replication write: `None` when no
/// contract applies (the source addressed no real version — an empty or nil
/// uuid travels as the literal "null", unversioned-source semantics),
/// otherwise whether the target echoed the source id or answered with
/// anything else — including nothing at all.
pub fn version_identity_capability_from_put(
source_version_id: &str,
assigned_version_id: Option<&str>,
) -> Option<VersionIdentityCapability> {
if source_version_id.is_empty() {
return None;
return false;
}
// A nil source uuid travels as the literal "null" (unversioned-source
// semantics); no identity contract applies to it.
if uuid::Uuid::parse_str(source_version_id)
.map(|uuid| uuid.is_nil())
.unwrap_or(true)
{
return None;
return false;
}
Some(if assigned_version_id == Some(source_version_id) {
VersionIdentityCapability::Adopts
} else {
VersionIdentityCapability::MintsOwn
})
assigned_version_id != Some(source_version_id)
}
const REPLICATION_TARGET_OFFLINE_ERROR_MARKERS: &[&str] = &[
@@ -421,9 +377,9 @@ mod tests {
use super::{
ReplicationSourceObject, ReplicationTargetObject, SsecPassthroughCapability, SsecPassthroughGate,
VersionIdentityCapability, content_matches_by_etag, is_replication_target_offline_error, replication_action_for_target,
replication_etags_match, single_part_replica_etag_mismatch, ssec_passthrough_evidence_present, ssec_passthrough_gate,
target_is_newer_than_source_null_version, version_identity_capability_from_put, version_identity_drifted,
content_matches_by_etag, is_replication_target_offline_error, replication_action_for_target, replication_etags_match,
single_part_replica_etag_mismatch, ssec_passthrough_evidence_present, ssec_passthrough_gate,
target_is_newer_than_source_null_version, version_identity_drifted,
};
use crate::filemeta::{ReplicationAction, ReplicationType};
use crate::http::AMZ_OBJECT_LOCK_MODE;
@@ -552,32 +508,6 @@ mod tests {
}
}
#[test]
fn version_identity_capability_is_judged_only_for_real_source_versions() {
let source = "8e4d2f4c-2d5c-4f1b-9d0a-9c8b7a6f5e4d";
assert_eq!(
version_identity_capability_from_put(source, Some(source)),
Some(VersionIdentityCapability::Adopts)
);
// Wasabi / AWS shape: a minted id, or no id at all, both mean the
// source-side id is not addressable on the target.
assert_eq!(
version_identity_capability_from_put(source, Some("001788697733811332140-fR6j6uXKV-")),
Some(VersionIdentityCapability::MintsOwn)
);
assert_eq!(
version_identity_capability_from_put(source, None),
Some(VersionIdentityCapability::MintsOwn)
);
// No contract for an unversioned source write.
assert_eq!(version_identity_capability_from_put("", Some("anything")), None);
assert_eq!(version_identity_capability_from_put("00000000-0000-0000-0000-000000000000", None), None);
assert_eq!(version_identity_capability_from_put("null", Some("null")), None);
assert!(VersionIdentityCapability::MintsOwn.version_addressing_unreliable());
assert!(!VersionIdentityCapability::Adopts.version_addressing_unreliable());
assert!(!VersionIdentityCapability::Unknown.version_addressing_unreliable());
}
#[test]
fn replication_target_offline_error_classifier_is_network_scoped() {
assert!(is_replication_target_offline_error("put_object dispatch failure: connector error"));
-1
View File
@@ -106,7 +106,6 @@ bytes.workspace = true
hex-simd.workspace = true
[dev-dependencies]
rustfs-heal.workspace = true
tracing-subscriber = { workspace = true, features = ["json", "env-filter", "time"] }
serial_test = { workspace = true }
temp-env = { workspace = true, features = ["async_closure"] }
+2 -2
View File
@@ -85,8 +85,8 @@ pub use rustfs_scanner_metrics::last_minute;
pub use scanner::{
ScannerCycleRecoveryMarker, ScannerCycleRecoveryStatus, ScannerCycleScheduleStatus, ScannerPauseBacklogAlertReason,
ScannerPauseBacklogPhase, ScannerPauseBacklogStatus, ScannerPauseBacklogThresholds, ScannerUsageStateResetResult,
init_data_scanner, init_scanner_with_recovery, reset_scanner_cycle_recovery, reset_scanner_usage_state_for_full_rebuild,
scanner_cycle_recovery_status, scanner_cycle_schedule_status, scanner_pause_backlog_status, scanner_topology_digest,
init_data_scanner, reset_scanner_cycle_recovery, reset_scanner_usage_state_for_full_rebuild, scanner_cycle_recovery_status,
scanner_cycle_schedule_status, scanner_pause_backlog_status, scanner_topology_digest,
};
pub use scanner_io::{
ScannerDirtyUsageAckError, ScannerDirtyUsageBucket, ScannerDirtyUsageSnapshot, ScannerDirtyUsageState,
+21 -115
View File
@@ -14,6 +14,7 @@
use std::collections::BTreeMap;
use std::future::Future;
#[cfg(test)]
use std::sync::Mutex as StdMutex;
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{Arc, LazyLock, RwLock};
@@ -947,33 +948,6 @@ pub async fn init_data_scanner(ctx: CancellationToken, storeapi: Arc<ECStore>) {
init_data_scanner_with_storage(ctx, storeapi).await;
}
/// Start normal scanning when enabled, or one resume-only cleanup attempt.
/// The disabled branch returns a finite task for the startup owner to join;
/// it never enables ordinary namespace scanning or accepts a new reset intent.
pub async fn init_scanner_with_recovery(
ctx: CancellationToken,
storeapi: Arc<ECStore>,
enabled: bool,
) -> Option<tokio::task::JoinHandle<()>> {
if enabled {
init_data_scanner(ctx, storeapi).await;
return None;
}
Some(tokio::spawn(async move {
if let Err(error) = resume_scanner_cycle_cleanup(ctx, storeapi).await {
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
state = "disabled_cleanup_deferred",
error = %error,
"Disabled scanner cleanup remains pending for an operator retry"
);
}
}))
}
async fn init_data_scanner_with_storage<S>(ctx: CancellationToken, storeapi: Arc<S>)
where
S: ScannerStorage,
@@ -1584,11 +1558,6 @@ async fn mark_scan_cycle_idle(cycle_info: &mut CurrentCycle, cycle_metrics_guard
cycle_metrics_guard.finish(cycle_info.clone()).await;
}
struct ScannerCycleScheduling {
requires_full_scan: bool,
service_cohort: Option<Arc<StdMutex<crate::scanner_io::ScannerServiceCohort>>>,
}
#[cfg(test)]
async fn run_data_scanner_cycle<S>(
ctx: &CancellationToken,
@@ -1601,19 +1570,7 @@ where
S: ScannerStorage,
{
let cycle_budget = ScannerCycleBudget::new(ctx, scanner_cycle_budget_config());
run_data_scanner_cycle_with_budget(
ctx,
storeapi,
cycle_info,
cycle_revision,
leader_epoch,
cycle_budget,
ScannerCycleScheduling {
requires_full_scan: true,
service_cohort: None,
},
)
.await
run_data_scanner_cycle_with_budget(ctx, storeapi, cycle_info, cycle_revision, leader_epoch, cycle_budget, true).await
}
#[instrument(skip_all)]
@@ -1625,7 +1582,7 @@ async fn run_data_scanner_cycle_with_budget<S>(
cycle_revision: &mut DataUsageCacheRevision,
leader_epoch: u64,
cycle_budget: Arc<ScannerCycleBudget>,
scheduling: ScannerCycleScheduling,
requires_full_scan: bool,
) -> ScannerCycleOutcome
where
S: ScannerStorage,
@@ -1730,7 +1687,6 @@ where
};
let baseline_publication_epoch = baseline_publication_guard.epoch();
let usage_persist_baseline_result = read_data_usage_persist_baseline(storeapi.clone()).await;
let observed_usage_candidate_result = read_config(storeapi.clone(), DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str()).await;
drop(baseline_publication_guard);
let usage_persist_baseline = match usage_persist_baseline_result {
Ok(baseline) => baseline,
@@ -1751,24 +1707,6 @@ where
return ScannerCycleOutcome::Failed;
}
};
let observed_usage_candidate = match observed_usage_candidate_result {
Ok(candidate) => Some(Bytes::from(candidate)),
Err(EcstoreError::ConfigNotFound) => None,
Err(err) => {
debug!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
path = %DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str(),
state = "observed_candidate_load_failed",
error = %err,
"Scanner skipped an unavailable observed usage candidate for scoped refresh"
);
None
}
};
let (sender, receiver) = mpsc::channel::<DataUsageInfo>(1);
let done_cycle = Metrics::time(Metric::ScanCycle);
@@ -1783,9 +1721,7 @@ where
scan_mode,
scan_scope: crate::scanner_io::ScannerBucketScanScope::default(),
persisted_usage_baseline: usage_persist_baseline.data.clone(),
observed_usage_candidate,
requires_full_scan: scheduling.requires_full_scan,
service_cohort: scheduling.service_cohort,
requires_full_scan,
#[cfg(test)]
resolved_scope_observer: None,
},
@@ -1910,10 +1846,10 @@ where
.as_ref()
.map(|(notification_system, grants)| (Arc::clone(notification_system), grants.clone()));
let remote_lease_release_safe = Arc::new(AtomicBool::new(true));
let mut usage_publication_result = match publication_defer_reason {
let mut usage_persist_outcome = match publication_defer_reason {
Some(reason) => {
drop(receiver);
DataUsagePublicationResult::from(DataUsagePersistOutcome::Deferred(reason))
DataUsagePersistOutcome::Deferred(reason)
}
None => {
// ScannerIO emits its complete or observational update only after
@@ -1928,11 +1864,6 @@ where
.as_ref()
.map(|(_, grants)| grants.iter().map(|grant| grant.lease.token).collect())
.unwrap_or_default();
let ack_expectation = scan_result
.as_ref()
.ok()
.filter(|result| result.has_dirty_usage_to_acknowledge())
.and_then(ScannerCycleResult::publication_expectation);
let mut usage_persist_task = AbortOnDropHandle::new(tokio::spawn(async move {
store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe_for_publication_epoch_and_lease_fence(
ctx_clone,
@@ -1945,7 +1876,6 @@ where
remote_lease_deadline,
remote_lease_fence,
)
.with_ack_expectation(ack_expectation)
.with_remote_lease_tokens(remote_lease_tokens)
.with_lease_release_flag(remote_lease_release_safe_for_task),
move || {
@@ -1979,7 +1909,7 @@ where
error = %err,
"Scanner data usage persistence task failed"
);
DataUsagePublicationResult::from(DataUsagePersistOutcome::Failed)
DataUsagePersistOutcome::Failed
}
DataUsagePersistTaskResult::Cancelled => {
debug!(
@@ -1991,7 +1921,7 @@ where
state = "usage_persist_task_cancelled",
"Scanner data usage persistence task cancelled"
);
DataUsagePublicationResult::from(DataUsagePersistOutcome::Failed)
DataUsagePersistOutcome::Failed
}
DataUsagePersistTaskResult::TimedOut => {
error!(
@@ -2004,12 +1934,11 @@ where
state = "usage_persist_task_timed_out",
"Scanner data usage persistence task timed out"
);
DataUsagePublicationResult::from(DataUsagePersistOutcome::Failed)
DataUsagePersistOutcome::Failed
}
}
}
};
let mut usage_persist_outcome = usage_publication_result.outcome();
let lease_expired = remote_publication_leases
.as_ref()
.is_some_and(|(_, grants)| grants.iter().any(|grant| !grant.lease.is_valid()));
@@ -2209,9 +2138,8 @@ where
};
}
usage_publication_result.restrict_outcome(usage_persist_outcome);
let (completion_outcome, scanner_pending_maintenance_work, remote_dirty_usage_acknowledgements) =
finalize_scanner_cycle_result(scan_cycle_result, usage_publication_result);
finalize_scanner_cycle_result(scan_cycle_result, usage_persist_outcome);
let remote_dirty_usage_pending = if remote_dirty_usage_acknowledgements.is_empty() {
false
} else if let Some(notification_system) = storeapi.scanner_notification_system() {
@@ -2655,7 +2583,6 @@ where
let mut clean_idle_backoff = ScannerCleanIdleBackoff::default();
let mut superseded_backoff = ScannerRetryBackoff::default();
let mut deferred_backoff = ScannerRetryBackoff::default();
let service_cohort = Arc::new(StdMutex::new(crate::scanner_io::ScannerServiceCohort::default()));
let initial_runtime_config = resolve_scanner_runtime_config();
if clean_idle_topology_supported && maintenance_generation_seen.is_none() {
let Some((features, generation)) = detect_stable_scanner_maintenance_features(&ctx, &storeapi).await else {
@@ -2865,10 +2792,7 @@ where
&mut cycle_revision,
leader_epoch,
cycle_budget.clone(),
ScannerCycleScheduling {
requires_full_scan: true,
service_cohort: Some(service_cohort.clone()),
},
true,
),
guard.lock_lost_notified(),
)
@@ -3159,14 +3083,11 @@ where
&mut cycle_revision,
leader_epoch,
cycle_budget.clone(),
ScannerCycleScheduling {
requires_full_scan: maintenance_features.requires_full_scan(
maintenance_generation_seen,
scanner_maintenance_generation(),
wake_reason,
),
service_cohort: Some(service_cohort.clone()),
},
maintenance_features.requires_full_scan(
maintenance_generation_seen,
scanner_maintenance_generation(),
wake_reason,
),
),
guard.lock_lost_notified(),
)
@@ -3445,35 +3366,21 @@ fn scanner_cycle_completion_outcome(
fn finalize_scanner_cycle_result(
scan_cycle_result: crate::scanner_io::ScannerCycleResult,
publication: DataUsagePublicationResult,
usage_persist_outcome: DataUsagePersistOutcome,
) -> (ScannerCycleOutcome, bool, Vec<ScannerDirtyUsageAcknowledgement>) {
let (usage_persist_outcome, proof) = publication.into_parts();
let completion_outcome = scanner_cycle_completion_outcome_for_result(&scan_cycle_result, usage_persist_outcome);
let pending_maintenance_work = scan_cycle_result.has_pending_maintenance_work();
let durable_complete_snapshot = scan_cycle_result.status == ScannerCycleStatus::Complete
&& matches!(
usage_persist_outcome,
DataUsagePersistOutcome::Saved | DataUsagePersistOutcome::AlreadyDurable
)
&& scan_cycle_result.publication_expectation().as_ref().is_some_and(|expected| {
proof
.as_ref()
.is_some_and(|proof| proof.verified_version_for(expected).is_some())
});
let pending_maintenance_work = scan_cycle_result.has_pending_maintenance_work()
|| (scan_cycle_result.has_dirty_usage_to_acknowledge() && !durable_complete_snapshot);
);
let remote_dirty_usage_acknowledgements = if durable_complete_snapshot {
match proof {
Some(proof) => scan_cycle_result.acknowledge_durable_usage(&proof),
None => Vec::new(),
}
scan_cycle_result.acknowledge_durable_usage()
} else {
Vec::new()
};
(
completion_outcome,
pending_maintenance_work || crate::scanner_io::dirty_usage_buckets_pending(),
remote_dirty_usage_acknowledgements,
)
(completion_outcome, pending_maintenance_work, remote_dirty_usage_acknowledgements)
}
fn scanner_cycle_completion_outcome_for_result(
@@ -3573,7 +3480,6 @@ use activity::*;
use backlog::*;
use cycle_state::*;
use leadership::*;
pub(crate) use usage_store::RootPublicationProof;
use usage_store::*;
pub use activity::scanner_topology_digest;
+19 -244
View File
@@ -18,7 +18,6 @@ use crate::data_usage_define::{
DATA_USAGE_BLOOM_RECOVERY_PATH, DATA_USAGE_RECOVERY_PATH, usage_floor_primary_read_error_allows_backup,
};
use crate::storage_api::owner::ObjectIO as _;
use std::sync::atomic::AtomicU64;
use tokio::io::AsyncReadExt as _;
const SCANNER_CYCLE_RECOVERY_SCHEMA_VERSION: u16 = 1;
@@ -35,86 +34,6 @@ const CACHE_CYCLE_AHEAD: &str = "cache_cycle_ahead";
const SCANNER_USAGE_STATE_RESET_MODE_FULL_REBUILD: &str = "full-rebuild";
#[cfg(test)]
pub(super) mod cleanup_io_fault {
use super::*;
#[derive(Clone, Copy, PartialEq, Eq)]
pub(in crate::scanner) enum Stage {
PrimaryRead,
PrimaryWrite,
UsageFence,
}
struct Injection {
store: std::sync::Weak<ECStore>,
stage: Stage,
fired: AtomicBool,
owned: AtomicBool,
newer_completion: bool,
}
static INJECTION: StdMutex<Option<Arc<Injection>>> = StdMutex::new(None);
pub(in crate::scanner) struct Guard(Arc<Injection>);
impl Guard {
pub(in crate::scanner) fn fired_while_owned(&self) -> bool {
self.0.fired.load(Ordering::Relaxed) && self.0.owned.load(Ordering::Relaxed)
}
}
impl Drop for Guard {
fn drop(&mut self) {
let mut slot = INJECTION.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
if slot.as_ref().is_some_and(|current| Arc::ptr_eq(current, &self.0)) {
*slot = None;
}
}
}
pub(in crate::scanner) fn install(store: &Arc<ECStore>, stage: Stage, newer_completion: bool) -> Guard {
let injection = Arc::new(Injection {
store: Arc::downgrade(store),
stage,
fired: AtomicBool::new(false),
owned: AtomicBool::new(false),
newer_completion,
});
let mut slot = INJECTION.lock().expect("cleanup injection slot");
assert!(slot.is_none(), "only one cleanup I/O injection may be installed");
*slot = Some(injection.clone());
Guard(injection)
}
pub(super) fn check(store: &Arc<ECStore>, stage: Stage, owned: bool) -> Result<(), ScannerError> {
let injection = {
let mut slot = INJECTION.lock().expect("cleanup injection slot");
if slot
.as_ref()
.is_some_and(|injection| injection.stage == stage && injection.store.ptr_eq(&Arc::downgrade(store)))
{
slot.take()
} else {
None
}
};
let Some(injection) = injection else {
return Ok(());
};
injection.fired.store(true, Ordering::Relaxed);
injection.owned.store(owned, Ordering::Relaxed);
if injection.newer_completion {
set_scanner_cycle_recovery_status(recovery_status("healthy", None, false));
}
let reason = match stage {
Stage::PrimaryRead => "injected primary read failure",
Stage::PrimaryWrite => "injected primary write failure",
Stage::UsageFence => "injected usage fence failure",
};
Err(ScannerError::Io(std::io::Error::other(reason)))
}
}
#[derive(Clone, Debug, Default, Serialize)]
pub struct ScannerCycleRecoveryStatus {
/// The immutable primary object whose revision is being guarded.
@@ -162,8 +81,6 @@ static SCANNER_CYCLE_RECOVERY_STATUS: LazyLock<RwLock<ScannerCycleRecoveryStatus
..Default::default()
})
});
// An old startup observation must not overwrite a newer explicit reset status.
static SCANNER_CYCLE_RECOVERY_STATUS_VERSION: AtomicU64 = AtomicU64::new(0);
pub fn scanner_cycle_recovery_status() -> ScannerCycleRecoveryStatus {
SCANNER_CYCLE_RECOVERY_STATUS
@@ -173,24 +90,6 @@ pub fn scanner_cycle_recovery_status() -> ScannerCycleRecoveryStatus {
}
fn set_scanner_cycle_recovery_status(status: ScannerCycleRecoveryStatus) {
let _ = publish_scanner_cleanup_status(status, None);
}
pub(super) fn scanner_cleanup_status_version() -> u64 {
let _status = SCANNER_CYCLE_RECOVERY_STATUS
.read()
.unwrap_or_else(|poisoned| poisoned.into_inner());
SCANNER_CYCLE_RECOVERY_STATUS_VERSION.load(Ordering::Relaxed)
}
pub(super) fn publish_scanner_cleanup_status(status: ScannerCycleRecoveryStatus, expected: Option<u64>) -> Option<u64> {
let mut current = SCANNER_CYCLE_RECOVERY_STATUS
.write()
.unwrap_or_else(|poisoned| poisoned.into_inner());
let version = SCANNER_CYCLE_RECOVERY_STATUS_VERSION.load(Ordering::Relaxed);
if expected.is_some_and(|expected| expected == u64::MAX || expected != version) {
return None;
}
let recovery_required = if matches!(
status.state.as_str(),
"blocked"
@@ -207,27 +106,9 @@ pub(super) fn publish_scanner_cleanup_status(status: ScannerCycleRecoveryStatus,
};
metrics::gauge!(METRIC_SCANNER_CYCLE_RECOVERY_REQUIRED).set(recovery_required);
metrics::gauge!(METRIC_SCANNER_CYCLE_RECOVERY_RETRY_COUNT).set(status.retry_count as f64);
*current = status;
let next = version.saturating_add(1);
SCANNER_CYCLE_RECOVERY_STATUS_VERSION.store(next, Ordering::Relaxed);
Some(next)
}
fn publish_scanner_cleanup_failure(reason: String, expected: u64) {
let mut status = {
let current = SCANNER_CYCLE_RECOVERY_STATUS
.read()
.unwrap_or_else(|poisoned| poisoned.into_inner());
if SCANNER_CYCLE_RECOVERY_STATUS_VERSION.load(Ordering::Relaxed) != expected {
return;
}
current.clone()
};
// Preserve the latest core progress, including a newly written primary's
// revision and epoch. The original marker may predate that durable write.
status.reason = Some(reason);
status.last_attempt_at_unix_secs = Some(unix_now_secs());
let _ = publish_scanner_cleanup_status(status, Some(expected));
*SCANNER_CYCLE_RECOVERY_STATUS
.write()
.unwrap_or_else(|poisoned| poisoned.into_inner()) = status;
}
pub(super) fn record_scanner_usage_floor_failure(reason: String) {
@@ -1051,81 +932,10 @@ pub(crate) async fn load_scanner_cycle_state_for_startup(
}
}
/// Reset a blocked cycle state after an explicit full-rescan request. Durable
/// cleanup state fences primary rewrites; marker removal retains its ETag and
/// usage-epoch checks.
/// Reset a blocked cycle state after an operator has explicitly requested a full
/// usage rebuild. The primary object is changed first with its observed ETag;
/// the recovery marker is removed only when its own ETag still matches.
pub async fn reset_scanner_cycle_recovery(ctx: CancellationToken, storeapi: Arc<ECStore>) -> Result<(), ScannerError> {
reset_scanner_cycle_recovery_for_intent(ctx, storeapi, None, None).await
}
/// Resume only an operator reset whose cleanup phase is already durable.
/// Missing or merely blocked markers never authorize an automatic reset.
pub(super) async fn resume_scanner_cycle_cleanup(ctx: CancellationToken, storeapi: Arc<ECStore>) -> Result<(), ScannerError> {
let status_version = scanner_cleanup_status_version();
let (marker, revision) = match read_scanner_cleanup_marker(storeapi.clone(), &ctx).await {
Ok(Some(marker)) => marker,
Ok(None) => return Ok(()),
Err(error) => {
let _ =
publish_scanner_cleanup_status(recovery_status("blocked", Some(&error.to_string()), false), Some(status_version));
return Err(error);
}
};
let mut observation =
publish_scanner_cleanup_status(recovery_status_from_marker(&marker, &marker.state), Some(status_version));
if marker.state != "cleanup-pending" {
return Ok(());
}
let result = reset_scanner_cycle_recovery_for_intent(ctx, storeapi, Some(revision), Some(&mut observation)).await;
if let Err(error) = &result
&& let Some(observation) = observation
{
publish_scanner_cleanup_failure(error.to_string(), observation);
}
result
}
pub(super) async fn read_scanner_cleanup_marker(
storeapi: Arc<impl ScannerObjectIO>,
ctx: &CancellationToken,
) -> Result<Option<(ScannerCycleRecoveryMarker, DataUsageCacheRevision)>, ScannerError> {
let (data, revision) = tokio::select! {
biased;
_ = ctx.cancelled() => return Err(ScannerError::Other("scanner cleanup recovery was cancelled".to_string())),
result = tokio::time::timeout(data_usage_persist_timeout(), read_cycle_recovery_marker_bytes(storeapi)) => {
result.map_err(|_| ScannerError::Other("scanner cleanup marker inspection timed out".to_string()))?
.map_err(|err| ScannerError::Other(format!("failed to inspect pending scanner cleanup: {err}")))?
}
};
let Some(data) = data else {
return Ok(None);
};
let marker: ScannerCycleRecoveryMarker = serde_json::from_slice(&data)
.map_err(|err| ScannerError::Other(format!("pending scanner cleanup marker is invalid: {err}")))?;
validate_recovery_marker(&marker)
.map_err(|err| ScannerError::Other(format!("pending scanner cleanup marker is invalid: {err}")))?;
Ok(Some((marker, revision)))
}
pub(super) async fn reset_scanner_cycle_recovery_for_intent(
ctx: CancellationToken,
storeapi: Arc<ECStore>,
expected_cleanup_revision: Option<DataUsageCacheRevision>,
mut observation: Option<&mut Option<u64>>,
) -> Result<(), ScannerError> {
#[cfg(test)]
let resume_only = expected_cleanup_revision.is_some();
// The outer Some distinguishes a tracked resume whose version may be
// invalidated from an explicit v3 reset with no observation owner.
let mut publish_status = |status| {
if let Some(version) = observation.as_deref_mut() {
if let Some(expected) = *version {
*version = publish_scanner_cleanup_status(status, Some(expected));
}
} else {
set_scanner_cycle_recovery_status(status);
}
};
let lock = storeapi
.new_ns_lock(RUSTFS_META_BUCKET, "leader.lock")
.await
@@ -1154,21 +964,6 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
}
Err(err) => return Err(ScannerError::Other(format!("failed to read cycle recovery marker: {err}"))),
};
if let Some(expected) = expected_cleanup_revision {
if marker_revision != expected || !owns_reset() {
return Err(ScannerError::Other(
"pending scanner cleanup changed before recovery acquired ownership".to_string(),
));
}
let marker: ScannerCycleRecoveryMarker = marker_data
.as_deref()
.and_then(|data| serde_json::from_slice(data).ok())
.filter(|marker| validate_recovery_marker(marker).is_ok() && marker.state == "cleanup-pending")
.ok_or_else(|| {
ScannerError::Other("scanner cleanup recovery requires an unchanged cleanup-pending marker".to_string())
})?;
publish_status(recovery_status_from_marker(&marker, "cleanup-pending"));
}
let Some(marker_data) = marker_data else {
// A delete may commit before its reply is lost. Confirm both durable
// fences before treating a retry without its marker as completed.
@@ -1186,7 +981,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
"scanner cycle recovery marker is absent without a completed reset fence".to_string(),
));
}
publish_status(recovery_status("healthy", None, false));
set_scanner_cycle_recovery_status(recovery_status("healthy", None, false));
super::notify_scanner_cycle_recovery_wake();
return Ok(());
};
@@ -1202,10 +997,6 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
));
}
#[cfg(test)]
if resume_only {
cleanup_io_fault::check(&storeapi, cleanup_io_fault::Stage::PrimaryRead, owns_reset())?;
}
let (mut primary_reader, primary_revision) = match storeapi
.get_object_reader(
RUSTFS_META_BUCKET,
@@ -1271,7 +1062,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
let (cleanup_marker, cleanup_marker_revision) =
mark_cycle_recovery_cleanup_pending(storeapi.clone(), marker.clone(), &marker_revision, reset_epoch, &owns_reset)
.await?;
publish_status(recovery_status_from_marker(&cleanup_marker, "cleanup-pending"));
set_scanner_cycle_recovery_status(recovery_status_from_marker(&cleanup_marker, "cleanup-pending"));
let usage_floor = persisted_usage_floor(storeapi.clone()).await?;
let fence_epoch = primary_epoch
.max(usage_floor.leader_epoch)
@@ -1291,10 +1082,6 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
));
}
verify_cycle_reset_intent(storeapi.clone(), &cleanup_marker_revision, &owns_reset).await?;
#[cfg(test)]
if resume_only {
cleanup_io_fault::check(&storeapi, cleanup_io_fault::Stage::PrimaryWrite, owns_reset())?;
}
let preserved_info = save_reset_config(
storeapi.clone(),
DATA_USAGE_BLOOM_NAME_PATH.as_str(),
@@ -1368,7 +1155,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
ScannerError::Other(format!("failed to clear stale cycle recovery marker: {err}"))
}
})?;
publish_status(recovery_status("healthy", None, false));
set_scanner_cycle_recovery_status(recovery_status("healthy", None, false));
super::notify_scanner_cycle_recovery_wake();
return Ok(());
} else if !force_full_rescan && !marker_cleanup_pending {
@@ -1441,23 +1228,11 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
));
}
verify_cycle_reset_intent(storeapi.clone(), &marker_revision, &owns_reset).await?;
let usage_fence = fence_scanner_usage_epoch_with_expected_epoch(
&ctx,
storeapi.clone(),
leader_epoch,
Some(reset_epoch),
false,
&owns_reset,
);
#[cfg(test)]
let usage_fence = async {
if resume_only {
cleanup_io_fault::check(&storeapi, cleanup_io_fault::Stage::UsageFence, owns_reset())?;
}
usage_fence.await
};
if let Err(err) = usage_fence.await {
publish_status(ScannerCycleRecoveryStatus {
if let Err(err) =
fence_scanner_usage_epoch_with_expected_epoch(&ctx, storeapi.clone(), leader_epoch, Some(reset_epoch), false, &owns_reset)
.await
{
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
quarantine_path: Some(DATA_USAGE_BLOOM_RECOVERY_PATH.clone()),
state: "cleanup-pending".to_string(),
@@ -1485,7 +1260,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
}
};
if current_revision != rebuilt_revision {
publish_status(ScannerCycleRecoveryStatus {
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
quarantine_path: Some(DATA_USAGE_BLOOM_RECOVERY_PATH.clone()),
state: "cleanup-pending".to_string(),
@@ -1506,7 +1281,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
}
if guard.is_lock_lost() {
publish_status(ScannerCycleRecoveryStatus {
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
quarantine_path: Some(DATA_USAGE_BLOOM_RECOVERY_PATH.clone()),
state: "cleanup-pending".to_string(),
@@ -1543,7 +1318,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
.await
{
if scanner_publication_epoch_changed(&err) {
publish_status(ScannerCycleRecoveryStatus {
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
quarantine_path: Some(DATA_USAGE_BLOOM_RECOVERY_PATH.clone()),
state: "cleanup-pending".to_string(),
@@ -1561,7 +1336,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
"scanner recovery reset deferred by a movement epoch change".to_string(),
));
}
publish_status(ScannerCycleRecoveryStatus {
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
quarantine_path: Some(DATA_USAGE_BLOOM_RECOVERY_PATH.clone()),
state: "cleanup-pending".to_string(),
@@ -1577,7 +1352,7 @@ pub(super) async fn reset_scanner_cycle_recovery_for_intent(
});
return Err(ScannerError::Other(format!("failed to clear cycle recovery marker: {err}")));
}
publish_status(ScannerCycleRecoveryStatus {
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
quarantine_path: Some(DATA_USAGE_BLOOM_RECOVERY_PATH.clone()),
state: "healthy".to_string(),
+17 -197
View File
@@ -36,11 +36,6 @@ use tokio::time::{Duration, advance};
const TEST_DEFAULT_SCANNER_CYCLE_SECS: u64 = 24 * 60 * 60;
mod quota_reset_preservation;
mod recovery_control;
mod scoped_ack_publication;
async fn setup_scanner_cycle_store() -> (tempfile::TempDir, Arc<ECStore>) {
setup_scanner_cycle_store_with_usage_baseline(true).await
}
@@ -1224,18 +1219,7 @@ async fn coordinator_walks_during_pending_put_without_persisting_or_acknowledgin
let mut revision = DataUsageCacheRevision::Missing;
let outcome = tokio::time::timeout(
Duration::from_secs(30),
run_data_scanner_cycle_with_budget(
&ctx,
&store,
&mut cycle_info,
&mut revision,
1,
Arc::clone(&budget),
ScannerCycleScheduling {
requires_full_scan: true,
service_cohort: None,
},
),
run_data_scanner_cycle_with_budget(&ctx, &store, &mut cycle_info, &mut revision, 1, Arc::clone(&budget), true),
)
.await
.expect("the coordinator must finish its namespace walk while a PUT is pending");
@@ -1279,18 +1263,7 @@ async fn coordinator_walks_during_pending_put_without_persisting_or_acknowledgin
let retry_budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
let outcome = tokio::time::timeout(
Duration::from_secs(30),
run_data_scanner_cycle_with_budget(
&ctx,
&store,
&mut cycle_info,
&mut revision,
1,
Arc::clone(&retry_budget),
ScannerCycleScheduling {
requires_full_scan: true,
service_cohort: None,
},
),
run_data_scanner_cycle_with_budget(&ctx, &store, &mut cycle_info, &mut revision, 1, Arc::clone(&retry_budget), true),
)
.await
.expect("the same cycle must converge after the pending PUT drains");
@@ -5291,103 +5264,6 @@ async fn scanner_usage_state_reset_resumes_every_cleanup_boundary_without_rewrit
}
}
#[tokio::test]
#[serial]
async fn scanner_usage_state_reset_resumes_real_store_cleanup_boundaries_after_reopen() {
let primary_path = DATA_USAGE_OBJ_NAME_PATH.as_str();
let cleanup_paths = [
format!("{primary_path}.bkp"),
LEGACY_DATA_USAGE_OBJ_NAME_PATH.as_str().to_string(),
format!("{}.bkp", LEGACY_DATA_USAGE_OBJ_NAME_PATH.as_str()),
DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str().to_string(),
];
for completed in 0..=cleanup_paths.len() {
let (_temp_dir, store) = setup_scanner_cycle_store().await;
let cycle = CurrentCycle {
current: 12,
next: 42,
cycle_completed: vec![Utc::now()],
started: Utc::now(),
};
save_config(
store.clone(),
DATA_USAGE_BLOOM_NAME_PATH.as_str(),
encode_scanner_cycle_state(&cycle, 3).expect("cycle state should encode"),
)
.await
.expect("cycle state should persist");
let marker = scanner_usage_bootstrap_marker(std::time::SystemTime::UNIX_EPOCH, Some(3));
save_config(
store.clone(),
primary_path,
serde_json::to_vec(&marker).expect("usage reset marker should encode"),
)
.await
.expect("usage reset marker should persist");
for path in cleanup_paths.iter().skip(completed) {
let mut usage = complete_usage_with_bucket_count(Some(std::time::SystemTime::UNIX_EPOCH), 0);
usage.scanner_epoch = Some(1);
usage.scanner_cycle = Some(12);
save_config(store.clone(), path, serde_json::to_vec(&usage).expect("cleanup slot should encode"))
.await
.expect("cleanup slot should persist");
}
for path in ["buckets/quota-reservations/ledger", "buckets/example/incarnation"] {
save_config(store.clone(), path, b"retain".to_vec())
.await
.expect("unrelated state should persist before reopen");
}
let restarted = restart_scanner_cycle_store_from(&store).await;
let intent_before = read_config_with_revision(restarted.clone(), primary_path)
.await
.expect("reopened reset intent should be readable");
let result = reset_scanner_usage_state_for_full_rebuild(CancellationToken::new(), restarted.clone())
.await
.expect("reopened usage reset should complete");
assert_eq!(result.leader_epoch, 3, "boundary {completed}");
assert_eq!(result.next_cycle, 42, "boundary {completed}");
assert_eq!(result.reset_paths.len(), cleanup_paths.len() + 1 - completed, "boundary {completed}");
assert_eq!(
read_config_with_revision(restarted.clone(), primary_path)
.await
.expect("completed reset intent should remain readable"),
intent_before,
"boundary {completed}: resumed cleanup must not rewrite the reset intent"
);
let (floor, state) = persisted_usage_floor_for_startup(restarted.clone(), false)
.await
.expect("completed reset marker should remain resumable");
assert_eq!(floor.leader_epoch, 3, "boundary {completed}");
assert_eq!(state, PersistedUsageFloorStartup::BootstrapPending, "boundary {completed}");
assert!(
persisted_usage_floor(restarted.clone()).await.is_err(),
"boundary {completed}: bootstrap marker must not become an authoritative floor"
);
for path in &cleanup_paths {
assert!(
matches!(read_config(restarted.clone(), path).await, Err(EcstoreError::ConfigNotFound)),
"boundary {completed}: reset should remove stale usage slot {path}"
);
}
for path in ["buckets/quota-reservations/ledger", "buckets/example/incarnation"] {
assert_eq!(
read_config(restarted.clone(), path)
.await
.expect("unrelated state should survive reopened reset"),
b"retain",
"boundary {completed}: reset must preserve non-scanner-state config"
);
}
}
}
#[tokio::test]
async fn scanner_usage_state_reset_stops_usage_fence_after_owner_loss() {
let store = Arc::new(MemoryConfigStore::default());
@@ -6284,7 +6160,7 @@ async fn coordinator_classifies_an_expired_publication_lease() {
.await;
assert_eq!(
outcome.outcome(),
outcome,
DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::PublicationLeaseDeadlineExceeded)
);
assert!(store.put_counts.lock().await.is_empty(), "expired lease must prevent a PUT");
@@ -7425,7 +7301,7 @@ fn scanner_cycle_cache_floor_stays_pending_during_deferred_usage_publication() {
#[test]
#[serial]
fn finalizing_a_saved_enum_without_proof_keeps_dirty_pending() {
fn finalizing_a_saved_cycle_acknowledges_its_exact_dirty_snapshot() {
crate::scanner_io::clear_dirty_usage_bucket("photos");
crate::scanner_io::record_dirty_usage_bucket("photos");
let dirty_snapshot = crate::scanner_io::dirty_usage_buckets_for_tests();
@@ -7437,19 +7313,17 @@ fn finalizing_a_saved_enum_without_proof_keeps_dirty_pending() {
};
let unsaved = crate::scanner_io::ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(dirty_snapshot.clone()))
.with_remote_dirty_usage_acknowledgements(vec![remote_acknowledgement.clone()]);
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(unsaved, DataUsagePersistOutcome::NoUpdate.into());
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(unsaved, DataUsagePersistOutcome::NoUpdate);
assert_eq!(outcome, ScannerCycleOutcome::Failed);
assert!(acknowledgements.is_empty());
assert!(crate::scanner_io::dirty_usage_buckets_pending());
let saved = crate::scanner_io::ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(dirty_snapshot))
.with_remote_dirty_usage_acknowledgements(vec![remote_acknowledgement]);
let (outcome, pending, acknowledgements) = finalize_scanner_cycle_result(saved, DataUsagePersistOutcome::Saved.into());
.with_remote_dirty_usage_acknowledgements(vec![remote_acknowledgement.clone()]);
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(saved, DataUsagePersistOutcome::Saved);
assert_eq!(outcome, ScannerCycleOutcome::Completed);
assert!(acknowledgements.is_empty());
assert!(pending);
assert!(crate::scanner_io::dirty_usage_buckets_pending());
crate::scanner_io::clear_dirty_usage_bucket("photos");
assert_eq!(acknowledgements, vec![remote_acknowledgement]);
assert!(!crate::scanner_io::dirty_usage_buckets_pending());
}
#[test]
@@ -7461,7 +7335,7 @@ fn finalizing_a_deferred_usage_save_keeps_dirty_work_pending() {
let deferred = crate::scanner_io::ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(dirty_snapshot));
let (outcome, _, acknowledgements) =
finalize_scanner_cycle_result(deferred, DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement).into());
finalize_scanner_cycle_result(deferred, DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert_eq!(outcome, ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert!(acknowledgements.is_empty());
@@ -7481,7 +7355,7 @@ fn finalizing_post_scan_observation_advances_partially_without_dirty_ack() {
)
.with_observational_snapshot_published(true);
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(observed, DataUsagePersistOutcome::Saved.into());
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(observed, DataUsagePersistOutcome::Saved);
assert_eq!(outcome, ScannerCycleOutcome::Partial);
assert!(acknowledgements.is_empty());
@@ -7517,20 +7391,17 @@ async fn scanner_cycle_keeps_remote_pending_acknowledgement() {
#[test]
#[serial]
fn finalizing_an_already_durable_enum_without_proof_keeps_dirty_pending() {
fn finalizing_an_already_durable_cycle_acknowledges_its_exact_dirty_snapshot() {
crate::scanner_io::clear_dirty_usage_bucket("photos");
crate::scanner_io::record_dirty_usage_bucket("photos");
let dirty_snapshot = crate::scanner_io::dirty_usage_buckets_for_tests();
let durable = crate::scanner_io::ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(dirty_snapshot));
let (outcome, pending, acknowledgements) =
finalize_scanner_cycle_result(durable, DataUsagePersistOutcome::AlreadyDurable.into());
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(durable, DataUsagePersistOutcome::AlreadyDurable);
assert_eq!(outcome, ScannerCycleOutcome::Completed);
assert!(acknowledgements.is_empty());
assert!(pending);
assert!(crate::scanner_io::dirty_usage_buckets_pending());
crate::scanner_io::clear_dirty_usage_bucket("photos");
assert!(!crate::scanner_io::dirty_usage_buckets_pending());
}
#[test]
@@ -7541,8 +7412,7 @@ fn finalizing_a_prior_same_cycle_snapshot_keeps_new_dirty_work_pending() {
let dirty_snapshot = crate::scanner_io::dirty_usage_buckets_for_tests();
let durable = crate::scanner_io::ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(dirty_snapshot));
let (outcome, _, acknowledgements) =
finalize_scanner_cycle_result(durable, DataUsagePersistOutcome::PriorCycleDurable.into());
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(durable, DataUsagePersistOutcome::PriorCycleDurable);
assert_eq!(outcome, ScannerCycleOutcome::Completed);
assert!(acknowledgements.is_empty());
@@ -7558,7 +7428,7 @@ fn finalizing_a_durable_superseded_snapshot_keeps_dirty_work_pending() {
let dirty_snapshot = crate::scanner_io::dirty_usage_buckets_for_tests();
let superseded = crate::scanner_io::ScannerCycleResult::new(ScannerCycleStatus::Superseded, Some(dirty_snapshot));
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(superseded, DataUsagePersistOutcome::Saved.into());
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(superseded, DataUsagePersistOutcome::Saved);
assert_eq!(outcome, ScannerCycleOutcome::Superseded);
assert!(acknowledgements.is_empty());
@@ -8644,56 +8514,6 @@ async fn test_wait_for_next_scanner_cycle_wakes_for_dirty_usage() {
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test(start_paused = true)]
#[serial]
async fn service_cohort_aging_preserves_explicit_cycle_wait() {
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
let config = ScannerRuntimeConfig {
cycle_interval: Duration::from_secs(3600),
cycle_interval_source: ScannerRuntimeConfigSource::Env,
..Default::default()
};
let observed = ScannerCycleObservedGenerations::for_wait(
&config,
None,
crate::scanner_io::dirty_usage_generation(),
crate::runtime_config::scanner_runtime_config_generation(),
crate::scanner_io::scanner_maintenance_generation(),
);
assert_eq!(observed.dirty_usage, None);
let inventory = HashMap::from([(
crate::data_usage_define::DataUsageCacheSource::new(0, 0),
vec![crate::storage_api::scanner_io::BucketInfo {
name: "waiting-bootstrap".to_string(),
..Default::default()
}],
)]);
let mut cohort = crate::scanner_io::ScannerServiceCohort::default();
cohort.refresh(&inventory);
let ctx = CancellationToken::new();
let mut wait = Box::pin(wait_for_next_scanner_cycle(
&ctx,
config.cycle_interval,
observed.dirty_usage,
observed.runtime_config,
observed.maintenance,
|| false,
));
assert!(matches!(futures::poll!(&mut wait), Poll::Pending));
for _ in 0..59 {
tokio::time::advance(Duration::from_secs(60)).await;
cohort.refresh(&inventory);
crate::scanner_io::record_dirty_usage_bucket("hot");
assert!(
matches!(futures::poll!(&mut wait), Poll::Pending),
"aging/dirty must not shorten the explicit hour"
);
}
tokio::time::advance(Duration::from_secs(60)).await;
assert_eq!(wait.await, ScannerCycleWakeReason::Timer);
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn test_wait_for_next_scanner_cycle_sees_unattempted_dirty_usage() {
@@ -8987,7 +8807,7 @@ fn post_lease_activity_proof_rejects_a_put_tail_that_finished_before_lease_acqui
]);
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(
result,
DataUsagePersistOutcome::Deferred(reason.expect("changed namespace should defer publication")).into(),
DataUsagePersistOutcome::Deferred(reason.expect("changed namespace should defer publication")),
);
assert_eq!(
outcome,
@@ -1,148 +0,0 @@
// Copyright 2026 RustFS Team
// Licensed under the Apache License, Version 2.0.
use super::*;
use crate::storage_api::owner::ObjectOperations as _;
const BUCKET: &str = "quota-reset-preservation";
const OPERATION: &str = "00000000-0000-0000-0000-000000000002";
async fn reservation_fixture() -> (tempfile::TempDir, Arc<ECStore>, Uuid, String, Vec<u8>) {
let (directory, store) = setup_scanner_cycle_store().await;
store
.make_bucket(BUCKET, &crate::storage_api::scan::MakeBucketOptions::default())
.await
.expect("create the reservation fixture bucket through its owner");
let incarnation = store
.bucket_incarnation_id_from_disk(BUCKET)
.await
.expect("durable bucket incarnation");
assert!(!incarnation.is_nil());
let path = format!("config/quota-ledger/{BUCKET}.json");
let bytes = serde_json::to_vec(&serde_json::json!({
"version": 1,
"bucket_incarnation": incarnation,
"quota_revision_unix_nanos": 1,
"accounted_usage": 100,
"reservations": {
OPERATION: {
"object": "pending-object",
"old_size": 0,
"new_size": 64,
"created_at": 1,
"pool_index": 0,
"set_index": 0,
"commit_started": true
}
}
}))
.expect("encode the committed reservation fixture");
save_config(store.clone(), &path, bytes.clone())
.await
.expect("persist reservation bytes through the real storage owner");
(directory, store, incarnation, path, bytes)
}
async fn assert_reservation_retained(store: &Arc<ECStore>, path: &str, expected: &[u8], incarnation: Uuid) {
let bytes = read_config(store.clone(), path)
.await
.expect("read the actual reservation ledger");
assert_eq!(bytes, expected, "scanner reset must not rewrite the reservation ledger");
let ledger: serde_json::Value = serde_json::from_slice(&bytes).expect("persisted ledger JSON");
assert_eq!(ledger["version"], 1);
assert_eq!(ledger["bucket_incarnation"], incarnation.to_string());
assert_eq!(ledger["accounted_usage"], 100);
let reservations = ledger["reservations"].as_object().expect("reservation map");
assert_eq!(reservations.len(), 1);
let pending = &reservations[OPERATION];
assert_eq!(pending["old_size"], 0);
assert_eq!(pending["new_size"], 64);
assert_eq!(pending["commit_started"], true);
assert_eq!(
store
.bucket_incarnation_id_from_disk(BUCKET)
.await
.expect("owner incarnation after restart"),
incarnation
);
}
#[tokio::test]
#[serial]
async fn quota_reset_preservation_survives_storage_owner_reconstruction() {
let (_directory, store, incarnation, path, bytes) = reservation_fixture().await;
let reset = reset_scanner_usage_state_for_full_rebuild(CancellationToken::new(), store.clone())
.await
.expect("reset scanner usage through the fenced production entry");
assert_eq!(reset.usage_state, "bootstrap-pending");
let restarted = restart_scanner_cycle_store_from(&store).await;
assert!(
!Arc::ptr_eq(&store, &restarted),
"the assertion must read through a newly constructed ECStore"
);
assert_reservation_retained(&restarted, &path, &bytes, incarnation).await;
let usage = read_config(restarted.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("read reset usage through the reconstructed owner");
let usage: DataUsageInfo = serde_json::from_slice(&usage).expect("bootstrap usage JSON");
assert!(data_usage_info_is_bootstrap_pending(&usage));
assert!(!data_usage_info_has_persisted_baseline_identity(&usage));
}
#[tokio::test]
#[serial]
async fn quota_reset_preservation_unknown_protocol_rejects_put_after_restart() {
for quota_shape in ["zero", "null", "missing"] {
let (_directory, store, incarnation, path, bytes) = reservation_fixture().await;
let mut quota = serde_json::json!({
"quota_type": "Hard",
"reservation_protocol": 2,
"reservation_quota": 1024
});
match quota_shape {
"zero" => quota["quota"] = serde_json::json!(0),
"null" => quota["quota"] = serde_json::Value::Null,
"missing" => {}
_ => unreachable!("fixed quota shapes"),
}
let unknown_quota = serde_json::to_vec(&quota).expect("unknown but syntactically valid quota protocol");
store
.update_bucket_metadata_config(BUCKET, rustfs_config::QUOTA_CONFIG_FILE, unknown_quota)
.await
.expect("persist a future protocol using the real metadata owner");
assert_eq!(
store
.bucket_incarnation_id_from_disk(BUCKET)
.await
.expect("same metadata owner incarnation"),
incarnation
);
reset_scanner_usage_state_for_full_rebuild(CancellationToken::new(), store.clone())
.await
.expect("scanner reset must not change quota metadata");
let restarted = restart_scanner_cycle_store_from(&store).await;
assert!(!Arc::ptr_eq(&store, &restarted));
assert_reservation_retained(&restarted, &path, &bytes, incarnation).await;
let mut reader = PutObjReader::from_vec(b"must-not-commit".to_vec());
let result = restarted.pools[0].disk_set[0]
.put_object(BUCKET, "rejected-object", &mut reader, &ObjectOptions::default())
.await;
let error = match result {
Err(error) => error,
Ok(_) => panic!("unknown reservation protocol with quota={quota_shape} must not admit a PUT"),
};
assert!(
matches!(error, EcstoreError::PartMissingOrCorrupt),
"unexpected protocol rejection: {error}"
);
let missing = restarted.pools[0].disk_set[0]
.get_object_info(BUCKET, "rejected-object", &ObjectOptions::default())
.await
.expect_err("the rejected PUT must not create an object");
assert!(
matches!(missing, EcstoreError::FileNotFound | EcstoreError::ObjectNotFound(_, _)),
"object absence must not be confused with another storage failure: {missing}"
);
assert_reservation_retained(&restarted, &path, &bytes, incarnation).await;
}
}
@@ -1,475 +0,0 @@
// Copyright 2026 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use super::super::cycle_state::cleanup_io_fault;
use super::*;
use crate::storage_api::owner::{EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta, EcstoreRebalanceStats};
async fn seed_cleanup(store: &Arc<ECStore>, state: &str) -> ScannerCycleRecoveryMarker {
let cycle = CurrentCycle {
current: 3,
next: 42,
..Default::default()
};
save_config(
store.clone(),
DATA_USAGE_BLOOM_NAME_PATH.as_str(),
encode_scanner_cycle_state(&cycle, 7).expect("cycle encoding"),
)
.await
.expect("persist cycle");
let usage = DataUsageInfo {
scanner_epoch: Some(7),
scanner_cycle: Some(41),
..complete_usage_with_bucket_count(Some(std::time::SystemTime::UNIX_EPOCH), 0)
};
save_config(
store.clone(),
DATA_USAGE_OBJ_NAME_PATH.as_str(),
serde_json::to_vec(&usage).expect("usage encoding"),
)
.await
.expect("persist usage floor");
let marker = ScannerCycleRecoveryMarker {
schema_version: 1,
primary_revision: "previous-primary".to_string(),
generation: 41,
leader_epoch: 7,
classification: "corrupt".to_string(),
first_detected_at_unix_secs: 1,
last_attempt_at_unix_secs: 2,
retry_count: 1,
reason: "operator reset in progress".to_string(),
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
quarantine_path: DATA_USAGE_BLOOM_RECOVERY_PATH.clone(),
state: state.to_string(),
};
save_config(
store.clone(),
DATA_USAGE_BLOOM_RECOVERY_PATH.as_str(),
serde_json::to_vec(&marker).expect("marker encoding"),
)
.await
.expect("persist operator marker");
marker
}
async fn persisted_state(store: &Arc<ECStore>) -> Vec<(Option<Vec<u8>>, DataUsageCacheRevision)> {
let mut state = Vec::new();
for path in [
DATA_USAGE_BLOOM_NAME_PATH.as_str(),
DATA_USAGE_BLOOM_RECOVERY_PATH.as_str(),
DATA_USAGE_OBJ_NAME_PATH.as_str(),
] {
state.push(
read_config_with_revision(store.clone(), path)
.await
.expect("read exact metadata revision"),
);
}
state
}
async fn run_disabled_startup(ctx: CancellationToken, store: Arc<ECStore>) {
let initialized_before = crate::scanner_runtime_initialized();
let cleanup = init_scanner_with_recovery(ctx, store, false).await;
if let Some(cleanup) = cleanup {
tokio::time::timeout(Duration::from_secs(15), cleanup)
.await
.expect("finite disabled cleanup attempt")
.expect("cleanup task should not panic");
}
assert_eq!(
crate::scanner_runtime_initialized(),
initialized_before,
"disabled recovery must not start the normal scanner runtime"
);
}
async fn assert_reset_fences(store: &Arc<ECStore>) {
let data = read_config(store.clone(), DATA_USAGE_BLOOM_NAME_PATH.as_str())
.await
.expect("cycle remains durable");
let (cycle, epoch) = decode_scanner_cycle_state(&data).expect("valid preserved cycle");
assert_eq!((cycle.current, cycle.next, epoch), (3, 42, 8));
let usage: DataUsageInfo = serde_json::from_slice(
&read_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("durable usage fence"),
)
.expect("valid usage");
assert_eq!(usage.scanner_epoch, Some(8));
assert_eq!(usage.scanner_cycle, Some(41));
assert!(matches!(
read_config(store.clone(), DATA_USAGE_BLOOM_RECOVERY_PATH.as_str()).await,
Err(EcstoreError::ConfigNotFound)
));
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_reopens_persisted_intent_without_starting_scanner() {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "cleanup-pending").await;
let restarted = restart_scanner_cycle_store_from(&store).await;
run_disabled_startup(CancellationToken::new(), restarted.clone()).await;
assert_reset_fences(&restarted).await;
let completed = persisted_state(&restarted).await;
run_disabled_startup(CancellationToken::new(), restarted.clone()).await;
assert_eq!(
persisted_state(&restarted).await,
completed,
"a later startup without an intent must not reset again"
);
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_does_not_authorize_blocked_unknown_or_corrupt_markers() {
let (_dir, store) = setup_scanner_cycle_store().await;
for kind in ["blocked", "unknown-phase", "future-version", "unknown-field", "corrupt"] {
let marker = seed_cleanup(&store, "blocked").await;
let mut value = serde_json::to_value(marker).expect("marker value");
match kind {
"unknown-phase" => value["state"] = "future-phase".into(),
"future-version" => value["schema_version"] = 99.into(),
"unknown-field" => value["future_hint"] = true.into(),
_ => {}
}
let bytes = if kind == "corrupt" {
b"{broken".to_vec()
} else {
serde_json::to_vec(&value).expect("marker JSON")
};
save_config(store.clone(), DATA_USAGE_BLOOM_RECOVERY_PATH.as_str(), bytes)
.await
.expect("persist rejected marker");
let before = persisted_state(&store).await;
run_disabled_startup(CancellationToken::new(), store.clone()).await;
assert_eq!(persisted_state(&store).await, before, "{kind} must not become an automatic full rescan");
}
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_rechecks_revision_after_waiting_for_leader_lock() {
let (_dir, store) = setup_scanner_cycle_store().await;
let mut marker = seed_cleanup(&store, "cleanup-pending").await;
let expected = read_config_with_revision(store.clone(), DATA_USAGE_BLOOM_RECOVERY_PATH.as_str())
.await
.expect("intent revision")
.1;
let lock = store
.new_ns_lock(RUSTFS_META_BUCKET, "leader.lock")
.await
.expect("leader lock");
let guard = lock
.get_write_lock_quiet(Duration::from_secs(1))
.await
.expect("hold leader ownership");
let mut recovery = Box::pin(reset_scanner_cycle_recovery_for_intent(
CancellationToken::new(),
store.clone(),
Some(expected),
None,
));
assert!(matches!(futures::poll!(&mut recovery), Poll::Pending));
marker.state = "blocked".to_string();
save_config(
store.clone(),
DATA_USAGE_BLOOM_RECOVERY_PATH.as_str(),
serde_json::to_vec(&marker).expect("replacement marker"),
)
.await
.expect("replace intent while the fixture owns leader lock");
let replaced = persisted_state(&store).await;
drop(guard);
let error = recovery.await.expect_err("old preflight cannot authorize replacement marker");
assert!(error.to_string().contains("changed before recovery acquired ownership"));
assert_eq!(persisted_state(&store).await, replaced);
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_requires_phase_even_when_revision_matches() {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "blocked").await;
let before = persisted_state(&store).await;
let error = reset_scanner_cycle_recovery_for_intent(CancellationToken::new(), store.clone(), Some(before[1].1.clone()), None)
.await
.expect_err("a matching ETag alone is not operator cleanup authorization");
assert!(error.to_string().contains("unchanged cleanup-pending"));
assert_eq!(persisted_state(&store).await, before);
reset_scanner_cycle_recovery(CancellationToken::new(), store.clone())
.await
.expect("explicit v3 core retains full reset authorization");
assert_reset_fences(&store).await;
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_lock_busy_preserves_intent_without_force_unlock() {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "cleanup-pending").await;
let before = persisted_state(&store).await;
let lock = store
.new_ns_lock(RUSTFS_META_BUCKET, "leader.lock")
.await
.expect("leader lock");
let guard = lock
.get_write_lock_quiet(Duration::from_secs(1))
.await
.expect("hold live leader");
let error = resume_scanner_cycle_cleanup(CancellationToken::new(), store.clone())
.await
.expect_err("busy leader must block recovery");
assert!(error.to_string().contains("leader lock is busy"));
let status = scanner_cycle_recovery_status();
assert_eq!(status.state, "cleanup-pending");
assert!(
status
.reason
.as_deref()
.is_some_and(|reason| reason.contains("leader lock is busy"))
);
assert!(!status.retryable, "disabled startup makes one attempt, not an automatic retry loop");
assert!(!guard.is_lock_lost(), "recovery must not revoke the live owner");
assert_eq!(persisted_state(&store).await, before);
drop(guard);
run_disabled_startup(CancellationToken::new(), store.clone()).await;
assert_reset_fences(&store).await;
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_movement_pause_preserves_intent_for_later_startup() {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "cleanup-pending").await;
let before = persisted_state(&store).await;
*store.rebalance_meta.write().await = Some(EcstoreRebalanceMeta {
id: "cleanup-movement".to_string(),
pool_stats: vec![EcstoreRebalanceStats {
participating: true,
info: EcstoreRebalanceInfo {
start_time: Some(time::OffsetDateTime::now_utc()),
status: EcstoreRebalStatus::Started,
..Default::default()
},
..Default::default()
}],
..Default::default()
});
let error = resume_scanner_cycle_cleanup(CancellationToken::new(), store.clone())
.await
.expect_err("movement must block reset publication");
assert!(error.to_string().contains("blocked by data movement"));
let status = scanner_cycle_recovery_status();
assert_eq!(status.state, "cleanup-pending");
assert!(
status
.reason
.as_deref()
.is_some_and(|reason| reason.contains("blocked by data movement"))
);
assert_eq!(persisted_state(&store).await, before);
*store.rebalance_meta.write().await = None;
run_disabled_startup(CancellationToken::new(), store.clone()).await;
assert_reset_fences(&store).await;
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_cancelled_startup_preserves_persisted_work() {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "cleanup-pending").await;
let before = persisted_state(&store).await;
let ctx = CancellationToken::new();
ctx.cancel();
run_disabled_startup(ctx, store.clone()).await;
assert_eq!(persisted_state(&store).await, before);
}
#[tokio::test(start_paused = true)]
#[serial]
async fn disabled_cleanup_probe_obeys_cancellation_and_existing_io_deadline() {
for cancel in [false, true] {
let store = Arc::new(MemoryConfigStore::default());
store.delayed_gets.lock().await.insert(
memory_config_key(RUSTFS_META_BUCKET, DATA_USAGE_BLOOM_RECOVERY_PATH.as_str()),
data_usage_persist_timeout().saturating_add(Duration::from_secs(60)),
);
let ctx = CancellationToken::new();
let mut probe = Box::pin(read_scanner_cleanup_marker(store.clone(), &ctx));
assert!(matches!(futures::poll!(&mut probe), Poll::Pending));
if cancel {
ctx.cancel();
} else {
advance(data_usage_persist_timeout()).await;
}
let error = probe.await.expect_err("pending read must be bounded");
assert!(error.to_string().contains(if cancel { "cancelled" } else { "timed out" }));
assert!(store.put_counts.lock().await.is_empty(), "probe must remain read-only");
}
}
#[test]
#[serial]
fn disabled_cleanup_old_observation_cannot_overwrite_a_new_completion() {
let original = scanner_cycle_recovery_status();
let healthy = ScannerCycleRecoveryStatus {
state: "healthy".to_string(),
..Default::default()
};
publish_scanner_cleanup_status(healthy.clone(), None).expect("first status version");
let old = scanner_cleanup_status_version();
publish_scanner_cleanup_status(healthy, None).expect("a newer completion may have identical fields");
assert!(
publish_scanner_cleanup_status(
ScannerCycleRecoveryStatus {
state: "cleanup-pending".to_string(),
reason: Some("old lock wait failed".to_string()),
..Default::default()
},
Some(old)
)
.is_none()
);
assert_eq!(scanner_cycle_recovery_status().state, "healthy");
publish_scanner_cleanup_status(original, None).expect("restore prior observation");
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_owned_read_and_write_failures_keep_specific_status() {
for (stage, newer_completion) in [
(cleanup_io_fault::Stage::PrimaryRead, false),
(cleanup_io_fault::Stage::PrimaryWrite, false),
(cleanup_io_fault::Stage::PrimaryWrite, true),
] {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "cleanup-pending").await;
let before = persisted_state(&store).await;
let injection = cleanup_io_fault::install(&store, stage, newer_completion);
let error = resume_scanner_cycle_cleanup(CancellationToken::new(), store.clone())
.await
.expect_err("injected owned I/O boundary");
assert!(
injection.fired_while_owned(),
"fault must occur after real leader ownership and marker validation"
);
let expected_error = match stage {
cleanup_io_fault::Stage::PrimaryRead => "injected primary read failure",
cleanup_io_fault::Stage::PrimaryWrite => "injected primary write failure",
cleanup_io_fault::Stage::UsageFence => "injected usage fence failure",
};
assert!(error.to_string().contains(expected_error));
let status = scanner_cycle_recovery_status();
if newer_completion {
assert_eq!(status.state, "healthy", "old error cannot overwrite a newer completion observation");
assert!(status.reason.is_none());
} else {
assert_eq!(status.state, "cleanup-pending");
assert!(
status.reason.as_deref().is_some_and(|reason| reason.contains(expected_error)),
"{status:?}"
);
}
let after = persisted_state(&store).await;
assert_eq!(after[0], before[0], "failed primary I/O must preserve its prior revision");
assert_eq!(after[2], before[2], "failed primary I/O must not advance the usage fence");
let marker: ScannerCycleRecoveryMarker =
serde_json::from_slice(after[1].0.as_deref().expect("durable marker retained")).expect("valid cleanup marker");
assert_eq!(marker.state, "cleanup-pending");
drop(injection);
run_disabled_startup(CancellationToken::new(), store.clone()).await;
assert_reset_fences(&store).await;
}
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_invalidated_observation_never_becomes_unconditional() {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "cleanup-pending").await;
let revision = read_config_with_revision(store.clone(), DATA_USAGE_BLOOM_RECOVERY_PATH.as_str())
.await
.expect("marker revision")
.1;
let newer = ScannerCycleRecoveryStatus {
state: "healthy".to_string(),
reason: Some("newer completion owner".to_string()),
..Default::default()
};
publish_scanner_cleanup_status(newer.clone(), None).expect("newer observation");
let mut invalidated = None;
reset_scanner_cycle_recovery_for_intent(CancellationToken::new(), store.clone(), Some(revision), Some(&mut invalidated))
.await
.expect("metadata cleanup may complete without owning the newest status observation");
assert!(invalidated.is_none());
assert_eq!(
serde_json::to_value(scanner_cycle_recovery_status()).expect("observed status"),
serde_json::to_value(newer).expect("newer status")
);
assert_reset_fences(&store).await;
}
#[tokio::test]
#[serial]
async fn disabled_cleanup_later_failure_preserves_rebuilt_primary_status_identity() {
for newer_completion in [false, true] {
let (_dir, store) = setup_scanner_cycle_store().await;
seed_cleanup(&store, "cleanup-pending").await;
save_config(store.clone(), DATA_USAGE_BLOOM_NAME_PATH.as_str(), b"corrupt-cycle".to_vec())
.await
.expect("force the full reconstruction branch");
let before = persisted_state(&store).await;
let injection = cleanup_io_fault::install(&store, cleanup_io_fault::Stage::UsageFence, newer_completion);
let error = resume_scanner_cycle_cleanup(CancellationToken::new(), store.clone())
.await
.expect_err("fail after primary publication");
assert!(injection.fired_while_owned());
assert!(error.to_string().contains("injected usage fence failure"));
let after = persisted_state(&store).await;
assert_ne!(after[0].1, before[0].1, "the primary write must actually commit before this failure");
let (cycle, epoch) =
decode_scanner_cycle_state(after[0].0.as_deref().expect("rebuilt primary")).expect("valid durable reconstruction");
assert_eq!((cycle.current, cycle.next, epoch), (0, 42, 8));
assert_eq!(after[2], before[2], "usage fence publication was rejected");
let marker: ScannerCycleRecoveryMarker =
serde_json::from_slice(after[1].0.as_deref().expect("cleanup marker retained")).expect("valid cleanup marker");
assert_eq!(marker.state, "cleanup-pending");
let status = scanner_cycle_recovery_status();
if newer_completion {
assert_eq!(status.state, "healthy");
assert!(
status.reason.is_none(),
"old core and outer failure must both retain invalidated ownership"
);
} else {
let DataUsageCacheRevision::Etag(etag) = &after[0].1 else {
panic!("rebuilt primary must have a revision");
};
assert_eq!(status.state, "cleanup-pending");
assert_eq!(status.primary_revision.as_deref(), Some(etag.as_str()));
assert_eq!(status.generation, Some(cycle.next));
assert_eq!(status.leader_epoch, Some(epoch));
assert!(
status
.reason
.as_deref()
.is_some_and(|reason| reason.contains("injected usage fence failure"))
);
}
}
}
@@ -1,490 +0,0 @@
// Copyright 2026 RustFS Team
// Licensed under the Apache License, Version 2.0.
use super::super::usage_store::DataUsagePublicationResult;
use super::*;
use crate::scanner_io::ScannerBucketScanScope;
use rustfs_utils::path::path_join_buf;
use sha2::Digest;
use std::time::SystemTime;
const PROOF_BUCKET: &str = "publication-proof-bucket";
const PROOF_EPOCH: u64 = 7;
const PROOF_CYCLE: u64 = 11;
async fn settle_namespace_commits(store: &ECStore) {
tokio::time::timeout(Duration::from_secs(30), async {
while store.scanner_data_usage_publication_blocked().await {
tokio::time::sleep(Duration::from_millis(1)).await;
}
})
.await
.expect("fixture namespace commits must settle before collecting complete coverage");
}
async fn complete_candidate(store: &Arc<ECStore>, cycle: u64) -> (crate::scanner_io::ScannerCycleResult, DataUsageInfo) {
settle_namespace_commits(store).await;
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new_with_progress_tracking(
&ctx,
ScannerCycleBudgetConfig {
max_objects: Some(8),
..Default::default()
},
);
let (updates, mut receiver) = mpsc::channel(1);
let result = crate::scanner_io::nsscanner_with_storage_status_scoped(
store.as_ref(),
crate::scanner_io::ScannerCycleRequest {
ctx,
budget,
updates,
want_cycle: cycle,
leader_epoch: PROOF_EPOCH,
scan_mode: HealScanMode::Normal,
scan_scope: ScannerBucketScanScope::default(),
persisted_usage_baseline: None,
observed_usage_candidate: None,
requires_full_scan: true,
service_cohort: None,
resolved_scope_observer: None,
},
)
.await
.expect("real scanner must produce the fixture candidate");
assert_eq!(result.status, ScannerCycleStatus::Complete);
let candidate = receiver.recv().await.expect("complete scanner snapshot");
assert!(candidate.usage_snapshot_complete);
assert_eq!(candidate.usage_snapshot_converged, Some(true));
assert_eq!(candidate.scanner_cycle, Some(cycle));
assert_eq!(candidate.scanner_epoch, Some(PROOF_EPOCH));
(result, candidate)
}
async fn candidate_store() -> (tempfile::TempDir, Arc<ECStore>) {
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
let (directory, store) = setup_scanner_cycle_store_with_usage_baseline(false).await;
store
.make_bucket(PROOF_BUCKET, &crate::storage_api::scan::MakeBucketOptions::default())
.await
.expect("create proof fixture bucket through the owner");
let mut reader = PutObjReader::from_vec(b"proof".to_vec());
store.pools[0].disk_set[0]
.put_object(PROOF_BUCKET, "initial", &mut reader, &ObjectOptions::default())
.await
.expect("persist fixture object through the owner");
crate::scanner_io::record_dirty_usage_bucket(PROOF_BUCKET);
settle_namespace_commits(&store).await;
(directory, store)
}
async fn read_root(store: &Arc<ECStore>) -> (Option<Vec<u8>>, DataUsageCacheRevision) {
read_config_with_revision(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("read actual v2 root bytes and revision")
}
async fn publish_candidate(
store: &Arc<ECStore>,
scan: &crate::scanner_io::ScannerCycleResult,
candidate: DataUsageInfo,
baseline: Option<DataUsagePersistBaseline>,
) -> DataUsagePublicationResult {
let expectation = scan.publication_expectation();
assert!(expectation.is_some(), "only a real complete scan may supply the expectation");
let (sender, receiver) = mpsc::channel(1);
sender.send(candidate).await.expect("enqueue the real scan candidate");
drop(sender);
store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe_for_publication_epoch_and_lease_fence(
CancellationToken::new(),
store.clone(),
receiver,
Some(PROOF_EPOCH),
baseline,
ScannerPublicationFence::new(scan.publication_epoch(), None, None).with_ack_expectation(expectation),
|| async { None },
)
.await
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_companion_only_does_not_authorize_root_ack() {
for companion in [
format!("{}.bkp", DATA_USAGE_OBJ_NAME_PATH.as_str()),
LEGACY_DATA_USAGE_OBJ_NAME_PATH.to_string(),
format!("{}.bkp", LEGACY_DATA_USAGE_OBJ_NAME_PATH.as_str()),
] {
let (_directory, store) = candidate_store().await;
let (scan, candidate) = complete_candidate(&store, PROOF_CYCLE).await;
let bytes = serde_json::to_vec(&candidate).expect("actual candidate JSON");
save_config(store.clone(), &companion, bytes.clone())
.await
.expect("persist the companion on real disks");
let baseline = read_data_usage_persist_baseline(store.clone())
.await
.expect("companion fallback baseline");
assert_eq!(baseline.data.as_deref(), Some(bytes.as_slice()));
assert_eq!(baseline.revision, DataUsageCacheRevision::Missing);
assert_eq!(read_root(&store).await.0, None);
let dirty = crate::scanner_io::dirty_usage_buckets_for_tests();
let publication = publish_candidate(&store, &scan, candidate, Some(baseline)).await;
assert_eq!(publication.outcome(), DataUsagePersistOutcome::AlreadyDurable);
let (_, pending, acknowledgements) = finalize_scanner_cycle_result(scan, publication);
assert!(pending, "unacknowledged durable companion work must remain pending");
assert!(acknowledgements.is_empty());
assert_eq!(
crate::scanner_io::dirty_usage_buckets_for_tests(),
dirty,
"a companion is not the v2 root target"
);
assert_eq!(read_root(&store).await, (None, DataUsageCacheRevision::Missing));
assert_eq!(read_config(store.clone(), &companion).await.expect("companion retained"), bytes);
}
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_actual_root_readback_accepts_semantic_json_equivalence() {
let (_directory, store) = candidate_store().await;
let (scan, candidate) = complete_candidate(&store, PROOF_CYCLE).await;
let canonical = serde_json::to_vec(&candidate).expect("candidate encoding");
let mut value = serde_json::to_value(&candidate).expect("candidate value");
value
.as_object_mut()
.expect("usage object")
.insert("fixture_unknown_field".into(), serde_json::json!({"retained": true}));
let different_bytes = serde_json::to_vec_pretty(&value).expect("noncanonical primary JSON");
assert_ne!(different_bytes, canonical);
assert_eq!(
serde_json::from_slice::<DataUsageInfo>(&different_bytes).expect("semantic primary"),
candidate
);
save_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str(), different_bytes.clone())
.await
.expect("persist actual primary representation");
let before = read_root(&store).await;
assert!(matches!(&before.1, DataUsageCacheRevision::Etag(etag) if !etag.is_empty()));
let baseline = read_data_usage_persist_baseline(store.clone())
.await
.expect("real primary revision");
assert!(crate::scanner_io::dirty_usage_buckets_pending());
let publication = publish_candidate(&store, &scan, candidate.clone(), Some(baseline)).await;
assert_eq!(publication.outcome(), DataUsagePersistOutcome::AlreadyDurable);
let (_, proof) = publication.into_parts();
let proof = proof.expect("actual primary readback must produce its own root proof");
let expected = scan.publication_expectation().expect("real scan expectation");
let (etag, raw_digest) = proof.verified_version_for(&expected).expect("proof must bind this candidate");
let DataUsageCacheRevision::Etag(expected_etag) = &before.1 else { panic!("actual root ETag") };
assert_eq!(etag, expected_etag);
let expected_digest: [u8; 32] = sha2::Sha256::digest(&different_bytes).into();
assert_eq!(
*raw_digest, expected_digest,
"proof must record actual bytes, not reserialized candidate bytes"
);
// Obtain another proof through the same real readback path rather than
// fabricating a publication result from the inspected proof above.
let publication = publish_candidate(&store, &scan, candidate, None).await;
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(scan, publication);
assert_eq!(outcome, ScannerCycleOutcome::Completed);
assert!(acknowledgements.is_empty(), "the single-node fixture has no remote targets");
assert!(
!crate::scanner_io::dirty_usage_buckets_pending(),
"actual root bytes plus a real revision authorize this scan"
);
assert_eq!(read_root(&store).await, before, "readback must not rewrite unknown fields or whitespace");
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_successful_root_cas_authorizes_its_scan() {
let (_directory, store) = candidate_store().await;
let (scan, candidate) = complete_candidate(&store, PROOF_CYCLE).await;
let baseline = read_data_usage_persist_baseline(store.clone())
.await
.expect("initial root revision");
assert_eq!(baseline.revision, DataUsageCacheRevision::Missing);
assert!(crate::scanner_io::dirty_usage_buckets_pending());
let publication = publish_candidate(&store, &scan, candidate.clone(), Some(baseline)).await;
assert_eq!(publication.outcome(), DataUsagePersistOutcome::Saved);
let (bytes, revision) = read_root(&store).await;
assert!(matches!(revision, DataUsageCacheRevision::Etag(etag) if !etag.is_empty()));
assert_eq!(
serde_json::from_slice::<DataUsageInfo>(&bytes.expect("actual saved root")).expect("root JSON"),
candidate
);
let (outcome, pending, acknowledgements) = finalize_scanner_cycle_result(scan, publication);
assert_eq!(outcome, ScannerCycleOutcome::Completed);
assert!(!pending);
assert!(acknowledgements.is_empty(), "the single-node fixture has no remote targets");
assert!(
!crate::scanner_io::dirty_usage_buckets_pending(),
"the real root CAS must authorize its matching scan"
);
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_observed_candidate_reuse_requires_a_new_root_proof() {
let (_directory, store) = candidate_store().await;
let bootstrap = scanner_usage_bootstrap_marker(SystemTime::now(), Some(PROOF_EPOCH));
save_config(
store.clone(),
DATA_USAGE_OBJ_NAME_PATH.as_str(),
serde_json::to_vec(&bootstrap).expect("bootstrap root encoding"),
)
.await
.expect("persist authoritative bootstrap root");
let (prior_scan, mut observed_candidate) = complete_candidate(&store, PROOF_CYCLE).await;
// Seed a complete but unconverged observation from real scanner coverage;
// the production writer attaches its authoritative baseline identity.
observed_candidate.usage_snapshot_converged = Some(false);
let observation = publish_candidate(&store, &prior_scan, observed_candidate, None).await;
let (outcome, proof) = observation.into_parts();
assert_eq!(outcome, DataUsagePersistOutcome::Saved);
assert!(proof.is_none(), "an observational write cannot authorize a root ACK");
let (root_before, revision_before) = read_root(&store).await;
let observed = read_config(store.clone(), DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str())
.await
.expect("read real persisted observation");
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default());
let (updates, mut receiver) = mpsc::channel(1);
let (observer, selected) = tokio::sync::oneshot::channel();
let scan = crate::scanner_io::nsscanner_with_storage_status_scoped(
store.as_ref(),
crate::scanner_io::ScannerCycleRequest {
ctx,
budget,
updates,
want_cycle: PROOF_CYCLE + 1,
leader_epoch: PROOF_EPOCH,
scan_mode: HealScanMode::Normal,
scan_scope: ScannerBucketScanScope::default(),
persisted_usage_baseline: root_before.clone().map(Bytes::from),
observed_usage_candidate: Some(Bytes::from(observed)),
requires_full_scan: false,
service_cohort: None,
resolved_scope_observer: Some(observer),
},
)
.await
.expect("observation-backed scope must run through the real scanner");
let scope = selected.await.expect("production resolver decision");
assert_eq!(scope.selected_buckets_for_tests(), Some(&HashSet::from([PROOF_BUCKET.to_string()])));
assert_eq!(scan.status, ScannerCycleStatus::Complete);
let expectation = scan.publication_expectation().expect("reused coverage must be revalidated");
assert!(
!expectation.same_candidate(&prior_scan.publication_expectation().expect("prior real candidate")),
"the observation cannot transfer the previous scan's expectation"
);
assert_eq!(read_root(&store).await, (root_before, revision_before));
assert!(crate::scanner_io::dirty_usage_buckets_pending());
let candidate = receiver.recv().await.expect("new validated root candidate");
assert_eq!(candidate.scanner_cycle, Some(PROOF_CYCLE + 1));
assert_eq!(candidate.usage_snapshot_converged, Some(true));
let publication = publish_candidate(&store, &scan, candidate, None).await;
assert_eq!(publication.outcome(), DataUsagePersistOutcome::Saved);
let (outcome, pending, acknowledgements) = finalize_scanner_cycle_result(scan, publication);
assert_eq!(outcome, ScannerCycleOutcome::Completed);
assert!(!pending);
assert!(acknowledgements.is_empty());
assert!(!crate::scanner_io::dirty_usage_buckets_pending());
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_stale_root_cas_keeps_dirty_after_bucket_save() {
let (_directory, store) = candidate_store().await;
let (scan, candidate) = complete_candidate(&store, PROOF_CYCLE).await;
let mut bucket_cache = DataUsageCache::default();
bucket_cache
.load(store.pools[0].disk_set[0].clone(), &path_join_buf(&[PROOF_BUCKET, DATA_USAGE_CACHE_NAME]))
.await
.expect("real bucket checkpoint must be persisted before root publication");
assert!(bucket_cache.info.snapshot_complete);
assert_eq!(
bucket_cache
.checked_flatten(PROOF_BUCKET)
.expect("persisted bucket root")
.objects,
1
);
let stale_baseline = read_data_usage_persist_baseline(store.clone())
.await
.expect("missing root revision");
assert_eq!(stale_baseline.revision, DataUsageCacheRevision::Missing);
let mut competing = candidate.clone();
competing.scanner_epoch = Some(PROOF_EPOCH + 1);
competing.scanner_cycle = Some(PROOF_CYCLE + 1);
for state in &mut competing.usage_snapshot_set_states {
state.scanner_epoch = Some(PROOF_EPOCH + 1);
state.scanner_cycle = Some(PROOF_CYCLE + 1);
}
let competing_bytes = serde_json::to_vec(&competing).expect("competing root");
save_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str(), competing_bytes.clone())
.await
.expect("another publisher wins the actual root slot");
let before = read_root(&store).await;
let dirty = crate::scanner_io::dirty_usage_buckets_for_tests();
let publication = publish_candidate(&store, &scan, candidate, Some(stale_baseline)).await;
assert_eq!(
publication.outcome(),
DataUsagePersistOutcome::Current,
"the old missing revision loses CAS and reconciles the newer root"
);
let (_, _, acknowledgements) = finalize_scanner_cycle_result(scan, publication);
assert!(acknowledgements.is_empty());
assert_eq!(crate::scanner_io::dirty_usage_buckets_for_tests(), dirty);
assert_eq!(
read_root(&store).await,
before,
"bucket durability must not authorize replacing the winning root"
);
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_cannot_transfer_proof_between_real_scan_results() {
let (_directory, store) = candidate_store().await;
let (first_scan, first_candidate) = complete_candidate(&store, PROOF_CYCLE).await;
let (second_scan, second_candidate) = complete_candidate(&store, PROOF_CYCLE).await;
assert_eq!(first_candidate.scanner_epoch, second_candidate.scanner_epoch);
assert_eq!(first_candidate.scanner_cycle, second_candidate.scanner_cycle);
assert_eq!(first_candidate.objects_total_count, second_candidate.objects_total_count);
let baseline = read_data_usage_persist_baseline(store.clone())
.await
.expect("initial root revision");
let dirty = crate::scanner_io::dirty_usage_buckets_for_tests();
let publication = publish_candidate(&store, &first_scan, first_candidate, Some(baseline)).await;
assert_eq!(publication.outcome(), DataUsagePersistOutcome::Saved);
assert!(read_root(&store).await.0.is_some(), "the first scan really published its root");
let (_, pending, acknowledgements) = finalize_scanner_cycle_result(second_scan, publication);
assert!(pending, "another scan's publication must not finish this scan's dirty maintenance work");
assert!(acknowledgements.is_empty());
assert_eq!(
crate::scanner_io::dirty_usage_buckets_for_tests(),
dirty,
"same counters and cycle cannot transfer another scan's proof"
);
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_stale_baseline_cannot_prove_a_replaced_root() {
let (_directory, store) = candidate_store().await;
let (first_scan, first_candidate) = complete_candidate(&store, PROOF_CYCLE).await;
save_config(
store.clone(),
DATA_USAGE_OBJ_NAME_PATH.as_str(),
serde_json::to_vec(&first_candidate).expect("first candidate"),
)
.await
.expect("persist the first candidate on real disks");
let stale_baseline = read_data_usage_persist_baseline(store.clone())
.await
.expect("capture the genuine first root revision");
let dirty = crate::scanner_io::dirty_usage_buckets_for_tests();
let mut reader = PutObjReader::from_vec(b"second".to_vec());
store.pools[0].disk_set[0]
.put_object(PROOF_BUCKET, "second", &mut reader, &ObjectOptions::default())
.await
.expect("commit a real namespace change");
assert_eq!(
crate::scanner_io::dirty_usage_buckets_for_tests(),
dirty,
"direct storage writes leave this fixture's scanner hint generation unchanged"
);
let (_, replacement) = complete_candidate(&store, PROOF_CYCLE).await;
assert_eq!(first_candidate.scanner_epoch, replacement.scanner_epoch);
assert_eq!(first_candidate.scanner_cycle, replacement.scanner_cycle);
assert_eq!((first_candidate.objects_total_count, replacement.objects_total_count), (1, 2));
save_config(
store.clone(),
DATA_USAGE_OBJ_NAME_PATH.as_str(),
serde_json::to_vec(&replacement).expect("replacement candidate"),
)
.await
.expect("publish the replacement root");
let current = read_root(&store).await;
assert_ne!(current.1, stale_baseline.revision);
// The supplied baseline still equals candidate A, but the actual target
// now contains B. Compatibility's AlreadyDurable outcome is not proof.
let publication = publish_candidate(&store, &first_scan, first_candidate, Some(stale_baseline)).await;
assert_eq!(publication.outcome(), DataUsagePersistOutcome::AlreadyDurable);
let (_, pending, acknowledgements) = finalize_scanner_cycle_result(first_scan, publication);
assert!(pending);
assert!(acknowledgements.is_empty());
assert_eq!(crate::scanner_io::dirty_usage_buckets_for_tests(), dirty);
assert_eq!(read_root(&store).await, current);
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn scoped_ack_publication_rejects_builder_mutation_after_real_root_publish() {
for mutation in ["remote_ack_target", "publication_epoch", "remote_lease_targets"] {
let (_directory, store) = candidate_store().await;
let (scan, candidate) = complete_candidate(&store, PROOF_CYCLE).await;
let baseline = read_data_usage_persist_baseline(store.clone())
.await
.expect("initial root revision");
let dirty = crate::scanner_io::dirty_usage_buckets_for_tests();
let changed_generation = dirty
.get(PROOF_BUCKET)
.expect("the real scan has dirty work")
.checked_add(1)
.expect("bounded fixture generation");
let changed_epoch = scan
.publication_epoch()
.expect("real scan publication epoch")
.checked_add(1)
.expect("bounded fixture epoch");
let publication = publish_candidate(&store, &scan, candidate.clone(), Some(baseline)).await;
assert_eq!(publication.outcome(), DataUsagePersistOutcome::Saved, "{mutation}");
let root_before = read_root(&store).await;
assert_eq!(
serde_json::from_slice::<DataUsageInfo>(root_before.0.as_deref().expect("actual saved root"))
.expect("persisted root JSON"),
candidate,
"{mutation}: the original candidate really reached root storage"
);
let changed = match mutation {
"remote_ack_target" => scan.with_remote_dirty_usage_acknowledgements(vec![ScannerDirtyUsageAcknowledgement {
host: "proof-peer:9000".to_string(),
instance_id: crate::scanner_activity_epoch().to_string(),
generation: changed_generation,
}]),
"publication_epoch" => scan.with_publication_epoch(Some(changed_epoch)),
"remote_lease_targets" => scan.with_remote_publication_lease_targets(vec![(
"proof-peer:9000".to_string(),
crate::scanner_activity_epoch().to_string(),
changed_generation,
)]),
_ => unreachable!("fixed mutation cases"),
};
let (_, pending, acknowledgements) = finalize_scanner_cycle_result(changed, publication);
assert!(
acknowledgements.is_empty(),
"{mutation}: the old root proof must not authorize changed ACK work"
);
assert!(pending, "{mutation}: changed maintenance work must remain pending");
assert_eq!(
crate::scanner_io::dirty_usage_buckets_for_tests(),
dirty,
"{mutation}: the changed scan must not clear local dirty work"
);
assert_eq!(read_root(&store).await, root_before, "{mutation}: the durable original root is retained");
}
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
+15 -192
View File
@@ -34,139 +34,6 @@ pub(super) enum DataUsagePersistOutcome {
Failed,
}
#[derive(Debug)]
pub(crate) struct RootPublicationProof {
candidate: crate::scanner_io::ScannerPublicationExpectation,
root_version: (String, [u8; 32]),
}
impl RootPublicationProof {
pub(crate) fn verified_version_for(
&self,
expected: &crate::scanner_io::ScannerPublicationExpectation,
) -> Option<&(String, [u8; 32])> {
self.candidate.same_candidate(expected).then_some(&self.root_version)
}
}
#[derive(Debug)]
pub(super) struct DataUsagePublicationResult {
outcome: DataUsagePersistOutcome,
proof: Option<RootPublicationProof>,
}
impl From<DataUsagePersistOutcome> for DataUsagePublicationResult {
fn from(outcome: DataUsagePersistOutcome) -> Self {
Self { outcome, proof: None }
}
}
impl DataUsagePublicationResult {
pub(super) fn outcome(&self) -> DataUsagePersistOutcome {
self.outcome
}
pub(super) fn restrict_outcome(&mut self, outcome: DataUsagePersistOutcome) {
if outcome != self.outcome {
self.proof = None;
}
self.outcome = outcome;
}
pub(super) fn into_parts(self) -> (DataUsagePersistOutcome, Option<RootPublicationProof>) {
(self.outcome, self.proof)
}
}
fn root_ack_write_is_confirmed<T, E>(
result: &std::result::Result<T, E>,
state: Option<ScannerPublicationCommitState>,
written_etag: Option<&str>,
) -> bool {
result.is_ok() && state == Some(ScannerPublicationCommitState::Committed) && written_etag.is_some_and(|etag| !etag.is_empty())
}
#[cfg(test)]
mod root_publication_confirmation_tests {
use super::*;
#[test]
fn root_publication_confirmation_requires_committed_state_and_write_revision() {
let saved = Ok::<(), ()>(());
for state in [
None,
Some(ScannerPublicationCommitState::Admitted),
Some(ScannerPublicationCommitState::InFlight),
Some(ScannerPublicationCommitState::AbortedBeforeCommit),
Some(ScannerPublicationCommitState::Indeterminate),
] {
assert!(!root_ack_write_is_confirmed(&saved, state, Some("revision")));
}
for etag in [None, Some("")] {
assert!(!root_ack_write_is_confirmed(&saved, Some(ScannerPublicationCommitState::Committed), etag));
}
assert!(root_ack_write_is_confirmed(
&saved,
Some(ScannerPublicationCommitState::Committed),
Some("revision")
));
}
#[test]
fn root_publication_confirmation_does_not_carry_state_across_cas_attempts() {
let attempts = [
(Err(()), Some(ScannerPublicationCommitState::Committed), Some("first")),
(Ok(()), Some(ScannerPublicationCommitState::AbortedBeforeCommit), Some("second")),
(Ok(()), None, Some("legacy")),
(Ok(()), Some(ScannerPublicationCommitState::Committed), Some("confirmed")),
];
let confirmations = attempts
.iter()
.map(|(result, state, etag)| root_ack_write_is_confirmed(result, *state, *etag))
.collect::<Vec<_>>();
assert_eq!(confirmations, [false, false, false, true]);
}
}
async fn read_root_publication_proof<S: ScannerObjectIO + ScannerConfigObjectDelete>(
store: Arc<S>,
ctx: &CancellationToken,
deadline: tokio::time::Instant,
epoch: u64,
expected: &crate::scanner_io::ScannerPublicationExpectation,
candidate: &DataUsageInfo,
written_etag: Option<&str>,
) -> Option<RootPublicationProof> {
let read = async {
let _admission = scanner_publication_admission_for_epoch(store.clone(), epoch).await?;
let (bytes, revision) = read_config_with_revision(store, DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.ok()?;
let bytes = bytes?;
let DataUsageCacheRevision::Etag(etag) = revision else {
return None;
};
if etag.is_empty() || written_etag.is_some_and(|written| written != etag) {
return None;
}
let persisted: DataUsageInfo = serde_json::from_slice(&bytes).ok()?;
if &persisted != candidate {
return None;
}
let root_digest = Sha256::digest(&bytes).into();
if ctx.is_cancelled() || tokio::time::Instant::now() >= deadline {
return None;
}
Some(RootPublicationProof {
candidate: expected.clone(),
root_version: (etag, root_digest),
})
};
tokio::select! {
biased;
_ = ctx.cancelled() => None,
result = tokio::time::timeout_at(deadline, read) => result.ok().flatten(),
}
}
fn remote_lease_expired(deadline: Option<std::time::Instant>) -> bool {
deadline.is_some_and(|deadline| std::time::Instant::now() >= deadline)
}
@@ -299,7 +166,6 @@ pub(super) struct ScannerPublicationFence {
pub(super) scanner_publication_lease_fence: Option<String>,
pub(super) remote_lease_tokens: Vec<Uuid>,
pub(super) lease_release_safe: Arc<AtomicBool>,
pub(super) ack_expectation: Option<crate::scanner_io::ScannerPublicationExpectation>,
}
impl ScannerPublicationFence {
@@ -314,7 +180,6 @@ impl ScannerPublicationFence {
scanner_publication_lease_fence,
remote_lease_tokens: Vec::new(),
lease_release_safe: Arc::new(AtomicBool::new(true)),
ack_expectation: None,
}
}
@@ -327,26 +192,21 @@ impl ScannerPublicationFence {
self.lease_release_safe = lease_release_safe;
self
}
pub(super) fn with_ack_expectation(mut self, expected: Option<crate::scanner_io::ScannerPublicationExpectation>) -> Self {
self.ack_expectation = expected;
self
}
}
#[derive(Debug)]
pub(super) enum DataUsagePersistTaskResult<T = DataUsagePersistOutcome> {
Completed(T),
pub(super) enum DataUsagePersistTaskResult {
Completed(DataUsagePersistOutcome),
Cancelled,
TimedOut,
JoinFailed(tokio::task::JoinError),
}
pub(super) async fn wait_for_data_usage_persist_task<T>(
pub(super) async fn wait_for_data_usage_persist_task(
ctx: &CancellationToken,
task: &mut AbortOnDropHandle<T>,
task: &mut AbortOnDropHandle<DataUsagePersistOutcome>,
timeout: Duration,
) -> DataUsagePersistTaskResult<T> {
) -> DataUsagePersistTaskResult {
tokio::select! {
biased;
result = &mut *task => match result {
@@ -460,7 +320,6 @@ where
route_probe,
)
.await
.outcome()
}
pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe_for_publication_epoch_and_lease_fence<
@@ -474,7 +333,7 @@ pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_basel
initial_baseline: Option<DataUsagePersistBaseline>,
publication_fence: ScannerPublicationFence,
route_probe: F,
) -> DataUsagePublicationResult
) -> DataUsagePersistOutcome
where
F: Fn() -> Fut + Send + Sync,
Fut: Future<Output = Option<ScannerCycleDeferReason>> + Send,
@@ -485,15 +344,11 @@ where
scanner_publication_lease_fence,
remote_lease_tokens,
lease_release_safe,
ack_expectation,
} = publication_fence;
let ack_deadline = scanner_publication_scope_deadline(data_usage_persist_timeout(), remote_lease_deadline);
let mut outcome = DataUsagePersistOutcome::NoUpdate;
let mut proof = None;
let mut next_baseline = initial_baseline;
'updates: while let Some(mut data_usage_info) = receiver.recv().await {
proof = None;
let _activity_guard = ScannerActivityGuard::new();
if ctx.is_cancelled() {
break;
@@ -668,14 +523,10 @@ where
continue;
}
};
let data_digest: [u8; 32] = Sha256::digest(&data).into();
let sha256hex = (!data.is_empty()).then(|| hex_simd::encode_to_string(data_digest, hex_simd::AsciiCase::Lower));
let sha256hex = (!data.is_empty()).then(|| hex_simd::encode_to_string(Sha256::digest(&data), hex_simd::AsciiCase::Lower));
let data = Bytes::from(data);
let backup_due = !observational && data_usage_backup_due(&data_usage_info);
let mut cas_retry = 0usize;
let mut ack_epoch = None;
let mut write_confirmed = false;
let mut written_etag = None;
let save_outcome = loop {
if ctx.is_cancelled() {
break 'updates;
@@ -706,7 +557,6 @@ where
} else {
None
};
ack_epoch = Some(publication_epoch_for_save);
let (existing_data, revision) = match baseline {
Some(baseline) => (baseline.data, baseline.revision),
None => match read_config_with_revision(storeapi.clone(), target_path).await {
@@ -795,7 +645,7 @@ where
}
let done_save = Metrics::time(Metric::SaveUsage);
let (save_result, commit_state) = {
let save_result = {
let publication_scope = storeapi
.scanner_data_usage_publication_commit_scope_with_release_flag(
publication_epoch_for_save,
@@ -831,33 +681,24 @@ where
.await;
drop(legacy_publication_admission);
if let Some(scope) = publication_scope {
let state = scope.wait_for_completion().await;
let result = match state {
ScannerPublicationCommitState::Committed => save_result,
ScannerPublicationCommitState::AbortedBeforeCommit => save_result,
match scope.wait_for_completion().await {
ScannerPublicationCommitState::Committed | ScannerPublicationCommitState::AbortedBeforeCommit => {
save_result
}
ScannerPublicationCommitState::Indeterminate
| ScannerPublicationCommitState::Admitted
| ScannerPublicationCommitState::InFlight => Err(EcstoreError::other(
"scanner publication commit scope did not reach a safe terminal state",
)),
};
(result, Some(state))
}
} else {
(save_result, None)
save_result
}
};
done_save();
let attempt_confirmed = root_ack_write_is_confirmed(
&save_result,
commit_state,
save_result.as_ref().ok().and_then(|info| info.etag.as_deref()),
);
match save_result {
Ok(object_info) => {
write_confirmed = attempt_confirmed;
written_etag = object_info.etag.as_ref().filter(|etag| !etag.is_empty()).cloned();
if !observational {
next_baseline = object_info
.etag
@@ -1068,27 +909,9 @@ where
break 'updates;
}
}
if !observational
&& data_usage_info.usage_snapshot_converged == Some(true)
&& matches!(outcome, DataUsagePersistOutcome::Saved | DataUsagePersistOutcome::AlreadyDurable)
&& (outcome == DataUsagePersistOutcome::AlreadyDurable || write_confirmed)
&& let (Some(expected), Some(epoch)) = (ack_expectation.as_ref(), ack_epoch)
&& expected.matches_encoded_candidate(&data_digest)
{
proof = read_root_publication_proof(
storeapi.clone(),
&ctx,
ack_deadline,
epoch,
expected,
&data_usage_info,
written_etag.as_deref(),
)
.await;
}
}
DataUsagePublicationResult { outcome, proof }
outcome
}
async fn cleanup_observed_data_usage_snapshot_for_epoch_and_lease(
+2 -15
View File
@@ -710,10 +710,6 @@ pub struct FolderScanner {
coverage_frontier: Option<String>,
resume_frontier: Option<String>,
coverage_gap: bool,
pending_heal_sync_deferred: bool,
pending_heal_batch_dirty: bool,
#[cfg(test)]
pending_heal_sync_count: usize,
pending_size_reconciliation_keys: HashSet<String>,
pending_size_reconciliation_scopes: HashSet<String>,
pending_size_reconciliation_truncated: bool,
@@ -1084,7 +1080,7 @@ impl FolderScanner {
scan_mode,
result,
);
if result.is_admitted() || matches!(priority, HealChannelPriority::Low) {
if result.is_admitted() {
return Ok(result);
}
@@ -1832,12 +1828,7 @@ impl FolderScanner {
}
}
FolderScanSource::Existing => {
// Usage sampling is not proof that a Deep check ran.
if self.scan_mode != HealScanMode::Deep
&& !forward_sweep
&& !into.compacted
&& self.old_cache.is_compacted(&h)
{
if !forward_sweep && !into.compacted && self.old_cache.is_compacted(&h) {
let next_cycle = self.old_cache.info.next_cycle as u32;
if !h.mod_(next_cycle, data_usage_update_dir_cycles()) {
// Transfer and add as child...
@@ -2444,10 +2435,6 @@ pub async fn scan_data_folder(
coverage_frontier: resume_frontier.clone(),
resume_frontier,
coverage_gap: false,
pending_heal_sync_deferred: false,
pending_heal_batch_dirty: false,
#[cfg(test)]
pending_heal_sync_count: 0,
pending_size_reconciliation_keys: HashSet::new(),
pending_size_reconciliation_scopes: HashSet::new(),
pending_size_reconciliation_truncated: false,
+83 -135
View File
@@ -11,58 +11,13 @@
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
/// Persisted best-effort retry hints. Existing TTL/count pruning applies only
/// to this hint cache, never to a committed durable repair obligation.
/// The pending-scanner-heal ledger: durable heal intents recorded during scans and retried after MRF consumption.
use super::*;
const PENDING_HEAL_RETRY_BASE_SECS: u64 = 15 * 60;
const PENDING_HEAL_RETRY_CAP_SECS: u64 = 6 * 60 * 60;
pub(super) struct PendingHealSyncBatch<'a> {
pub(super) scanner: &'a mut FolderScanner,
}
impl<'a> PendingHealSyncBatch<'a> {
pub(super) fn new(scanner: &'a mut FolderScanner) -> Self {
scanner.pending_heal_sync_deferred = true;
scanner.pending_heal_batch_dirty = false;
Self { scanner }
}
}
impl Drop for PendingHealSyncBatch<'_> {
fn drop(&mut self) {
self.scanner.pending_heal_sync_deferred = false;
if self.scanner.pending_heal_batch_dirty {
self.scanner.pending_heal_batch_dirty = false;
self.scanner.sync_pending_heals();
}
}
}
pub(super) fn record_pending_heal_attempt(entry: &mut PendingScannerHeal, now: u64) {
entry.last_attempt = now;
entry.attempts = entry.attempts.saturating_add(1);
}
pub(super) fn observe_pending_heal_admission(entry: &mut PendingScannerHeal, result: HealAdmissionResult) {
// Rediscovery and coalesced admissions must not postpone an armed retry.
entry.last_admission_result = result.result_label().to_string();
entry.last_admission_reason = result.reason_label().to_string();
}
impl FolderScanner {
pub(super) fn sync_pending_heals(&mut self) {
self.pending_heals_changed = true;
if self.pending_heal_sync_deferred {
self.pending_heal_batch_dirty = true;
return;
}
self.update_cache.info.pending_heals = self.new_cache.info.pending_heals.clone();
#[cfg(test)]
{
self.pending_heal_sync_count += 1;
}
self.pending_heals_changed = true;
}
pub(super) fn clear_pending_scanner_heal(
@@ -82,6 +37,32 @@ impl FolderScanner {
}
}
/// Batched variant of [`Self::clear_pending_scanner_heal`] for repaired
/// notices (backlog#1894 axis B): one retain pass and one ledger sync
/// for the whole notice set, so a mass-recovery first sweep cannot turn
/// into thousands of full-table clones on the scan task. Only Object
/// entries match — bucket-level heals are never the MRF consumer's work.
pub(super) fn clear_pending_scanner_heals_for_repaired(&mut self, events: &[rustfs_common::mrf_channel::MrfRepairedEvent]) {
// Pre-resolve the notice version strings once; each ledger entry then
// compares against plain Option<&str>.
let targets: Vec<(&str, &str, Option<String>)> = events
.iter()
.map(|event| (event.bucket.as_ref(), event.object.as_ref(), mrf_repaired_version_id(event.version_id)))
.collect();
let before = self.new_cache.info.pending_heals.len();
self.new_cache.info.pending_heals.retain(|entry| {
entry.kind != PendingScannerHealKind::Object
|| !targets.iter().any(|(bucket, object, version)| {
entry.bucket.as_str() == *bucket
&& entry.object.as_deref() == Some(*object)
&& entry.version_id.as_deref() == version.as_deref()
})
});
if self.new_cache.info.pending_heals.len() != before {
self.sync_pending_heals();
}
}
pub(super) fn record_pending_scanner_heal(
&mut self,
kind: PendingScannerHealKind,
@@ -99,7 +80,10 @@ impl FolderScanner {
.iter_mut()
.find(|entry| pending_scanner_heal_matches(entry, kind, bucket, object, version_id))
{
observe_pending_heal_admission(entry, result);
entry.last_attempt = now;
entry.attempts = entry.attempts.saturating_add(1);
entry.last_admission_result = result.result_label().to_string();
entry.last_admission_reason = result.reason_label().to_string();
self.sync_pending_heals();
return;
}
@@ -214,54 +198,47 @@ impl FolderScanner {
result: HealAdmissionResult,
) {
match result {
HealAdmissionResult::Accepted | HealAdmissionResult::Merged => {
self.clear_pending_scanner_heal(kind, bucket, object, version_id);
}
HealAdmissionResult::Full | HealAdmissionResult::Dropped(HealAdmissionDropReason::QueueFull) => {
self.record_pending_scanner_heal(kind, bucket, object, version_id, scan_mode, result);
}
HealAdmissionResult::Accepted
| HealAdmissionResult::Merged
| HealAdmissionResult::Dropped(HealAdmissionDropReason::PolicyDropped)
| HealAdmissionResult::Dropped(HealAdmissionDropReason::AlreadyRunning)
HealAdmissionResult::Dropped(HealAdmissionDropReason::PolicyDropped) => {
self.clear_pending_scanner_heal(kind, bucket, object, version_id);
}
// Admin-only overlap rejections (HS-06); the scanner never sees
// them, but if it ever does, treat them as terminal like any
// other policy drop rather than endlessly retrying.
HealAdmissionResult::Dropped(HealAdmissionDropReason::AlreadyRunning)
| HealAdmissionResult::Dropped(HealAdmissionDropReason::OverlappingPaths) => {
// Admission is neither repair completion nor a durable
// successor receipt. Preserve existing responsibility without
// turning every newly admitted hint into a persisted intent.
if let Some(entry) = self
.new_cache
.info
.pending_heals
.iter_mut()
.find(|entry| pending_scanner_heal_matches(entry, kind, bucket, object, version_id))
{
observe_pending_heal_admission(entry, result);
self.sync_pending_heals();
}
self.clear_pending_scanner_heal(kind, bucket, object, version_id);
}
}
}
pub(super) async fn retry_pending_scanner_heals(&mut self) -> Result<(), ScannerError> {
let batch = PendingHealSyncBatch::new(self);
let scanner = &mut *batch.scanner;
if !scanner.should_heal().await {
if !self.should_heal().await {
return Ok(());
}
let bucket = scanner.new_cache.info.name.clone();
// Legacy notices cannot bind a verified disposition to the current
// incarnation, kind, set scope and responsibility generation.
let bucket = self.new_cache.info.name.clone();
// Backlog#1894 axis B: repairs the MRF consumer landed hand the
// manager the heal task, so the matching pending-ledger entries are
// retried nowhere — drop them here. Best-effort: a lost notice just
// leaves the entry to expire through its own attempts/age limits.
let repaired = rustfs_common::mrf_channel::take_mrf_repaired_events_for(&bucket);
if !repaired.is_empty() {
counter!("rustfs_scanner_unverified_repair_notices_total")
.increment(u64::try_from(repaired.len()).unwrap_or(u64::MAX));
self.clear_pending_scanner_heals_for_repaired(&repaired);
}
scanner.prune_pending_scanner_heals();
for pending in pending_scanner_heal_retry_candidates(&scanner.new_cache.info.pending_heals, &bucket) {
if !scanner.should_heal().await {
self.prune_pending_scanner_heals();
for pending in pending_scanner_heal_retry_candidates(&self.new_cache.info.pending_heals, &bucket) {
if !self.should_heal().await {
break;
}
let Some(request) = build_pending_scanner_heal_request(&pending) else {
scanner.clear_pending_scanner_heal(pending.kind, &pending.bucket, None, pending.version_id.as_deref());
self.clear_pending_scanner_heal(pending.kind, &pending.bucket, None, pending.version_id.as_deref());
counter!(
METRIC_SCANNER_PENDING_HEAL_MALFORMED_TOTAL,
"bucket" => pending.bucket.clone(),
@@ -280,27 +257,14 @@ impl FolderScanner {
continue;
};
if let Some(entry) = scanner.new_cache.info.pending_heals.iter_mut().find(|entry| {
pending_scanner_heal_matches(
entry,
pending.kind,
&pending.bucket,
pending.object.as_deref(),
pending.version_id.as_deref(),
)
}) {
record_pending_heal_attempt(entry, Self::now_secs());
scanner.sync_pending_heals();
}
scanner
.send_required_scanner_heal_request(
pending.kind,
pending.bucket.clone(),
pending.object.clone(),
pending.version_id.clone(),
request,
)
.await?;
self.send_required_scanner_heal_request(
pending.kind,
pending.bucket.clone(),
pending.object.clone(),
pending.version_id.clone(),
request,
)
.await?;
}
Ok(())
@@ -331,6 +295,16 @@ pub(super) fn pending_scanner_heal_identity(entry: &PendingScannerHeal) -> (u8,
(kind, entry.bucket.as_str(), entry.object.as_deref(), entry.version_id.as_deref())
}
/// Decode an MRF repaired-notice version id for ledger matching. A nil UUID
/// means "no value" per the repo-wide defensive-UUID invariant, so it maps
/// to `None` and matches unversioned ledger entries only.
pub(super) fn mrf_repaired_version_id(version_id: Option<[u8; 16]>) -> Option<String> {
version_id
.map(uuid::Uuid::from_bytes)
.filter(|uuid| !uuid.is_nil())
.map(|uuid| uuid.to_string())
}
pub(super) fn sort_pending_scanner_heals_for_retry(entries: &mut [PendingScannerHeal]) {
entries.sort_by(|a, b| {
a.last_attempt
@@ -344,56 +318,30 @@ pub(super) fn pending_scanner_heal_retry_candidates(
pending_heals: &[PendingScannerHeal],
bucket: &str,
) -> Vec<PendingScannerHeal> {
pending_scanner_heal_retry_candidates_at(pending_heals, bucket, FolderScanner::now_secs())
}
pub(super) fn pending_scanner_heal_retry_candidates_at(
pending_heals: &[PendingScannerHeal],
bucket: &str,
now: u64,
) -> Vec<PendingScannerHeal> {
// Schedule across scanner cycles rather than allocating a timer per hint.
// A later Full response must not reset an already retried hint's backoff.
let mut entries: Vec<&PendingScannerHeal> = pending_heals
.iter()
.filter(|entry| {
let exponent = entry.attempts.saturating_sub(1).min(31);
let delay = PENDING_HEAL_RETRY_BASE_SECS
.saturating_mul(1_u64 << exponent)
.min(PENDING_HEAL_RETRY_CAP_SECS);
entry.bucket == bucket && now.checked_sub(entry.last_attempt).is_some_and(|age| age >= delay)
})
.collect();
entries.sort_by(|a, b| {
a.last_attempt
.cmp(&b.last_attempt)
.then_with(|| a.attempts.cmp(&b.attempts))
.then_with(|| pending_scanner_heal_identity(a).cmp(&pending_scanner_heal_identity(b)))
});
let mut entries: Vec<PendingScannerHeal> = pending_heals.iter().filter(|entry| entry.bucket == bucket).cloned().collect();
sort_pending_scanner_heals_for_retry(&mut entries);
entries.truncate(MAX_PENDING_SCANNER_HEAL_RETRIES_PER_BUCKET);
entries.into_iter().cloned().collect()
entries
}
pub(super) fn build_pending_scanner_heal_request(entry: &PendingScannerHeal) -> Option<HealChannelRequest> {
let priority = if entry.last_admission_result == "full"
|| (entry.last_admission_result == "dropped" && entry.last_admission_reason == "queue_full")
{
HealChannelPriority::High
} else {
HealChannelPriority::Low
};
match entry.kind {
PendingScannerHealKind::Bucket => Some(build_bucket_heal_request(entry.bucket.clone(), priority)),
PendingScannerHealKind::Bucket => Some(build_bucket_heal_request(entry.bucket.clone(), HealChannelPriority::High)),
PendingScannerHealKind::Object => entry.object.as_ref().map(|object| {
if entry.version_id.is_none() {
build_non_destructive_object_heal_request(entry.bucket.clone(), object.clone(), entry.scan_mode, priority)
build_non_destructive_object_heal_request(
entry.bucket.clone(),
object.clone(),
entry.scan_mode,
HealChannelPriority::High,
)
} else {
build_object_heal_request(
entry.bucket.clone(),
object.clone(),
entry.version_id.clone(),
entry.scan_mode,
priority,
HealChannelPriority::High,
)
}
}),
+28 -19
View File
@@ -15,8 +15,6 @@
use crate::SCANNER_SLEEPER;
use super::*;
mod mrf_ownership;
use crate::storage_api::VersionPurgeStatusType;
use crate::{DiskOption, Endpoint, STORAGE_FORMAT_FILE, TierStats, new_disk, storageclass};
use rustfs_filemeta::{FileInfo, FileMeta, MetadataResolutionParams};
@@ -352,9 +350,6 @@ async fn build_test_scanner() -> (FolderScanner, std::path::PathBuf) {
coverage_frontier: None,
resume_frontier: None,
coverage_gap: false,
pending_heal_sync_deferred: false,
pending_heal_batch_dirty: false,
pending_heal_sync_count: 0,
pending_size_reconciliation_keys: HashSet::new(),
pending_size_reconciliation_scopes: HashSet::new(),
pending_size_reconciliation_truncated: false,
@@ -1137,8 +1132,23 @@ fn pending_heal(
}
}
/// The nil-UUID branch of the defensive-UUID invariant: a nil version in
/// a repaired notice means "no value" and must match unversioned ledger
/// entries only.
#[test]
fn test_mrf_repaired_version_id_maps_nil_to_none() {
assert_eq!(mrf_repaired_version_id(None), None);
assert_eq!(mrf_repaired_version_id(Some([0u8; 16])), None);
let uuid = Uuid::new_v4();
assert_eq!(mrf_repaired_version_id(Some(*uuid.as_bytes())), Some(uuid.to_string()));
}
/// Full wiring of backlog#1894 axis B: notes taken for the scanned bucket
/// clear exactly the matching Object ledger entries — bucket-level
/// entries, other buckets' entries, and version-mismatched entries
/// survive; a real (non-nil) version matches only the same version.
#[tokio::test]
async fn mrf_ownership_legacy_notices_preserve_pending_entries() {
async fn test_mrf_repaired_notices_clear_matching_ledger_entries() {
use rustfs_common::mrf_channel::note_mrf_repaired;
let (mut scanner, temp_dir) = build_test_scanner().await;
@@ -1166,7 +1176,8 @@ async fn mrf_ownership_legacy_notices_preserve_pending_entries() {
note_mrf_repaired("bucket", "object-a", None);
note_mrf_repaired("bucket", "object-b", Some(*Uuid::parse_str(&version).unwrap().as_bytes()));
// Neither nil nor a matching version proves incarnation, scope or owner.
// A nil-UUID notice for object-c means "no value": it clears the
// unversioned entry but must not touch the versioned one.
note_mrf_repaired("bucket", "object-c", Some([0u8; 16]));
// A notice for a target the ledger does not track must be a no-op.
note_mrf_repaired("bucket", "object-untracked", None);
@@ -1183,12 +1194,12 @@ async fn mrf_ownership_legacy_notices_preserve_pending_entries() {
.iter()
.map(|entry| (entry.kind, entry.bucket.as_str(), entry.object.as_deref(), entry.version_id.as_deref()))
.collect();
// Cleared: object-a (no version), object-b (exact version match), and
// object-c's unversioned entry (the nil branch matched no-version
// only — the versioned object-c entry survives).
assert_eq!(
survivors,
vec![
(PendingScannerHealKind::Object, "bucket", Some("object-a"), None),
(PendingScannerHealKind::Object, "bucket", Some("object-b"), Some(version.as_str())),
(PendingScannerHealKind::Object, "bucket", Some("object-c"), None),
(
PendingScannerHealKind::Object,
"bucket",
@@ -1327,7 +1338,7 @@ async fn test_pending_heal_update_keeps_stale_entry_until_retry_prune() {
);
assert_eq!(scanner.new_cache.info.pending_heals.len(), 1);
assert_eq!(scanner.new_cache.info.pending_heals[0].attempts, 1);
assert_eq!(scanner.new_cache.info.pending_heals[0].attempts, 2);
assert_eq!(scanner.new_cache.info.pending_heals[0].object.as_deref(), Some("object"));
assert_eq!(scanner.update_cache.info.pending_heals, scanner.new_cache.info.pending_heals);
}
@@ -1360,7 +1371,7 @@ async fn test_pending_heal_queue_full_deduplicates_object_entry() {
let pending = &scanner.new_cache.info.pending_heals[0];
assert_eq!(pending.object.as_deref(), Some("object"));
assert_eq!(pending.version_id.as_deref(), Some("version-a"));
assert_eq!(pending.attempts, 1);
assert_eq!(pending.attempts, 2);
assert_eq!(pending.last_admission_result, "dropped");
assert_eq!(pending.last_admission_reason, "queue_full");
assert_eq!(scanner.update_cache.info.pending_heals, scanner.new_cache.info.pending_heals);
@@ -1368,7 +1379,7 @@ async fn test_pending_heal_queue_full_deduplicates_object_entry() {
}
#[tokio::test]
async fn mrf_ownership_admission_preserves_existing_pending() {
async fn test_pending_heal_admitted_results_clear_matching_entry() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(u64::MAX, usize::MAX, &mut scanner, temp_dir);
@@ -1389,8 +1400,7 @@ async fn mrf_ownership_admission_preserves_existing_pending() {
HealAdmissionResult::Accepted,
);
assert_eq!(scanner.new_cache.info.pending_heals.len(), 1);
assert_eq!(scanner.new_cache.info.pending_heals[0].last_admission_result, "accepted");
assert!(scanner.new_cache.info.pending_heals.is_empty());
scanner.update_pending_scanner_heal_after_admission(
PendingScannerHealKind::Bucket,
@@ -1409,12 +1419,11 @@ async fn mrf_ownership_admission_preserves_existing_pending() {
HealAdmissionResult::Merged,
);
assert_eq!(scanner.new_cache.info.pending_heals.len(), 2);
assert_eq!(scanner.new_cache.info.pending_heals[1].last_admission_result, "merged");
assert!(scanner.new_cache.info.pending_heals.is_empty());
}
#[tokio::test]
async fn mrf_ownership_policy_drop_does_not_discharge_existing_pending() {
async fn test_pending_heal_policy_dropped_clears_without_creating_entry() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(u64::MAX, usize::MAX, &mut scanner, temp_dir);
@@ -1445,7 +1454,7 @@ async fn mrf_ownership_policy_drop_does_not_discharge_existing_pending() {
HealAdmissionResult::Dropped(HealAdmissionDropReason::PolicyDropped),
);
assert_eq!(scanner.new_cache.info.pending_heals.len(), 1);
assert!(scanner.new_cache.info.pending_heals.is_empty());
}
#[test]
@@ -20,7 +20,6 @@ use crate::{DataUsageCacheSource, DataUsageScanPlanDigest};
use std::io::Cursor;
use tokio::io::AsyncReadExt;
mod deep_compacted;
mod segment_observation;
const CACHE_NAME: &str = "bucket/checkpoint-fixture.bin";
@@ -1,204 +0,0 @@
// Copyright 2026 RustFS Team
// Licensed under the Apache License, Version 2.0.
use super::*;
const PREFIX: &str = "bucket/prefix";
const OBJECTS: u64 = 4;
struct CompactedFixture {
disk: Arc<Disk>,
root: std::path::PathBuf,
cache: DataUsageCache,
identity: crate::DataUsageScanIdentity,
store: Arc<FixtureStore>,
_cleanup: TestGuard,
}
async fn scan(
disk: &Arc<Disk>,
cache: DataUsageCache,
mode: HealScanMode,
max_objects: u64,
) -> (ScannerDiskScanOutcome, Arc<ScannerCycleBudget>) {
let budget = ScannerCycleBudget::new_with_progress_tracking(
&CancellationToken::new(),
ScannerCycleBudgetConfig {
max_objects: Some(max_objects),
..Default::default()
},
);
let outcome = disk
.clone()
.nsscanner_disk(budget.token(), budget.clone(), vec![disk.clone()], cache, None, mode)
.await
.expect("bounded real disk scan");
(outcome, budget)
}
async fn save_reload(store: &Arc<FixtureStore>, cache: &DataUsageCache) -> DataUsageCache {
let revisions = DataUsageCache::default()
.load_with_revisions(store.clone(), CACHE_NAME)
.await
.expect("fixture save revisions");
cache
.save_with_revisions_for_epoch(store.clone(), CACHE_NAME, &revisions, 0)
.await
.expect("save compacted checkpoint through real codec and revision checks");
let loaded = store.strict_load().await;
assert_eq!(loaded.info.snapshot_complete, cache.info.snapshot_complete);
assert_eq!(loaded.info.scan_checkpoint, cache.info.scan_checkpoint);
assert_eq!(loaded.info.scan_identity, cache.info.scan_identity);
assert_eq!(loaded.info.scan_plan_digest, cache.info.scan_plan_digest);
assert_eq!(
loaded.checked_flatten("bucket").expect("reloaded root").size,
cache.checked_flatten("bucket").expect("returned root").size
);
loaded
}
impl CompactedFixture {
async fn new(mode: HealScanMode) -> Self {
let (scanner, root) = build_test_scanner().await;
let cleanup = TestGuard {
temp_dir: Some(root.clone()),
};
for index in 0..OBJECTS {
write_checkpoint_object(&root, &format!("prefix/{index:04}"), &[(None, 1)]).await;
}
let identity = crate::DataUsageScanIdentity {
scan_mode: mode,
tier_registry_generation: crate::runtime_tier_registry_for_cycle(11, 7).await.generation,
..bound_checkpoint().1
};
let mut cache = DataUsageCache::default();
// The first real scan builds coverage; the second same-plan scan takes
// the normal compaction path. No synthetic compacted cache is injected.
for cycle in [11, 12] {
cache.prepare_bucket_checkpoint("bucket", cycle, 7, SOURCE, PLAN, identity);
cache.info.skip_healing = true;
let (outcome, budget) = scan(&scanner.local_disk, cache, mode, OBJECTS + 1).await;
let ScannerDiskScanOutcome::Complete(completed) = outcome else {
panic!("fixture baseline must complete")
};
assert_eq!(budget.progress().0, OBJECTS, "baseline must read every metadata object");
cache = completed;
}
let store = FixtureStore::new();
cache = save_reload(&store, &cache).await;
assert!(cache.find(PREFIX).expect("baseline prefix").compacted);
assert_eq!(cache.checked_flatten("bucket").expect("complete baseline").size, 4);
assert!(cache.info.scan_progress.is_none());
Self {
disk: scanner.local_disk,
root,
cache,
identity,
store,
_cleanup: cleanup,
}
}
fn next_cycle(&self, sampled: bool) -> u64 {
let first = self.cache.info.next_cycle + 1;
(first..first + 16)
.find(|cycle| hash_path(PREFIX).mod_(u32::try_from(*cycle).expect("bounded cycle"), 16) == sampled)
.expect("one selected cycle and non-selected cycles exist within the fixed rotation")
}
fn prepare(&mut self, cycle: u64) {
let state = crate::scanner_io::current_cache_root_or_prepare_with_generation(
&mut self.cache,
"bucket",
SOURCE,
cycle,
7,
PLAN,
crate::scanner_io::DataUsageCacheReuseOptions {
checkpoint_identity: Some(self.identity),
..Default::default()
},
);
assert!(matches!(state, crate::scanner_io::DataUsageCacheScanState::Prepared { .. }));
assert_eq!(self.cache.info.scan_identity, Some(self.identity));
assert_eq!(self.cache.info.scan_plan_digest, Some(PLAN));
assert!(
self.cache.info.scan_progress.is_none(),
"same-strength complete baseline uses the existing tree"
);
assert!(self.cache.find(PREFIX).expect("prepared prefix").compacted);
}
async fn change_metadata_without_activity_event(&self) {
// This models a local metadata change not announced by a segment
// producer. Deep traversal must not depend on a usage-clean signal.
// Healing is disabled: the oracle proves metadata re-entry, not repair.
write_checkpoint_object(&self.root, "prefix/0000", &[(None, 7)]).await;
}
}
#[tokio::test]
#[serial]
async fn deep_compacted_same_plan_rechecks_unsampled_prefix() {
temp_env::async_with_vars([(ENV_DATA_USAGE_UPDATE_DIR_CYCLES, Some("16"))], async {
let mut fixture = CompactedFixture::new(HealScanMode::Deep).await;
fixture.change_metadata_without_activity_event().await;
fixture.prepare(fixture.next_cycle(false));
let (outcome, budget) = scan(&fixture.disk, fixture.cache.clone(), HealScanMode::Deep, OBJECTS + 1).await;
assert_eq!(
budget.progress().0,
OBJECTS,
"Deep must inspect compacted children even outside the usage sample cycle"
);
let ScannerDiskScanOutcome::Complete(cache) = outcome else {
panic!("bounded Deep scan must complete")
};
let loaded = save_reload(&fixture.store, &cache).await;
let root = loaded.checked_flatten("bucket").expect("Deep scan root");
assert_eq!((root.objects, root.size), (4, 10), "Deep must observe the changed metadata");
})
.await;
}
#[tokio::test]
#[serial]
async fn deep_compacted_normal_scan_preserves_periodic_sampling() {
temp_env::async_with_vars([(ENV_DATA_USAGE_UPDATE_DIR_CYCLES, Some("16"))], async {
let mut fixture = CompactedFixture::new(HealScanMode::Normal).await;
fixture.change_metadata_without_activity_event().await;
fixture.prepare(fixture.next_cycle(false));
let (outcome, budget) = scan(&fixture.disk, fixture.cache.clone(), HealScanMode::Normal, OBJECTS + 1).await;
assert_eq!(budget.progress().0, 0, "Normal retains its existing unsampled-subtree policy");
let ScannerDiskScanOutcome::Complete(cache) = outcome else { panic!("normal sampling completes") };
assert_eq!(cache.checked_flatten("bucket").expect("sampled root").size, 4);
fixture.cache = save_reload(&fixture.store, &cache).await;
fixture.prepare(fixture.next_cycle(true));
let (outcome, budget) = scan(&fixture.disk, fixture.cache.clone(), HealScanMode::Normal, OBJECTS + 1).await;
assert_eq!(budget.progress().0, OBJECTS);
let ScannerDiskScanOutcome::Complete(cache) = outcome else {
panic!("selected Normal rotation completes")
};
assert_eq!(cache.checked_flatten("bucket").expect("refreshed root").size, 10);
})
.await;
}
#[tokio::test]
#[serial]
async fn deep_compacted_budget_preserves_partial_checkpoint() {
temp_env::async_with_vars([(ENV_DATA_USAGE_UPDATE_DIR_CYCLES, Some("16"))], async {
let mut fixture = CompactedFixture::new(HealScanMode::Deep).await;
fixture.prepare(fixture.next_cycle(false));
let (outcome, budget) = scan(&fixture.disk, fixture.cache.clone(), HealScanMode::Deep, 2).await;
assert_eq!(budget.progress().0, 2);
assert_eq!(budget.reason(), Some(ScannerCycleBudgetReason::Objects));
let ScannerDiskScanOutcome::Partial(cache) = outcome else {
panic!("Deep must retain budget interruption as partial")
};
assert!(!cache.info.snapshot_complete);
let loaded = save_reload(&fixture.store, &cache).await;
assert!(!loaded.info.snapshot_complete);
assert!(loaded.checked_flatten("bucket").expect("partial root").objects > 0);
})
.await;
}
@@ -1,410 +0,0 @@
// Copyright 2026 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use super::*;
use crate::storage_api::EcstoreHealResultItem as HealItem;
use crate::storage_api::scanner_io::BucketInfo;
use rustfs_common::mrf_channel::{
MrfIngressResult, MrfKind, MrfScope, note_mrf_repaired, take_mrf_repaired_events_for, try_send_mrf_intent_typed,
};
use rustfs_heal::heal::{
manager::{HealConfig, HealManager},
mrf_queue::spawn_mrf_consumer,
storage::{HealListItem, HealObjectInfo, HealStorageAPI},
};
use rustfs_heal_contracts::heal_channel::HealOpts;
#[tokio::test]
async fn mrf_ownership_admission_observation_does_not_postpone_retry() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(u64::MAX, usize::MAX, &mut scanner, temp_dir);
scanner.new_cache.info.pending_heals.push(pending_heal(
PendingScannerHealKind::Object,
"bucket",
Some("object"),
None,
100,
2,
));
for result in [
HealAdmissionResult::Accepted,
HealAdmissionResult::Merged,
HealAdmissionResult::Full,
HealAdmissionResult::Dropped(HealAdmissionDropReason::QueueFull),
HealAdmissionResult::Dropped(HealAdmissionDropReason::PolicyDropped),
] {
scanner.update_pending_scanner_heal_after_admission(
PendingScannerHealKind::Object,
"bucket",
Some("object"),
None,
HealScanMode::Deep,
result,
);
let entry = &scanner.new_cache.info.pending_heals[0];
assert_eq!((entry.last_attempt, entry.attempts), (100, 2));
assert!(pending_scanner_heal_retry_candidates_at(&scanner.new_cache.info.pending_heals, "bucket", 1899).is_empty());
assert_eq!(
pending_scanner_heal_retry_candidates_at(&scanner.new_cache.info.pending_heals, "bucket", 1900).len(),
1
);
}
scanner.update_pending_scanner_heal_after_admission(
PendingScannerHealKind::Object,
"bucket",
Some("new-object"),
None,
HealScanMode::Deep,
HealAdmissionResult::Accepted,
);
assert_eq!(
scanner.new_cache.info.pending_heals.len(),
1,
"successful admission does not create a new ledger"
);
}
#[test]
fn mrf_ownership_retry_due_boundaries_and_priority_are_bounded() {
let mut entry = pending_heal(PendingScannerHealKind::Object, "bucket", Some("object"), None, 100, 1);
entry.last_admission_result = "accepted".to_string();
assert!(pending_scanner_heal_retry_candidates_at(std::slice::from_ref(&entry), "bucket", 999).is_empty());
assert_eq!(
pending_scanner_heal_retry_candidates_at(std::slice::from_ref(&entry), "bucket", 1000).len(),
1
);
assert_eq!(
build_pending_scanner_heal_request(&entry).expect("request").priority,
HealChannelPriority::Low
);
record_pending_heal_attempt(&mut entry, 1000);
observe_pending_heal_admission(&mut entry, HealAdmissionResult::Full);
assert!(pending_scanner_heal_retry_candidates_at(std::slice::from_ref(&entry), "bucket", 2799).is_empty());
assert_eq!(
pending_scanner_heal_retry_candidates_at(std::slice::from_ref(&entry), "bucket", 2800).len(),
1
);
assert_eq!(
build_pending_scanner_heal_request(&entry).expect("request").priority,
HealChannelPriority::High
);
entry.attempts = u32::MAX;
assert!(pending_scanner_heal_retry_candidates_at(std::slice::from_ref(&entry), "bucket", 22599).is_empty());
assert_eq!(
pending_scanner_heal_retry_candidates_at(std::slice::from_ref(&entry), "bucket", 22600).len(),
1
);
entry.last_attempt = u64::MAX;
assert!(pending_scanner_heal_retry_candidates_at(&[entry], "bucket", 22600).is_empty());
}
#[tokio::test]
async fn mrf_ownership_full_hint_table_has_bounded_multicycle_work_and_sync() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(u64::MAX, usize::MAX, &mut scanner, temp_dir);
let base = 1_700_000_000;
scanner.new_cache.info.pending_heals = (0..MAX_PENDING_SCANNER_HEALS_PER_BUCKET)
.map(|index| {
let mut entry =
pending_heal(PendingScannerHealKind::Object, "bucket", Some(&format!("object-{index}")), None, base, 1);
entry.first_seen = base;
entry.last_admission_result = "accepted".to_string();
entry
})
.collect();
let indices: HashMap<String, usize> = scanner
.new_cache
.info
.pending_heals
.iter()
.enumerate()
.map(|(index, entry)| (entry.object.clone().expect("object identity"), index))
.collect();
scanner.sync_pending_heals();
let initial_syncs = scanner.pending_heal_sync_count;
let mut requests = 0usize;
let mut nonempty_batches = 0;
for minute in 0..24 * 60 {
let now = base + minute * 60;
let candidates = pending_scanner_heal_retry_candidates_at(&scanner.new_cache.info.pending_heals, "bucket", now);
assert!(candidates.len() <= MAX_PENDING_SCANNER_HEAL_RETRIES_PER_BUCKET);
if candidates.is_empty() {
continue;
}
nonempty_batches += 1;
let before = scanner.pending_heal_sync_count;
{
let batch = PendingHealSyncBatch::new(&mut scanner);
for candidate in candidates {
assert_eq!(
build_pending_scanner_heal_request(&candidate)
.expect("retry request")
.priority,
HealChannelPriority::Low
);
let index = indices[candidate.object.as_ref().expect("object identity")];
record_pending_heal_attempt(&mut batch.scanner.new_cache.info.pending_heals[index], now);
observe_pending_heal_admission(
&mut batch.scanner.new_cache.info.pending_heals[index],
HealAdmissionResult::Accepted,
);
batch.scanner.sync_pending_heals();
requests += 1;
}
}
assert_eq!(scanner.pending_heal_sync_count, before + 1, "one table clone per changed retry batch");
assert_eq!(scanner.new_cache.info.pending_heals.len(), MAX_PENDING_SCANNER_HEALS_PER_BUCKET);
}
assert!(requests >= MAX_PENDING_SCANNER_HEALS_PER_BUCKET, "every retained hint receives a retry");
assert!(
requests <= 7 * MAX_PENDING_SCANNER_HEALS_PER_BUCKET,
"15min..6h backoff bounds repeated work within 24h"
);
assert!(scanner.new_cache.info.pending_heals.iter().all(|entry| entry.attempts >= 2));
assert_eq!(scanner.pending_heal_sync_count - initial_syncs, nonempty_batches);
assert_eq!(scanner.update_cache.info.pending_heals, scanner.new_cache.info.pending_heals);
}
#[tokio::test]
async fn mrf_ownership_cancelled_batch_restores_sync_without_per_item_clones() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(u64::MAX, usize::MAX, &mut scanner, temp_dir);
scanner
.new_cache
.info
.pending_heals
.push(pending_heal(PendingScannerHealKind::Object, "bucket", Some("object"), None, 1, 1));
let before = scanner.pending_heal_sync_count;
let mut work = Box::pin(async {
let batch = PendingHealSyncBatch::new(&mut scanner);
record_pending_heal_attempt(&mut batch.scanner.new_cache.info.pending_heals[0], 100);
batch.scanner.sync_pending_heals();
batch.scanner.sync_pending_heals();
std::future::pending::<()>().await;
});
assert!(futures::poll!(&mut work).is_pending());
drop(work);
assert!(!scanner.pending_heal_sync_deferred);
assert!(!scanner.pending_heal_batch_dirty);
assert_eq!(scanner.pending_heal_sync_count, before + 1);
assert_eq!(scanner.new_cache.info.pending_heals[0].attempts, 2);
assert!(pending_scanner_heal_retry_candidates_at(&scanner.new_cache.info.pending_heals, "bucket", 101).is_empty());
assert_eq!(scanner.update_cache.info.pending_heals, scanner.new_cache.info.pending_heals);
let result: std::result::Result<(), &'static str> = async {
let batch = PendingHealSyncBatch::new(&mut scanner);
record_pending_heal_attempt(&mut batch.scanner.new_cache.info.pending_heals[0], 200);
observe_pending_heal_admission(&mut batch.scanner.new_cache.info.pending_heals[0], HealAdmissionResult::Merged);
batch.scanner.sync_pending_heals();
Err("injected retry batch failure")
}
.await;
assert!(result.is_err());
assert_eq!(scanner.pending_heal_sync_count, before + 2);
assert!(!scanner.pending_heal_sync_deferred);
assert_eq!(scanner.update_cache.info.pending_heals, scanner.new_cache.info.pending_heals);
}
#[derive(Default)]
struct NoticeStorage {
calls: std::sync::Mutex<HashMap<String, u32>>,
retry_started: tokio::sync::Notify,
}
#[async_trait::async_trait]
impl HealStorageAPI for NoticeStorage {
async fn get_object_meta(&self, _: &str, _: &str) -> rustfs_heal::Result<Option<HealObjectInfo>> {
Ok(None)
}
async fn ec_decode_rebuild(&self, _: &str, _: &str) -> rustfs_heal::Result<Vec<u8>> {
Err(rustfs_heal::Error::other("unused decode fixture"))
}
async fn get_bucket_info(&self, bucket: &str) -> rustfs_heal::Result<Option<BucketInfo>> {
Ok(Some(BucketInfo {
name: bucket.to_string(),
..Default::default()
}))
}
async fn list_buckets(&self) -> rustfs_heal::Result<Vec<BucketInfo>> {
Ok(Vec::new())
}
async fn object_exists(&self, _: &str, _: &str) -> rustfs_heal::Result<bool> {
Ok(true)
}
async fn heal_object(
&self,
_: &str,
object: &str,
_: Option<&str>,
_: &HealOpts,
) -> rustfs_heal::Result<(HealItem, Option<rustfs_heal::Error>)> {
let retry = {
let mut calls = self.calls.lock().expect("fixture calls");
let count = calls.entry(object.to_string()).or_default();
*count += 1;
*count > 1
};
if retry {
self.retry_started.notify_one();
std::future::pending::<()>().await;
}
match object {
"grace" => Ok((
HealItem::default(),
Some(rustfs_heal::Error::Disk(crate::DiskError::other(
"dangling object deletion deferred by heal grace window; retry_after_secs=3599; grace_secs=3600",
))),
)),
"failed" => Err(rustfs_heal::Error::other("permanent fixture failure")),
"cancelled" => Err(rustfs_heal::Error::TaskCancelled),
_ => Ok((HealItem::default(), None)),
}
}
async fn heal_bucket(&self, _: &str, _: &HealOpts) -> rustfs_heal::Result<HealItem> {
Ok(HealItem::default())
}
async fn heal_format(&self, _: bool) -> rustfs_heal::Result<(HealItem, Option<rustfs_heal::Error>)> {
Ok((HealItem::default(), None))
}
async fn list_objects_for_heal_page(
&self,
_: &str,
_: &str,
_: Option<&str>,
_: bool,
) -> rustfs_heal::Result<(Vec<HealListItem>, Option<String>, bool)> {
Ok((Vec::new(), None, false))
}
async fn get_disk_for_resume(&self, _: &str) -> rustfs_heal::Result<crate::DiskStore> {
Err(rustfs_heal::Error::other("unused resume fixture"))
}
}
#[tokio::test]
#[serial]
async fn mrf_ownership_manager_completion_preserves_scanner_pending() {
const CHILD: &str = "RUSTFS_MRF_OWNERSHIP_TEST_CHILD";
if std::env::var_os(CHILD).is_none() {
let output = std::process::Command::new(std::env::current_exe().expect("test executable"))
.args([
"--exact",
"scanner_folder::tests::mrf_ownership::mrf_ownership_manager_completion_preserves_scanner_pending",
"--nocapture",
])
.env(CHILD, "1")
.env("RUSTFS_HEAL_MRF_ENABLE", "true")
.output()
.expect("isolated ingress test process");
let stdout = String::from_utf8_lossy(&output.stdout);
assert!(
output.status.success() && stdout.contains("1 passed;"),
"{stdout}\n{}",
String::from_utf8_lossy(&output.stderr)
);
return;
}
// The production ingress channel is a process singleton; isolation keeps
// its receiver and lease generations independent from other scanner tests.
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(u64::MAX, usize::MAX, &mut scanner, temp_dir);
let bucket = format!("mrf-ownership-{}", Uuid::new_v4());
scanner.new_cache.info.name = bucket.clone();
scanner.update_cache.info.name = bucket.clone();
scanner.heal_object_select = 1;
let storage = Arc::new(NoticeStorage::default());
let manager = Arc::new(HealManager::new(
storage.clone(),
Some(HealConfig {
enable_auto_heal: false,
mainline_throttle_enable: false,
heal_interval: Duration::from_millis(10),
..Default::default()
}),
));
manager.start().await.expect("production manager starts");
spawn_mrf_consumer(manager.clone());
for (index, object) in ["grace", "unknown", "failed", "cancelled"].iter().enumerate() {
let version = Uuid::new_v4();
scanner.new_cache.info.pending_heals.push(pending_heal(
PendingScannerHealKind::Object,
&bucket,
Some(object),
Some(&version.to_string()),
1,
1,
));
let scope = Some(MrfScope {
pool_index: 0,
set_index: 0,
});
assert_eq!(
try_send_mrf_intent_typed(MrfKind::PartialWrite, &bucket, object, Some(version), scope),
MrfIngressResult::Enqueued
);
// Re-admission establishes that the first terminal callback released
// its ingress lease. Statistics alone precede notice publication.
tokio::time::timeout(Duration::from_secs(5), async {
loop {
match try_send_mrf_intent_typed(MrfKind::PartialWrite, &bucket, object, Some(version), scope) {
MrfIngressResult::Enqueued => break,
MrfIngressResult::Coalesced => tokio::task::yield_now().await,
other => panic!("unexpected retry ingress result: {other:?}"),
}
}
})
.await
.expect("production terminal releases its ingress lease");
tokio::time::timeout(Duration::from_secs(5), storage.retry_started.notified())
.await
.expect("the real consumer starts the second generation");
assert!(
take_mrf_repaired_events_for(&bucket).is_empty(),
"{object}: task completion must not emit an unproved repair"
);
if *object == "unknown" {
assert_eq!(
manager.get_statistics().await.total_objects_healed,
1,
"legacy healed count is not repair proof"
);
}
assert_eq!(
try_send_mrf_intent_typed(MrfKind::PartialWrite, &bucket, object, Some(version), scope),
MrfIngressResult::Coalesced,
"the in-flight retry retains its new ingress lease"
);
note_mrf_repaired(&bucket, object, Some(*version.as_bytes()));
let syncs_before_retry = scanner.pending_heal_sync_count;
scanner
.retry_pending_scanner_heals()
.await
.expect("real scanner ledger retry");
assert_eq!(scanner.pending_heal_sync_count, syncs_before_retry + 1, "the real retry batch syncs once");
assert_eq!(
scanner.new_cache.info.pending_heals.len(),
index + 1,
"{object}: pending responsibility survives"
);
let restored = DataUsageCache::unmarshal(&scanner.new_cache.marshal_msg().expect("serialize pending cache"))
.expect("decode pending cache");
assert_eq!(restored.info.pending_heals.len(), index + 1);
assert_eq!(
manager
.cancel_tasks_for_path(&format!("{bucket}/{object}"))
.await
.expect("cancel blocked retry"),
1
);
}
manager.stop().await.expect("production manager stops");
}
+14 -84
View File
@@ -27,7 +27,7 @@ use metrics::counter;
use rand::seq::SliceRandom as _;
#[cfg(test)]
use rustfs_config::{ENV_SCANNER_MAX_CONCURRENT_DISK_SCANS, ENV_SCANNER_MAX_CONCURRENT_SET_SCANS};
use rustfs_data_usage::{BucketTargetUsageInfo, BucketUsageInfo, observed_data_usage_is_newer};
use rustfs_data_usage::{BucketTargetUsageInfo, BucketUsageInfo};
use rustfs_filemeta::FileMeta;
use rustfs_heal_contracts::heal_channel::HealScanMode;
use rustfs_lock::{LockError, NamespaceLockGuard};
@@ -114,11 +114,6 @@ pub(crate) struct ScannerBucketScanScope {
}
impl ScannerBucketScanScope {
#[cfg(test)]
pub(crate) fn selected_buckets_for_tests(&self) -> Option<&HashSet<String>> {
self.selected_buckets.as_deref()
}
fn is_default(&self) -> bool {
self.selected_buckets.is_none() && self.baseline_scan_plan_digest.is_none()
}
@@ -133,8 +128,7 @@ impl ScannerBucketScanScope {
#[derive(Clone, Copy)]
pub(super) struct ScannerCacheBaselineProof<'a> {
pub(super) authoritative_data: Option<&'a Bytes>,
pub(super) observed_candidate_data: Option<&'a Bytes>,
pub(super) data: Option<&'a Bytes>,
pub(super) expected_sources: &'a HashSet<DataUsageCacheSource>,
pub(super) leader_epoch: u64,
pub(super) want_cycle: u64,
@@ -177,23 +171,21 @@ fn verified_remote_dirty_usage_buckets(
(received_peers.len() == expected_peers.len()).then_some(dirty_buckets)
}
fn complete_scanner_cache_snapshot_plan_digest(
snapshot: &DataUsageInfo,
proof: ScannerCacheBaselineProof<'_>,
expected_converged: bool,
) -> Option<DataUsageScanPlanDigest> {
if !snapshot.is_complete_bucket_usage_snapshot()
|| snapshot.usage_snapshot_partial
|| snapshot.usage_snapshot_converged != Some(expected_converged)
|| snapshot.scanner_epoch != Some(proof.leader_epoch)
|| snapshot.usage_snapshot_set_states.len() != proof.expected_sources.len()
fn complete_scanner_cache_baseline_plan_digest(proof: ScannerCacheBaselineProof<'_>) -> Option<DataUsageScanPlanDigest> {
let data = proof.data?;
let baseline = serde_json::from_slice::<DataUsageInfo>(data).ok()?;
if !baseline.is_complete_bucket_usage_snapshot()
|| baseline.usage_snapshot_partial
|| baseline.usage_snapshot_converged != Some(true)
|| baseline.scanner_epoch != Some(proof.leader_epoch)
|| baseline.usage_snapshot_set_states.len() != proof.expected_sources.len()
{
return None;
}
// Completed maintenance also covers ordinary usage. Keep its exact stored
// proof for cache reuse, and reject mixtures of different set work proofs.
let baseline_plan_digest = DataUsageScanPlanDigest(snapshot.usage_snapshot_set_states.first()?.scan_plan_digest?);
let baseline_plan_digest = DataUsageScanPlanDigest(baseline.usage_snapshot_set_states.first()?.scan_plan_digest?);
if ![
proof.scan_plan_digest,
scanner_bucket_work_digest(proof.scan_plan_digest, HealScanMode::Normal, true),
@@ -203,8 +195,8 @@ fn complete_scanner_cache_snapshot_plan_digest(
{
return None;
}
let mut states = HashSet::with_capacity(snapshot.usage_snapshot_set_states.len());
for state in &snapshot.usage_snapshot_set_states {
let mut states = HashSet::with_capacity(baseline.usage_snapshot_set_states.len());
for state in &baseline.usage_snapshot_set_states {
let source = DataUsageCacheSource::new(usize::try_from(state.pool_index).ok()?, usize::try_from(state.set_index).ok()?);
if !proof.expected_sources.contains(&source)
|| !states.insert(source)
@@ -221,30 +213,6 @@ fn complete_scanner_cache_snapshot_plan_digest(
(states == *proof.expected_sources).then_some(baseline_plan_digest)
}
fn complete_scanner_cache_baseline_plan_digest(proof: ScannerCacheBaselineProof<'_>) -> Option<DataUsageScanPlanDigest> {
let authoritative = serde_json::from_slice::<DataUsageInfo>(proof.authoritative_data?).ok()?;
if let Some(validated_digest) = complete_scanner_cache_snapshot_plan_digest(&authoritative, proof, true) {
return Some(validated_digest);
}
// A complete but superseded observation may reuse its per-set cache only
// when it was explicitly tied to the durable authoritative baseline. It
// remains observational: this proof grants bucket-scope reuse only and
// never changes authoritative usage publication or dirty acknowledgement.
let authoritative_has_identity = (crate::scanner::data_usage_info_has_persisted_baseline_identity(&authoritative)
&& authoritative.usage_snapshot_converged != Some(false))
|| crate::scanner::data_usage_info_is_bootstrap_pending(&authoritative);
if !authoritative_has_identity {
return None;
}
let observed = serde_json::from_slice::<DataUsageInfo>(proof.observed_candidate_data?).ok()?;
if !observed_data_usage_is_newer(&observed, &authoritative) {
return None;
}
complete_scanner_cache_snapshot_plan_digest(&observed, proof, false)
}
fn scoped_scan_scope_from_dirty_buckets(
requested_scope: ScannerBucketScanScope,
dirty_buckets: HashSet<String>,
@@ -318,7 +286,6 @@ pub struct ScannerBucketScanPlan {
/// Includes mutation generations even when the set planner uses a structural digest.
bucket_coverage_digest: DataUsageScanPlanDigest,
requires_full_scan: bool,
service_cohort: Option<Arc<StdMutex<ScannerServiceCohort>>>,
// Cache work must invalidate on namespace completion even when its scoped baseline remains reusable.
execution_digest: DataUsageScanPlanDigest,
leader_epoch: u64,
@@ -901,7 +868,6 @@ pub(crate) struct ScannerCycleResult {
failed_dirty_usage: bool,
pending_maintenance_work: bool,
required_cycle_floor: Option<u64>,
publication_expectation: Option<ScannerPublicationExpectation>,
}
impl ScannerCycleResult {
@@ -917,12 +883,10 @@ impl ScannerCycleResult {
failed_dirty_usage: false,
pending_maintenance_work: false,
required_cycle_floor: None,
publication_expectation: None,
}
}
pub(crate) fn with_publication_epoch(mut self, publication_epoch: Option<u64>) -> Self {
self.publication_expectation = None;
self.publication_epoch = publication_epoch;
self
}
@@ -932,7 +896,6 @@ impl ScannerCycleResult {
}
fn with_activity_digest(mut self, activity_digest: [u8; 32]) -> Self {
self.publication_expectation = None;
self.activity_digest = Some(activity_digest);
self
}
@@ -942,7 +905,6 @@ impl ScannerCycleResult {
}
pub(crate) fn with_observational_snapshot_published(mut self, published: bool) -> Self {
self.publication_expectation = None;
self.observational_snapshot_published = published;
self
}
@@ -952,19 +914,16 @@ impl ScannerCycleResult {
}
fn with_failed_dirty_usage(mut self, failed_dirty_usage: bool) -> Self {
self.publication_expectation = None;
self.failed_dirty_usage = failed_dirty_usage;
self
}
fn with_pending_maintenance_work(mut self, pending_maintenance_work: bool) -> Self {
self.publication_expectation = None;
self.pending_maintenance_work = pending_maintenance_work;
self
}
fn with_required_cycle_floor(mut self, required_cycle_floor: Option<u64>) -> Self {
self.publication_expectation = None;
self.required_cycle_floor = required_cycle_floor;
self
}
@@ -973,13 +932,11 @@ impl ScannerCycleResult {
mut self,
acknowledgements: Vec<crate::scanner::ScannerDirtyUsageAcknowledgement>,
) -> Self {
self.publication_expectation = None;
self.remote_dirty_usage_acknowledgements = acknowledgements;
self
}
pub(crate) fn with_remote_publication_lease_targets(mut self, targets: Vec<(String, String, u64)>) -> Self {
self.publication_expectation = None;
self.remote_publication_lease_targets = targets;
self
}
@@ -988,32 +945,7 @@ impl ScannerCycleResult {
&self.remote_publication_lease_targets
}
pub(crate) fn publication_expectation(&self) -> Option<ScannerPublicationExpectation> {
self.publication_expectation.clone()
}
fn with_publication_expectation(mut self, expectation: Option<ScannerPublicationExpectation>) -> Self {
// Seal only after all coverage and acknowledgement inputs are final.
self.publication_expectation = expectation;
self
}
pub(crate) fn acknowledge_durable_usage(
self,
proof: &crate::scanner::RootPublicationProof,
) -> Vec<crate::scanner::ScannerDirtyUsageAcknowledgement> {
if self.status != ScannerCycleStatus::Complete
|| self
.publication_expectation
.as_ref()
.is_none_or(|expected| proof.verified_version_for(expected).is_none())
{
return Vec::new();
}
self.clear_verified_usage()
}
fn clear_verified_usage(self) -> Vec<crate::scanner::ScannerDirtyUsageAcknowledgement> {
pub(crate) fn acknowledge_durable_usage(self) -> Vec<crate::scanner::ScannerDirtyUsageAcknowledgement> {
if let Some(snapshot) = self.dirty_usage_clear {
clear_dirty_usage_buckets(&snapshot);
}
@@ -1041,7 +973,6 @@ impl ScannerCycleResult {
mod cache;
mod dirty_usage;
mod guards;
pub(crate) use guards::ScannerServiceCohort;
mod io_cache;
mod io_cycle;
#[cfg(test)]
@@ -1053,7 +984,6 @@ mod publish_gate_tests;
#[cfg(test)]
mod tests;
pub(crate) use cache::ScannerPublicationExpectation;
use cache::*;
use dirty_usage::*;
use guards::*;
+1 -98
View File
@@ -308,86 +308,6 @@ impl<'a> ValidatedScannerSnapshot<'a> {
}
}
#[derive(Clone, Debug)]
pub(crate) struct ScannerPublicationExpectation {
candidate: Arc<([u8; 32], DataUsageScanPlanDigest)>,
}
impl ScannerPublicationExpectation {
pub(crate) fn matches_encoded_candidate(&self, digest: &[u8; 32]) -> bool {
&self.candidate.0 == digest
}
pub(crate) fn same_candidate(&self, other: &Self) -> bool {
Arc::ptr_eq(&self.candidate, &other.candidate) && self.candidate.1 == other.candidate.1
}
}
pub(super) struct ValidatedUsageCandidate {
data: DataUsageInfo,
#[cfg(test)]
last_update: SystemTime,
coverage_digest: DataUsageScanPlanDigest,
}
pub(super) fn empty_namespace_usage_candidate(
all_buckets: &[BucketInfo],
sources: &HashSet<DataUsageCacheSource>,
buckets_by_source: &HashMap<DataUsageCacheSource, Vec<BucketInfo>>,
identity: ScannerSnapshotIdentity,
) -> Option<ValidatedUsageCandidate> {
if !all_buckets.is_empty()
|| sources.is_empty()
|| sources.len() != buckets_by_source.len()
|| sources
.iter()
.any(|source| buckets_by_source.get(source).is_none_or(|buckets| !buckets.is_empty()))
{
return None;
}
let last_update = SystemTime::now();
Some(ValidatedUsageCandidate {
data: DataUsageInfo {
last_update: Some(last_update),
scanner_cycle: Some(identity.cycle),
scanner_epoch: Some(identity.leader_epoch),
usage_snapshot_complete: true,
..Default::default()
},
#[cfg(test)]
last_update,
coverage_digest: identity.coverage_digest,
})
}
impl ValidatedUsageCandidate {
pub(super) fn prepare(mut self, status: ScannerCycleStatus) -> (DataUsageInfo, Option<ScannerPublicationExpectation>) {
self.data.usage_snapshot_converged = Some(status == ScannerCycleStatus::Complete);
let expectation = if status == ScannerCycleStatus::Complete {
struct DigestWriter(Sha256);
impl std::io::Write for DigestWriter {
fn write(&mut self, bytes: &[u8]) -> std::io::Result<usize> {
self.0.update(bytes);
Ok(bytes.len())
}
fn flush(&mut self) -> std::io::Result<()> {
Ok(())
}
}
let mut writer = DigestWriter(Sha256::new());
serde_json::to_writer(&mut writer, &self.data)
.ok()
.map(|()| ScannerPublicationExpectation {
candidate: Arc::new((writer.0.finalize().into(), self.coverage_digest)),
})
} else {
None
};
(self.data, expectation)
}
}
#[cfg(test)]
pub(super) fn completed_data_usage_info(
results: &[DataUsageCache],
scope: &ScannerSnapshotScope<'_>,
@@ -396,18 +316,6 @@ pub(super) fn completed_data_usage_info(
budget_elapsed: bool,
cancelled: bool,
) -> Option<(DataUsageInfo, SystemTime)> {
completed_usage_candidate(results, scope, tier_registry_names, bucket_plan_complete, budget_elapsed, cancelled)
.map(|candidate| (candidate.data, candidate.last_update))
}
pub(super) fn completed_usage_candidate(
results: &[DataUsageCache],
scope: &ScannerSnapshotScope<'_>,
tier_registry_names: &[String],
bucket_plan_complete: bool,
budget_elapsed: bool,
cancelled: bool,
) -> Option<ValidatedUsageCandidate> {
if !bucket_plan_complete {
return None;
}
@@ -485,12 +393,7 @@ pub(super) fn completed_usage_candidate(
usage_snapshot_set_states,
..Default::default()
};
Some(ValidatedUsageCandidate {
data: data_usage_info,
#[cfg(test)]
last_update: merged_last_update,
coverage_digest: scope.identity.coverage_digest,
})
Some((data_usage_info, merged_last_update))
}
fn tier_accounting_proof_is_publishable(
-605
View File
@@ -14,312 +14,6 @@
/// scan concurrency accounting: gauge recorders, RAII guards, and worker limits.
use super::*;
const SCANNER_SERVICE_COHORT_MAX_MEMBERS: usize = 4096;
const SCANNER_SERVICE_COHORT_MAX_NAME_BYTES: usize = 128 * 1024;
static SERVICE_COHORT_METRICS_OWNER: StdMutex<std::sync::Weak<()>> = StdMutex::new(std::sync::Weak::new());
struct ScannerCohortMetricsOwner(Arc<()>);
impl Default for ScannerCohortMetricsOwner {
fn default() -> Self {
let owner = Arc::new(());
let mut current = SERVICE_COHORT_METRICS_OWNER
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
*current = Arc::downgrade(&owner);
write_service_cohort_metrics(0, 0.0, false);
Self(owner)
}
}
impl Drop for ScannerCohortMetricsOwner {
fn drop(&mut self) {
let mut current = SERVICE_COHORT_METRICS_OWNER
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
if current.ptr_eq(&Arc::downgrade(&self.0)) {
*current = std::sync::Weak::new();
write_service_cohort_metrics(0, 0.0, false);
}
}
}
fn write_service_cohort_metrics(waiting: usize, oldest: f64, overflowed: bool) {
metrics::gauge!("rustfs_scanner_service_cohort_waiting").set(waiting as f64);
metrics::gauge!("rustfs_scanner_service_cohort_oldest_wait_seconds").set(oldest);
metrics::gauge!("rustfs_scanner_service_cohort_capacity_fallback").set(if overflowed { 1.0 } else { 0.0 });
}
struct ScannerCohortWait {
order: u64,
queued_at: Instant,
admitted: bool,
present: bool,
}
/// Leader-local admission order, never evidence of completed scan coverage.
/// Retains at most 4096 members and 128 KiB of name payload, including the
/// cursor shared with its last member. Candidate selection borrows at most
/// 4096 inventory entries; the existing full inventory is not bounded here.
pub(crate) struct ScannerServiceCohort {
members: HashMap<DataUsageCacheSource, HashMap<Arc<str>, ScannerCohortWait>>,
cursor: Option<(DataUsageCacheSource, Arc<str>)>,
next_order: u64,
max_members: usize,
max_name_bytes: usize,
overflowed: bool,
metrics_owner: ScannerCohortMetricsOwner,
waiting: usize,
oldest_wait_at_refresh: f64,
#[cfg(test)]
metric_members_examined: usize,
}
impl Default for ScannerServiceCohort {
fn default() -> Self {
Self {
members: HashMap::new(),
cursor: None,
next_order: 0,
max_members: SCANNER_SERVICE_COHORT_MAX_MEMBERS,
max_name_bytes: SCANNER_SERVICE_COHORT_MAX_NAME_BYTES,
overflowed: false,
metrics_owner: ScannerCohortMetricsOwner::default(),
waiting: 0,
oldest_wait_at_refresh: 0.0,
#[cfg(test)]
metric_members_examined: 0,
}
}
}
impl ScannerServiceCohort {
pub(crate) fn refresh(&mut self, inventory: &HashMap<DataUsageCacheSource, Vec<BucketInfo>>) {
let count = inventory
.values()
.fold(0usize, |count, buckets| count.saturating_add(buckets.len()));
let name_bytes = inventory
.values()
.flatten()
.fold(0usize, |bytes, bucket| bytes.saturating_add(bucket.name.len()));
self.overflowed = count > self.max_members || name_bytes > self.max_name_bytes;
for wait in self.members.values_mut().flat_map(HashMap::values_mut) {
wait.present = false;
}
for (source, buckets) in inventory {
for bucket in buckets {
if let Some(wait) = self
.members
.get_mut(source)
.and_then(|members| members.get_mut(bucket.name.as_str()))
{
wait.present = true;
}
}
}
self.members.retain(|_, buckets| {
buckets.retain(|_, wait| wait.present);
!buckets.is_empty()
});
if self.members.values().flat_map(HashMap::values).all(|wait| wait.admitted) {
self.members.clear();
self.next_order = 0;
}
if count == 0 {
self.cursor = None;
}
let mut member_count = self.members.values().map(HashMap::len).sum::<usize>();
let mut retained_bytes = self
.members
.values()
.flat_map(HashMap::keys)
.map(|name| name.len())
.sum::<usize>();
let mut incoming = self.admission_candidates(inventory, true);
if incoming.is_empty() {
incoming = self.admission_candidates(inventory, false);
}
for (pool, set, bucket) in incoming {
if member_count >= self.max_members {
break;
}
if retained_bytes.saturating_add(bucket.len()) > self.max_name_bytes {
continue;
}
let Some(next_order) = self.next_order.checked_add(1) else {
self.overflowed = true;
break;
};
let source = DataUsageCacheSource::new(pool, set);
let members = self.members.entry(source).or_default();
if members.contains_key(bucket) {
continue;
}
let name: Arc<str> = bucket.into();
retained_bytes += name.len();
member_count += 1;
members.insert(
name.clone(),
ScannerCohortWait {
order: self.next_order,
queued_at: Instant::now(),
admitted: false,
present: true,
},
);
self.cursor = Some((source, name));
self.next_order = next_order;
}
self.refresh_metrics();
}
fn admission_candidates<'a>(
&self,
inventory: &'a HashMap<DataUsageCacheSource, Vec<BucketInfo>>,
after_cursor: bool,
) -> Vec<(usize, usize, &'a str)> {
let mut candidates = std::collections::BinaryHeap::new();
for (source, buckets) in inventory {
for bucket in buckets {
let key = (source.pool_index, source.set_index, bucket.name.as_str());
let after = self
.cursor
.as_ref()
.is_none_or(|(source, name)| key > (source.pool_index, source.set_index, name.as_ref()));
if after != after_cursor
|| bucket.name.len() > self.max_name_bytes
|| self
.members
.get(source)
.is_some_and(|members| members.contains_key(bucket.name.as_str()))
{
continue;
}
if candidates.len() < self.max_members {
candidates.push(key);
} else if candidates.peek().is_some_and(|last| key < *last) {
candidates.pop();
candidates.push(key);
}
}
}
candidates.into_sorted_vec()
}
pub(crate) fn order_set_indices(&self, sets: &[Arc<SetDisks>]) -> Vec<usize> {
let ranks = self
.members
.iter()
.map(|(source, buckets)| {
(
*source,
buckets
.values()
.filter(|wait| !wait.admitted)
.map(|wait| wait.order)
.min()
.unwrap_or(u64::MAX),
)
})
.collect::<HashMap<_, _>>();
let mut indices = (0..sets.len()).collect::<Vec<_>>();
indices.sort_by_key(|index| {
(
ranks
.get(&DataUsageCacheSource::new(sets[*index].pool_index, sets[*index].set_index))
.copied()
.unwrap_or(u64::MAX),
*index,
)
});
indices
}
pub(crate) fn order_buckets(&self, source: DataUsageCacheSource, buckets: &mut [BucketInfo]) {
let rank = |bucket: &str| {
self.members
.get(&source)
.and_then(|members| members.get(bucket))
.filter(|wait| !wait.admitted)
.map_or(u64::MAX, |wait| wait.order)
};
// Stable sorting preserves the existing dispatch order in the tail.
buckets.sort_by_key(|bucket| rank(&bucket.name));
}
pub(crate) fn record_admitted(&mut self, source: DataUsageCacheSource, bucket: &str) {
let Some(wait) = self.members.get_mut(&source).and_then(|members| members.get_mut(bucket)) else {
return;
};
if wait.admitted {
return;
}
wait.admitted = true;
self.waiting -= 1;
if self.waiting == 0 {
self.oldest_wait_at_refresh = 0.0;
}
self.record_metrics();
}
#[cfg(test)]
pub(super) fn admitted_members(&self) -> Vec<(DataUsageCacheSource, String)> {
self.members
.iter()
.flat_map(|(source, buckets)| {
buckets
.iter()
.filter(|(_, wait)| wait.admitted)
.map(|(bucket, _)| (*source, bucket.to_string()))
})
.collect()
}
fn refresh_metrics(&mut self) {
// Oldest age is an inventory-refresh snapshot, not a per-admission
// scan of the cohort. Clear it immediately when no waiters remain.
let (mut waiting, mut oldest) = (0usize, 0.0f64);
for wait in self.members.values().flat_map(HashMap::values) {
#[cfg(test)]
{
self.metric_members_examined += 1;
}
if !wait.admitted {
waiting += 1;
oldest = oldest.max(wait.queued_at.elapsed().as_secs_f64());
}
}
self.waiting = waiting;
self.oldest_wait_at_refresh = oldest;
self.record_metrics();
}
fn record_metrics(&self) {
// Serialize owner replacement, publication and retirement. A retired
// scanner must neither publish nor clear a replacement's gauges.
let current = SERVICE_COHORT_METRICS_OWNER
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
if current.ptr_eq(&Arc::downgrade(&self.metrics_owner.0)) {
write_service_cohort_metrics(self.waiting, self.oldest_wait_at_refresh, self.overflowed);
}
}
}
pub(super) async fn wait_for_bucket_scan_permit(
semaphore: &Arc<Semaphore>,
ctx: &CancellationToken,
complete: &CancellationToken,
) -> Option<tokio::sync::OwnedSemaphorePermit> {
tokio::select! {
biased;
_ = complete.cancelled() => None,
_ = ctx.cancelled() => None,
permit = semaphore.clone().acquire_owned() => permit.ok(),
}
}
pub(super) fn bucket_usage_scan_order(
buckets: &[BucketInfo],
old_cache: &DataUsageCache,
@@ -594,305 +288,6 @@ mod tests {
use rustfs_scanner_metrics::metrics::{ScannerWorkSource, global_metrics};
use tokio::sync::oneshot;
#[derive(Default)]
struct RecordedGauge(AtomicU64);
impl metrics::GaugeFn for RecordedGauge {
fn increment(&self, value: f64) {
self.set(f64::from_bits(self.0.load(Ordering::Relaxed)) + value);
}
fn decrement(&self, value: f64) {
self.increment(-value);
}
fn set(&self, value: f64) {
self.0.store(value.to_bits(), Ordering::Relaxed);
}
}
#[derive(Default)]
struct CohortGaugeRecorder(StdMutex<HashMap<String, Arc<RecordedGauge>>>);
impl metrics::Recorder for CohortGaugeRecorder {
fn describe_counter(&self, _: metrics::KeyName, _: Option<metrics::Unit>, _: metrics::SharedString) {}
fn describe_gauge(&self, _: metrics::KeyName, _: Option<metrics::Unit>, _: metrics::SharedString) {}
fn describe_histogram(&self, _: metrics::KeyName, _: Option<metrics::Unit>, _: metrics::SharedString) {}
fn register_counter(&self, _: &metrics::Key, _: &metrics::Metadata<'_>) -> metrics::Counter {
metrics::Counter::noop()
}
fn register_histogram(&self, _: &metrics::Key, _: &metrics::Metadata<'_>) -> metrics::Histogram {
metrics::Histogram::noop()
}
fn register_gauge(&self, key: &metrics::Key, _: &metrics::Metadata<'_>) -> metrics::Gauge {
metrics::Gauge::from_arc(
self.0
.lock()
.expect("gauge recorder")
.entry(key.name().to_string())
.or_default()
.clone(),
)
}
}
impl CohortGaugeRecorder {
fn value(&self, name: &str) -> f64 {
f64::from_bits(self.0.lock().expect("gauge recorder")[name].0.load(Ordering::Relaxed))
}
}
#[test]
#[serial_test::serial]
fn service_cohort_metrics_retire_only_the_current_owner() {
let recorder = CohortGaugeRecorder::default();
metrics::with_local_recorder(&recorder, || {
let mut old = ScannerServiceCohort::default();
old.refresh(&cohort_inventory(&["old"]));
let mut current = ScannerServiceCohort {
max_members: 1,
..Default::default()
};
current.refresh(&cohort_inventory(&["a", "b"]));
for wait in current.members.values_mut().flat_map(HashMap::values_mut) {
wait.queued_at = Instant::now() - Duration::from_secs(60);
}
current.refresh_metrics();
old.refresh(&cohort_inventory(&["old", "more"]));
drop(old);
assert_eq!(recorder.value("rustfs_scanner_service_cohort_waiting"), 1.0);
assert_eq!(recorder.value("rustfs_scanner_service_cohort_capacity_fallback"), 1.0);
assert!(recorder.value("rustfs_scanner_service_cohort_oldest_wait_seconds") >= 60.0);
drop(current);
for metric in [
"rustfs_scanner_service_cohort_waiting",
"rustfs_scanner_service_cohort_oldest_wait_seconds",
"rustfs_scanner_service_cohort_capacity_fallback",
] {
assert_eq!(recorder.value(metric), 0.0, "owner retirement must clear {metric}");
}
});
}
#[test]
#[serial_test::serial]
fn service_cohort_admission_metrics_do_not_rescan_a_full_window() {
let recorder = CohortGaugeRecorder::default();
metrics::with_local_recorder(&recorder, || {
let source = DataUsageCacheSource::new(0, 0);
let names = (0..SCANNER_SERVICE_COHORT_MAX_MEMBERS)
.map(|index| format!("bucket-{index:04}"))
.collect::<Vec<_>>();
let inventory = cohort_inventory(&names.iter().map(String::as_str).collect::<Vec<_>>());
let mut cohort = ScannerServiceCohort::default();
cohort.refresh(&inventory);
assert_eq!(cohort.metric_members_examined, SCANNER_SERVICE_COHORT_MAX_MEMBERS);
assert_eq!(cohort.waiting, SCANNER_SERVICE_COHORT_MAX_MEMBERS);
for (index, name) in names.iter().enumerate() {
cohort.record_admitted(source, name);
for _ in 0..10 {
cohort.record_admitted(source, name);
cohort.record_admitted(source, "untracked-overflow-name");
cohort.record_admitted(DataUsageCacheSource::new(99, 0), name);
}
assert_eq!(cohort.waiting, SCANNER_SERVICE_COHORT_MAX_MEMBERS - index - 1);
assert_eq!(
cohort.metric_members_examined, SCANNER_SERVICE_COHORT_MAX_MEMBERS,
"tracked, repeated and overflow admissions must not scan cohort members"
);
}
assert_eq!(recorder.value("rustfs_scanner_service_cohort_waiting"), 0.0);
assert_eq!(recorder.value("rustfs_scanner_service_cohort_oldest_wait_seconds"), 0.0);
cohort.refresh(&inventory);
assert_eq!(
cohort.metric_members_examined,
2 * SCANNER_SERVICE_COHORT_MAX_MEMBERS,
"one inventory refresh performs one metrics traversal"
);
});
}
#[tokio::test]
#[serial_test::serial]
async fn service_cohort_queued_permit_cancel_and_drop_return_the_same_capacity() {
let semaphore = Arc::new(Semaphore::new(1));
let active = Arc::new(AtomicUsize::new(0));
let mut cohort = ScannerServiceCohort::default();
cohort.refresh(&cohort_inventory(&["waiting"]));
let gauge_reset = DiskBucketScanGaugeReset::new("cohort-wait".to_string(), "0".to_string());
record_disk_bucket_scans_queued(1, "cohort-wait", "0");
record_disk_bucket_scans_active(0, "cohort-wait", "0");
for cancel in [true, false] {
let held = semaphore
.clone()
.acquire_owned()
.await
.expect("hold the sole permit as a barrier");
let ctx = CancellationToken::new();
let complete = CancellationToken::new();
let mut waiter = Box::pin(wait_for_bucket_scan_permit(&semaphore, &ctx, &complete));
assert!(
futures::poll!(&mut waiter).is_pending(),
"the production wait must actually enqueue behind the barrier"
);
assert_eq!(semaphore.available_permits(), 0);
if cancel {
ctx.cancel();
assert!(waiter.as_mut().await.is_none());
}
drop(waiter);
assert_eq!(semaphore.available_permits(), 0, "cancelling a waiter must not release the held permit");
assert!(cohort.admitted_members().is_empty());
drop(held);
assert_eq!(semaphore.available_permits(), 1, "no queued waiter may leak or steal released capacity");
}
let ctx = CancellationToken::new();
let complete = CancellationToken::new();
let permit = wait_for_bucket_scan_permit(&semaphore, &ctx, &complete)
.await
.expect("same semaphore remains usable");
let active_guard = DiskBucketScanActiveGuard::new(active.clone(), "cohort-wait".to_string(), "0".to_string());
assert_eq!(active.load(Ordering::Relaxed), 1);
drop(active_guard);
drop(permit);
drop(gauge_reset);
assert_eq!(active.load(Ordering::Relaxed), 0);
assert_eq!(semaphore.available_permits(), 1);
let state = global_metrics()
.scanner_runtime_details_report()
.disk_bucket_scan_states
.into_iter()
.find(|state| state.pool == "cohort-wait" && state.set == "0")
.expect("fixture gauges");
assert_eq!((state.queued, state.active), (0, 0));
assert!(cohort.admitted_members().is_empty(), "permit ownership alone does not admit a bucket");
}
fn cohort_inventory(names: &[&str]) -> HashMap<DataUsageCacheSource, Vec<BucketInfo>> {
HashMap::from([(
DataUsageCacheSource::new(0, 0),
names
.iter()
.map(|name| BucketInfo {
name: (*name).to_string(),
..Default::default()
})
.collect(),
)])
}
#[test]
#[serial_test::serial]
fn service_cohort_visits_fixed_members_within_service_round_bound() {
let inventory = cohort_inventory(&["a", "b", "c", "d", "e"]);
let source = DataUsageCacheSource::new(0, 0);
let mut cohort = ScannerServiceCohort::default();
let mut admitted = HashSet::new();
for _ in 0..3 {
cohort.refresh(&inventory);
let mut buckets = inventory[&source].clone();
cohort.order_buckets(source, &mut buckets);
for bucket in buckets.iter().take(2) {
admitted.insert(bucket.name.clone());
cohort.record_admitted(source, &bucket.name);
}
}
assert_eq!(admitted.len(), 5, "ceil(5/2) service rounds must include every member");
}
#[test]
#[serial_test::serial]
fn service_cohort_keeps_waiting_bootstrap_ahead_of_new_work() {
let source = DataUsageCacheSource::new(0, 0);
let mut cohort = ScannerServiceCohort::default();
cohort.refresh(&cohort_inventory(&["a-hot", "z-bootstrap"]));
cohort.record_admitted(source, "a-hot");
let queued_at = cohort.members[&source]["z-bootstrap"].queued_at;
let inventory = cohort_inventory(&["a-hot", "aaa-new-bootstrap", "z-bootstrap"]);
for _ in 0..10 {
cohort.refresh(&inventory);
let mut buckets = inventory[&source].clone();
cohort.order_buckets(source, &mut buckets);
assert_eq!(buckets[0].name, "z-bootstrap");
assert_eq!(buckets[1].name, "aaa-new-bootstrap");
assert_eq!(cohort.members[&source]["z-bootstrap"].queued_at, queued_at);
}
}
#[test]
#[serial_test::serial]
fn service_cohort_overflow_preserves_waiters_and_rotates_finished_windows() {
let source = DataUsageCacheSource::new(0, 0);
let mut cohort = ScannerServiceCohort {
max_members: 2,
max_name_bytes: 4,
..Default::default()
};
cohort.refresh(&cohort_inventory(&["aa", "bb"]));
cohort.record_admitted(source, "aa");
for names in [["aa", "bb", "c"], ["aa", "bb", "d"]] {
let inventory = cohort_inventory(&names);
cohort.refresh(&inventory);
assert!(cohort.overflowed);
assert_eq!(cohort.members[&source].len(), 2);
assert!(cohort.members[&source].contains_key("aa"));
let mut fallback = inventory[&source].clone();
fallback.reverse();
cohort.order_buckets(source, &mut fallback);
assert_eq!(fallback[0].name, "bb", "overflow must not discard a waiting member's priority");
assert_eq!(fallback.len(), 3, "unknown tail must remain dispatchable");
}
cohort.record_admitted(source, "bb");
let inventory = cohort_inventory(&["aa", "bb", "c", "d"]);
cohort.refresh(&inventory);
assert_eq!(
cohort.members[&source].keys().map(AsRef::as_ref).collect::<HashSet<&str>>(),
HashSet::from(["c", "d"])
);
cohort.record_admitted(source, "c");
cohort.record_admitted(source, "d");
cohort.refresh(&inventory);
assert!(
cohort.members[&source].contains_key("aa"),
"finite inventory must wrap after the last window"
);
cohort.refresh(&cohort_inventory(&["bb"]));
assert!(!cohort.overflowed);
assert_eq!(cohort.members[&source].len(), 1);
cohort.next_order = u64::MAX;
cohort.refresh(&cohort_inventory(&["bb", "c"]));
assert!(cohort.overflowed);
assert_eq!(cohort.members[&source].len(), 1);
}
#[test]
#[serial_test::serial]
fn service_cohort_bounds_names_and_does_not_reset_duplicate_dirty_age() {
let source = DataUsageCacheSource::new(0, 0);
let mut cohort = ScannerServiceCohort {
max_members: 2,
max_name_bytes: 4,
..Default::default()
};
cohort.refresh(&cohort_inventory(&["aa", "bb", "long-name"]));
let queued_at = cohort.members[&source]["bb"].queued_at;
for _ in 0..10 {
cohort.refresh(&cohort_inventory(&["aa", "aa", "bb", "long-name"]));
assert_eq!(cohort.members.values().map(HashMap::len).sum::<usize>(), 2);
assert_eq!(
cohort
.members
.values()
.flat_map(HashMap::keys)
.map(|name| name.len())
.sum::<usize>(),
4
);
assert_eq!(cohort.members[&source]["bb"].queued_at, queued_at);
}
cohort.refresh(&cohort_inventory(&[]));
assert!(cohort.members.is_empty());
assert!(cohort.cursor.is_none());
}
fn active_bucket_drive_count(source: ScannerWorkSource, bucket: &str, drive: &str) -> u64 {
global_metrics()
.scanner_runtime_details_report()
+25 -31
View File
@@ -117,7 +117,6 @@ impl ScannerIOCache for SetDisks {
digest: scan_plan_digest,
bucket_coverage_digest,
requires_full_scan,
service_cohort,
execution_digest,
leader_epoch,
tier_registry_generation,
@@ -501,13 +500,7 @@ impl ScannerIOCache for SetDisks {
let mut permutes = buckets.clone();
permutes.shuffle(&mut rand::rng());
let mut scan_order = bucket_usage_scan_order(&permutes, &old_cache, &dirty_usage_buckets);
if let Some(cohort) = &service_cohort {
cohort
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.order_buckets(source, &mut scan_order);
}
let scan_order = bucket_usage_scan_order(&permutes, &old_cache, &dirty_usage_buckets);
for bucket in scan_order.iter() {
if let Some(c) = old_cache.find(&bucket.name) {
@@ -565,7 +558,6 @@ impl ScannerIOCache for SetDisks {
let remaining_bucket_work = Arc::new(AtomicUsize::new(buckets.len()));
let bucket_work_complete = CancellationToken::new();
for (disk, worker_mode) in workers {
let service_cohort_clone = service_cohort.clone();
let bucket_rx_mutex_clone = bucket_rx_mutex.clone();
let bucket_tx_clone = bucket_tx.clone();
let remaining_bucket_work_clone = remaining_bucket_work.clone();
@@ -595,18 +587,6 @@ impl ScannerIOCache for SetDisks {
let remote_session_id = uuid::Uuid::new_v4();
let mut remote_session_sequence = 0_u64;
loop {
// Do not prefetch a FIFO member into an independently
// scheduled permit waiter: that can reorder admissions.
let permit_wait_start = Instant::now();
let Some(_permit) =
wait_for_bucket_scan_permit(&disk_scan_semaphore_clone, &ctx_clone, &bucket_work_complete_clone).await
else {
break;
};
if ctx_clone.is_cancelled() || budget_clone.budget_elapsed() {
break;
}
let permit_wait_elapsed = permit_wait_start.elapsed();
let bucket = tokio::select! {
_ = bucket_work_complete_clone.cancelled() => break,
_ = ctx_clone.cancelled() => break,
@@ -620,27 +600,41 @@ impl ScannerIOCache for SetDisks {
let mut work_guard =
BucketWorkGuard::new(remaining_bucket_work_clone.clone(), bucket_work_complete_clone.clone());
let permit_wait = ctx_clone.clone();
let permit_wait_start = Instant::now();
let _permit = tokio::select! {
permit = disk_scan_semaphore_clone.clone().acquire_owned() => match permit {
Ok(permit) => permit,
Err(_) => {
decrement_disk_bucket_scans_queued(
&queued_disk_bucket_scans_clone,
&pool_label_clone,
&set_label_clone,
);
break;
},
},
_ = permit_wait.cancelled() => {
decrement_disk_bucket_scans_queued(
&queued_disk_bucket_scans_clone,
&pool_label_clone,
&set_label_clone,
);
break;
},
};
metrics::histogram!(
METRIC_SCANNER_DISK_SCAN_WAIT_SECONDS,
"pool" => pool_label_clone.clone(),
"set" => set_label_clone.clone()
)
.record(permit_wait_elapsed.as_secs_f64());
.record(permit_wait_start.elapsed().as_secs_f64());
decrement_disk_bucket_scans_queued(&queued_disk_bucket_scans_clone, &pool_label_clone, &set_label_clone);
let _active_guard = DiskBucketScanActiveGuard::new(
active_disk_bucket_scans_clone.clone(),
pool_label_clone.clone(),
set_label_clone.clone(),
);
if ctx_clone.is_cancelled() || budget_clone.budget_elapsed() {
break;
}
if let Some(cohort) = &service_cohort_clone {
cohort
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.record_admitted(source, &bucket.name);
}
debug!(
target: "rustfs::scanner::io",
+21 -68
View File
@@ -72,9 +72,7 @@ where
scan_mode,
scan_scope: ScannerBucketScanScope::default(),
persisted_usage_baseline: None,
observed_usage_candidate: None,
requires_full_scan: true,
service_cohort: None,
#[cfg(test)]
resolved_scope_observer: None,
};
@@ -90,10 +88,8 @@ pub(crate) struct ScannerCycleRequest {
pub(crate) scan_mode: HealScanMode,
pub(crate) scan_scope: ScannerBucketScanScope,
pub(crate) persisted_usage_baseline: Option<Bytes>,
pub(crate) observed_usage_candidate: Option<Bytes>,
/// Scheduled maintenance must visit clean buckets even with a valid dirty scope.
pub(crate) requires_full_scan: bool,
pub(crate) service_cohort: Option<Arc<StdMutex<ScannerServiceCohort>>>,
#[cfg(test)]
pub(crate) resolved_scope_observer: Option<tokio::sync::oneshot::Sender<ScannerBucketScanScope>>,
}
@@ -187,9 +183,7 @@ where
scan_mode,
scan_scope,
persisted_usage_baseline,
observed_usage_candidate,
requires_full_scan,
service_cohort,
#[cfg(test)]
resolved_scope_observer,
} = request;
@@ -281,12 +275,6 @@ where
}
bucket_plan_complete &= buckets_by_source.keys().copied().collect::<HashSet<_>>() == *expected_sources;
bucket_plan_complete &= scanner_bucket_inventory_is_complete(&all_buckets, &buckets_by_source);
if bucket_plan_complete && let Some(cohort) = &service_cohort {
cohort
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.refresh(&buckets_by_source);
}
let structural_scan_plan_digest =
scanner_bucket_plan_digest(&all_buckets, crate::scanner::scanner_activity_structural_digest(&activity_before));
let scan_plan_digest = scanner_bucket_work_digest(structural_scan_plan_digest, scan_mode, requires_full_scan);
@@ -300,8 +288,7 @@ where
ScannerBucketScopeResolution {
requested_scope: scan_scope,
baseline_proof: ScannerCacheBaselineProof {
authoritative_data: persisted_usage_baseline.as_ref(),
observed_candidate_data: observed_usage_candidate.as_ref(),
data: persisted_usage_baseline.as_ref(),
expected_sources: &expected_sources,
leader_epoch,
want_cycle,
@@ -338,21 +325,12 @@ where
dirty_usage_status,
activity_status,
);
let Some(candidate) = empty_namespace_usage_candidate(
&all_buckets,
&expected_sources,
&buckets_by_source,
ScannerSnapshotIdentity {
cycle: want_cycle,
leader_epoch,
plan_digest: scan_plan_digest,
coverage_digest: bucket_coverage_digest,
tier_registry_generation: Some(tier_registry_generation),
},
) else {
return Ok(ScannerCycleResult::new(ScannerCycleStatus::Incomplete, None).with_publication_epoch(publication_epoch));
let empty_usage = DataUsageInfo {
last_update: Some(SystemTime::now()),
scanner_cycle: Some(want_cycle),
usage_snapshot_complete: true,
..Default::default()
};
let (empty_usage, publication_expectation) = candidate.prepare(status);
let observational_snapshot_published = if should_publish_observational_snapshot(status) {
publish_observational_snapshot(&updates, empty_usage).await?
} else {
@@ -375,8 +353,7 @@ where
.with_activity_digest(activity_digest)
.with_observational_snapshot_published(observational_snapshot_published)
.with_remote_publication_lease_targets(remote_publication_lease_targets)
.with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements)
.with_publication_expectation(publication_expectation));
.with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements));
}
let total_results = expected_sources.len();
@@ -422,32 +399,7 @@ where
let first_err_mutex: Arc<Mutex<Option<Error>>> = Arc::new(Mutex::new(None));
let mut wait_futs = Vec::new();
let set_order = service_cohort.as_ref().map_or_else(
|| (0..set_disks.len()).collect::<Vec<_>>(),
|cohort| {
cohort
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.order_set_indices(&set_disks)
},
);
for results_index in set_order {
let set = &set_disks[results_index];
// Acquire in dispatch order, not in independently scheduled tasks.
// A whole set still shares the existing parent budget; this is not
// a per-bucket quantum or a cross-source completion guarantee.
let permit_wait_start = Instant::now();
let permit = tokio::select! {
biased;
_ = child_token.cancelled() => break,
permit = set_scan_semaphore.clone().acquire_owned() => match permit {
Ok(permit) => permit,
Err(_) => break,
},
};
if child_token.is_cancelled() || budget.budget_elapsed() {
break;
}
for (results_index, set) in set_disks.iter().enumerate() {
let results_index_clone = results_index;
// Clone the Arc to move it into the spawned task
let set_clone: Arc<SetDisks> = Arc::clone(set);
@@ -462,6 +414,7 @@ where
let scan_mode_clone = scan_mode;
let results_mutex_clone = results_mutex.clone();
let first_err_mutex_clone = first_err_mutex.clone();
let set_scan_semaphore_clone = set_scan_semaphore.clone();
let queued_set_scans_clone = queued_set_scans.clone();
let active_set_scans_clone = active_set_scans.clone();
@@ -484,7 +437,6 @@ where
digest: structural_scan_plan_digest,
bucket_coverage_digest,
requires_full_scan,
service_cohort: service_cohort.clone(),
execution_digest,
leader_epoch,
tier_registry_generation,
@@ -496,10 +448,15 @@ where
};
// Spawn task to run the scanner
let scanner_fut = tokio::spawn(async move {
let _permit = permit;
if child_token_clone.is_cancelled() || budget_clone.budget_elapsed() {
return;
}
let permit_wait = child_token_clone.clone();
let permit_wait_start = Instant::now();
let _permit = tokio::select! {
permit = set_scan_semaphore_clone.acquire_owned() => match permit {
Ok(permit) => permit,
Err(_) => return,
},
_ = permit_wait.cancelled() => return,
};
metrics::histogram!(
METRIC_SCANNER_SET_SCAN_WAIT_SECONDS,
"pool" => pool_label.clone(),
@@ -605,7 +562,7 @@ where
let (activity_status, remote_publication_lease_targets) =
scanner_cycle_activity_status(store, distributed, &activity_before).await;
let all_bucket_names = all_buckets.iter().map(|bucket| bucket.name.clone()).collect::<Vec<_>>();
let completed_usage = completed_usage_candidate(
let completed_usage = completed_data_usage_info(
&results,
&ScannerSnapshotScope {
sources: &expected_sources,
@@ -646,10 +603,7 @@ where
dirty_usage_status,
activity_status,
);
let mut publication_expectation = None;
let observational_snapshot_published = if let Some(candidate) = completed_usage {
let (data_usage_info, expectation) = candidate.prepare(cycle_status);
publication_expectation = expectation;
let observational_snapshot_published = if let Some((data_usage_info, _)) = completed_usage {
if should_publish_observational_snapshot(cycle_status) {
publish_observational_snapshot(&updates, data_usage_info).await?
} else {
@@ -687,6 +641,5 @@ where
.with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements)
.with_failed_dirty_usage(!failed_buckets.is_empty())
.with_pending_maintenance_work(pending_maintenance_work)
.with_required_cycle_floor(required_cycle_floor)
.with_publication_expectation(publication_expectation))
.with_required_cycle_floor(required_cycle_floor))
}
+8 -136
View File
@@ -40,7 +40,6 @@ use time::OffsetDateTime;
use uuid::Uuid;
mod scoped_entry_fallback;
mod service_cohort;
#[derive(Clone)]
struct FixedWorkloadProvider {
@@ -397,10 +396,8 @@ async fn scoped_scan_production_entry_preserves_deep_and_full_maintenance_work()
scan_mode,
scan_scope: requested_scope,
persisted_usage_baseline: baseline,
observed_usage_candidate: None,
requires_full_scan,
resolved_scope_observer: Some(observer),
service_cohort: None,
},
),
)
@@ -453,16 +450,7 @@ async fn scoped_scan_same_cycle_maintenance_rewalks_after_root_delivery_failure(
.put_object(bucket, "initial", &mut reader, &ScannerObjectOptions::default())
.await
.expect("initial object should persist");
let lock = store.pools[0].disk_set[0]
.new_ns_lock(bucket, "initial")
.await
.expect("fixture namespace lock should be created");
let _settled = lock
.get_write_lock(Duration::from_secs(30))
.await
.expect("fixture rename tail should finish before the usage scan");
}
wait_for_namespace_commit_tails(&store).await;
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default());
let (updates, receiver) = mpsc::channel(1);
@@ -480,10 +468,8 @@ async fn scoped_scan_same_cycle_maintenance_rewalks_after_root_delivery_failure(
scan_mode: HealScanMode::Normal,
scan_scope: ScannerBucketScanScope::default(),
persisted_usage_baseline: None,
observed_usage_candidate: None,
requires_full_scan: false,
resolved_scope_observer: None,
service_cohort: None,
},
),
)
@@ -512,7 +498,6 @@ async fn scoped_scan_same_cycle_maintenance_rewalks_after_root_delivery_failure(
.put_object("cold-bucket", "new", &mut reader, &ScannerObjectOptions::default())
.await
.expect("new cold object should persist");
wait_for_namespace_commit_tails(&store).await;
record_dirty_usage_bucket("hot-bucket");
if scan_mode == HealScanMode::Normal && !requires_full_scan {
record_dirty_usage_bucket("cold-bucket");
@@ -533,10 +518,8 @@ async fn scoped_scan_same_cycle_maintenance_rewalks_after_root_delivery_failure(
scan_mode,
scan_scope: ScannerBucketScanScope::default(),
persisted_usage_baseline: None,
observed_usage_candidate: None,
requires_full_scan,
resolved_scope_observer: None,
service_cohort: None,
},
),
)
@@ -1004,8 +987,8 @@ fn dirty_usage_snapshot_clears_a_stably_absent_bucket_after_durable_save() {
assert!(dirty_usage_buckets().contains_key("temporarily-omitted"));
assert_eq!(dirty_usage_snapshot_status(&snapshot), DirtyUsageSnapshotStatus::Current);
let acknowledgements =
ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(snapshot.buckets.as_ref().clone())).clear_verified_usage();
let acknowledgements = ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(snapshot.buckets.as_ref().clone()))
.acknowledge_durable_usage();
assert!(acknowledgements.is_empty());
assert!(!dirty_usage_buckets().contains_key("temporarily-omitted"));
clear_dirty_usage_buckets_for_tests();
@@ -1111,7 +1094,7 @@ fn dirty_usage_is_acknowledged_only_after_durable_usage_confirmation() {
assert!(dirty_usage_buckets().contains_key("photos"));
let confirmed = ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(snapshot.buckets.as_ref().clone()));
let acknowledgements = confirmed.clear_verified_usage();
let acknowledgements = confirmed.acknowledge_durable_usage();
assert!(acknowledgements.is_empty());
assert!(!dirty_usage_buckets().contains_key("photos"));
clear_dirty_usage_buckets_for_tests();
@@ -1278,7 +1261,6 @@ async fn set_snapshot_reuse_requires_execution_identity_and_fences_stale_writers
ctx.clone(),
ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default()),
ScannerBucketScanPlan {
service_cohort: None,
buckets: Vec::new(),
all_buckets: Arc::new(Vec::new()),
scope: ScannerBucketScanScope::default(),
@@ -1344,8 +1326,7 @@ fn scoped_scan_requires_a_converged_complete_baseline_with_exact_set_provenance(
assert_eq!(
complete_scanner_cache_baseline_plan_digest(ScannerCacheBaselineProof {
authoritative_data: Some(&baseline),
observed_candidate_data: None,
data: Some(&baseline),
expected_sources: &expected_sources,
leader_epoch: 11,
want_cycle: 8,
@@ -1359,8 +1340,7 @@ fn scoped_scan_requires_a_converged_complete_baseline_with_exact_set_provenance(
let incomplete = bytes::Bytes::from(serde_json::to_vec(&incomplete).expect("test baseline should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(ScannerCacheBaselineProof {
authoritative_data: Some(&incomplete),
observed_candidate_data: None,
data: Some(&incomplete),
expected_sources: &expected_sources,
leader_epoch: 11,
want_cycle: 8,
@@ -1374,8 +1354,7 @@ fn scoped_scan_requires_a_converged_complete_baseline_with_exact_set_provenance(
let wrong_provenance = bytes::Bytes::from(serde_json::to_vec(&wrong_provenance).expect("test baseline should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(ScannerCacheBaselineProof {
authoritative_data: Some(&wrong_provenance),
observed_candidate_data: None,
data: Some(&wrong_provenance),
expected_sources: &expected_sources,
leader_epoch: 11,
want_cycle: 8,
@@ -1385,111 +1364,6 @@ fn scoped_scan_requires_a_converged_complete_baseline_with_exact_set_provenance(
);
}
#[test]
fn scoped_scan_accepts_only_a_complete_observation_tied_to_the_authoritative_baseline() {
let source = DataUsageCacheSource::new(1, 2);
let expected_sources = HashSet::from([source]);
let scan_plan_digest = DataUsageScanPlanDigest([9; 32]);
let authoritative = complete_usage_baseline(source, scan_plan_digest, 7, 11);
let authoritative_info =
serde_json::from_slice::<DataUsageInfo>(&authoritative).expect("authoritative baseline should decode");
let bootstrap_authoritative_info =
crate::scanner::scanner_usage_bootstrap_marker(SystemTime::UNIX_EPOCH + Duration::from_secs(9), Some(11));
let bootstrap_authoritative =
bytes::Bytes::from(serde_json::to_vec(&bootstrap_authoritative_info).expect("bootstrap baseline should encode"));
let mut observed_info = authoritative_info.clone();
observed_info.last_update = Some(SystemTime::UNIX_EPOCH + Duration::from_secs(11));
observed_info.scanner_cycle = Some(8);
observed_info.usage_snapshot_converged = Some(false);
observed_info.usage_snapshot_authoritative_baseline = Some(bootstrap_authoritative_info.snapshot_identity());
observed_info.usage_snapshot_set_states[0].scanner_cycle = Some(8);
let observed = bytes::Bytes::from(serde_json::to_vec(&observed_info).expect("observation should encode"));
macro_rules! proof {
($authoritative:expr, $candidate:expr) => {
ScannerCacheBaselineProof {
authoritative_data: Some($authoritative),
observed_candidate_data: $candidate,
expected_sources: &expected_sources,
leader_epoch: 11,
want_cycle: 9,
scan_plan_digest,
}
};
}
assert_eq!(
complete_scanner_cache_baseline_plan_digest(proof!(&bootstrap_authoritative, Some(&observed))),
Some(scan_plan_digest)
);
observed_info.usage_snapshot_authoritative_baseline = Some(DataUsageInfo::default().snapshot_identity());
let mismatched_baseline = bytes::Bytes::from(serde_json::to_vec(&observed_info).expect("observation should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(proof!(&bootstrap_authoritative, Some(&mismatched_baseline))),
None
);
observed_info = serde_json::from_slice(&observed).expect("observation should decode");
observed_info.usage_snapshot_partial = true;
let partial = bytes::Bytes::from(serde_json::to_vec(&observed_info).expect("partial observation should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(proof!(&bootstrap_authoritative, Some(&partial))),
None
);
observed_info = serde_json::from_slice(&observed).expect("observation should decode");
observed_info.usage_snapshot_converged = Some(true);
let converged = bytes::Bytes::from(serde_json::to_vec(&observed_info).expect("converged observation should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(proof!(&bootstrap_authoritative, Some(&converged))),
None
);
let mut legacy_authoritative = authoritative_info.clone();
legacy_authoritative.usage_snapshot_converged = None;
let mut stale_info = serde_json::from_slice::<DataUsageInfo>(&observed).expect("observation should decode");
stale_info.scanner_cycle = Some(7);
stale_info.usage_snapshot_set_states[0].scanner_cycle = Some(7);
stale_info.usage_snapshot_authoritative_baseline = Some(legacy_authoritative.snapshot_identity());
let legacy_authoritative =
bytes::Bytes::from(serde_json::to_vec(&legacy_authoritative).expect("legacy baseline should encode"));
let stale = bytes::Bytes::from(serde_json::to_vec(&stale_info).expect("stale observation should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(proof!(&legacy_authoritative, Some(&stale))),
None
);
let malformed = bytes::Bytes::from_static(b"not data usage json");
assert_eq!(
complete_scanner_cache_baseline_plan_digest(proof!(&bootstrap_authoritative, Some(&malformed))),
None
);
let mut nonconverged_authoritative = authoritative_info;
nonconverged_authoritative.usage_snapshot_converged = Some(false);
let mut observation_of_nonconverged_authoritative = nonconverged_authoritative.clone();
observation_of_nonconverged_authoritative.last_update = Some(SystemTime::UNIX_EPOCH + Duration::from_secs(12));
observation_of_nonconverged_authoritative.scanner_cycle = Some(8);
observation_of_nonconverged_authoritative.usage_snapshot_set_states[0].scanner_cycle = Some(8);
observation_of_nonconverged_authoritative.usage_snapshot_authoritative_baseline =
Some(nonconverged_authoritative.snapshot_identity());
let nonconverged_authoritative =
bytes::Bytes::from(serde_json::to_vec(&nonconverged_authoritative).expect("nonconverged baseline should encode"));
let observation_of_nonconverged_authoritative =
bytes::Bytes::from(serde_json::to_vec(&observation_of_nonconverged_authoritative).expect("observation should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(ScannerCacheBaselineProof {
authoritative_data: Some(&nonconverged_authoritative),
observed_candidate_data: Some(&observation_of_nonconverged_authoritative),
expected_sources: &expected_sources,
leader_epoch: 11,
want_cycle: 9,
scan_plan_digest,
}),
None
);
}
#[test]
fn scoped_scan_selects_only_current_dirty_buckets_after_baseline_validation() {
let source = DataUsageCacheSource::new(1, 2);
@@ -1503,8 +1377,7 @@ fn scoped_scan_selects_only_current_dirty_buckets_after_baseline_validation() {
true,
&[bucket_info("photos")],
ScannerCacheBaselineProof {
authoritative_data: Some(&baseline),
observed_candidate_data: None,
data: Some(&baseline),
expected_sources: &expected_sources,
leader_epoch: 11,
want_cycle: 8,
@@ -1542,8 +1415,7 @@ fn scoped_scan_baseline_work_proof_requires_uniform_known_set_identity() {
let data = Bytes::from(serde_json::to_vec(&candidate).expect("candidate should encode"));
assert_eq!(
complete_scanner_cache_baseline_plan_digest(ScannerCacheBaselineProof {
authoritative_data: Some(&data),
observed_candidate_data: None,
data: Some(&data),
expected_sources: &sources,
leader_epoch: 11,
want_cycle: 8,
@@ -126,9 +126,7 @@ async fn run_entry(store: &Arc<ECStore>, cycle: u64, selected: Option<&str>, exp
scan_mode: HealScanMode::Normal,
scan_scope: ScannerBucketScanScope::default(),
persisted_usage_baseline: root_before.0.clone().map(Bytes::from),
observed_usage_candidate: None,
requires_full_scan: false,
service_cohort: None,
resolved_scope_observer: Some(observer),
},
),
@@ -1,243 +0,0 @@
// Copyright 2026 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use super::*;
use crate::data_usage_define::{DATA_USAGE_OBJ_NAME_PATH, read_config_with_revision};
async fn create_cohort_bucket(store: &ECStore, bucket: &str) {
store
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("fixture bucket");
for set in store.all_set_disks() {
let mut reader = ScannerPutObjReader::from_vec(b"cohort".to_vec());
set.put_object(
bucket,
"initial",
&mut reader,
&ScannerObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("fixture object and all rename tails should persist");
}
}
async fn run_cohort_cycle(
store: &Arc<ECStore>,
cohort: Arc<StdMutex<ScannerServiceCohort>>,
cycle: u64,
budget: Arc<ScannerCycleBudget>,
) -> (ScannerCycleResult, Option<DataUsageInfo>) {
let root_before = read_config_with_revision(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("root before candidate");
let dirty_before = dirty_usage_buckets_for_tests();
let generation_before = dirty_usage_generation();
let (updates, mut receiver) = mpsc::channel(1);
let result = tokio::time::timeout(
Duration::from_secs(30),
nsscanner_with_storage_status_scoped(
store.as_ref(),
ScannerCycleRequest {
ctx: budget.token(),
budget,
updates,
want_cycle: cycle,
leader_epoch: 11,
scan_mode: HealScanMode::Normal,
scan_scope: ScannerBucketScanScope::default(),
persisted_usage_baseline: None,
observed_usage_candidate: None,
requires_full_scan: false,
service_cohort: Some(cohort),
resolved_scope_observer: None,
},
),
)
.await
.expect("cohort cycle should finish")
.expect("cohort cycle should return its status");
let usage = receiver.recv().await;
assert!(receiver.recv().await.is_none());
assert_eq!(
read_config_with_revision(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("root after candidate"),
root_before
);
assert_eq!(
dirty_usage_buckets_for_tests(),
dirty_before,
"candidate production must not ACK pending work"
);
assert_eq!(dirty_usage_generation(), generation_before);
(result, usage)
}
#[tokio::test]
#[serial]
async fn service_cohort_production_dispatch_services_waiters_across_sources() {
let (_dir, store) = setup_two_pool_scanner_store().await;
clear_dirty_usage_buckets_for_tests();
let hot = format!("a-hot-{}", Uuid::new_v4().simple());
let bootstrap = format!("z-bootstrap-{}", Uuid::new_v4().simple());
create_cohort_bucket(&store, &hot).await;
create_cohort_bucket(&store, &bootstrap).await;
let cohort = Arc::new(StdMutex::new(ScannerServiceCohort::default()));
let expected = store
.all_set_disks()
.iter()
.flat_map(|set| {
let source = DataUsageCacheSource::new(set.pool_index, set.set_index);
[(source, hot.clone()), (source, bootstrap.clone())]
})
.collect::<HashSet<_>>();
let mut seen = HashSet::new();
for cycle in 1..=4 {
// Both newly bootstrapped names and repeated dirty work sort ahead of
// the original bootstrap in the old dirty-first policy.
if cycle > 1 {
create_cohort_bucket(&store, &format!("b-new-{cycle}")).await;
}
record_dirty_usage_bucket(&hot);
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new_with_progress_tracking(
&ctx,
ScannerCycleBudgetConfig {
max_objects: Some(1),
..Default::default()
},
);
run_cohort_cycle(&store, cohort.clone(), cycle, budget.clone()).await;
assert!(budget.budget_elapsed());
assert_eq!(
budget.progress().0,
1,
"each round must reach one real object, not just mark an admission"
);
let admitted = cohort
.lock()
.expect("cohort lock")
.admitted_members()
.into_iter()
.collect::<HashSet<_>>();
let newly_admitted = admitted.difference(&seen).cloned().collect::<Vec<_>>();
assert_eq!(
newly_admitted.len(),
1,
"serial parent object budget must stop before another bucket admission"
);
seen.extend(newly_admitted);
}
assert!(
expected.is_subset(&seen),
"ongoing dirty/new bootstrap must not displace the original cohort"
);
// Admission fairness is not completed coverage: prior budgeted prefixes
// were observed under changing plans. A clean tail must not certify them.
let ctx = CancellationToken::new();
let (result, usage) =
run_cohort_cycle(&store, cohort, 5, ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default())).await;
assert_eq!(result.status, ScannerCycleStatus::Incomplete);
assert!(usage.is_none(), "neither source has a complete mixed-plan baseline to publish");
for set in store.all_set_disks() {
let mut cache = DataUsageCache::default();
cache
.load(set, &path_join_buf(&[&hot, DATA_USAGE_CACHE_NAME]))
.await
.expect("retained hot prefix");
assert!(!cache.info.snapshot_complete);
assert!(cache.info.scan_progress.is_some());
assert!(cache.info.scan_plan_digest.is_none(), "mixed coverage must remain non-authoritative");
}
clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn service_cohort_fresh_complete_aggregate_preserves_reordered_sources() {
let (_dir, store) = setup_two_pool_scanner_store().await;
clear_dirty_usage_buckets_for_tests();
for bucket in ["cohort-first", "cohort-second"] {
create_cohort_bucket(&store, bucket).await;
}
let sets = store.all_set_disks();
let listing = store
.list_bucket_for_scanner(&BucketOptions::default())
.await
.expect("fresh inventory");
let inventory = listing
.set_buckets
.into_iter()
.map(|set| (DataUsageCacheSource::new(set.pool_index, set.set_index), set.buckets))
.collect::<HashMap<_, _>>();
let cohort = Arc::new(StdMutex::new(ScannerServiceCohort::default()));
{
let mut cohort = cohort.lock().expect("cohort lock");
cohort.refresh(&inventory);
for bucket in &inventory[&DataUsageCacheSource::new(0, 0)] {
cohort.record_admitted(DataUsageCacheSource::new(0, 0), &bucket.name);
}
assert_eq!(cohort.order_set_indices(&sets), vec![1, 0]);
}
let ctx = CancellationToken::new();
let (result, usage) =
run_cohort_cycle(&store, cohort, 1, ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default())).await;
assert_eq!(result.status, ScannerCycleStatus::Complete);
let usage = usage.expect("fresh complete aggregate");
assert_eq!(usage.objects_total_count, 4);
assert!(usage.buckets_usage.values().all(|bucket| bucket.objects_count == 2));
assert_eq!(usage.usage_snapshot_set_states.len(), 2);
assert_eq!(
usage
.usage_snapshot_set_states
.iter()
.map(|set| (set.pool_index, set.set_index))
.collect::<HashSet<_>>(),
HashSet::from([(0, 0), (1, 0)])
);
clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn service_cohort_cancelled_dispatch_does_not_consume_waiters_or_leak_permits() {
let (_dir, store) = setup_two_pool_scanner_store().await;
clear_dirty_usage_buckets_for_tests();
let bucket = format!("cancel-{}", Uuid::new_v4().simple());
create_cohort_bucket(&store, &bucket).await;
let cohort = Arc::new(StdMutex::new(ScannerServiceCohort::default()));
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default());
ctx.cancel();
run_cohort_cycle(&store, cohort.clone(), 1, budget).await;
assert!(cohort.lock().expect("cohort lock").admitted_members().is_empty());
let report = rustfs_scanner_metrics::metrics::global_metrics().scanner_runtime_details_report();
assert!(report.active_bucket_drive_scans.is_empty());
let ctx = CancellationToken::new();
let (result, usage) =
run_cohort_cycle(&store, cohort, 1, ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default())).await;
assert_eq!(
result.status,
ScannerCycleStatus::Complete,
"cancellation must not block a subsequent scan"
);
assert_eq!(usage.expect("retry aggregate").objects_total_count, 2);
clear_dirty_usage_buckets_for_tests();
}
-3
View File
@@ -126,9 +126,6 @@ pub(crate) use rustfs_lifecycle::{
};
use rustfs_storage_api as storage_contracts;
#[cfg(test)]
pub(crate) type EcstoreHealResultItem = <EcstoreStore as storage_contracts::HealOperations>::HealResultItem;
pub(crate) mod owner {
#[cfg(test)]
pub(crate) use rustfs_ecstore::api::set_disk::test_util::hold_namespace_commit as ecstore_hold_namespace_commit;
@@ -168,8 +168,6 @@ New intents use a 15-minute expiry. A peer-only terminal tombstone is retained u
The coordinator creates its durable record and peer `Prepare` blocks new reference creation, drains exact tier-operation leases, and proves that edit/remove/clear will not strand authoritative references. Prepare, Commit, and Abort use all-node fanout rather than quorum: independent peer calls use a work-conserving concurrency limit of four, a 30-second per-peer deadline, and a 30-second fanout-wide deadline; Prepare is additionally capped by the intent expiry. The coordinator collects every completed outcome. A timed-out or otherwise ambiguous started Prepare is included in compensating Abort because cancellation does not prove the peer failed to persist its fence; peers not started before the fanout deadline make Prepare fail but do not require Abort. The coordinator then conditionally writes tier config, durably commits the coordinator intent, releases its exclusive guards, requires every prepared peer to commit, publishes the runtime candidate, and clears the block. Per-mutation sharded mutexes serialize local phases only; persisted intent plus tier-config ETag is authoritative.
Terminal recovery does not reinstall a process-local operation fence when the published in-memory manager has the exact committed candidate digest. This exception affects only ordinary tier-operation leases: recovery still replays peer `Commit`, retains and conditionally cleans the durable evidence, and blocks a new tier configuration mutation until the recovery snapshot is quiescent. A different local digest remains fenced until the committed candidate is safely published.
### Recovery decisions
| Observed durable state/input | Unique current owner | Current recovery decision | Destructive/config admission |
@@ -55,14 +55,6 @@ Standard S3 areas that must not be described as complete:
`excluded_tests.txt` holds tests that must not block the compatibility gate: vendor-specific or non-portable behavior, and intentionally unsupported product behavior such as ACL authorization.
## Intentional Deviations From AWS S3
Object keys are stored as file-system paths under each drive (`{drive}/{bucket}/{object}/xl.meta`), the same layout MinIO uses. The rules below exist to keep that layout unambiguous and are not compatibility gaps to close; clients that need the AWS behavior must adapt on their side.
| Behavior | RustFS | AWS S3 | Why |
|---|---|---|---|
| Object key with a `.` or `..` path segment, or an empty segment (`//`), such as `a//b/./c/../d` | `400 InvalidArgument` (`check_object_args` in `crates/ecstore/src/bucket/utils.rs`, mirroring MinIO `IsValidObjectPrefix`) | Accepted as an opaque key | A `..` segment would resolve to a parent directory and `.`/`//` segments would alias other keys on disk; encoding them would change the MinIO-compatible on-disk format. |
## Update Rule
When a feature starts passing, move its test entries from `unimplemented_tests.txt` to `implemented_tests.txt` and update the row here in the same PR. Do not change README wording beyond the supported coverage. Handler-level status (missing, stubbed, or diverging endpoints) is tracked in [minio-rustfs-router-compatibility.md](minio-rustfs-router-compatibility.md).
-2
View File
@@ -255,8 +255,6 @@ Source-backed responses report only what the source can vouch for: its ETag, its
- The source ETag is always recorded in internal metadata regardless of the policy, so an audit or a later comparison can still see it.
- `Last-Modified` of a pulled object is the local write time, not the source's. The source timestamp is preserved in metadata.
A preserved source ETag does not describe the local part layout. Part reads and object attributes use the stored parts, and replication uses multipart transport when their logical boundaries are available. Legacy compressed or encrypted objects without a multipart ETag can lack those boundaries; they retain their existing streaming PUT path and its 5 GiB limit. Their replication target may therefore store a different part layout.
## Metadata mapping
Copied to the local object:
@@ -179,14 +179,6 @@ The reset does not delete metadata files by hand and does not publish an authori
| data movement | wait for decommission or rebalance to leave the scanner metadata path, then retry |
| invalid scanner cycle state | run `POST /v3/scanner/cycle-state/reset` with `{"mode":"full-rescan"}` first |
## Cleanup With The Scanner Disabled
With `RUSTFS_SCANNER_ENABLED=false`, startup makes one controlled attempt to finish a previously persisted cycle reset whose validated recovery marker is already `cleanup-pending`. This is metadata cleanup only: it does not start the ordinary scanner loop, scan namespaces, accept a new reset request, or automatically perform a usage-state `full-rebuild`. Missing, merely `blocked`, unknown-version, unknown-phase, or corrupt markers do not authorize an automatic reset.
The attempt uses the existing leader lock and revalidates the observed marker revision and phase after acquiring it. A busy leader or data-movement pause leaves the marker intact and is reported through `cycle_recovery.state` and `cycle_recovery.reason` in the existing scanner status response. There is no automatic retry loop while disabled. After resolving the blocker, explicitly retry `POST /v3/scanner/cycle-state/reset` with `{"mode":"full-rescan"}`, or restart to make another controlled attempt. The v3 reset routes remain synchronous and return their existing successful HTTP 200 responses; no asynchronous HTTP 202 acceptance is introduced.
The startup probe is cancellation-aware and uses the existing cache persistence I/O timeout. Shutdown waits only for the existing server shutdown timeout. If the cleanup task cannot join in that window, the `scanner_cleanup_not_joined` warning means completion is unconfirmed, not drained. The task is not force-aborted or force-unlocked while its runtime remains alive; it retains its existing namespace/admission guards, and durable marker/fence state remains authoritative. Inspect status before retrying. This does not establish a hard deadline for an unresponsive storage operation or prove that I/O has drained when the process or runtime subsequently exits. Task-ownership timeout tests are not storage fsync, commit-tail, or process-crash durability evidence.
## Data Movement Pauses
RustFS uses a `global_pause` policy while pool decommission or rebalance can hide scanner metadata: usage publication, lifecycle discovery, tier cleanup discovery, scanner-originated heal and bitrot checks, and replication discovery are deferred together. A failed or canceled decommission remains a publication barrier until an operator retries or clears it. The same pause and estimate objects are included in `GET /v3/ilm/expiry/status`.
@@ -299,24 +291,6 @@ Heal knobs are environment-only and read by `HealConfig::default` (`crates/heal/
| `RUSTFS_HEAL_MRF_REPLAY_BATCH` | `256` (`DEFAULT_HEAL_MRF_REPLAY_BATCH`) | Intents per replay push round. |
| `RUSTFS_HEAL_DANGLING_DELETE_GRACE_SECS` | `3600` (`DEFAULT_HEAL_DANGLING_DELETE_GRACE_SECS`, `crates/ecstore/src/set_disk/core/io_primitives.rs`) | A recently modified object is never deleted as dangling inside this window; `0` disables the grace window. |
### Admin heal start, retries, and budgets
Admin heal has three separate budgets. Increasing one does not extend the others:
| Budget | Existing behavior |
|---|---|
| Control request | A start/query/cancel envelope has a bounded lifetime derived from the internode RPC timeout, with room for the transport response. It bounds control execution, not the admitted repair task. The HTTP caller can also stop waiting independently. |
| Task execution | `RUSTFS_HEAL_TASK_TIMEOUT_SECS` supplies the default of 300 seconds when the execution has no explicit timeout. Elapsed execution time is deducted before a recoverable scheduler retry, which keeps the request identity and remaining budget. Object/listing backoff and pressure pacing inside an execution consume that budget; time queued or in the scheduler's between-attempt backoff is not a new absolute wall-clock deadline. Zero is not an unlimited execution budget. |
| Object/listing retry | A recursive bucket/prefix traversal retries recoverable failures at most three times after the initial attempt, with 2/4/8-second delays plus the existing task-derived jitter. Object retries also obey the bounded delayed window and its 30-second age, checked at safe boundaries; listing retries keep their cursor. These limits do not extend the task execution budget or force an in-flight storage operation to finish within the retry age. |
For a large admin heal, select a finite configured execution budget appropriate to the expected work and contention, and inspect progress while it runs. Investigate stalled work and object failures before increasing this budget; a larger timeout must not hide lock or quorum problems. A longer HTTP or internode timeout does not keep a repair task alive past its execution budget. A successful start response returns the canonical token for accepted or merged work; it does not prove that repair has completed. Query that token without `forceStart`; terminal timeout/cancellation reports retain completed progress while the task report remains available.
Capability preflight runs before constructing and admitting a start request. However, the public `cluster heal coordination unavailable` error can also follow a transport failure or an invalid coordinator response after a request was sent. An unknown or lost HTTP response is therefore not proof that no task was admitted.
The coordinator can replay the result of the exact original RPC envelope within its existing replay lifetime and coordinator epoch. Reusing an ID with changed parameters or nonce is rejected. This bounded, process-local receipt cache is not a general HTTP idempotency key or a restart-surviving admission receipt. Each new HTTP start constructs a new request/envelope, and a fresh `forceStart` intentionally requests a distinct start: do not automatically resend it after an ambiguous response. A fresh non-forced request follows the configured overlap policy, rather than recovering the original receipt. The v3 API rejects `forceStart` combined with `clientToken` or `forceStop`.
Implementation references: `rustfs/src/admin/handlers/heal.rs` (`submit_cluster_heal_start`, `new_heal_control_metadata`), `rustfs/src/storage/rpc/node_service.rs` (`execute_heal_control_envelope_with_manager`), `crates/protos/src/lib.rs` (`heal_control_execution_timeout`), and `crates/heal/src/heal/task.rs` (`retry_request_with_remaining_timeout`, `remaining_timeout`, `bucket_object_retry_delay`). These contracts do not replace the separate real response-loss, long-task, or start-latency validation lanes.
### Running admin heal pacing
The manager passes its existing workload provider and a configuration snapshot into each admin execution. Bucket/prefix listing and object boundaries resample foreground pressure; erasure-set page workers also resample after earlier work releases page capacity. `High`, `Urgent`, and `force_start` do not exempt ordinary admin execution from this runtime pacing. The existing start-time bypass and overlap-control meanings are unchanged.
@@ -327,24 +301,6 @@ The pacing gate holds neither namespace locks nor I/O/page permits while sleepin
A missing provider, zero pause, or both class thresholds set to zero preserves unpaced execution. Missing counts follow the existing shared pressure interpreter; they are observations, not health, quorum or resource-ownership proof. The current provider exposes node-level workload classes, so this does not claim independent per-set foreground measurements or a hard global resource budget. Runtime waits increment `rustfs_heal_mainline_throttle_total` with `source=admin`, `result=delayed`, and a foreground-pressure or `recovery_window` reason. Real p99/throughput protection requires the separate W20 fixed-load ABBA measurements.
## Pending Heal Hints
The scanner's persisted pending-heal cache is a best-effort retry backstop, bounded to 10,000 hints per bucket and a 24-hour age limit. These limits do not authorize garbage collection of committed durable repair obligations. A durable owner must retain its independent replay record until verified object completion or an equivalent durable successor permits removal.
Accepted, merged, or policy-dropped admission does not clear an existing hint. Task completion and legacy repair notices also lack the incarnation, set scope, generation, and storage verification needed to prove repair responsibility was discharged. The legacy success path therefore retains hints conservatively; this is not a complete durable MRF handoff protocol.
Pending retries use their persisted attempt count and last-attempt timestamp, starting at 15 minutes and doubling up to six hours. Each bucket submits at most 128 due hints per cycle. Already admitted or policy-dropped hints retry at Low priority; queue-full hints keep High priority but obey the same due-time bound. A changed retry batch synchronizes its pending cache once, including when cancelled, rather than copying the entire table after every admission.
Rediscovery and admission observations update the recorded result but do not postpone an already armed retry. The actual retry loop advances the attempt count and timestamp before awaiting admission, so cancellation or repeated queue-full results cannot reset the retry budget.
| Producer | Current identity and compensation | Durable handoff boundary |
|---|---|---|
| Scanner corrupt metadata | Metadata kind with no invented version/set; an existing pending-cache hint remains available for bounded retries. | Cache publication is separate from MRF ingress. Its age/count limits mean it is not an irrevocable repair-obligation ledger. |
| Read decode failure | Decode kind, available version and erasure-set scope; a later failing read can rediscover the repair. | Nonblocking ingress and in-memory read-repair admission do not acknowledge durable acceptance. |
| Partial write | Partial-write kind, available version and erasure-set scope; an in-memory heal request is the fast path. | The caller's documented restart-survival requirement is not fulfilled by ignoring the ingress result or by removing the unaccepted journal record at manager admission. Verified durable ownership remains pending. |
Legacy notices carry only bucket/object/version, not a verified storage disposition, incarnation, scope, or durable responsibility generation. They are drained without clearing hints. Terminal callbacks release only their exact node-local ingress lease so rediscovery remains possible; lease generations are not durable successor receipts. Pending migration staging is not activated, and this change does not enable durable tombstones or garbage collection. Positive cleanup requires a storage-owner receipt with the complete responsibility identity and validated commit/fence evidence; neither task status nor the bounded diagnostic outcome window supplies it.
## Deliberate non-parity with MinIO
These differences from MinIO are design decisions, recorded so they are not re-filed as gaps.
+1 -1
View File
@@ -68,7 +68,7 @@ license = []
io-scheduler-debug = [] # Enable debug information in I/O scheduler
tracing-chunk-debug = [] # Enable per-chunk tracing in data plane (high noise, for debugging only)
full = ["metrics-gpu", "ftps", "swift", "webdav", "sftp", "pyroscope", "gcs"]
e2e-test-hooks = []
e2e-test-hooks = ["rustfs-ecstore/e2e-test-hooks"]
# Shortens Connect credentials only in debug E2E builds.
connect-e2e-short-credentials = []
# Builds the dedicated rustfs-cli-e2e target with a build-time public enrollment root.
+56 -144
View File
@@ -237,13 +237,18 @@ struct HealStartSuccess {
#[derive(Debug, Serialize)]
#[serde(rename_all = "camelCase")]
struct HealTaskStatus {
#[serde(flatten)]
payload: HealTaskStatusPayload,
summary: String,
#[serde(rename = "detail")]
failure_detail: String,
start_time: String,
#[serde(rename = "settings")]
heal_settings: HealOpts,
#[serde(skip_serializing_if = "Vec::is_empty")]
items: Vec<rustfs_madmin::heal_commands::HealResultItem>,
#[serde(skip_serializing_if = "std::ops::Not::not")]
truncated: bool,
#[serde(skip_serializing_if = "Option::is_none")]
progress: Option<serde_json::Value>,
}
#[derive(Debug, Serialize)]
@@ -1050,23 +1055,15 @@ async fn submit_cluster_heal_channel_command(
}
}
#[derive(Debug, Default, Serialize, Deserialize)]
#[derive(Debug, Deserialize)]
struct HealTaskStatusPayload {
#[serde(skip)]
adapted_detail: Option<String>,
summary: String,
#[serde(default, skip_serializing_if = "Vec::is_empty")]
#[serde(default)]
items: Vec<rustfs_madmin::heal_commands::HealResultItem>,
#[serde(default, skip_serializing_if = "std::ops::Not::not")]
#[serde(default)]
truncated: bool,
#[serde(default, skip_serializing_if = "Option::is_none")]
#[serde(default)]
progress: Option<serde_json::Value>,
#[serde(default, skip_serializing_if = "Option::is_none")]
outcome: Option<serde_json::Value>,
#[serde(default, rename = "nextSeq", alias = "next_seq", skip_serializing_if = "Option::is_none")]
next_seq: Option<u64>,
#[serde(default, rename = "minSeq", alias = "min_seq", skip_serializing_if = "Option::is_none")]
min_seq: Option<u64>,
}
#[cfg(test)]
@@ -1106,16 +1103,21 @@ fn encode_heal_start_success(client_token: String, client_address: String) -> S3
}
fn encode_heal_task_status(
mut payload: HealTaskStatusPayload,
summary: String,
failure_detail: String,
heal_settings: HealOpts,
items: Vec<rustfs_madmin::heal_commands::HealResultItem>,
truncated: bool,
progress: Option<serde_json::Value>,
) -> S3Result<Vec<u8>> {
let failure_detail = payload.adapted_detail.take().unwrap_or(failure_detail);
encode_json(&HealTaskStatus {
payload,
summary,
failure_detail,
start_time: current_rfc3339_time()?,
heal_settings,
items,
truncated,
progress,
})
}
@@ -1160,63 +1162,42 @@ fn build_heal_channel_request(hip: &HealInitParams) -> HealChannelRequest {
fn heal_channel_response_status(
response: &rustfs_heal_contracts::heal_channel::HealChannelResponse,
) -> S3Result<HealTaskStatusPayload> {
) -> (String, Vec<rustfs_madmin::heal_commands::HealResultItem>, bool, Option<serde_json::Value>) {
let Some(data) = response.data.as_deref() else {
return Ok(HealTaskStatusPayload {
summary: "running".to_string(),
..Default::default()
});
return ("running".to_string(), Vec::new(), false, None);
};
if let Ok(mut payload) = serde_json::from_slice::<HealTaskStatusPayload>(data)
&& matches!(payload.summary.as_str(), "running" | "finished" | "stopped" | "notFound")
if let Ok(payload) = serde_json::from_slice::<HealTaskStatusPayload>(data)
&& !payload.summary.is_empty()
{
let adapted = payload
.outcome
.as_ref()
.map(|outcome| {
rustfs_heal::heal::outcome::legacy_wire_status(&payload.summary, outcome, payload.truncated)
.map(|(summary, detail)| (summary.to_string(), detail))
})
.transpose();
if let Ok(adapted) = adapted {
if let Some((summary, detail)) = adapted {
payload.summary = summary;
payload.adapted_detail = detail;
}
return Ok(payload);
}
return (payload.summary, payload.items, payload.truncated, payload.progress);
}
if let Ok(summary @ ("running" | "finished" | "stopped" | "notFound")) = std::str::from_utf8(data) {
return Ok(HealTaskStatusPayload {
summary: summary.to_string(),
..Default::default()
});
}
Err(s3s::S3Error::with_message(
s3s::S3ErrorCode::InternalError,
"invalid heal status payload or unsupported summary",
))
let summary = std::str::from_utf8(data)
.ok()
.filter(|summary| !summary.is_empty())
.unwrap_or("running")
.to_string();
(summary, Vec::new(), false, None)
}
#[cfg(test)]
fn heal_channel_response_summary(response: &rustfs_heal_contracts::heal_channel::HealChannelResponse) -> String {
heal_channel_response_status(response).expect("valid status fixture").summary
heal_channel_response_status(response).0
}
#[cfg(test)]
fn heal_channel_response_items(
response: &rustfs_heal_contracts::heal_channel::HealChannelResponse,
) -> Vec<rustfs_madmin::heal_commands::HealResultItem> {
heal_channel_response_status(response).expect("valid status fixture").items
heal_channel_response_status(response).1
}
#[cfg(test)]
fn heal_channel_response_progress(
response: &rustfs_heal_contracts::heal_channel::HealChannelResponse,
) -> Option<serde_json::Value> {
heal_channel_response_status(response).expect("valid status fixture").progress
heal_channel_response_status(response).3
}
fn encode_background_heal_status(
@@ -1404,8 +1385,15 @@ impl Operation for HealHandler {
response.error.unwrap_or_else(|| "query heal status failed".to_string())
));
}
let payload = heal_channel_response_status(&response)?;
let body = encode_heal_task_status(payload, response.error.unwrap_or_default(), HealOpts::default())?;
let (summary, items, truncated, progress) = heal_channel_response_status(&response);
let body = encode_heal_task_status(
summary,
response.error.unwrap_or_default(),
HealOpts::default(),
items,
truncated,
progress,
)?;
info!(
event = EVENT_ADMIN_RESPONSE_EMITTED,
component = LOG_COMPONENT_ADMIN_API,
@@ -1442,8 +1430,8 @@ impl Operation for HealHandler {
let body = if client_token.is_empty() {
encode_heal_start_success(response.request_id, client_address)?
} else {
let payload = heal_channel_response_status(&response)?;
encode_heal_task_status(payload, response.error.unwrap_or_default(), hip.hs)?
let (summary, items, truncated, progress) = heal_channel_response_status(&response);
encode_heal_task_status(summary, response.error.unwrap_or_default(), hip.hs, items, truncated, progress)?
};
info!(
event = EVENT_ADMIN_RESPONSE_EMITTED,
@@ -2773,12 +2761,12 @@ mod tests {
#[test]
fn test_encode_heal_task_status_uses_client_wire_shape() {
let encoded = encode_heal_task_status(
super::HealTaskStatusPayload {
summary: "Heal status query accepted".to_string(),
..Default::default()
},
"Heal status query accepted".to_string(),
String::new(),
HealOpts::default(),
Vec::new(),
false,
None,
)
.expect("status response should serialize");
let json: serde_json::Value = serde_json::from_slice(&encoded).expect("json should deserialize");
@@ -2795,13 +2783,12 @@ mod tests {
#[test]
fn test_encode_heal_task_status_reports_truncated_items() {
let encoded = encode_heal_task_status(
super::HealTaskStatusPayload {
summary: "running".to_string(),
truncated: true,
..Default::default()
},
"running".to_string(),
"heal result items were truncated".to_string(),
HealOpts::default(),
Vec::new(),
true,
None,
)
.expect("truncated status response should serialize");
let json: serde_json::Value = serde_json::from_slice(&encoded).expect("json should deserialize");
@@ -2817,13 +2804,12 @@ mod tests {
"currentObject": "bucket-a/object-a"
});
let encoded = encode_heal_task_status(
super::HealTaskStatusPayload {
summary: "running".to_string(),
progress: Some(progress.clone()),
..Default::default()
},
"running".to_string(),
String::new(),
HealOpts::default(),
Vec::new(),
false,
Some(progress.clone()),
)
.expect("status response should serialize");
let json: serde_json::Value = serde_json::from_slice(&encoded).expect("json should deserialize");
@@ -2831,80 +2817,6 @@ mod tests {
assert_eq!(json["progress"], progress);
}
#[test]
fn outcome_v3_admin_forwards_outcome_cursors_and_progress_without_recounting() {
let cases: serde_json::Value =
serde_json::from_str(include_str!("../../../../crates/madmin/tests/fixtures/heal-outcome-v3.json"))
.expect("shared fixtures");
for case in cases.as_array().expect("cases") {
let expected = &case["response"];
let mut channel_payload = case.get("remoteResponse").unwrap_or(expected).clone();
let payload = channel_payload.as_object_mut().expect("payload");
let next_seq = payload.remove("nextSeq").expect("cursor");
let min_seq = payload.remove("minSeq").expect("cursor");
payload.insert("next_seq".to_string(), next_seq);
payload.insert("min_seq".to_string(), min_seq);
let response = rustfs_heal_contracts::heal_channel::HealChannelResponse {
request_id: "token".into(),
success: true,
data: Some(serde_json::to_vec(&channel_payload).expect("channel bytes")),
error: None,
};
let payload = super::heal_channel_response_status(&response).expect("valid owner payload");
let encoded =
encode_heal_task_status(payload, expected["detail"].as_str().expect("detail").into(), HealOpts::default())
.expect("public response");
let actual: serde_json::Value = serde_json::from_slice(&encoded).expect("public JSON");
for key in ["summary", "detail", "outcome", "progress", "truncated", "nextSeq", "minSeq"] {
assert_eq!(actual[key], expected[key], "{}: {key}", case["name"]);
}
assert_eq!(actual["progress"]["objectsHealed"], 7);
assert_eq!(actual["outcome"]["counters"]["healed"], 0);
}
}
#[test]
fn outcome_v3_rejects_corrupt_status_without_inventing_a_terminal() {
for data in [
br#"{"summary":"future_state"}"#.as_slice(),
br#"{"summary":"finished","nextSeq":9,"next_seq":8}"#.as_slice(),
br#"{"outcome":{"execution":{"state":"completed"}}}"#.as_slice(),
b"future_state".as_slice(),
] {
let response = rustfs_heal_contracts::heal_channel::HealChannelResponse {
request_id: "token".into(),
success: true,
data: Some(data.to_vec()),
error: None,
};
assert!(super::heal_channel_response_status(&response).is_err());
}
}
#[test]
fn outcome_v3_admin_preserves_future_nonterminal_without_validating_success() {
let outcome = serde_json::json!({
"execution": {"state": "future_execution", "extension": {"value": 7}},
"futureCounter": 9
});
let mut wire = serde_json::json!({"summary": "running", "outcome": outcome, "next_seq": 9, "min_seq": 4});
let mut response = rustfs_heal_contracts::heal_channel::HealChannelResponse {
request_id: "token".into(),
success: true,
data: Some(serde_json::to_vec(&wire).expect("future running wire")),
error: None,
};
let payload = super::heal_channel_response_status(&response).expect("future nonterminal is opaque");
let bytes = encode_heal_task_status(payload, String::new(), HealOpts::default()).expect("public nonterminal");
let public: serde_json::Value = serde_json::from_slice(&bytes).expect("public JSON");
assert_eq!(public["summary"], "running");
assert_eq!(public["outcome"], outcome);
assert_eq!((public["nextSeq"].as_u64(), public["minSeq"].as_u64()), (Some(9), Some(4)));
wire["summary"] = serde_json::json!("finished");
response.data = Some(serde_json::to_vec(&wire).expect("unprovable success wire"));
assert!(super::heal_channel_response_status(&response).is_err());
}
#[test]
fn test_build_heal_channel_request_preserves_safe_client_options() {
let hip = HealInitParams {
File diff suppressed because it is too large Load Diff
+4 -49
View File
@@ -63,7 +63,6 @@ const EVENT_ADMIN_REQUEST_STATE: &str = "admin_request_state";
const EVENT_ADMIN_REQUEST_REJECTED: &str = "admin_request_rejected";
const EVENT_ADMIN_REQUEST_FAILED: &str = "admin_request_failed";
const EVENT_ADMIN_RESPONSE_EMITTED: &str = "admin_response_emitted";
const POOL_ACTIVATION_FLEET_PROOF_REQUIRED: &str = "pool activation requires a live fleet capability proof";
fn admin_request_id(headers: &HeaderMap) -> Option<&str> {
headers
@@ -322,17 +321,6 @@ fn contextualize_admin_pool_api_error(
}
}
fn decommission_start_api_error(err: crate::storage_api::error::StorageError) -> ApiError {
if crate::storage_api::capacity::is_pool_activation_fleet_proof_error(&err) {
return ApiError {
code: S3ErrorCode::InternalError,
message: POOL_ACTIVATION_FLEET_PROOF_REQUIRED.to_string(),
source: Some(Box::new(err)),
};
}
ApiError::from(err)
}
fn decommission_admin_not_initialized_error_with_audit(operation: &str, audit: PoolAuditContext<'_>) -> S3Error {
error!(
event = EVENT_ADMIN_REQUEST_FAILED,
@@ -802,24 +790,7 @@ impl Operation for StartDecommission {
store
.decommission(ctx.clone(), pools_indices.clone())
.await
.map_err(|err| {
error!(
event = EVENT_ADMIN_REQUEST_FAILED,
component = LOG_COMPONENT_ADMIN_API,
subsystem = LOG_SUBSYSTEM_POOL_ADMIN,
operation = "start_decommission",
action = "start_decommission",
result = "failed",
reason = "storage_decommission_failed",
request_id = %request_id,
actor = %actor,
remote_addr = %remote_addr,
pool_indices = ?pools_indices,
error = %err,
"admin request failed"
);
decommission_start_api_error(err)
})
.map_err(ApiError::from)
.map_err(|err| contextualize_admin_pool_api_error(err, "start decommission", &pool_context))?;
}
}
@@ -1047,10 +1018,9 @@ impl Operation for ClearDecommission {
#[cfg(test)]
mod pools_handler_tests {
use super::{
AdminPoolStatus, Body, CancelDecommission, ClearDecommission, HeaderMap, ListPools, Method, Operation,
POOL_ACTIVATION_FLEET_PROOF_REQUIRED, Params, PoolAuditContext, S3ErrorCode, S3Request, StartDecommission,
StatusDecommission, StatusPool, Uri, contextualize_admin_pool_api_error,
decommission_admin_not_initialized_error_with_audit, decommission_peer_target, decommission_start_api_error,
AdminPoolStatus, Body, CancelDecommission, ClearDecommission, HeaderMap, ListPools, Method, Operation, Params,
PoolAuditContext, S3ErrorCode, S3Request, StartDecommission, StatusDecommission, StatusPool, Uri,
contextualize_admin_pool_api_error, decommission_admin_not_initialized_error_with_audit, decommission_peer_target,
has_duplicate_indices, parse_mutation_pool_query, parse_pool_idx_by_id, parse_status_pool_query,
pool_admin_missing_credentials_error, pool_admin_missing_credentials_error_with_request,
pool_admin_pool_index_error_with_audit, pool_admin_pool_not_found_error_with_audit,
@@ -1239,21 +1209,6 @@ mod pools_handler_tests {
);
}
#[test]
fn test_decommission_start_api_error_preserves_fleet_proof_retry_marker() {
let err = crate::storage_api::error::StorageError::other(POOL_ACTIVATION_FLEET_PROOF_REQUIRED);
let err = decommission_start_api_error(err);
assert_eq!(err.code, s3s::S3ErrorCode::InternalError);
assert_eq!(err.message, POOL_ACTIVATION_FLEET_PROOF_REQUIRED);
assert!(err.source.is_some());
let unrelated = decommission_start_api_error(crate::storage_api::error::StorageError::other("disk read failed"));
assert_eq!(unrelated.code, s3s::S3ErrorCode::InternalError);
assert_eq!(unrelated.message, "We encountered an internal error, please try again.");
}
#[test]
fn test_contextualize_admin_pool_api_error_preserves_source() {
let err = contextualize_admin_pool_api_error(
-14
View File
@@ -507,18 +507,6 @@ pub const ADMIN_ROUTE_POLICY_SPECS: &[AdminRouteSpec] = &[
LIST_TIER,
RouteRiskLevel::Sensitive,
),
admin(
HttpMethod::Post,
"/rustfs/admin/v3/ilm/recovery/records/{control_id}",
SET_TIER,
RouteRiskLevel::High,
),
admin(
HttpMethod::Get,
"/rustfs/admin/v3/ilm/recovery/exports/{export_id}",
SET_TIER,
RouteRiskLevel::High,
),
admin(HttpMethod::Post, "/rustfs/admin/v3/ilm/transition/run", SET_TIER, RouteRiskLevel::High),
admin(
HttpMethod::Get,
@@ -2185,8 +2173,6 @@ mod tests {
fn route_policy_uses_tier_actions_for_transition_routes() {
assert_action(HttpMethod::Get, "/rustfs/admin/v3/ilm/recovery/records", LIST_TIER);
assert_action(HttpMethod::Get, "/rustfs/admin/v3/ilm/recovery/records/{control_id}", LIST_TIER);
assert_action(HttpMethod::Post, "/rustfs/admin/v3/ilm/recovery/records/{control_id}", SET_TIER);
assert_action(HttpMethod::Get, "/rustfs/admin/v3/ilm/recovery/exports/{export_id}", SET_TIER);
assert_action(HttpMethod::Post, "/rustfs/admin/v3/ilm/transition/run", SET_TIER);
assert_action(HttpMethod::Get, "/rustfs/admin/v3/ilm/transition/jobs/{job_id}", SET_TIER);
assert_action(HttpMethod::Delete, "/rustfs/admin/v3/ilm/transition/jobs/{job_id}", SET_TIER);
@@ -215,16 +215,6 @@ fn expected_admin_route_matrix() -> Vec<RouteMatrixEntry> {
"/v3/ilm/recovery/records/{control_id}",
"/v3/ilm/recovery/records/aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
),
admin_route_sample(
Method::POST,
"/v3/ilm/recovery/records/{control_id}",
"/v3/ilm/recovery/records/aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa",
),
admin_route_sample(
Method::GET,
"/v3/ilm/recovery/exports/{export_id}",
"/v3/ilm/recovery/exports/bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb",
),
admin_route(Method::POST, "/v3/ilm/transition/run"),
admin_route_sample(
Method::GET,
@@ -952,16 +942,6 @@ fn test_register_routes_cover_representative_admin_paths() {
Method::GET,
&admin_path("/v3/ilm/recovery/records/aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"),
);
assert_route(
&router,
Method::POST,
&admin_path("/v3/ilm/recovery/records/aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"),
);
assert_route(
&router,
Method::GET,
&admin_path("/v3/ilm/recovery/exports/bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"),
);
assert_route(&router, Method::POST, &admin_path("/v3/ilm/transition/run"));
assert_route(
&router,
+1 -13
View File
@@ -18,7 +18,7 @@ use super::storage_api::bucket::replication::{self, BucketReplicationResyncStatu
use super::storage_api::bucket::target::{BucketTarget, BucketTargetType, BucketTargets};
use super::storage_api::bucket::target_sys::{
BucketTargetSys, PutObjectOptions, RemoveObjectOptions, S3ClientError, SsecPassthroughCapability, TargetClient,
VersionIdentityCapability, append_version_id_query,
append_version_id_query,
};
use super::storage_api::bucket::versioning_sys::BucketVersioningSys;
use super::storage_api::bucket::{AdminReplicationConfigExt as _, AdminVersioningConfigExt as _};
@@ -2100,18 +2100,6 @@ async fn check_replication_target(
}
_ => {}
}
// Same for the identity verdict: once a target is known to mint its own
// version ids, the worker locates replicas by content identity instead of
// re-driving PUTs whenever a version-addressed HEAD answers 404.
match (result.phases.version_fidelity.status, result.phases.version_fidelity.code) {
("OK", _) => {
BucketTargetSys::get().record_version_identity_capability(&target.arn, VersionIdentityCapability::Adopts);
}
("FAILED", Some(REPLICATION_CHECK_CODE_VERSION_MISMATCH)) => {
BucketTargetSys::get().record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn);
}
_ => {}
}
result
}
+1 -5
View File
@@ -211,7 +211,6 @@ pub(crate) mod bucket_target_sys {
pub(crate) type RemoveObjectOptions = super::ecstore_bucket::bucket_target_sys::RemoveObjectOptions;
pub(crate) type S3ClientError = super::ecstore_bucket::bucket_target_sys::S3ClientError;
pub(crate) type SsecPassthroughCapability = super::ecstore_bucket::bucket_target_sys::SsecPassthroughCapability;
pub(crate) type VersionIdentityCapability = super::ecstore_bucket::bucket_target_sys::VersionIdentityCapability;
pub(crate) type TargetClient = super::ecstore_bucket::bucket_target_sys::TargetClient;
}
@@ -234,10 +233,7 @@ pub(crate) mod lifecycle {
super::ecstore_bucket::lifecycle::bucket_lifecycle_ops::ManualTransitionRunOptions;
pub(crate) type ManualTransitionRunReport = super::ecstore_bucket::lifecycle::bucket_lifecycle_ops::ManualTransitionRunReport;
pub(crate) use super::ecstore_bucket::lifecycle::recovery_control::{
IlmRecoveryClassification, IlmRecoveryControlView, IlmRecoveryProtocol, inspect_recovery_control, list_recovery_controls,
};
pub(crate) use super::ecstore_bucket::lifecycle::recovery_export::{
IlmRecoveryExportObservation, create_recovery_export, inspect_recovery_export_observation, load_recovery_export,
IlmRecoveryClassification, IlmRecoveryProtocol, inspect_recovery_control, list_recovery_controls,
};
pub(crate) use super::ecstore_bucket::lifecycle::transition_transaction::{
TransitionOperatorDeleteResult, TransitionOperatorError, delete_transition_candidate_for_operator,
+6 -5
View File
@@ -4070,13 +4070,14 @@ mod tests {
abort_incomplete_multipart_upload: None,
del_marker_expiration: None,
filter: Some(s3s::dto::LifecycleRuleFilter {
and: Some(s3s::dto::LifecycleRuleAndOperator {
prefix: Some("logs/".to_string()),
..Default::default()
prefix: Some("logs/".to_string()),
tag: Some(s3s::dto::Tag {
key: Some("env".to_string()),
value: Some("prod".to_string()),
}),
..Default::default()
}),
id: Some("one-member-and".to_string()),
id: Some("two-predicates".to_string()),
noncurrent_version_expiration: None,
noncurrent_version_transitions: None,
prefix: None,
@@ -4086,7 +4087,7 @@ mod tests {
&ObjectLockConfiguration::default(),
)
.await
.expect_err("a Filter And with one predicate is a schema violation");
.expect_err("a Filter with two predicates is a schema violation");
assert_eq!(*lifecycle_validation_error(&malformed).code(), S3ErrorCode::MalformedXML);
let invalid_value = validate_lifecycle_config(
+35 -2
View File
@@ -26,13 +26,14 @@
//! server is not ready rather than that another server's global context applies.
use super::global::{AppContext, get_global_app_context};
use crate::app::storage_api::context::ECStore;
use crate::app::storage_api::context::{BootstrapLocalTarget, ECStore, InstanceContext};
use std::sync::{Arc, OnceLock};
/// Late-bound, per-server handle to the application context.
#[derive(Default)]
pub struct ServerContextSlot {
app_context: OnceLock<Arc<AppContext>>,
bootstrap_target: Option<BootstrapLocalTarget>,
heal_topology_fingerprint: Arc<tokio::sync::OnceCell<String>>,
}
@@ -50,15 +51,47 @@ impl ServerContextSlot {
pub fn new() -> Arc<Self> {
Arc::new(Self {
app_context: OnceLock::new(),
bootstrap_target: None,
heal_topology_fingerprint: Arc::new(tokio::sync::OnceCell::new()),
})
}
/// Bind the listener to its foundation before it can accept requests.
pub fn with_instance_context(ctx: Arc<InstanceContext>) -> Arc<Self> {
Arc::new(Self {
bootstrap_target: Some(BootstrapLocalTarget::new(ctx)),
..Self::default()
})
}
/// Install this server's application context (once). Returns `false` if
/// the slot was already installed; the first installation wins, matching
/// the process-global singleton's `get_or_init` semantics.
pub fn install(&self, context: Arc<AppContext>) -> bool {
self.app_context.set(context).is_ok()
self.try_install(context).is_ok()
}
/// Claim the slot before any process-global application publication.
/// Repeated installation, even of the same Arc, is an explicit conflict.
pub fn try_install(&self, context: Arc<AppContext>) -> std::io::Result<()> {
if self
.bootstrap_target
.as_ref()
.is_some_and(|target| !target.is_for_store(&context.object_store()))
{
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
"application context does not belong to this server foundation",
));
}
self.app_context.set(context).map_err(|_| {
std::io::Error::new(std::io::ErrorKind::AlreadyExists, "server application context is already installed")
})
}
/// Immutable, restricted startup capability; never resolves an ambient store.
pub fn bootstrap_target(&self) -> Option<BootstrapLocalTarget> {
self.bootstrap_target.clone()
}
/// This server's installed application context, if startup has completed.
+2 -2
View File
@@ -37,8 +37,8 @@ impl AppContext {
// also publishes to the process default (first server wins) so legacy
// free-function readers keep resolving the first server's context.
let context = Arc::new(AppContext::with_default_interfaces(store, iam, kms_interface));
publish_global_app_context(context.clone());
let _ = server_ctx.install(context);
server_ctx.try_install(context.clone())?;
publish_global_app_context(context);
Ok(())
}
}
+1 -1
View File
@@ -1261,7 +1261,7 @@ pub(crate) mod context {
pub(crate) use super::EndpointServerPools;
pub(crate) use super::bucket;
pub(crate) use super::runtime;
pub(crate) use crate::storage::storage_api::{ECStore, EndpointServerPools};
pub(crate) use crate::storage::storage_api::{BootstrapLocalTarget, ECStore, EndpointServerPools, InstanceContext};
#[cfg(test)]
pub(crate) use crate::storage::storage_api::{Endpoint, Endpoints, PoolEndpoints};
}
+3 -1
View File
@@ -36,7 +36,9 @@ use crate::server::{
};
use crate::storage_api::server::http as storage;
use crate::storage_api::server::http::rpc::InternodeRpcService;
#[cfg(test)]
use crate::storage_api::server::http::tonic_service::make_server;
use crate::storage_api::server::http::tonic_service::make_server_for_slot;
use crate::storage_api::server::http::{
ServerContextSlot, TONIC_RPC_PREFIX, normalize_tonic_rpc_audience, tonic_boot_epoch_challenge,
tonic_boot_epoch_response_headers, verify_tonic_rpc_signature_with_bootstrap,
@@ -1834,7 +1836,7 @@ fn process_connection(
// each service in the auth interceptor.
let rpc_max_message_size = rustfs_protos::internode_rpc_max_message_size();
let node_service = InterceptedService::new(
NodeServiceServer::new(make_server())
NodeServiceServer::new(make_server_for_slot(Arc::clone(&server_ctx)))
.max_decoding_message_size(rpc_max_message_size)
.max_encoding_message_size(rpc_max_message_size),
check_auth,
+1 -3
View File
@@ -124,9 +124,6 @@ pub(crate) async fn run_embedded_startup(args: EmbeddedStartupArgs) -> Result<Em
} else {
bootstrap_instance_ctx()
};
// This server's request-path context slot (backlog#1052 S2).
let server_ctx = ServerContextSlot::new();
let EmbeddedStartupConfig {
config,
identity,
@@ -151,6 +148,7 @@ pub(crate) async fn run_embedded_startup(args: EmbeddedStartupArgs) -> Result<Em
.await
.map_err(init_error)?;
let server_ctx = ServerContextSlot::with_instance_context(instance_ctx.clone());
let http_server = start_embedded_http_server(&config, listen_context.readiness.clone(), server_ctx.clone()).await?;
let shutdown_handle = http_server.shutdown_handle;
let bound_addr = http_server.bound_addr;
+51 -4
View File
@@ -62,6 +62,29 @@ fn emit_fatal_stderr(context: &str, error: impl std::fmt::Display) {
}
async fn async_main() -> Result<()> {
#[cfg(feature = "e2e-test-hooks")]
if let Ok(nonce) = std::env::var("RUSTFS_E2E_STARTUP_CAS_PROBE") {
let nonce = uuid::Uuid::parse_str(&nonce).map_err(Error::other)?;
// This precedes CLI parsing and observability, including `--help`.
println!(
"RUSTFS_E2E_STARTUP_CAS {}",
serde_json::json!({
"kind": "capability", "schema": "fresh-startup-cas/v1", "nonce": nonce,
})
);
return Ok(());
}
#[cfg(feature = "e2e-test-hooks")]
if let Ok(nonce) = std::env::var("RUSTFS_E2E_STARTUP_CAS_NONCE") {
let nonce = uuid::Uuid::parse_str(&nonce).map_err(Error::other)?;
let line = format!(
"RUSTFS_E2E_STARTUP_CAS {}\n",
serde_json::json!({
"kind": "observer-ready", "nonce": nonce, "pid": std::process::id(),
})
);
let _ = std::io::Write::write_all(&mut std::io::stderr().lock(), line.as_bytes());
}
hotpath::tokio_runtime!();
// Log container resource detection early in startup
@@ -141,10 +164,6 @@ async fn run(config: Config) -> Result<()> {
// the storage path explicitly (Phase 5 follow-up, backlog#1052); a future
// multi-instance server constructs its own context here instead.
let instance_ctx = bootstrap_instance_ctx();
// This server's request-path context slot (backlog#1052 S2): handed to the
// HTTP service now, installed once IAM bootstrap completes.
let server_ctx = ServerContextSlot::new();
let StartupListenContext {
readiness,
server_addr,
@@ -152,6 +171,7 @@ async fn run(config: Config) -> Result<()> {
} = init_startup_listen_context(&config, &instance_ctx).await?;
let endpoint_pools = init_startup_storage_foundation(&server_address, &config.volumes, &instance_ctx).await?;
let server_ctx = ServerContextSlot::with_instance_context(instance_ctx.clone());
let StartupHttpServers {
state_manager,
s3_shutdown_tx,
@@ -163,6 +183,33 @@ async fn run(config: Config) -> Result<()> {
shutdown_token: ctx,
} = init_startup_storage_runtime(server_addr, &endpoint_pools, readiness.clone(), instance_ctx).await?;
#[cfg(feature = "e2e-test-hooks")]
if let Ok(nonce) = std::env::var("RUSTFS_E2E_STARTUP_CAS_NONCE") {
let nonce = uuid::Uuid::parse_str(&nonce).map_err(Error::other)?;
let release = std::path::PathBuf::from(
std::env::var_os("RUSTFS_E2E_STARTUP_CAS_RELEASE")
.ok_or_else(|| Error::other("startup CAS fixture requires a release path"))?,
);
if server_ctx.installed_object_store().is_some() {
return Err(Error::other("startup CAS gate reached an installed slot"));
}
let line = format!(
"RUSTFS_E2E_STARTUP_CAS {}\n",
serde_json::json!({
"kind": "gate", "nonce": nonce, "pid": std::process::id(), "slot_installed": false,
})
);
let _ = std::io::Write::write_all(&mut std::io::stderr().lock(), line.as_bytes());
tokio::time::timeout(std::time::Duration::from_secs(180), async {
while !tokio::fs::try_exists(&release).await? {
tokio::time::sleep(std::time::Duration::from_millis(25)).await;
}
Ok::<_, Error>(())
})
.await
.map_err(|_| Error::other("startup CAS gate release timed out"))??;
}
let capacity_tasks = crate::capacity::capacity_integration::init_capacity_management_managed().await;
let service_runtime = init_startup_runtime_services(

Some files were not shown because too many files have changed in this diff Show More