mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-07 20:46:11 +00:00
Compare commits
99 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 0486ca9877 | |||
| 54716aa61c | |||
| 1a5e2b6256 | |||
| 474fcf78fb | |||
| 752d4a81ab | |||
| 0686277ee4 | |||
| 7373a5902e | |||
| e95a0ed6d9 | |||
| f0e0f5307d | |||
| 6855640192 | |||
| f11697d6e2 | |||
| 736b6a366c | |||
| 7b40b9503b | |||
| e22b879996 | |||
| b0dac1ac24 | |||
| 90e3ed701e | |||
| fd92853ac4 | |||
| bc0f608431 | |||
| f20a575994 | |||
| cc91483fac | |||
| c8fc50ada2 | |||
| 9badf3939c | |||
| c71c686e2a | |||
| 45f57ca57a | |||
| 265f429358 | |||
| 47f640f23e | |||
| 237c96dd3b | |||
| 647fe1a292 | |||
| 0c6314babc | |||
| fc76099280 | |||
| 899f81f3ad | |||
| 37bda24e1c | |||
| 4c2a0cdf9a | |||
| b370e746a0 | |||
| 95e6c89c1c | |||
| aafa7e2b7f | |||
| 5211b56277 | |||
| c9c6bb7a24 | |||
| ad0c44dc63 | |||
| 2d6f9417ff | |||
| 638da7a2e3 | |||
| 7c7b48aaca | |||
| c40988dae7 | |||
| 481c1b7939 | |||
| 26d827f946 | |||
| 376cb037a0 | |||
| 64829c704d | |||
| 17ba30f648 | |||
| 21015cfac8 | |||
| c04cd089e9 | |||
| bffdf0809f | |||
| 19a69ee897 | |||
| 1499295393 | |||
| 4c4dcb6f5e | |||
| 672087ec0d | |||
| 9dbeae1b45 | |||
| 6018dd372f | |||
| d633a635ec | |||
| fbd30a6f43 | |||
| a32360198c | |||
| 05efc584b0 | |||
| a70c96d520 | |||
| b0f68f5d0a | |||
| df554439b0 | |||
| 3b404e56c0 | |||
| ff00872922 | |||
| 4d1ce9618a | |||
| a2bad0953f | |||
| 14c99a994c | |||
| f6c2a9bfe0 | |||
| 640d7e0e3c | |||
| 409ac3de66 | |||
| 2a63fcbea6 | |||
| f4049598e4 | |||
| 1651541d38 | |||
| 85849788af | |||
| 975983abdd | |||
| cf1c45eb91 | |||
| 536c283716 | |||
| a04ddc237e | |||
| 7de6ac82e1 | |||
| 7993e11058 | |||
| 37ead2f69c | |||
| 1b1e590df7 | |||
| 086ee8e48a | |||
| 22bff27aee | |||
| 8a20498705 | |||
| c3e6d90c6f | |||
| d2af8f073e | |||
| 2baba1bda7 | |||
| bf2ca9113f | |||
| f474ea30d4 | |||
| 002ac9544c | |||
| d74d970e24 | |||
| cc15eae479 | |||
| c130d00d4b | |||
| e49d9cdea2 | |||
| f5778d8b97 | |||
| 9da42a899a |
@@ -47,6 +47,7 @@ script-tests: ## Run shell script tests
|
||||
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test
|
||||
./scripts/validate_object_data_cache_cold_stampede.sh --self-test
|
||||
./scripts/run_scanner_heal_evidence_case.sh --self-test
|
||||
|
||||
.PHONY: test
|
||||
test: core-deps script-tests ## Run all tests (needs cargo-nextest; RUSTFS_ALLOW_CARGO_TEST_FALLBACK=1 to override)
|
||||
|
||||
@@ -8,10 +8,26 @@
|
||||
"suite": "e2e_test",
|
||||
"name": "heal_erasure_disk_rebuild_test::tests::test_cluster_root_heal_recovers_remote_shards_after_background_target_restart",
|
||||
"oracle": "background-target-restart.json",
|
||||
"evidence": "process-restart",
|
||||
"unclean_shutdown_marker": false,
|
||||
"min_objects": 9,
|
||||
"max_objects": 65,
|
||||
"topology": {"nodes": 4, "drives_per_node": 1},
|
||||
"scope": "Target process restart, exact unversioned S3 bodies and replacement-disk shards; not power loss or EC8+4."
|
||||
},
|
||||
"background-target-crash": {
|
||||
"gate": "G14",
|
||||
"task": "W21",
|
||||
"lane": "e2e-nightly",
|
||||
"suite": "e2e_test",
|
||||
"name": "heal_erasure_disk_rebuild_test::tests::test_cluster_root_heal_recovers_remote_shards_after_background_target_crash",
|
||||
"oracle": "background-target-crash.json",
|
||||
"evidence": "process-crash-restart",
|
||||
"unclean_shutdown_marker": true,
|
||||
"min_objects": 9,
|
||||
"max_objects": 65,
|
||||
"topology": {"nodes": 4, "drives_per_node": 1},
|
||||
"scope": "Target process killed during partial background rebuild, real unclean-shutdown marker, exact unversioned S3 bodies and replacement-disk shards; not power loss or EC8+4."
|
||||
}
|
||||
},
|
||||
"release_pending": {
|
||||
|
||||
@@ -582,59 +582,13 @@ jobs:
|
||||
install-build-packaging-tools: 'false'
|
||||
|
||||
- name: Build debug binary
|
||||
run: |
|
||||
python3 - <<'PYBUILD'
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import subprocess
|
||||
|
||||
def git(*args):
|
||||
return subprocess.check_output(["git", *args], text=True).strip()
|
||||
|
||||
def sha256(path):
|
||||
digest = hashlib.sha256()
|
||||
with pathlib.Path(path).open("rb") as source:
|
||||
for chunk in iter(lambda: source.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
argv = ["cargo", "build", "-p", "rustfs", "--bins", "--features", "e2e-test-hooks"]
|
||||
commit, tree = git("rev-parse", "HEAD"), git("rev-parse", "HEAD^{tree}")
|
||||
clean_before = not git("status", "--porcelain", "--untracked-files=normal")
|
||||
if not clean_before:
|
||||
raise SystemExit("hooks binary requires a clean build checkout")
|
||||
lock_sha256 = sha256("Cargo.lock")
|
||||
lock_git_blob = git("hash-object", "Cargo.lock")
|
||||
rustc = subprocess.check_output(["rustc", "-vV"], text=True)
|
||||
host = next(line.removeprefix("host: ") for line in rustc.splitlines() if line.startswith("host: "))
|
||||
if os.environ.get("CARGO_BUILD_TARGET") or pathlib.Path(os.environ.get("CARGO_TARGET_DIR", "target")).resolve() != pathlib.Path("target").resolve():
|
||||
raise SystemExit("this artifact requires the native target/debug output")
|
||||
subprocess.run(argv, check=True)
|
||||
clean_after = not git("status", "--porcelain", "--untracked-files=normal")
|
||||
if not clean_after or commit != git("rev-parse", "HEAD") or tree != git("rev-parse", "HEAD^{tree}") or lock_sha256 != sha256("Cargo.lock"):
|
||||
raise SystemExit("hooks binary source changed while building")
|
||||
manifest = {
|
||||
"schema": 1, "commit": commit, "tree": tree,
|
||||
"clean_before": clean_before, "clean_after": clean_after,
|
||||
"lock_sha256": lock_sha256, "lock_git_blob": lock_git_blob,
|
||||
"argv": argv, "profile": "debug", "target": host,
|
||||
"features": ["e2e-test-hooks"],
|
||||
"rustc_verbose": rustc,
|
||||
"build_flags": {key: os.environ[key] for key in ("RUSTFLAGS", "CARGO_ENCODED_RUSTFLAGS", "CARGO_BUILD_TARGET", "CARGO_TARGET_DIR", "RUSTUP_TOOLCHAIN") if key in os.environ},
|
||||
"binary_sha256": sha256("target/debug/rustfs"),
|
||||
}
|
||||
pathlib.Path("target/debug/rustfs.e2e-startup-cas-build.json").write_text(json.dumps(manifest, indent=2) + "\n")
|
||||
PYBUILD
|
||||
run: cargo build -p rustfs --bins --features e2e-test-hooks
|
||||
|
||||
- name: Upload debug binary
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-debug-binary
|
||||
path: |
|
||||
target/debug/rustfs
|
||||
target/debug/rustfs.e2e-startup-cas-build.json
|
||||
path: target/debug/rustfs
|
||||
if-no-files-found: error
|
||||
retention-days: 1
|
||||
|
||||
@@ -900,6 +854,12 @@ jobs:
|
||||
run: |
|
||||
sudo apt-get install -y iptables
|
||||
sudo -n iptables --version
|
||||
# The endpoint-blackhole heal scenario needs CAP_NET_ADMIN. Containerised
|
||||
# runners can run iptables but not touch the rule set; the test then logs
|
||||
# a skip instead of failing, so surface that here where it is visible.
|
||||
if ! sudo -n iptables -w 5 -S OUTPUT >/dev/null 2>&1; then
|
||||
echo "::warning::iptables cannot read the OUTPUT chain on this runner (no CAP_NET_ADMIN); the endpoint-blackhole heal scenario will be skipped"
|
||||
fi
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6.3.0
|
||||
@@ -952,36 +912,6 @@ jobs:
|
||||
- name: Make binary executable
|
||||
run: chmod +x ./target/debug/rustfs
|
||||
|
||||
- name: Preserve startup CAS binary input
|
||||
env:
|
||||
STARTUP_CAS_INPUT: ${{ runner.temp }}/rustfs-startup-cas-input
|
||||
run: |
|
||||
python3 - <<'PYINPUT'
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import shutil
|
||||
import subprocess
|
||||
|
||||
source = pathlib.Path("target/debug/rustfs")
|
||||
manifest_path = source.with_name("rustfs.e2e-startup-cas-build.json")
|
||||
manifest = json.loads(manifest_path.read_text())
|
||||
target = pathlib.Path(os.environ["STARTUP_CAS_INPUT"])
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
binary = target / "rustfs"
|
||||
shutil.copy2(source, binary)
|
||||
digest = hashlib.sha256()
|
||||
with binary.open("rb") as stream:
|
||||
for chunk in iter(lambda: stream.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
commit = subprocess.check_output(["git", "rev-parse", "HEAD"], text=True).strip()
|
||||
if manifest["binary_sha256"] != digest.hexdigest() or manifest["commit"] != commit:
|
||||
raise SystemExit("downloaded hooks binary identity mismatch")
|
||||
shutil.copy2(manifest_path, target / manifest_path.name)
|
||||
binary.chmod(0o755)
|
||||
PYINPUT
|
||||
|
||||
- name: Verify e2e full membership
|
||||
env:
|
||||
NEXTEST_LISTING: ${{ runner.temp }}/rustfs-e2e-full-list.json
|
||||
@@ -994,10 +924,6 @@ jobs:
|
||||
# extend that filter, never add ad-hoc e2e jobs here. Reuses the downloaded
|
||||
# debug binary; each test spawns its own rustfs server on a random port.
|
||||
- name: Run e2e full suite
|
||||
env:
|
||||
RUSTFS_E2E_STARTUP_CAS_BINARY: ${{ runner.temp }}/rustfs-startup-cas-input/rustfs
|
||||
RUSTFS_E2E_STARTUP_CAS_BUILD_MANIFEST: ${{ runner.temp }}/rustfs-startup-cas-input/rustfs.e2e-startup-cas-build.json
|
||||
RUSTFS_E2E_STARTUP_CAS_ARTIFACT_DIR: ${{ runner.temp }}/rustfs-startup-cas-evidence
|
||||
run: cargo nextest run --profile e2e-full -p e2e_test
|
||||
|
||||
- name: Upload junit
|
||||
@@ -1010,17 +936,6 @@ jobs:
|
||||
${{ runner.temp }}/rustfs-e2e-full-list.json
|
||||
retention-days: 7
|
||||
|
||||
- name: Upload startup CAS evidence
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: fresh-startup-cas-evidence-${{ github.run_number }}
|
||||
path: |
|
||||
${{ runner.temp }}/rustfs-startup-cas-evidence
|
||||
${{ runner.temp }}/rustfs-startup-cas-input/rustfs.e2e-startup-cas-build.json
|
||||
if-no-files-found: warn
|
||||
retention-days: 7
|
||||
|
||||
e2e-tests-rio-v2:
|
||||
name: End-to-End Tests (rio-v2)
|
||||
# Inherits the schedule/dispatch-only gate through needs: on every other
|
||||
|
||||
@@ -121,6 +121,22 @@ jobs:
|
||||
create_latest=false
|
||||
source_ref="$GITHUB_SHA"
|
||||
|
||||
# Pre-GA policy: until the first stable (vX.Y.Z) tag exists, every
|
||||
# prerelease (alpha/beta/rc) also moves `latest`, so users pulling
|
||||
# `latest` get the newest test build. Once a stable tag is published
|
||||
# this returns false and `latest` follows stable releases only.
|
||||
prerelease_moves_latest() {
|
||||
local stable_tags
|
||||
stable_tags=$(git ls-remote --tags --refs origin 2>/dev/null \
|
||||
| awk '{print $2}' \
|
||||
| grep -E '^refs/tags/v?[0-9]+\.[0-9]+\.[0-9]+$' || true)
|
||||
if [[ -z "$stable_tags" ]]; then
|
||||
return 0
|
||||
fi
|
||||
echo "ℹ️ Stable release tag(s) already exist; prereleases no longer update latest"
|
||||
return 1
|
||||
}
|
||||
|
||||
if [[ "${{ github.event_name }}" == "workflow_run" ]]; then
|
||||
# Triggered by build workflow completion
|
||||
echo "🔗 Triggered by build workflow completion"
|
||||
@@ -184,8 +200,8 @@ jobs:
|
||||
if [[ "$version" == *"alpha"* ]] || [[ "$version" == *"beta"* ]] || [[ "$version" == *"rc"* ]]; then
|
||||
build_type="prerelease"
|
||||
is_prerelease=true
|
||||
# Current policy: create latest tags for stable releases and selected prereleases (alpha/beta).
|
||||
if [[ "$version" == *"alpha"* ]] || [[ "$version" == *"beta"* ]]; then
|
||||
# Pre-GA policy: prereleases update latest until the first stable tag exists.
|
||||
if prerelease_moves_latest; then
|
||||
create_latest=true
|
||||
echo "🧪 Building Docker image for prerelease: $version (creating latest tag)"
|
||||
else
|
||||
@@ -243,8 +259,8 @@ jobs:
|
||||
v*alpha*|v*beta*|v*rc*|*alpha*|*beta*|*rc*)
|
||||
build_type="prerelease"
|
||||
is_prerelease=true
|
||||
# Current policy: create latest tags for stable releases and selected prereleases (alpha/beta).
|
||||
if [[ "$version" == *"alpha"* ]] || [[ "$version" == *"beta"* ]]; then
|
||||
# Pre-GA policy: prereleases update latest until the first stable tag exists.
|
||||
if prerelease_moves_latest; then
|
||||
create_latest=true
|
||||
echo "🧪 Building with prerelease version: $input_version (creating latest tag)"
|
||||
else
|
||||
@@ -394,11 +410,13 @@ jobs:
|
||||
TAG_BASE="${VERSION}${VARIANT_SUFFIX}"
|
||||
TAGS="${{ env.REGISTRY_DOCKERHUB }}:$TAG_BASE,${{ env.REGISTRY_GHCR }}:$TAG_BASE,${{ env.REGISTRY_QUAY }}:$TAG_BASE"
|
||||
|
||||
# Add channel tags for prereleases and latest for stable
|
||||
# Add latest when requested (stable releases, and prereleases before GA)
|
||||
if [[ "$CREATE_LATEST" == "true" ]]; then
|
||||
# Create latest tags for stable releases and selected prereleases when CREATE_LATEST=true.
|
||||
TAGS="$TAGS,${{ env.REGISTRY_DOCKERHUB }}:latest${VARIANT_SUFFIX},${{ env.REGISTRY_GHCR }}:latest${VARIANT_SUFFIX},${{ env.REGISTRY_QUAY }}:latest${VARIANT_SUFFIX}"
|
||||
elif [[ "$BUILD_TYPE" == "prerelease" ]]; then
|
||||
fi
|
||||
|
||||
# Always add the channel tag for prereleases, independent of latest
|
||||
if [[ "$BUILD_TYPE" == "prerelease" ]]; then
|
||||
# Prerelease channel tags (alpha, beta, rc)
|
||||
if [[ "$VERSION" == *"alpha"* ]]; then
|
||||
CHANNEL="alpha"
|
||||
@@ -555,7 +573,7 @@ jobs:
|
||||
"prerelease")
|
||||
echo "🧪 Prerelease Docker image has been built with ${VERSION} tags"
|
||||
echo "⚠️ This is a prerelease image - use with caution"
|
||||
# Create latest tags for stable releases and selected prereleases when CREATE_LATEST=true.
|
||||
# Prereleases move latest until the first stable tag exists (pre-GA policy).
|
||||
if [[ "$CREATE_LATEST" == "true" ]]; then
|
||||
echo "🏷️ Latest tag has been created for prerelease: $VERSION"
|
||||
else
|
||||
|
||||
@@ -17,7 +17,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: overtrue/repo-visuals-action@fd79cba437ecfac933d00a69add17eb95d3939c3 # v1.3.1
|
||||
- uses: overtrue/repo-visuals-action@ee2c632f6ce617e851fb46ea935ee8af762ebb93 # v1.4.0
|
||||
with:
|
||||
github-token: ${{ github.token }}
|
||||
output-branch: star-history
|
||||
|
||||
@@ -8,6 +8,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
- **Multipart admission queue**: an `UploadPart` waiting for a foreground write permit now waits at most 10 s by default (`RUSTFS_PUT_MULTIPART_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS`, previously 30 s), so a queued part returns S3 `SlowDown` before the client's socket write timeout drops the connection. Separately, the API listener no longer forces a 4 MiB `SO_RCVBUF` on every accepted socket (kernel autotuning applies; `RUSTFS_HTTP_SOCKET_RECV_BUFFER_BYTES` restores a fixed size), so a queued part no longer lets up to 8 MiB of unread body accumulate in kernel memory per connection, which is what throttled whole nodes under SDK-default multipart concurrency. Fixes #7385.
|
||||
- **Helm Ingress**: `customAnnotations` are now merged with class-specific annotations (nginx/traefik) instead of being ignored when `ingress.className` is set.
|
||||
- **Per-pool erasure parity**: Erasure parity (STANDARD and reduced-redundancy) is now resolved independently for every pool instead of reusing the first pool's value. A heterogeneous topology — for example a 4-drive pool plus a 2-drive pool created during expansion — previously inherited the first pool's parity and could resolve to zero data shards in the smaller pool, panicking Reed-Solomon construction on write. Automatic parity now resolves per pool (for example `2+2` in the 4-drive pool and `1+1` in the 2-drive pool). Fixes #4801.
|
||||
|
||||
|
||||
Generated
+28
-28
@@ -2527,18 +2527,18 @@ checksum = "790eea4361631c5e7d22598ecd5723ff611904e3344ce8720784c93e3d83d40b"
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-channel"
|
||||
version = "0.5.16"
|
||||
version = "0.5.17"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d85363c37faeca707aef026efa9f3b34d077bce547e48f770770625c6013679e"
|
||||
checksum = "98b0cc327b5bc766e7fda9c9260cc0fa81b43a8e240440422dff70788e3f9ef1"
|
||||
dependencies = [
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-deque"
|
||||
version = "0.8.7"
|
||||
version = "0.8.8"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5181e0de7b61eb03a81e347d6dd8797bae9da5146707b51077e2d71a54ec0ceb"
|
||||
checksum = "622f3fc73690be383c7214310406f28a90e6edeadc3cea882f9d71e495b9711a"
|
||||
dependencies = [
|
||||
"crossbeam-epoch",
|
||||
"crossbeam-utils",
|
||||
@@ -2546,27 +2546,27 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-epoch"
|
||||
version = "0.9.20"
|
||||
version = "0.9.21"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2d6914041f254d6e9176c01941b21115dcfb7089e55135a35411081bd106ef3f"
|
||||
checksum = "dc74980687109a3b14c72fd458107bf0baa1da1a1a805e178d15501ba9b86d9d"
|
||||
dependencies = [
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-queue"
|
||||
version = "0.3.13"
|
||||
version = "0.3.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "803d13fb3b09d88be9f4dbc29062c66b19bf7170867ceb746d2a8689bf6c7a26"
|
||||
checksum = "03e8bd762f7479489c70ed6c768ddca99d7296857de437a68dcb2a94365b3fae"
|
||||
dependencies = [
|
||||
"crossbeam-utils",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "crossbeam-utils"
|
||||
version = "0.8.22"
|
||||
version = "0.8.23"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17"
|
||||
checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6"
|
||||
|
||||
[[package]]
|
||||
name = "crunchy"
|
||||
@@ -3673,9 +3673,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "der"
|
||||
version = "0.8.1"
|
||||
version = "0.8.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a69dedd701da44b0536442edf09c81a64b0ab97a7a4a5e3d1971f00027cbc63d"
|
||||
checksum = "a878c850e9e421b20262e9b41f9c860e4785fa07541c266b62ff9d1ef998a80a"
|
||||
dependencies = [
|
||||
"const-oid 0.10.2",
|
||||
"pem-rfc7468 1.0.0",
|
||||
@@ -4057,7 +4057,6 @@ dependencies = [
|
||||
"sha1 0.11.0",
|
||||
"sha2 0.11.0",
|
||||
"suppaftp",
|
||||
"tempfile",
|
||||
"time",
|
||||
"tokio",
|
||||
"tokio-stream",
|
||||
@@ -4091,7 +4090,7 @@ version = "0.17.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c0681a4fc24c767085329728d8dfba959af91228aa4610cca4f8ce317ba46ae0"
|
||||
dependencies = [
|
||||
"der 0.8.1",
|
||||
"der 0.8.2",
|
||||
"digest 0.11.3",
|
||||
"elliptic-curve 0.14.1",
|
||||
"rfc6979 0.6.0",
|
||||
@@ -5735,9 +5734,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "ipnet"
|
||||
version = "2.12.1"
|
||||
version = "2.12.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6a756c3fac73139e83f14c2d742155dd2b78d3ee56597b419a0579b7bdd6dd78"
|
||||
checksum = "791930b43c0d5973160d90a8f3894509f2b273430f5c5c73b668636d0287c5c0"
|
||||
dependencies = [
|
||||
"serde",
|
||||
]
|
||||
@@ -6140,9 +6139,9 @@ checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||
|
||||
[[package]]
|
||||
name = "libflate"
|
||||
version = "2.3.1"
|
||||
version = "2.3.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a4da9b700e758e57152a1fd1c52cbdc5727c1aa6d8743dc1acda917398f1d76c"
|
||||
checksum = "561a8da1a50e1428d3c51321dafeca849df992a5bb67720c386131234caba82e"
|
||||
dependencies = [
|
||||
"adler32",
|
||||
"crc32fast",
|
||||
@@ -7934,7 +7933,7 @@ version = "0.8.0-rc.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "986d2e952779af96ea048f160fd9194e1751b4faea78bcf3ceb456efe008088e"
|
||||
dependencies = [
|
||||
"der 0.8.1",
|
||||
"der 0.8.2",
|
||||
"spki 0.8.0",
|
||||
]
|
||||
|
||||
@@ -7977,7 +7976,7 @@ dependencies = [
|
||||
"aes 0.9.3",
|
||||
"aes-gcm",
|
||||
"cbc 0.2.1",
|
||||
"der 0.8.1",
|
||||
"der 0.8.2",
|
||||
"pbkdf2 0.13.0",
|
||||
"rand_core 0.10.1",
|
||||
"scrypt 0.12.0",
|
||||
@@ -8001,7 +8000,7 @@ version = "0.11.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "451913da69c775a56034ea8d9003d27ee8948e12443eae7c038ba100a4f21cb7"
|
||||
dependencies = [
|
||||
"der 0.8.1",
|
||||
"der 0.8.2",
|
||||
"pkcs5 0.8.1",
|
||||
"rand_core 0.10.1",
|
||||
"spki 0.8.0",
|
||||
@@ -8943,9 +8942,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "redis"
|
||||
version = "1.6.0"
|
||||
version = "1.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e37a4ca5c6ca42aa3e6df2fd32b987a65d32a4c2159a6f3fe0fd1df306a2658f"
|
||||
checksum = "2acbc41a996f7652b2ddd9dfd98cc4ff602cfd742ae35382f07f608405ab50ed"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arcstr",
|
||||
@@ -9327,7 +9326,7 @@ dependencies = [
|
||||
"curve25519-dalek 5.0.0",
|
||||
"data-encoding",
|
||||
"delegate",
|
||||
"der 0.8.1",
|
||||
"der 0.8.2",
|
||||
"digest 0.11.3",
|
||||
"ecdsa 0.17.0",
|
||||
"ed25519-dalek 3.0.0",
|
||||
@@ -10000,6 +9999,7 @@ dependencies = [
|
||||
"rustls-pki-types",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"serde_with",
|
||||
"serial_test",
|
||||
"sha1 0.11.0",
|
||||
"sha2 0.11.0",
|
||||
@@ -10944,9 +10944,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-uring"
|
||||
version = "0.2.1"
|
||||
version = "0.2.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0486e62d0efe25db95c00aeacb2da84368adcba299216cda99fcb11328061c84"
|
||||
checksum = "b29bc57b4bd62a73f4fae408b536adf578332e50e464797d09dc2382c7cb68c2"
|
||||
dependencies = [
|
||||
"io-uring",
|
||||
"libc",
|
||||
@@ -11421,7 +11421,7 @@ checksum = "d56d437c2f19203ce5f7122e507831de96f3d2d4d3be5af44a0b0a09d8a80e4d"
|
||||
dependencies = [
|
||||
"base16ct 1.0.0",
|
||||
"ctutils",
|
||||
"der 0.8.1",
|
||||
"der 0.8.2",
|
||||
"hybrid-array",
|
||||
"subtle",
|
||||
"zeroize",
|
||||
@@ -11995,7 +11995,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1d9efca8738c78ee9484207732f728b1ef517bbb1833d6fc0879ca898a522f6f"
|
||||
dependencies = [
|
||||
"base64ct",
|
||||
"der 0.8.1",
|
||||
"der 0.8.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
+6
-5
@@ -191,6 +191,7 @@ rmp = { version = "0.8.15" }
|
||||
rmp-serde = { version = "1.3.1" }
|
||||
serde = { version = "1.0.229" }
|
||||
serde_ignored = { version = "0.1" }
|
||||
serde_with = { version = "3", default-features = false, features = ["macros", "std"] }
|
||||
serde_json = { version = "1.0.151" }
|
||||
serde_urlencoded = "0.7.1"
|
||||
|
||||
@@ -256,10 +257,10 @@ clap = { version = "4.6.6" }
|
||||
const-str = { version = "1.1.0" }
|
||||
convert_case = "0.12.0"
|
||||
criterion = { version = "0.8" }
|
||||
crossbeam-queue = "0.3.13"
|
||||
crossbeam-channel = "0.5.16"
|
||||
crossbeam-deque = "0.8.7"
|
||||
crossbeam-utils = "0.8.22"
|
||||
crossbeam-queue = "0.3.14"
|
||||
crossbeam-channel = "0.5.17"
|
||||
crossbeam-deque = "0.8.8"
|
||||
crossbeam-utils = "0.8.23"
|
||||
datafusion = { default-features = false, version = "55.0.0" }
|
||||
derive_builder = "0.20.2"
|
||||
enumset = "1.1.14"
|
||||
@@ -306,7 +307,7 @@ rustfs-erasure-codec = { version = "8.0.2" }
|
||||
reed-solomon-simd = "3.1.0"
|
||||
regex = { version = "1.13.1" }
|
||||
rumqttc = { package = "rumqttc-next", version = "0.34.0" }
|
||||
redis = { version = "1.6.0" }
|
||||
redis = { version = "1.7.0" }
|
||||
rustify = { version = "0.7", default-features = false }
|
||||
rustix = { version = "1.1.4" }
|
||||
rust-embed = { version = "8.12.0" }
|
||||
|
||||
@@ -109,6 +109,15 @@ Star RustFS on GitHub and be instantly notified of new releases.
|
||||
|
||||
## Quickstart
|
||||
|
||||
> [!IMPORTANT]
|
||||
> **Pool expansion notice:**
|
||||
>
|
||||
> - A single-node single-drive (SNSD) deployment is supported only as a standalone local path. It cannot expand in place or be added as a Pool. To move to a multi-drive topology, create a new deployment and migrate data through S3.
|
||||
> - Keep an existing multi-drive Pool's endpoints and Erasure Set width unchanged; expand by appending a new Pool. With ellipsis-based expansion, every Pool argument must contain an ellipsis expression and expand to at least two drive endpoints.
|
||||
> - Single-node multi-drive Pools and multi-node Pools with one drive per node are allowed, subject to valid Erasure Set geometry and EC settings; acceptance does not guarantee host-failure tolerance.
|
||||
>
|
||||
> These topology rules follow MinIO, but automatic parity selection differs between the projects. See the [Pool layout compatibility and regression tests](docs/testing/pool-layout-compatibility.md) before expanding a deployment.
|
||||
|
||||
To get started with RustFS, follow these steps:
|
||||
|
||||
### 1. One-click Installation (Option 1)
|
||||
|
||||
@@ -89,6 +89,15 @@ RustFS 是一个基于 Rust 构建的高性能分布式对象存储系统。Rust
|
||||
|
||||
## 快速开始
|
||||
|
||||
> [!IMPORTANT]
|
||||
> **Pool 扩容 Notice:**
|
||||
>
|
||||
> - 单节点单盘(SNSD)部署仅支持使用本地路径独立运行,不支持原地扩容,也不能作为 Pool 加入集群。如需改为多盘拓扑,请创建新部署并通过 S3 迁移数据。
|
||||
> - 已有多盘 Pool 的端点和 Erasure Set 宽度应保持不变,扩容应追加新的 Pool。使用省略号表达式扩容时,每个 Pool 参数都必须包含省略号表达式,并展开为至少两个磁盘端点。
|
||||
> - 允许单节点多盘 Pool,也允许多节点、每节点一盘的 Pool,但必须满足 Erasure Set 布局和 EC 配置要求;配置合法不代表能够容忍整台主机故障。
|
||||
>
|
||||
> 这些拓扑规则与 MinIO 一致,但两者的默认 parity 选择方式存在差异。扩容前请阅读 [Pool 布局兼容性与回归测试说明](docs/testing/pool-layout-compatibility.md)。
|
||||
|
||||
请按照以下步骤快速上手 RustFS:
|
||||
|
||||
### 1. 一键安装脚本 (选项 1)
|
||||
|
||||
@@ -66,6 +66,11 @@ Current guidance:
|
||||
|
||||
- `RUSTFS_BROWSER_REDIRECT_URL` sets the externally reachable browser origin used for OIDC callback, console success redirect, and logout fallback URLs. Configure it to the public scheme and authority without a path, for example `https://console.example.com`. In load-balancer deployments, keep OIDC authorize and callback requests on the same backend node because the in-flight OIDC `state` is local to the RustFS node.
|
||||
|
||||
## S3 API environment variables
|
||||
|
||||
- `RUSTFS_API_OBJECT_MAX_VERSIONS` caps the number of retained versions for a single object. It defaults to `9223372036854775807`, matching MinIO's practical-unlimited default. Set a positive integer to enforce a lower per-object metadata bound.
|
||||
- `MINIO_API_OBJECT_MAX_VERSIONS` is accepted as a compatibility alias when the canonical RustFS variable is not set.
|
||||
|
||||
## Distributed endpoint locality
|
||||
|
||||
- `RUSTFS_LOCAL_ENDPOINT_HOST` identifies this server's host in a distributed `RUSTFS_VOLUMES` topology without resolving every peer during startup. Set it to exactly one host, without a scheme, port, or path. It is accepted only for orchestrated URL topologies and must match at least one endpoint on the RustFS server port; invalid or unmatched values fail startup. Leave it unset to retain DNS-based locality discovery.
|
||||
@@ -130,6 +135,47 @@ Scanner cycle budget controls:
|
||||
- timeout returns S3 `SlowDown`, so clients should use normal SDK retry handling.
|
||||
- this is not a fdatasync or group-commit switch. Track fdatasync batching separately with `rustfs_s3_put_object_rename_fdatasync_batch_files`.
|
||||
|
||||
## Foreground write admission environment variables
|
||||
|
||||
Large direct `PutObject` requests and multipart `UploadPart` requests share one
|
||||
per-process permit pool that bounds how many bodies are ingested and written
|
||||
concurrently. Small direct PUTs stay on the legacy path.
|
||||
|
||||
- `RUSTFS_PUT_LARGE_FOREGROUND_ADMISSION_ENABLE`
|
||||
- enables the default-on pool; `false` keeps only the soft request counter.
|
||||
- default is `true`.
|
||||
- `RUSTFS_PUT_LARGE_FOREGROUND_ADMISSION_LIMIT`
|
||||
- permits in the pool; `0` derives half of `RUSTFS_OBJECT_MAX_CONCURRENT_DISK_READS`, clamped to `32`.
|
||||
- default is `0` (32 permits at stock settings).
|
||||
- `RUSTFS_PUT_LARGE_FOREGROUND_ADMISSION_MIN_SIZE_BYTES`
|
||||
- smallest direct `PutObject` that takes a permit; unknown-size requests always do.
|
||||
- default is `33554432` (32 MiB).
|
||||
- `RUSTFS_PUT_LARGE_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS`
|
||||
- how long a direct `PutObject` waits for a permit before returning S3 `SlowDown`.
|
||||
- default is `250`.
|
||||
- `RUSTFS_PUT_MULTIPART_FOREGROUND_ADMISSION_MIN_SIZE_BYTES`
|
||||
- smallest `UploadPart` that takes a permit; `0` gates every part.
|
||||
- default is `0`.
|
||||
- `RUSTFS_PUT_MULTIPART_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS`
|
||||
- how long an `UploadPart` waits in the bounded queue for a permit before returning S3 `SlowDown`; `0` rejects immediately when the pool is full.
|
||||
- default is `10000`. Parts wait before body ingest, so SDK-default clients that send every part of an upload concurrently drain through the pool instead of failing on a full pool.
|
||||
- RustFS does not read the request body while a part is queued, so the client's socket write stalls for the whole wait and whatever timeout the client or an intermediary has configured competes with this value. Keep it with margin below the shortest such timeout in use (botocore applies its 60 s `connect_timeout` to the body write; the AWS SDK for Java v2 has a 30 s socket write timeout; reverse proxies add their own body timeouts); a wait that outlives the client timeout surfaces as a dropped connection instead of `SlowDown`.
|
||||
- `RUSTFS_PUT_MULTIPART_FOREGROUND_ADMISSION_MAX_PENDING`
|
||||
- maximum `UploadPart` requests waiting for a permit at once; parts beyond it return `SlowDown` without waiting.
|
||||
- default is `0`, which derives 16 times the permit limit (512 at stock settings).
|
||||
- each queued HTTP/1 part holds whatever unread body the client already pushed into the connection's kernel receive buffer (an HTTP/2 part holds up to its flow-control window in process memory), so this depth also bounds that memory. RustFS leaves the receive buffer to kernel autotuning (see `RUSTFS_HTTP_SOCKET_RECV_BUFFER_BYTES` below), which keeps an unread connection at the kernel's initial size (128 KiB on current Linux).
|
||||
- `RUSTFS_PUT_FOREGROUND_ADMISSION_ENABLE`, `RUSTFS_PUT_FOREGROUND_ADMISSION_LIMIT`, `RUSTFS_PUT_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS`
|
||||
- experimental strict gate that applies to every foreground write regardless of size and replaces the pool above when enabled.
|
||||
- default is disabled; enabling it with limit `0` disables foreground write admission entirely.
|
||||
|
||||
## HTTP listener socket environment variables
|
||||
|
||||
- `RUSTFS_HTTP_SOCKET_RECV_BUFFER_BYTES`
|
||||
- fixed `SO_RCVBUF` for the API listener, inherited by every accepted socket; `0` leaves the receive buffer to kernel autotuning.
|
||||
- default is `0`. Earlier releases hard-coded 4 MiB, which Linux doubles to 8 MiB and which disables autotuning, so every connection whose body was not being read yet (a multipart part queued for a foreground write permit) could accumulate up to 8 MiB of unread body in kernel memory; at SDK-default multipart concurrency that was enough to push a node into TCP memory pressure.
|
||||
- with autotuning the per-connection receive ceiling is the kernel's (`net.ipv4.tcp_rmem` max, 6 MiB on stock Linux) instead of the former fixed 8 MiB, so a single very high-bandwidth-delay connection may see a somewhat lower ceiling; raise `net.ipv4.tcp_rmem` first, and set this variable only on kernels without receive-buffer autotuning (illumos/Solaris) or where the sysctl cannot be changed.
|
||||
- the send buffer stays fixed at 4 MiB because the stock Linux send autotuning ceiling (`net.ipv4.tcp_wmem` max, 4 MiB) is lower than a GB-level response stream needs.
|
||||
|
||||
## Remote tier timeout environment variables
|
||||
|
||||
- `RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS`
|
||||
|
||||
@@ -90,3 +90,15 @@ pub const ENV_API_MAX_CONNECTIONS: &str = "RUSTFS_API_MAX_CONNECTIONS";
|
||||
|
||||
/// Default for `RUSTFS_API_MAX_CONNECTIONS` (`0` = unlimited).
|
||||
pub const DEFAULT_API_MAX_CONNECTIONS: usize = 0;
|
||||
|
||||
/// Maximum retained versions per object.
|
||||
///
|
||||
/// The default follows MinIO and is effectively unlimited for practical
|
||||
/// deployments. Operators can lower it to bound per-object metadata growth.
|
||||
/// Environment variable: RUSTFS_API_OBJECT_MAX_VERSIONS
|
||||
/// MinIO-compatible alias: MINIO_API_OBJECT_MAX_VERSIONS
|
||||
/// Example: RUSTFS_API_OBJECT_MAX_VERSIONS=50000
|
||||
pub const ENV_API_OBJECT_MAX_VERSIONS: &str = "RUSTFS_API_OBJECT_MAX_VERSIONS";
|
||||
|
||||
/// Default for `RUSTFS_API_OBJECT_MAX_VERSIONS`.
|
||||
pub const DEFAULT_API_OBJECT_MAX_VERSIONS: u64 = 9_223_372_036_854_775_807;
|
||||
|
||||
@@ -365,13 +365,54 @@ pub const ENV_PUT_MULTIPART_FOREGROUND_ADMISSION_MIN_SIZE_BYTES: &str =
|
||||
"RUSTFS_PUT_MULTIPART_FOREGROUND_ADMISSION_MIN_SIZE_BYTES";
|
||||
pub const DEFAULT_PUT_MULTIPART_FOREGROUND_ADMISSION_MIN_SIZE_BYTES: usize = 0;
|
||||
|
||||
/// Time in milliseconds an automatic foreground write waits for a permit.
|
||||
/// Time in milliseconds an automatic foreground direct PutObject waits for a permit.
|
||||
///
|
||||
/// A short wait smooths transient bursts while still returning S3
|
||||
/// `SlowDown`/503 before body ingest when the node is already saturated.
|
||||
pub const ENV_PUT_LARGE_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS: &str = "RUSTFS_PUT_LARGE_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS";
|
||||
pub const DEFAULT_PUT_LARGE_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS: u64 = 250;
|
||||
|
||||
/// Time in milliseconds a multipart UploadPart waits for a foreground write permit.
|
||||
///
|
||||
/// SDK-default multipart clients send every part of an upload concurrently, so
|
||||
/// a single node routinely sees several times more parts in flight than the
|
||||
/// permit pool allows. A queued part waits before body ingest, so the pool
|
||||
/// still bounds the number of parts being written, but the wait is not free:
|
||||
/// RustFS does not read the request body while the part is queued (hyper only
|
||||
/// sends `100 Continue` once the body is first polled, and the AWS SDKs send
|
||||
/// the body after a 1-3 s `Expect: 100-continue` grace anyway), so the
|
||||
/// client's socket write stalls once the kernel buffers fill, and whatever
|
||||
/// timeout the client or an intermediary has configured decides the outcome.
|
||||
/// botocore applies its `connect_timeout` (60 s) to the body write, the AWS
|
||||
/// SDK for Java v2 has a 30 s socket write timeout, and MinIO bounds the same
|
||||
/// wait with a 10 s request deadline. The wait must leave margin under the
|
||||
/// shortest of those, not merely fall below an SDK default, so the part
|
||||
/// receives S3 `SlowDown`/503 for the client to retry instead of losing its
|
||||
/// connection (issue #7385). `0` rejects immediately when the pool is full.
|
||||
pub const ENV_PUT_MULTIPART_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS: &str =
|
||||
"RUSTFS_PUT_MULTIPART_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS";
|
||||
pub const DEFAULT_PUT_MULTIPART_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS: u64 = 10_000;
|
||||
|
||||
// A queued part holds the client's body write open for the whole wait. The
|
||||
// shortest write timeout among mainstream S3 SDKs is the AWS SDK for Java v2's
|
||||
// 30 s socket write timeout; keep the compiled default at no more than a third
|
||||
// of it. This locks only the default; the environment variable may still raise
|
||||
// the wait past any client timeout.
|
||||
const _: () = assert!(DEFAULT_PUT_MULTIPART_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS * 3 <= 30_000);
|
||||
|
||||
/// Maximum multipart UploadPart requests waiting for a foreground write permit per process.
|
||||
///
|
||||
/// Parts beyond this queue depth are rejected with S3 `SlowDown`/503 without
|
||||
/// waiting, so a genuinely saturated node still fails fast instead of holding
|
||||
/// an unbounded set of connections open for the whole wait timeout. Each
|
||||
/// queued HTTP/1 part also holds whatever unread body the client already
|
||||
/// pushed into that connection's kernel receive buffer, and a queued HTTP/2
|
||||
/// part holds up to its flow-control window in process memory, so the depth
|
||||
/// bounds socket and window memory as well as connections.
|
||||
/// `0` derives the depth from the permit limit.
|
||||
pub const ENV_PUT_MULTIPART_FOREGROUND_ADMISSION_MAX_PENDING: &str = "RUSTFS_PUT_MULTIPART_FOREGROUND_ADMISSION_MAX_PENDING";
|
||||
pub const DEFAULT_PUT_MULTIPART_FOREGROUND_ADMISSION_MAX_PENDING: usize = 0;
|
||||
|
||||
const _: () = assert!(DEFAULT_PUT_LARGE_FOREGROUND_ADMISSION_ENABLE);
|
||||
|
||||
/// Environment variable for minimum GetObject timeout in seconds.
|
||||
|
||||
@@ -159,6 +159,24 @@ pub const DEFAULT_HTTP1_HEADER_READ_TIMEOUT: u64 = 75;
|
||||
pub const ENV_HTTP1_MAX_BUF_SIZE: &str = "RUSTFS_HTTP1_MAX_BUF_SIZE";
|
||||
pub const DEFAULT_HTTP1_MAX_BUF_SIZE: usize = 64 * 1024; // 64 KB
|
||||
|
||||
/// Environment variable for a fixed kernel receive buffer (`SO_RCVBUF`, bytes)
|
||||
/// on the API listener. Default: 0, which leaves the buffer to kernel
|
||||
/// autotuning.
|
||||
///
|
||||
/// A fixed `SO_RCVBUF` is inherited by every accepted socket and disables
|
||||
/// receive-buffer autotuning, so a connection whose request body is not being
|
||||
/// read yet (a multipart part queued for a foreground write permit) lets up to
|
||||
/// the fixed size of unread body accumulate in kernel memory — Linux doubles
|
||||
/// the requested value, so the former hard-coded 4 MiB held up to 8 MiB per
|
||||
/// queued connection (issue #7385). Autotuning keeps an unread connection at
|
||||
/// the kernel's initial size and grows only connections that are being
|
||||
/// drained. Set this only on kernels without receive-buffer autotuning
|
||||
/// (illumos/Solaris) or on very high-bandwidth-delay links where the kernel's
|
||||
/// autotuning ceiling (`net.ipv4.tcp_rmem` on Linux) is too low and cannot be
|
||||
/// raised.
|
||||
pub const ENV_HTTP_SOCKET_RECV_BUFFER_BYTES: &str = "RUSTFS_HTTP_SOCKET_RECV_BUFFER_BYTES";
|
||||
pub const DEFAULT_HTTP_SOCKET_RECV_BUFFER_BYTES: usize = 0;
|
||||
|
||||
/// Environment variable for the S3 request-body inter-chunk read timeout
|
||||
/// (seconds). Default: 300. Set to 0 to disable.
|
||||
///
|
||||
|
||||
@@ -144,6 +144,3 @@ russh = { workspace = true, features = ["serde"] }
|
||||
russh-sftp = { workspace = true }
|
||||
zip.workspace = true
|
||||
clap = { workspace = true, features = ["derive", "env"] }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile.workspace = true
|
||||
|
||||
@@ -184,6 +184,8 @@ the wiring source of truth. Committed test-ID digests under
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
**Endpoint blackhole scenario skipped** — `heal_erasure_disk_rebuild_test::tests::test_cluster_root_heal_recovers_after_target_endpoint_blackhole` installs a loopback `iptables` DROP rule and therefore needs `CAP_NET_ADMIN` (root or passwordless `sudo -n iptables`). A host where `iptables` is missing or cannot read the OUTPUT chain (typical inside an unprivileged container, where the nf_tables backend reports "Permission denied" even under `sudo`) logs a `heal_interruption_skipped` warning and returns without exercising heal. Set `RUSTFS_E2E_REQUIRE_NET_FAULT_INJECTION=1` on lanes that do provision the capability so a broken runner fails instead of skipping.
|
||||
|
||||
**Reproduce a CI failure locally** — run the exact profile/lane:
|
||||
|
||||
```bash
|
||||
|
||||
@@ -57,6 +57,8 @@ const RUSTFS_FULL_FEATURE: &str = "full";
|
||||
const TEST_PORT_MIN: u16 = 20_000;
|
||||
// Keep allocator ports below the ephemeral range used by bind(..., 0) test helpers.
|
||||
const TEST_PORT_RANGE: u16 = 10_000;
|
||||
const TEST_PORT_MIN_ENV: &str = "RUSTFS_E2E_TEST_PORT_MIN";
|
||||
const TEST_PORT_RANGE_ENV: &str = "RUSTFS_E2E_TEST_PORT_RANGE";
|
||||
const TEST_PORT_COUNTER_PATH: &str = "/tmp/rustfs_e2e_next_port";
|
||||
const TEST_PORT_LOCK_DIR: &str = "/tmp/rustfs_e2e_port_allocator.lock";
|
||||
const TEST_PORT_LOCK_STALE_AFTER: Duration = Duration::from_secs(30);
|
||||
@@ -99,22 +101,74 @@ impl Drop for PortAllocatorGuard {
|
||||
}
|
||||
}
|
||||
|
||||
fn advance_test_port(port: u16) -> u16 {
|
||||
let offset = (port - TEST_PORT_MIN + 1) % TEST_PORT_RANGE;
|
||||
TEST_PORT_MIN + offset
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||
struct TestPortAllocatorConfig {
|
||||
min: u16,
|
||||
range: u16,
|
||||
}
|
||||
|
||||
fn seeded_test_port() -> u16 {
|
||||
let offset = (Uuid::new_v4().as_u128() % u128::from(TEST_PORT_RANGE)) as u16;
|
||||
TEST_PORT_MIN + offset
|
||||
impl TestPortAllocatorConfig {
|
||||
fn max_exclusive(self) -> u32 {
|
||||
u32::from(self.min) + u32::from(self.range)
|
||||
}
|
||||
|
||||
fn contains(self, port: &u16) -> bool {
|
||||
(u32::from(self.min)..self.max_exclusive()).contains(&u32::from(*port))
|
||||
}
|
||||
}
|
||||
|
||||
fn read_next_test_port() -> u16 {
|
||||
fn parse_test_port_allocator_config(
|
||||
min_override: Option<&str>,
|
||||
range_override: Option<&str>,
|
||||
) -> Result<TestPortAllocatorConfig, Box<dyn std::error::Error + Send + Sync>> {
|
||||
let min = match min_override {
|
||||
Some(value) => value
|
||||
.parse::<u16>()
|
||||
.map_err(|err| format!("{TEST_PORT_MIN_ENV} must be a valid u16: {err}"))?,
|
||||
None => TEST_PORT_MIN,
|
||||
};
|
||||
let range = match range_override {
|
||||
Some(value) => value
|
||||
.parse::<u16>()
|
||||
.map_err(|err| format!("{TEST_PORT_RANGE_ENV} must be a valid u16: {err}"))?,
|
||||
None => TEST_PORT_RANGE,
|
||||
};
|
||||
if range == 0 {
|
||||
return Err(format!("{TEST_PORT_RANGE_ENV} must be greater than zero").into());
|
||||
}
|
||||
if min < 1024 {
|
||||
return Err(format!("{TEST_PORT_MIN_ENV} must be at least 1024").into());
|
||||
}
|
||||
let max_exclusive = u32::from(min) + u32::from(range);
|
||||
if max_exclusive > u32::from(u16::MAX) + 1 {
|
||||
return Err(format!("{TEST_PORT_MIN_ENV} + {TEST_PORT_RANGE_ENV} exceeds u16 port space").into());
|
||||
}
|
||||
Ok(TestPortAllocatorConfig { min, range })
|
||||
}
|
||||
|
||||
fn test_port_allocator_config() -> Result<TestPortAllocatorConfig, Box<dyn std::error::Error + Send + Sync>> {
|
||||
parse_test_port_allocator_config(
|
||||
std::env::var(TEST_PORT_MIN_ENV).ok().as_deref(),
|
||||
std::env::var(TEST_PORT_RANGE_ENV).ok().as_deref(),
|
||||
)
|
||||
}
|
||||
|
||||
fn advance_test_port(port: u16, config: TestPortAllocatorConfig) -> u16 {
|
||||
let offset = (port - config.min + 1) % config.range;
|
||||
config.min + offset
|
||||
}
|
||||
|
||||
fn seeded_test_port(config: TestPortAllocatorConfig) -> u16 {
|
||||
let offset = (Uuid::new_v4().as_u128() % u128::from(config.range)) as u16;
|
||||
config.min + offset
|
||||
}
|
||||
|
||||
fn read_next_test_port(config: TestPortAllocatorConfig) -> u16 {
|
||||
stdfs::read_to_string(TEST_PORT_COUNTER_PATH)
|
||||
.ok()
|
||||
.and_then(|value| value.trim().parse::<u16>().ok())
|
||||
.filter(|port| (TEST_PORT_MIN..TEST_PORT_MIN + TEST_PORT_RANGE).contains(port))
|
||||
.unwrap_or_else(seeded_test_port)
|
||||
.filter(|port| config.contains(port))
|
||||
.unwrap_or_else(|| seeded_test_port(config))
|
||||
}
|
||||
|
||||
fn remove_stale_port_allocator_lock() {
|
||||
@@ -629,11 +683,12 @@ impl RustFSTestEnvironment {
|
||||
pub async fn find_available_port() -> Result<u16, Box<dyn std::error::Error + Send + Sync>> {
|
||||
use std::net::TcpListener;
|
||||
let _guard = PortAllocatorGuard::acquire().await?;
|
||||
let mut next_port = read_next_test_port();
|
||||
let config = test_port_allocator_config()?;
|
||||
let mut next_port = read_next_test_port(config);
|
||||
|
||||
for _ in 0..TEST_PORT_RANGE {
|
||||
for _ in 0..config.range {
|
||||
let port = next_port;
|
||||
next_port = advance_test_port(next_port);
|
||||
next_port = advance_test_port(next_port, config);
|
||||
write_next_test_port(next_port)?;
|
||||
|
||||
if let Ok(listener) = TcpListener::bind(("127.0.0.1", port)) {
|
||||
@@ -2108,6 +2163,35 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn e2e_port_allocator_uses_default_range() {
|
||||
assert_eq!(
|
||||
parse_test_port_allocator_config(None, None).expect("default port allocator config"),
|
||||
TestPortAllocatorConfig {
|
||||
min: TEST_PORT_MIN,
|
||||
range: TEST_PORT_RANGE
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn e2e_port_allocator_accepts_explicit_test_range() {
|
||||
let config = parse_test_port_allocator_config(Some("31000"), Some("128")).expect("explicit port range");
|
||||
|
||||
assert_eq!(advance_test_port(31127, config), 31000);
|
||||
assert!(config.contains(&31000));
|
||||
assert!(config.contains(&31127));
|
||||
assert!(!config.contains(&31128));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn e2e_port_allocator_rejects_invalid_override() {
|
||||
assert!(parse_test_port_allocator_config(Some("1023"), Some("1")).is_err());
|
||||
assert!(parse_test_port_allocator_config(Some("65000"), Some("1000")).is_err());
|
||||
assert!(parse_test_port_allocator_config(Some("31000"), Some("0")).is_err());
|
||||
assert!(parse_test_port_allocator_config(Some("not-a-port"), Some("128")).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolves_rustfs_binary_in_configured_cargo_target_directory() {
|
||||
let workspace = Path::new("workspace");
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -34,12 +34,14 @@ use s3s::dto::{
|
||||
AbortMultipartUploadInput, AbortMultipartUploadOutput, CommonPrefix, CompleteMultipartUploadInput,
|
||||
CompleteMultipartUploadOutput, CreateMultipartUploadInput, CreateMultipartUploadOutput, DeleteMarkerEntry, DeleteObjectInput,
|
||||
DeleteObjectOutput, DeleteObjectTaggingInput, DeleteObjectTaggingOutput, ETag, GetBucketVersioningInput,
|
||||
GetBucketVersioningOutput, GetObjectInput, GetObjectLockConfigurationInput, GetObjectLockConfigurationOutput,
|
||||
GetObjectOutput, GetObjectTaggingInput, GetObjectTaggingOutput, HeadBucketInput, HeadBucketOutput, HeadObjectInput,
|
||||
GetBucketVersioningOutput, GetObjectInput, GetObjectLegalHoldInput, GetObjectLegalHoldOutput,
|
||||
GetObjectLockConfigurationInput, GetObjectLockConfigurationOutput, GetObjectOutput, GetObjectRetentionInput,
|
||||
GetObjectRetentionOutput, GetObjectTaggingInput, GetObjectTaggingOutput, HeadBucketInput, HeadBucketOutput, HeadObjectInput,
|
||||
HeadObjectOutput, ListObjectVersionsInput, ListObjectVersionsOutput, ListObjectsV2Input, ListObjectsV2Output, Object,
|
||||
ObjectLockConfiguration, ObjectLockEnabled, ObjectStorageClass, ObjectVersionId, PutObjectInput, PutObjectOutput,
|
||||
PutObjectTaggingInput, PutObjectTaggingOutput, Range, StreamingBlob, Tag, TagSet, Timestamp, TimestampFormat,
|
||||
UploadPartInput, UploadPartOutput,
|
||||
ObjectLockConfiguration, ObjectLockEnabled, ObjectLockLegalHold, ObjectLockLegalHoldStatus, ObjectLockMode,
|
||||
ObjectLockRetention, ObjectLockRetentionMode, ObjectStorageClass, ObjectVersionId, PutObjectInput, PutObjectLegalHoldInput,
|
||||
PutObjectLegalHoldOutput, PutObjectOutput, PutObjectRetentionInput, PutObjectRetentionOutput, PutObjectTaggingInput,
|
||||
PutObjectTaggingOutput, Range, StreamingBlob, Tag, TagSet, Timestamp, TimestampFormat, UploadPartInput, UploadPartOutput,
|
||||
};
|
||||
use s3s::service::{S3Service, S3ServiceBuilder};
|
||||
use s3s::validation::{AwsNameValidation, NameValidation};
|
||||
@@ -127,6 +129,10 @@ pub enum Operation {
|
||||
GetObjectTagging,
|
||||
PutObjectTagging,
|
||||
DeleteObjectTagging,
|
||||
GetObjectRetention,
|
||||
PutObjectRetention,
|
||||
GetObjectLegalHold,
|
||||
PutObjectLegalHold,
|
||||
ListObjectVersions,
|
||||
ListObjectsV2,
|
||||
CreateMultipartUpload,
|
||||
@@ -501,6 +507,10 @@ struct StoreState {
|
||||
/// PutObject carrying any `x-amz-object-lock-*` header must also carry
|
||||
/// `Content-MD5` or an `x-amz-checksum-*` header.
|
||||
require_checksum_for_object_lock: bool,
|
||||
/// Models Wasabi (rustfs/backlog#2340): a version-addressed DELETE of a
|
||||
/// version id the target never had answers 404 `NoSuchVersion` instead of
|
||||
/// the idempotent 204 RustFS/MinIO give.
|
||||
reject_unknown_version_deletes: bool,
|
||||
limits: StoreLimits,
|
||||
buckets: HashMap<String, BucketState>,
|
||||
uploads: HashMap<String, MultipartState>,
|
||||
@@ -565,6 +575,41 @@ struct ObjectVersion {
|
||||
/// SSE-C passthrough transport headers stored with the version (RustFS
|
||||
/// target behavior); empty when the drop mode discarded them.
|
||||
replication_sse_headers: Vec<(String, String)>,
|
||||
/// Object Lock state of the version: retention (mode, retain-until) from
|
||||
/// the PUT / CreateMultipartUpload headers or PutObjectRetention, and the
|
||||
/// legal hold flag; replayed on HEAD.
|
||||
lock: VersionLock,
|
||||
}
|
||||
|
||||
#[derive(Clone, Default)]
|
||||
struct VersionLock {
|
||||
retention: Option<(String, Timestamp)>,
|
||||
/// `None` until a legal hold status was ever set; like S3, HEAD then
|
||||
/// reports nothing, while an explicit OFF is reported as `OFF`.
|
||||
legal_hold: Option<bool>,
|
||||
}
|
||||
|
||||
impl VersionLock {
|
||||
fn from_headers(
|
||||
mode: Option<ObjectLockMode>,
|
||||
retain_until: Option<Timestamp>,
|
||||
legal_hold: Option<ObjectLockLegalHoldStatus>,
|
||||
) -> Self {
|
||||
Self {
|
||||
retention: mode.zip(retain_until).map(|(mode, until)| (mode.as_str().to_string(), until)),
|
||||
legal_hold: legal_hold.map(|status| status.as_str().eq_ignore_ascii_case("ON")),
|
||||
}
|
||||
}
|
||||
|
||||
fn legal_hold_status(&self) -> Option<ObjectLockLegalHoldStatus> {
|
||||
self.legal_hold.map(|on| {
|
||||
ObjectLockLegalHoldStatus::from_static(if on {
|
||||
ObjectLockLegalHoldStatus::ON
|
||||
} else {
|
||||
ObjectLockLegalHoldStatus::OFF
|
||||
})
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
@@ -576,6 +621,7 @@ struct MultipartState {
|
||||
metadata: Option<HashMap<String, String>>,
|
||||
standard_headers: StandardHeaders,
|
||||
replication_sse_headers: Vec<(String, String)>,
|
||||
lock: VersionLock,
|
||||
parts: BTreeMap<i32, MultipartPart>,
|
||||
}
|
||||
|
||||
@@ -584,6 +630,10 @@ struct MultipartPart {
|
||||
body: Bytes,
|
||||
e_tag: String,
|
||||
digest: [u8; 16],
|
||||
/// Plaintext length declared by an SSE-C passthrough sender
|
||||
/// (`x-rustfs-replication-part-actual-size`); RustFS validates the 5 MiB
|
||||
/// minimum against it rather than against the stored bytes.
|
||||
actual_size: Option<usize>,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
@@ -845,6 +895,7 @@ impl FakeS3Target {
|
||||
standard_headers: seed.standard_headers.clone(),
|
||||
tags: Vec::new(),
|
||||
replication_sse_headers: Vec::new(),
|
||||
lock: VersionLock::default(),
|
||||
};
|
||||
upsert_version(&mut state, bucket, key.into(), version).expect("seed object must fit the storage budget");
|
||||
e_tag
|
||||
@@ -918,6 +969,12 @@ impl FakeS3Target {
|
||||
/// PutObject that carries Object Lock parameters (AWS S3 / MinIO rule,
|
||||
/// rustfs#7082). `Content-MD5`, when present, is always verified against
|
||||
/// the body regardless of this mode.
|
||||
/// Wasabi-like mode: DELETE of an unknown version id answers 404
|
||||
/// `NoSuchVersion` (the default 204 models RustFS/MinIO).
|
||||
pub fn reject_unknown_version_deletes(&self, enabled: bool) {
|
||||
lock(&self.backend.store).reject_unknown_version_deletes = enabled;
|
||||
}
|
||||
|
||||
pub fn require_checksum_for_object_lock(&self, enabled: bool) {
|
||||
lock(&self.backend.store).require_checksum_for_object_lock = enabled;
|
||||
}
|
||||
@@ -1152,6 +1209,10 @@ fn operation_from_s3_name(name: &str) -> Operation {
|
||||
"GetObjectTagging" => Operation::GetObjectTagging,
|
||||
"PutObjectTagging" => Operation::PutObjectTagging,
|
||||
"DeleteObjectTagging" => Operation::DeleteObjectTagging,
|
||||
"GetObjectRetention" => Operation::GetObjectRetention,
|
||||
"PutObjectRetention" => Operation::PutObjectRetention,
|
||||
"GetObjectLegalHold" => Operation::GetObjectLegalHold,
|
||||
"PutObjectLegalHold" => Operation::PutObjectLegalHold,
|
||||
"ListObjectsV2" => Operation::ListObjectsV2,
|
||||
"CreateMultipartUpload" => Operation::CreateMultipartUpload,
|
||||
"UploadPart" => Operation::UploadPart,
|
||||
@@ -1289,6 +1350,18 @@ fn parse_request(method: &Method, uri: &Uri) -> ParsedRequest {
|
||||
(&Method::DELETE, true) if query.contains_key("tagging") && only_query_keys(&["tagging", "versionId"]) => {
|
||||
Operation::DeleteObjectTagging
|
||||
}
|
||||
(&Method::GET, true) if query.contains_key("retention") && only_query_keys(&["retention", "versionId"]) => {
|
||||
Operation::GetObjectRetention
|
||||
}
|
||||
(&Method::PUT, true) if query.contains_key("retention") && only_query_keys(&["retention", "versionId"]) => {
|
||||
Operation::PutObjectRetention
|
||||
}
|
||||
(&Method::GET, true) if query.contains_key("legal-hold") && only_query_keys(&["legal-hold", "versionId"]) => {
|
||||
Operation::GetObjectLegalHold
|
||||
}
|
||||
(&Method::PUT, true) if query.contains_key("legal-hold") && only_query_keys(&["legal-hold", "versionId"]) => {
|
||||
Operation::PutObjectLegalHold
|
||||
}
|
||||
// A replication PUT addresses the source version via `?versionId=`.
|
||||
(&Method::PUT, true) if only_query_keys(&["versionId"]) => Operation::PutObject,
|
||||
(&Method::GET, true) if only_query_keys(&["versionId"]) => Operation::GetObject,
|
||||
@@ -1842,6 +1915,28 @@ fn set_version_tags(
|
||||
Ok(resolved)
|
||||
}
|
||||
|
||||
fn update_version_lock(
|
||||
state: &mut StoreState,
|
||||
bucket: &str,
|
||||
key: &str,
|
||||
version_id: Option<&str>,
|
||||
update: impl FnOnce(&mut VersionLock),
|
||||
) -> S3Result<String> {
|
||||
let resolved = find_version(state, bucket, key, version_id)?.version_id;
|
||||
let version = state
|
||||
.buckets
|
||||
.get_mut(bucket)
|
||||
.expect("bucket existence checked by find_version")
|
||||
.objects
|
||||
.get_mut(key)
|
||||
.expect("key existence checked by find_version")
|
||||
.iter_mut()
|
||||
.find(|version| version.version_id == resolved)
|
||||
.expect("version existence checked by find_version");
|
||||
update(&mut version.lock);
|
||||
Ok(resolved)
|
||||
}
|
||||
|
||||
/// Whether version ids are surfaced for this bucket. Unknown buckets report
|
||||
/// `true`; the caller's lookup raises `NoSuchBucket` first.
|
||||
fn bucket_versioned(state: &StoreState, bucket: &str) -> bool {
|
||||
@@ -2281,6 +2376,11 @@ impl S3 for FakeBackend {
|
||||
standard_headers,
|
||||
tags: Vec::new(),
|
||||
replication_sse_headers: captured_replication_sse_headers(&headers, drop_unlisted),
|
||||
lock: VersionLock::from_headers(
|
||||
input.object_lock_mode,
|
||||
input.object_lock_retain_until_date,
|
||||
input.object_lock_legal_hold_status,
|
||||
),
|
||||
};
|
||||
upsert_version(&mut lock(&self.store), &input.bucket, input.key, version)?;
|
||||
Ok(apply_response_fault(
|
||||
@@ -2339,6 +2439,13 @@ impl S3 for FakeBackend {
|
||||
last_modified: Some(version.last_modified.clone()),
|
||||
version_id: versioned.then_some(version.version_id),
|
||||
sse_customer_algorithm,
|
||||
object_lock_mode: version
|
||||
.lock
|
||||
.retention
|
||||
.as_ref()
|
||||
.map(|(mode, _)| ObjectLockMode::from(mode.clone())),
|
||||
object_lock_retain_until_date: version.lock.retention.as_ref().map(|(_, until)| until.clone()),
|
||||
object_lock_legal_hold_status: version.lock.legal_hold_status(),
|
||||
..Default::default()
|
||||
});
|
||||
response.status = served.status;
|
||||
@@ -2373,6 +2480,13 @@ impl S3 for FakeBackend {
|
||||
last_modified: Some(version.last_modified.clone()),
|
||||
version_id: versioned.then_some(version.version_id),
|
||||
sse_customer_algorithm,
|
||||
object_lock_mode: version
|
||||
.lock
|
||||
.retention
|
||||
.as_ref()
|
||||
.map(|(mode, _)| ObjectLockMode::from(mode.clone())),
|
||||
object_lock_retain_until_date: version.lock.retention.as_ref().map(|(_, until)| until.clone()),
|
||||
object_lock_legal_hold_status: version.lock.legal_hold_status(),
|
||||
..Default::default()
|
||||
});
|
||||
response.status = served.status;
|
||||
@@ -2432,6 +2546,82 @@ impl S3 for FakeBackend {
|
||||
))
|
||||
}
|
||||
|
||||
async fn get_object_retention(
|
||||
&self,
|
||||
req: S3Request<GetObjectRetentionInput>,
|
||||
) -> S3Result<S3Response<GetObjectRetentionOutput>> {
|
||||
let fault = request_fault(&req);
|
||||
apply_non_body_fault(fault.as_ref(), &self.control).await?;
|
||||
let input = req.input;
|
||||
let version = find_version(&lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref())?;
|
||||
Ok(apply_response_fault(
|
||||
S3Response::new(GetObjectRetentionOutput {
|
||||
retention: version.lock.retention.map(|(mode, until)| ObjectLockRetention {
|
||||
mode: Some(ObjectLockRetentionMode::from(mode)),
|
||||
retain_until_date: Some(until),
|
||||
}),
|
||||
}),
|
||||
fault.as_ref(),
|
||||
))
|
||||
}
|
||||
|
||||
async fn put_object_retention(
|
||||
&self,
|
||||
req: S3Request<PutObjectRetentionInput>,
|
||||
) -> S3Result<S3Response<PutObjectRetentionOutput>> {
|
||||
let fault = request_fault(&req);
|
||||
apply_non_body_fault(fault.as_ref(), &self.control).await?;
|
||||
let input = req.input;
|
||||
let retention = input
|
||||
.retention
|
||||
.and_then(|retention| retention.mode.zip(retention.retain_until_date))
|
||||
.map(|(mode, until)| (mode.as_str().to_string(), until));
|
||||
update_version_lock(&mut lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref(), |lock| {
|
||||
lock.retention = retention;
|
||||
})?;
|
||||
Ok(apply_response_fault(S3Response::new(PutObjectRetentionOutput::default()), fault.as_ref()))
|
||||
}
|
||||
|
||||
async fn get_object_legal_hold(
|
||||
&self,
|
||||
req: S3Request<GetObjectLegalHoldInput>,
|
||||
) -> S3Result<S3Response<GetObjectLegalHoldOutput>> {
|
||||
let fault = request_fault(&req);
|
||||
apply_non_body_fault(fault.as_ref(), &self.control).await?;
|
||||
let input = req.input;
|
||||
let version = find_version(&lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref())?;
|
||||
Ok(apply_response_fault(
|
||||
S3Response::new(GetObjectLegalHoldOutput {
|
||||
legal_hold: Some(ObjectLockLegalHold {
|
||||
status: Some(
|
||||
version
|
||||
.lock
|
||||
.legal_hold_status()
|
||||
.unwrap_or_else(|| ObjectLockLegalHoldStatus::from_static(ObjectLockLegalHoldStatus::OFF)),
|
||||
),
|
||||
}),
|
||||
}),
|
||||
fault.as_ref(),
|
||||
))
|
||||
}
|
||||
|
||||
async fn put_object_legal_hold(
|
||||
&self,
|
||||
req: S3Request<PutObjectLegalHoldInput>,
|
||||
) -> S3Result<S3Response<PutObjectLegalHoldOutput>> {
|
||||
let fault = request_fault(&req);
|
||||
apply_non_body_fault(fault.as_ref(), &self.control).await?;
|
||||
let input = req.input;
|
||||
let legal_hold_on = input
|
||||
.legal_hold
|
||||
.and_then(|hold| hold.status)
|
||||
.is_some_and(|status| status.as_str().eq_ignore_ascii_case("ON"));
|
||||
update_version_lock(&mut lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref(), |lock| {
|
||||
lock.legal_hold = Some(legal_hold_on);
|
||||
})?;
|
||||
Ok(apply_response_fault(S3Response::new(PutObjectLegalHoldOutput::default()), fault.as_ref()))
|
||||
}
|
||||
|
||||
async fn delete_object_tagging(
|
||||
&self,
|
||||
req: S3Request<DeleteObjectTaggingInput>,
|
||||
@@ -2485,6 +2675,7 @@ impl S3 for FakeBackend {
|
||||
return Ok(apply_response_fault(S3Response::new(DeleteObjectOutput::default()), fault.as_ref()));
|
||||
}
|
||||
if let Some(version_id) = input.version_id {
|
||||
let reject_unknown = state.reject_unknown_version_deletes;
|
||||
let (removed_bytes, removed_versions, delete_marker, remove_key) = {
|
||||
let Some(versions) = state
|
||||
.buckets
|
||||
@@ -2493,6 +2684,9 @@ impl S3 for FakeBackend {
|
||||
.objects
|
||||
.get_mut(&input.key)
|
||||
else {
|
||||
if reject_unknown {
|
||||
return Err(s3s::s3_error!(NoSuchVersion, "The specified version does not exist."));
|
||||
}
|
||||
return Ok(apply_response_fault(
|
||||
S3Response::new(DeleteObjectOutput {
|
||||
version_id: Some(version_id),
|
||||
@@ -2501,6 +2695,9 @@ impl S3 for FakeBackend {
|
||||
fault.as_ref(),
|
||||
));
|
||||
};
|
||||
if reject_unknown && !versions.iter().any(|version| version.version_id == version_id) {
|
||||
return Err(s3s::s3_error!(NoSuchVersion, "The specified version does not exist."));
|
||||
}
|
||||
let mut removed_bytes = 0usize;
|
||||
let mut removed_versions = 0usize;
|
||||
let mut delete_marker = None;
|
||||
@@ -2554,6 +2751,7 @@ impl S3 for FakeBackend {
|
||||
standard_headers: StandardHeaders::default(),
|
||||
tags: Vec::new(),
|
||||
replication_sse_headers: Vec::new(),
|
||||
lock: VersionLock::default(),
|
||||
},
|
||||
)?;
|
||||
Ok(apply_response_fault(
|
||||
@@ -2608,6 +2806,11 @@ impl S3 for FakeBackend {
|
||||
metadata: input.metadata,
|
||||
standard_headers,
|
||||
replication_sse_headers: captured_replication_sse_headers(&headers, drop_unlisted),
|
||||
lock: VersionLock::from_headers(
|
||||
input.object_lock_mode,
|
||||
input.object_lock_retain_until_date,
|
||||
input.object_lock_legal_hold_status,
|
||||
),
|
||||
parts: BTreeMap::new(),
|
||||
},
|
||||
);
|
||||
@@ -2624,6 +2827,11 @@ impl S3 for FakeBackend {
|
||||
|
||||
async fn upload_part(&self, req: S3Request<UploadPartInput>) -> S3Result<S3Response<UploadPartOutput>> {
|
||||
let fault = request_fault(&req);
|
||||
let declared_actual_size = req
|
||||
.headers
|
||||
.get("x-rustfs-replication-part-actual-size")
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.and_then(|value| value.parse::<usize>().ok());
|
||||
let _body_permit = timeout(MAX_FAULT_DURATION, Arc::clone(&self.body_limit).acquire_owned())
|
||||
.await
|
||||
.map_err(|_| s3s::s3_error!(RequestTimeout, "fake target body limiter wait exceeded 30 seconds"))?
|
||||
@@ -2665,6 +2873,7 @@ impl S3 for FakeBackend {
|
||||
body,
|
||||
e_tag: e_tag.clone(),
|
||||
digest,
|
||||
actual_size: declared_actual_size,
|
||||
},
|
||||
);
|
||||
Ok(apply_response_fault(
|
||||
@@ -2734,7 +2943,9 @@ impl S3 for FakeBackend {
|
||||
if requested_etag != &stored.e_tag {
|
||||
return Err(s3s::s3_error!(InvalidPart, "part ETag does not match"));
|
||||
}
|
||||
if index + 1 != requested_parts.len() && stored.body.len() < MIN_MULTIPART_PART_BYTES {
|
||||
if index + 1 != requested_parts.len()
|
||||
&& stored.actual_size.unwrap_or(stored.body.len()) < MIN_MULTIPART_PART_BYTES
|
||||
{
|
||||
return Err(s3s::s3_error!(EntityTooSmall, "non-final multipart part is smaller than 5 MiB"));
|
||||
}
|
||||
selected.push((*number, stored.clone()));
|
||||
@@ -2748,6 +2959,7 @@ impl S3 for FakeBackend {
|
||||
metadata: upload.metadata.clone(),
|
||||
standard_headers: upload.standard_headers.clone(),
|
||||
replication_sse_headers: upload.replication_sse_headers.clone(),
|
||||
lock: upload.lock.clone(),
|
||||
parts: BTreeMap::new(),
|
||||
},
|
||||
selected,
|
||||
@@ -2778,6 +2990,7 @@ impl S3 for FakeBackend {
|
||||
standard_headers: upload.standard_headers,
|
||||
tags: Vec::new(),
|
||||
replication_sse_headers: upload.replication_sse_headers,
|
||||
lock: upload.lock,
|
||||
};
|
||||
let mut state = lock(&self.store);
|
||||
let versioned = bucket_versioned(&state, &input.bucket);
|
||||
@@ -4609,6 +4822,7 @@ mod tests {
|
||||
metadata: None,
|
||||
standard_headers: StandardHeaders::default(),
|
||||
replication_sse_headers: Vec::new(),
|
||||
lock: VersionLock::default(),
|
||||
parts: BTreeMap::new(),
|
||||
},
|
||||
);
|
||||
|
||||
@@ -34,6 +34,8 @@ mod tests {
|
||||
use tokio::net::TcpStream;
|
||||
use tokio::time::{Duration, Instant, sleep, timeout};
|
||||
use tracing::info;
|
||||
#[cfg(target_os = "linux")]
|
||||
use tracing::warn;
|
||||
|
||||
const POOL_METADATA_OBJECT: &str = "pool.bin";
|
||||
|
||||
@@ -52,6 +54,34 @@ mod tests {
|
||||
test_binary: EvidenceBuild,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct ScannerHealEvidenceCase {
|
||||
id: &'static str,
|
||||
oracle: &'static str,
|
||||
evidence: &'static str,
|
||||
unclean_shutdown_marker: bool,
|
||||
}
|
||||
|
||||
const BACKGROUND_TARGET_RESTART_EVIDENCE: ScannerHealEvidenceCase = ScannerHealEvidenceCase {
|
||||
id: "background-target-restart",
|
||||
oracle: "background-target-restart.json",
|
||||
evidence: "process-restart",
|
||||
unclean_shutdown_marker: false,
|
||||
};
|
||||
|
||||
const BACKGROUND_TARGET_CRASH_EVIDENCE: ScannerHealEvidenceCase = ScannerHealEvidenceCase {
|
||||
id: "background-target-crash",
|
||||
oracle: "background-target-crash.json",
|
||||
evidence: "process-crash-restart",
|
||||
unclean_shutdown_marker: true,
|
||||
};
|
||||
|
||||
struct RestartEvidenceContext {
|
||||
directory: PathBuf,
|
||||
run: RestartEvidenceRun,
|
||||
case: ScannerHealEvidenceCase,
|
||||
}
|
||||
|
||||
fn file_sha256(path: &Path) -> Result<String, Box<dyn Error + Send + Sync>> {
|
||||
let mut file = std::fs::File::open(path)?;
|
||||
let mut digest = Sha256::new();
|
||||
@@ -66,10 +96,24 @@ mod tests {
|
||||
Ok(digest.finalize().iter().map(|byte| format!("{byte:02x}")).collect())
|
||||
}
|
||||
|
||||
fn restart_evidence_run(binary: &Path) -> Result<Option<(PathBuf, RestartEvidenceRun)>, Box<dyn Error + Send + Sync>> {
|
||||
fn restart_evidence_run(
|
||||
binary: &Path,
|
||||
case: ScannerHealEvidenceCase,
|
||||
) -> Result<Option<RestartEvidenceContext>, Box<dyn Error + Send + Sync>> {
|
||||
let Some(directory) = std::env::var_os("RUSTFS_SCANNER_HEAL_RUN_DIR") else {
|
||||
return Ok(None);
|
||||
};
|
||||
if case.id.is_empty()
|
||||
|| case.oracle.is_empty()
|
||||
|| !case.oracle.ends_with(".json")
|
||||
|| case.oracle.contains('/')
|
||||
|| case.oracle.contains('\\')
|
||||
|| case.oracle.contains("..")
|
||||
|| !matches!(case.evidence, "process-restart" | "process-crash-restart")
|
||||
|| (case.evidence == "process-crash-restart") != case.unclean_shutdown_marker
|
||||
{
|
||||
return Err("invalid scanner/heal evidence case".into());
|
||||
}
|
||||
let directory = PathBuf::from(directory);
|
||||
let receipt = directory.join("run.json");
|
||||
if receipt.metadata()?.len() > 1024 * 1024 {
|
||||
@@ -89,10 +133,10 @@ mod tests {
|
||||
run.test_binary.sha256,
|
||||
"test executable must match the run receipt"
|
||||
);
|
||||
if directory.join("background-target-restart.json").exists() {
|
||||
if directory.join(case.oracle).exists() {
|
||||
return Err("scanner/heal oracle already exists; create a new execution receipt".into());
|
||||
}
|
||||
Ok(Some((directory, run)))
|
||||
Ok(Some(RestartEvidenceContext { directory, run, case }))
|
||||
}
|
||||
|
||||
fn compiled_test_identity() -> serde_json::Value {
|
||||
@@ -115,6 +159,49 @@ mod tests {
|
||||
}
|
||||
|
||||
impl TcpPortBlackhole {
|
||||
/// Environment flag that turns an unusable fault-injection host into a
|
||||
/// hard failure instead of a logged skip. Lanes that provision
|
||||
/// `CAP_NET_ADMIN` set it so a broken runner cannot pass silently.
|
||||
#[cfg(target_os = "linux")]
|
||||
const REQUIRE_ENV: &str = "RUSTFS_E2E_REQUIRE_NET_FAULT_INJECTION";
|
||||
|
||||
/// Probe whether this host can manipulate the OUTPUT chain at all.
|
||||
///
|
||||
/// Returns `Ok(Some(reason))` when `iptables` is missing or lacks
|
||||
/// `CAP_NET_ADMIN` (the nf_tables backend reports "Permission denied"
|
||||
/// even under `sudo` inside an unprivileged container) and the lane did
|
||||
/// not demand fault injection; returns an error when the lane demands
|
||||
/// it; returns `Ok(None)` when the blackhole can be installed.
|
||||
#[cfg(target_os = "linux")]
|
||||
fn unavailable_reason() -> Result<Option<String>, Box<dyn Error + Send + Sync>> {
|
||||
let id = Command::new("id").arg("-u").output()?;
|
||||
if !id.status.success() {
|
||||
return Err(format!("failed to determine the test process uid: {}", String::from_utf8_lossy(&id.stderr)).into());
|
||||
}
|
||||
let use_sudo = String::from_utf8_lossy(&id.stdout).trim() != "0";
|
||||
let mut command = if use_sudo {
|
||||
let mut command = Command::new("sudo");
|
||||
command.args(["-n", "iptables"]);
|
||||
command
|
||||
} else {
|
||||
Command::new("iptables")
|
||||
};
|
||||
let probe = command.args(["-w", "5", "-S", "OUTPUT"]).output();
|
||||
let reason = match probe {
|
||||
Ok(output) if output.status.success() => return Ok(None),
|
||||
Ok(output) => format!(
|
||||
"iptables cannot read the OUTPUT chain (status {}): {}",
|
||||
output.status,
|
||||
String::from_utf8_lossy(&output.stderr).trim()
|
||||
),
|
||||
Err(err) => format!("iptables is not runnable: {err}"),
|
||||
};
|
||||
if std::env::var_os(Self::REQUIRE_ENV).is_some() {
|
||||
return Err(format!("{} is set but network fault injection is unavailable: {reason}", Self::REQUIRE_ENV).into());
|
||||
}
|
||||
Ok(Some(reason))
|
||||
}
|
||||
|
||||
fn install(address: &str) -> Result<Self, Box<dyn Error + Send + Sync>> {
|
||||
let address = address.parse::<SocketAddr>()?;
|
||||
if !address.ip().is_loopback() {
|
||||
@@ -199,6 +286,27 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// Remove a disk directory underneath a running server. Background writers
|
||||
/// (scanner, usage cache, heal markers) can recreate entries between the
|
||||
/// recursive listing and the final `rmdir`, which surfaces as
|
||||
/// `DirectoryNotEmpty` on macOS; retry briefly so the wipe reflects the
|
||||
/// operator action rather than a listing race.
|
||||
fn wipe_directory_while_server_runs(disk: &Path) -> std::io::Result<()> {
|
||||
let mut last_err = None;
|
||||
for _ in 0..20 {
|
||||
match std::fs::remove_dir_all(disk) {
|
||||
Ok(()) => return Ok(()),
|
||||
Err(err) if err.kind() == std::io::ErrorKind::NotFound => return Ok(()),
|
||||
Err(err) if err.kind() == std::io::ErrorKind::DirectoryNotEmpty => {
|
||||
last_err = Some(err);
|
||||
std::thread::sleep(std::time::Duration::from_millis(100));
|
||||
}
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
}
|
||||
Err(last_err.expect("retry loop only exits without success after recording an error"))
|
||||
}
|
||||
|
||||
fn has_file_under(path: &Path) -> bool {
|
||||
let Ok(entries) = std::fs::read_dir(path) else {
|
||||
return false;
|
||||
@@ -481,7 +589,7 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
std::fs::remove_dir_all(&disk0).expect("disk0 wipe should succeed while server is running");
|
||||
wipe_directory_while_server_runs(&disk0).expect("disk0 wipe should succeed while server is running");
|
||||
std::fs::create_dir_all(&disk0).expect("disk0 should be recreated empty while server is running");
|
||||
assert!(!has_file_under(&disk0), "disk0 must be empty immediately after runtime wipe");
|
||||
|
||||
@@ -855,6 +963,16 @@ mod tests {
|
||||
.await?
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn test_cluster_root_heal_recovers_remote_shards_after_background_target_crash()
|
||||
-> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
timeout(
|
||||
Duration::from_secs(420),
|
||||
run_cluster_root_heal_interruption(InterruptionScenario::BackgroundTargetCrash),
|
||||
)
|
||||
.await?
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn test_cluster_root_heal_recovers_remote_shards_after_coordinator_restart() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
@@ -868,6 +986,18 @@ mod tests {
|
||||
#[cfg(target_os = "linux")]
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn test_cluster_root_heal_recovers_after_target_endpoint_blackhole() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
if let Some(reason) = TcpPortBlackhole::unavailable_reason()? {
|
||||
init_logging();
|
||||
warn!(
|
||||
event = "heal_interruption_skipped",
|
||||
component = "e2e_test",
|
||||
subsystem = "heal",
|
||||
interruption_kind = "target_endpoint_blackhole",
|
||||
reason,
|
||||
"Skipping endpoint blackhole scenario: network fault injection is unavailable on this host"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
timeout(
|
||||
Duration::from_secs(420),
|
||||
run_cluster_root_heal_interruption(InterruptionScenario::TargetEndpointBlackhole),
|
||||
@@ -879,21 +1009,27 @@ mod tests {
|
||||
enum InterruptionScenario {
|
||||
IsolatedTargetRestart,
|
||||
BackgroundTargetRestart,
|
||||
BackgroundTargetCrash,
|
||||
BackgroundCoordinatorRestart,
|
||||
TargetEndpointBlackhole,
|
||||
}
|
||||
|
||||
async fn run_cluster_root_heal_interruption(scenario: InterruptionScenario) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let server_binary = rustfs_binary_path();
|
||||
let evidence_run = if scenario == InterruptionScenario::BackgroundTargetRestart {
|
||||
restart_evidence_run(&server_binary)?
|
||||
} else {
|
||||
None
|
||||
let evidence_run = match scenario {
|
||||
InterruptionScenario::BackgroundTargetRestart => {
|
||||
restart_evidence_run(&server_binary, BACKGROUND_TARGET_RESTART_EVIDENCE)?
|
||||
}
|
||||
InterruptionScenario::BackgroundTargetCrash => {
|
||||
restart_evidence_run(&server_binary, BACKGROUND_TARGET_CRASH_EVIDENCE)?
|
||||
}
|
||||
_ => None,
|
||||
};
|
||||
let mut evidence_objects = Vec::new();
|
||||
let (background_enabled, interruption_node, interruption_kind) = match scenario {
|
||||
InterruptionScenario::IsolatedTargetRestart => (false, 1, "target_restart"),
|
||||
InterruptionScenario::BackgroundTargetRestart => (true, 1, "background_target_restart"),
|
||||
InterruptionScenario::BackgroundTargetCrash => (true, 1, "background_target_crash"),
|
||||
InterruptionScenario::BackgroundCoordinatorRestart => (true, 0, "coordinator_restart"),
|
||||
InterruptionScenario::TargetEndpointBlackhole => (false, 1, "target_endpoint_blackhole"),
|
||||
};
|
||||
@@ -960,6 +1096,7 @@ mod tests {
|
||||
.unwrap_or(4 * 1024 * 1024)
|
||||
.clamp(1024 * 1024, 16 * 1024 * 1024);
|
||||
let mut expected_manifests = Vec::with_capacity(online_object_count);
|
||||
let mut unclean_shutdown_marker_observed = None;
|
||||
for index in 0..online_object_count {
|
||||
let key = format!("cluster/online/object-{index:04}.bin");
|
||||
let payload_seed = u8::try_from(index + 1).expect("clamped object count must fit in u8");
|
||||
@@ -1326,7 +1463,11 @@ mod tests {
|
||||
"Restored target endpoint forwarding"
|
||||
);
|
||||
} else {
|
||||
cluster.stop_node(interruption_node)?;
|
||||
if scenario == InterruptionScenario::BackgroundTargetRestart {
|
||||
cluster.stop_node_gracefully(interruption_node).await?;
|
||||
} else {
|
||||
cluster.stop_node(interruption_node)?;
|
||||
}
|
||||
let stopped_count = metadata_count(&replaced_disk, bucket, &expected_manifests);
|
||||
assert!(
|
||||
stopped_count > 0 && stopped_count < expected_manifests.len(),
|
||||
@@ -1342,9 +1483,12 @@ mod tests {
|
||||
.join(".rustfs.sys")
|
||||
.join("unclean-shutdown");
|
||||
if background_enabled {
|
||||
let marker_exists = unclean_shutdown_marker.is_file();
|
||||
unclean_shutdown_marker_observed = Some(marker_exists);
|
||||
let expected_marker = !matches!(scenario, InterruptionScenario::BackgroundTargetRestart);
|
||||
assert!(
|
||||
unclean_shutdown_marker.is_file(),
|
||||
"background restart must retain the real unclean-shutdown marker"
|
||||
marker_exists == expected_marker,
|
||||
"background restart/crash lane observed unexpected unclean-shutdown marker state"
|
||||
);
|
||||
} else {
|
||||
match std::fs::remove_file(&unclean_shutdown_marker) {
|
||||
@@ -1535,17 +1679,23 @@ mod tests {
|
||||
return Err(format!("heal data rebuilt but task did not finish successfully: {task_status}").into());
|
||||
}
|
||||
|
||||
if let Some((directory, run)) = evidence_run {
|
||||
if let Some(evidence_context) = evidence_run {
|
||||
let restarted_pid = cluster.nodes[1].process.as_ref().ok_or("restarted target is absent")?.id();
|
||||
assert_ne!(target_pid, restarted_pid, "target must be a new process");
|
||||
assert_eq!(file_sha256(&server_binary)?, run.binary.sha256, "server build changed during restart");
|
||||
assert_eq!(
|
||||
file_sha256(&server_binary)?,
|
||||
evidence_context.run.binary.sha256,
|
||||
"server build changed during restart"
|
||||
);
|
||||
let evidence = serde_json::json!({
|
||||
"schema": 1, "case": "background-target-restart", "evidence": "process-restart",
|
||||
"run_id": run.run_id, "source_revision": run.source_revision,
|
||||
"schema": 1, "case": evidence_context.case.id, "evidence": evidence_context.case.evidence,
|
||||
"run_id": evidence_context.run.run_id, "source_revision": evidence_context.run.source_revision,
|
||||
"test_build": compiled_test_identity(),
|
||||
"binary_sha256": run.binary.sha256, "test_binary_sha256": run.test_binary.sha256,
|
||||
"binary_sha256": evidence_context.run.binary.sha256,
|
||||
"test_binary_sha256": evidence_context.run.test_binary.sha256,
|
||||
"topology": {"nodes": cluster.nodes.len(), "drives_per_node": cluster.nodes[0].data_dirs.len()},
|
||||
"pid_before": target_pid, "pid_after": restarted_pid,
|
||||
"unclean_shutdown_marker": unclean_shutdown_marker_observed.unwrap_or(false),
|
||||
"objects": evidence_objects, "node_listings": node_listings,
|
||||
});
|
||||
let data = serde_json::to_vec(&evidence)?;
|
||||
@@ -1555,7 +1705,7 @@ mod tests {
|
||||
let mut output = std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
.create_new(true)
|
||||
.open(directory.join("background-target-restart.json"))?;
|
||||
.open(evidence_context.directory.join(evidence_context.case.oracle))?;
|
||||
output.write_all(&data)?;
|
||||
output.sync_all()?;
|
||||
}
|
||||
|
||||
@@ -21,7 +21,7 @@
|
||||
use super::common::{BoxError, OdmSourceSpec, OdmTestEnv, SeedObject};
|
||||
use crate::fake_s3_target::{BucketMode, Operation};
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::types::{BucketVersioningStatus, VersioningConfiguration};
|
||||
use aws_sdk_s3::types::{BucketVersioningStatus, ObjectAttributes, VersioningConfiguration};
|
||||
use bytes::Bytes;
|
||||
use std::time::Duration;
|
||||
|
||||
@@ -103,11 +103,19 @@ async fn get_miss_pulls_inline_and_serves_locally_afterwards() -> TestResult {
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_large_object_streams_through_and_backfills_in_background() -> TestResult {
|
||||
const PART_SIZE: usize = 5 * 1024 * 1024;
|
||||
let bucket = "odm-get-large";
|
||||
let env = configured_env(bucket, |spec| spec.policy.inline_max_bytes = 4096).await?;
|
||||
let env = configured_env(bucket, |spec| {
|
||||
spec.policy.inline_max_bytes = 4096;
|
||||
spec.policy.multipart_part_size_bytes = u64::try_from(PART_SIZE).expect("part size fits in u64");
|
||||
})
|
||||
.await?;
|
||||
let key = "large/archive.bin";
|
||||
let body = payload(512 * 1024);
|
||||
env.seed_source(SOURCE_BUCKET, &[SeedObject::new(key, body.clone())]);
|
||||
let body = payload(PART_SIZE + 4096);
|
||||
let etag = env
|
||||
.seed_source(SOURCE_BUCKET, &[SeedObject::new(key, body.clone())])
|
||||
.remove(0);
|
||||
assert_eq!(etag.len(), 32, "the source fixture has a plain MD5 ETag");
|
||||
|
||||
let response = env.raw_get(bucket, key).await?;
|
||||
assert_eq!(response.status, 200, "{}", String::from_utf8_lossy(&response.body));
|
||||
@@ -125,6 +133,68 @@ async fn get_large_object_streams_through_and_backfills_in_background() -> TestR
|
||||
vec![None, None],
|
||||
"one passthrough GET plus one background pull, both unranged"
|
||||
);
|
||||
|
||||
let source_requests = env.source.requests().len();
|
||||
let second_part = env.client.get_object().bucket(bucket).key(key).part_number(2).send().await?;
|
||||
assert_eq!(second_part.content_length(), Some(4096), "the completed second part is the tail");
|
||||
assert_eq!(
|
||||
second_part.content_range(),
|
||||
Some(format!("bytes {PART_SIZE}-{}/{}", body.len() - 1, body.len()).as_str()),
|
||||
"partNumber reads the stored multipart boundary"
|
||||
);
|
||||
assert_eq!(
|
||||
second_part.body.collect().await?.into_bytes(),
|
||||
body.slice(PART_SIZE..),
|
||||
"the local second part contains the exact source tail"
|
||||
);
|
||||
let third_part = env
|
||||
.client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key(key)
|
||||
.part_number(3)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("the completed object has exactly two parts");
|
||||
assert_eq!(third_part.code(), Some("InvalidPart"));
|
||||
|
||||
let mut part_marker = None;
|
||||
for (part_number, part_size) in [(1, PART_SIZE), (2, 4096)] {
|
||||
let attributes = env
|
||||
.client
|
||||
.get_object_attributes()
|
||||
.bucket(bucket)
|
||||
.key(key)
|
||||
.object_attributes(ObjectAttributes::ObjectParts)
|
||||
.object_attributes(ObjectAttributes::Etag)
|
||||
.max_parts(1)
|
||||
.set_part_number_marker(part_marker.clone())
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(
|
||||
attributes.e_tag().map(|value| value.trim_matches('"')),
|
||||
Some(etag.as_str()),
|
||||
"multipart write-back preserves the source MD5 ETag"
|
||||
);
|
||||
let parts = attributes
|
||||
.object_parts()
|
||||
.expect("RustFS must expose the stored multipart layout");
|
||||
assert_eq!(parts.total_parts_count(), Some(2));
|
||||
assert_eq!(parts.max_parts(), Some(1));
|
||||
assert_eq!(parts.is_truncated(), Some(part_number == 1));
|
||||
assert_eq!(parts.parts().len(), 1, "RustFS returns one stored part per requested page");
|
||||
assert_eq!(parts.parts()[0].part_number(), Some(part_number));
|
||||
assert_eq!(parts.parts()[0].size(), Some(i64::try_from(part_size).expect("part size fits in i64")));
|
||||
part_marker = parts.next_part_number_marker().map(str::to_owned);
|
||||
if part_number == 1 {
|
||||
assert_eq!(part_marker.as_deref(), Some("1"), "the next request continues after the first part");
|
||||
}
|
||||
}
|
||||
assert_eq!(
|
||||
env.source.requests().len(),
|
||||
source_requests,
|
||||
"local part reads must not consult the source"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
|
||||
@@ -23,16 +23,20 @@
|
||||
|
||||
use super::common::{
|
||||
ALLOW_LOOPBACK_SOURCE_ENV, AdminResponse, BackfillOp, BackfillRequest, BoxError, ODM_MODULE_SWITCH_ENV, ODM_SERVER_ENV,
|
||||
OdmEnvOptions, OdmSourceSpec, OdmTestEnv, SeedObject, start_configured_env, start_configured_env_with,
|
||||
OdmEnvOptions, OdmSourceSpec, OdmTestEnv, SeedObject, start_configured_env, start_configured_env_with, start_source_rustfs,
|
||||
};
|
||||
use crate::common::{RustFSTestEnvironment, replication_fast_env, signed_request};
|
||||
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY, FakeS3Target, Operation};
|
||||
use crate::object_lock::common::put_object_lock_configuration;
|
||||
use crate::replication_extension_test::{
|
||||
ReplicationTargetOptions, enable_bucket_versioning, set_replication_target_with_options,
|
||||
};
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::types::{
|
||||
BucketVersioningStatus, Event, FilterRule, FilterRuleName, NotificationConfiguration, NotificationConfigurationFilter,
|
||||
ObjectLockRetentionMode, QueueConfiguration, S3KeyFilter, ServerSideEncryption, ServerSideEncryptionByDefault,
|
||||
ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging, VersioningConfiguration,
|
||||
ObjectAttributes, ObjectLockRetentionMode, QueueConfiguration, S3KeyFilter, ServerSideEncryption,
|
||||
ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging,
|
||||
VersioningConfiguration,
|
||||
};
|
||||
use bytes::Bytes;
|
||||
use local_ip_address::local_ip;
|
||||
@@ -580,6 +584,130 @@ async fn test_odm_pulled_object_replicates_and_target_as_source_is_rejected() ->
|
||||
"a bucket may not migrate from its own replication target: {}",
|
||||
rejected.body
|
||||
);
|
||||
Box::pin(assert_odm_multipart_replicates_to_rustfs(&env, bucket)).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn assert_odm_multipart_replicates_to_rustfs(env: &OdmTestEnv, bucket: &str) -> TestResult {
|
||||
const PART_SIZE: usize = 5 * 1024 * 1024;
|
||||
let replica = start_source_rustfs().await?;
|
||||
let replica_bucket = "odm-real-replica";
|
||||
replica.create_test_bucket(replica_bucket).await?;
|
||||
enable_bucket_versioning(&replica, replica_bucket).await?;
|
||||
let arn = set_replication_target_with_options(
|
||||
&env.rustfs,
|
||||
bucket,
|
||||
ReplicationTargetOptions {
|
||||
endpoint: &replica.address,
|
||||
access_key: &replica.access_key,
|
||||
secret_key: &replica.secret_key,
|
||||
target_bucket: replica_bucket,
|
||||
secure: false,
|
||||
skip_tls_verify: false,
|
||||
ca_cert_pem: None,
|
||||
},
|
||||
)
|
||||
.await?;
|
||||
put_bucket_replication(&env.rustfs, bucket, &arn).await?;
|
||||
let mut spec = env.fake_source_spec(SOURCE_BUCKET);
|
||||
// Below the 16 MiB inline default the pull is one tee'd PUT with a single
|
||||
// part; force the passthrough + background multipart write-back instead.
|
||||
spec.policy.inline_max_bytes = 4096;
|
||||
spec.policy.multipart_part_size_bytes = PART_SIZE as u64;
|
||||
spec.policy.preserve_etag = true;
|
||||
env.configure_and_wait(bucket, &spec).await?;
|
||||
|
||||
let key = "replicated/preserved-md5-multipart.bin";
|
||||
let body = payload(PART_SIZE + 4096);
|
||||
let source_put = env
|
||||
.source_client()
|
||||
.put_object()
|
||||
.bucket(SOURCE_BUCKET)
|
||||
.key(key)
|
||||
.body(aws_sdk_s3::primitives::ByteStream::from(body.clone()))
|
||||
.send()
|
||||
.await?;
|
||||
let etag = source_put.e_tag().ok_or("source PUT omitted its ETag")?.trim_matches('"');
|
||||
assert_eq!(etag.len(), 32, "the source must retain a single-PUT MD5 ETag");
|
||||
assert!(etag.bytes().all(|byte| byte.is_ascii_hexdigit()));
|
||||
let pulled = env.raw_get(bucket, key).await?;
|
||||
assert_eq!(pulled.status, 200, "{}", String::from_utf8_lossy(&pulled.body));
|
||||
assert_eq!(pulled.body, body);
|
||||
assert!(env.wait_local_listed(bucket, key, SETTLE).await?, "the multipart pull must persist");
|
||||
|
||||
let deadline = Instant::now() + SETTLE;
|
||||
let source_head = loop {
|
||||
let head = env.client.head_object().bucket(bucket).key(key).send().await?;
|
||||
match head.replication_status().map(|status| status.as_str()) {
|
||||
Some("COMPLETED") => break head,
|
||||
Some("FAILED") => return Err("the ODM multipart copy failed replication to RustFS".into()),
|
||||
_ => {
|
||||
assert!(Instant::now() < deadline, "the ODM multipart copy never completed replication to RustFS");
|
||||
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||
}
|
||||
}
|
||||
};
|
||||
let version = source_head
|
||||
.version_id()
|
||||
.ok_or("the versioned ODM copy omitted its version id")?;
|
||||
assert_ne!(version, "null");
|
||||
let replica_client = replica.create_s3_client();
|
||||
for (client, object_bucket) in [(&env.client, bucket), (&replica_client, replica_bucket)] {
|
||||
let attributes = client
|
||||
.get_object_attributes()
|
||||
.bucket(object_bucket)
|
||||
.key(key)
|
||||
.version_id(version)
|
||||
.object_attributes(ObjectAttributes::Etag)
|
||||
.object_attributes(ObjectAttributes::ObjectParts)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(attributes.e_tag().map(|value| value.trim_matches('"')), Some(etag));
|
||||
let parts = attributes
|
||||
.object_parts()
|
||||
.ok_or("the local copy and replica must both expose two parts")?;
|
||||
assert_eq!(parts.total_parts_count(), Some(2));
|
||||
assert_eq!(
|
||||
parts
|
||||
.parts()
|
||||
.iter()
|
||||
.map(|part| (part.part_number(), part.size()))
|
||||
.collect::<Vec<_>>(),
|
||||
[(Some(1), Some(PART_SIZE as i64)), (Some(2), Some(4096))]
|
||||
);
|
||||
}
|
||||
// REPLICA status surfaces on HEAD, like the other inbound-replica checks.
|
||||
let replica_head = replica_client
|
||||
.head_object()
|
||||
.bucket(replica_bucket)
|
||||
.key(key)
|
||||
.version_id(version)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(replica_head.replication_status().map(|status| status.as_str()), Some("REPLICA"));
|
||||
let replica_get = replica_client
|
||||
.get_object()
|
||||
.bucket(replica_bucket)
|
||||
.key(key)
|
||||
.version_id(version)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(replica_get.version_id(), Some(version));
|
||||
assert_eq!(replica_get.body.collect().await?.into_bytes(), body);
|
||||
let boundary = replica_client
|
||||
.get_object()
|
||||
.bucket(replica_bucket)
|
||||
.key(key)
|
||||
.version_id(version)
|
||||
.range(format!("bytes={}-{}", PART_SIZE - 32, PART_SIZE + 31))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(boundary.body.collect().await?.into_bytes(), body.slice(PART_SIZE - 32..PART_SIZE + 32));
|
||||
assert_eq!(
|
||||
env.source.count_requests(Operation::GetObject, key),
|
||||
2,
|
||||
"one passthrough GET plus one background pull; replication and local reads must not fetch the migration source again"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
|
||||
@@ -368,23 +368,23 @@ impl Drop for SlowReplicationTargetGuard {
|
||||
// Mirrors madmin-go `ResyncTargetsInfo`/`ResyncTarget` json tags — the same
|
||||
// shape `mc replicate resync status` decodes.
|
||||
#[derive(Debug, Clone, serde::Deserialize)]
|
||||
struct ReplicationResetStatusResponse {
|
||||
pub(crate) struct ReplicationResetStatusResponse {
|
||||
#[serde(rename = "target", default)]
|
||||
targets: Vec<ReplicationResetStatusTarget>,
|
||||
pub(crate) targets: Vec<ReplicationResetStatusTarget>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, serde::Deserialize)]
|
||||
struct ReplicationResetStatusTarget {
|
||||
pub(crate) struct ReplicationResetStatusTarget {
|
||||
#[serde(rename = "arn", default)]
|
||||
arn: String,
|
||||
pub(crate) arn: String,
|
||||
#[serde(rename = "resetid", default)]
|
||||
reset_id: String,
|
||||
pub(crate) reset_id: String,
|
||||
#[serde(rename = "resyncStatus", default)]
|
||||
status: String,
|
||||
pub(crate) status: String,
|
||||
#[serde(rename = "replicationCount", default)]
|
||||
replicated_count: i64,
|
||||
pub(crate) replicated_count: i64,
|
||||
#[serde(rename = "object", default)]
|
||||
object: String,
|
||||
pub(crate) object: String,
|
||||
}
|
||||
|
||||
fn extract_xml_tag(xml: &str, tag: &str) -> Option<String> {
|
||||
@@ -512,7 +512,7 @@ pub(crate) async fn put_bucket_replication(
|
||||
put_bucket_replication_with_delete_statuses(env, bucket, target_arn, "Enabled", None).await
|
||||
}
|
||||
|
||||
async fn put_bucket_replication_with_delete_statuses(
|
||||
pub(crate) async fn put_bucket_replication_with_delete_statuses(
|
||||
env: &RustFSTestEnvironment,
|
||||
bucket: &str,
|
||||
target_arn: &str,
|
||||
@@ -627,7 +627,7 @@ async fn put_bucket_replication_rules(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn delete_bucket_replication(
|
||||
pub(crate) async fn delete_bucket_replication(
|
||||
env: &RustFSTestEnvironment,
|
||||
bucket: &str,
|
||||
) -> Result<reqwest::Response, Box<dyn Error + Send + Sync>> {
|
||||
@@ -2294,7 +2294,7 @@ async fn site_replication_state_edit(
|
||||
/// return the target `(arn, reset_id)`, asserting the response carries the
|
||||
/// madmin `ResyncTargetsInfo` shape (`target[0].arn` / `target[0].resetid`)
|
||||
/// that `mc replicate resync start` decodes.
|
||||
async fn start_bucket_replication_reset(
|
||||
pub(crate) async fn start_bucket_replication_reset(
|
||||
env: &RustFSTestEnvironment,
|
||||
bucket: &str,
|
||||
) -> Result<(String, String), Box<dyn Error + Send + Sync>> {
|
||||
@@ -2314,7 +2314,7 @@ async fn start_bucket_replication_reset(
|
||||
Ok((arn, reset_id))
|
||||
}
|
||||
|
||||
async fn get_replication_reset_status(
|
||||
pub(crate) async fn get_replication_reset_status(
|
||||
env: &RustFSTestEnvironment,
|
||||
bucket: &str,
|
||||
arn: &str,
|
||||
@@ -3837,6 +3837,244 @@ async fn test_bucket_replication_converges_delete_marker_and_version_purge() ->
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Regression for rustfs/backlog#2340 (not Wasabi specific): a directory
|
||||
/// marker (`prefix/` with a body) in a versioned bucket is stored as the null
|
||||
/// version, like MinIO (`putOpts`: "for directory objects skip creating new
|
||||
/// versions"), and must still replicate to completion instead of staying
|
||||
/// `PENDING`.
|
||||
#[tokio::test]
|
||||
async fn test_bucket_replication_replicates_directory_marker_in_versioned_bucket() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let mut source_env = RustFSTestEnvironment::new().await?;
|
||||
let mut source_env_vars = replication_fast_env();
|
||||
source_env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||
source_env.start_rustfs_server_with_env(vec![], &source_env_vars).await?;
|
||||
|
||||
let mut target_env = RustFSTestEnvironment::new().await?;
|
||||
target_env.start_rustfs_server_without_cleanup(vec![]).await?;
|
||||
|
||||
let source_bucket = "replication-dir-marker-src";
|
||||
let target_bucket = "replication-dir-marker-dst";
|
||||
let source_client = source_env.create_s3_client();
|
||||
let target_client = target_env.create_s3_client();
|
||||
|
||||
source_client.create_bucket().bucket(source_bucket).send().await?;
|
||||
target_client.create_bucket().bucket(target_bucket).send().await?;
|
||||
enable_bucket_versioning(&source_env, source_bucket).await?;
|
||||
enable_bucket_versioning(&target_env, target_bucket).await?;
|
||||
let target_arn = set_replication_target(&source_env, source_bucket, &target_env, target_bucket).await?;
|
||||
put_bucket_replication(&source_env, source_bucket, &target_arn).await?;
|
||||
|
||||
let marker_key = "dir/trailing/";
|
||||
let body = b"directory marker body";
|
||||
let put = source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(marker_key)
|
||||
.body(ByteStream::from_static(body))
|
||||
.send()
|
||||
.await?;
|
||||
assert!(
|
||||
put.version_id()
|
||||
.is_none_or(|id| id == "null" || id == uuid::Uuid::nil().to_string()),
|
||||
"a directory marker is the null version even in a versioned bucket: {:?}",
|
||||
put.version_id()
|
||||
);
|
||||
|
||||
wait_for_source_replication_status(&source_client, source_bucket, marker_key, "COMPLETED", false).await?;
|
||||
|
||||
let replica = target_client
|
||||
.get_object()
|
||||
.bucket(target_bucket)
|
||||
.key(marker_key)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(replica.body.collect().await?.into_bytes().as_ref(), body);
|
||||
let listed = target_client
|
||||
.list_object_versions()
|
||||
.bucket(target_bucket)
|
||||
.prefix(marker_key)
|
||||
.send()
|
||||
.await?;
|
||||
let marker_versions: Vec<_> = listed.versions().iter().filter(|v| v.key() == Some(marker_key)).collect();
|
||||
assert_eq!(marker_versions.len(), 1, "the marker must land exactly once: {marker_versions:?}");
|
||||
assert_eq!(
|
||||
marker_versions[0].version_id(),
|
||||
Some("null"),
|
||||
"the replica keeps the null version identity"
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Regression for rustfs/backlog#2340 (not Wasabi specific): permanently
|
||||
/// deleting a version whose payload lives in a data dir must leave the source
|
||||
/// clean once the purge replicates. Managed-SSE objects are never inlined and a
|
||||
/// plain object above the inline threshold takes the same layout. The version
|
||||
/// retained with a pending purge used to lose its data dir, so the purge state
|
||||
/// could never be applied (`VersionNotFound` on every retry) and the bucket
|
||||
/// stayed `BucketNotEmpty` while `ListObjectVersions` was already empty.
|
||||
#[tokio::test]
|
||||
async fn test_bucket_replication_version_purge_of_non_inline_object_releases_source_bucket() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let (source_env, target_env, source_bucket, target_bucket) = build_sse_replication_pair("purge-datadir", true, true).await?;
|
||||
let target_arn = wait_for_remote_target_arn(&source_env, &source_bucket).await?;
|
||||
put_bucket_replication_with_delete_statuses(&source_env, &source_bucket, &target_arn, "Enabled", Some("Enabled")).await?;
|
||||
let source_client = source_env.create_s3_client();
|
||||
let target_client = target_env.create_s3_client();
|
||||
|
||||
let sse_key = "sse-object.bin";
|
||||
let large_key = "large-object.bin";
|
||||
let sse_put = source_client
|
||||
.put_object()
|
||||
.bucket(&source_bucket)
|
||||
.key(sse_key)
|
||||
.body(ByteStream::from_static(b"encrypted source payload"))
|
||||
.server_side_encryption(ServerSideEncryption::Aes256)
|
||||
.send()
|
||||
.await?;
|
||||
let large_put = source_client
|
||||
.put_object()
|
||||
.bucket(&source_bucket)
|
||||
.key(large_key)
|
||||
.body(ByteStream::from(vec![0x5a; 2 * 1024 * 1024]))
|
||||
.send()
|
||||
.await?;
|
||||
let purged = [
|
||||
(sse_key, sse_put.version_id().ok_or("SSE PUT omitted version ID")?.to_string()),
|
||||
(large_key, large_put.version_id().ok_or("large PUT omitted version ID")?.to_string()),
|
||||
];
|
||||
assert_replication_converged(&source_client, &source_bucket, &target_client, &target_bucket).await?;
|
||||
|
||||
for (key, version_id) in &purged {
|
||||
source_client
|
||||
.delete_object()
|
||||
.bucket(&source_bucket)
|
||||
.key(*key)
|
||||
.version_id(version_id)
|
||||
.send()
|
||||
.await?;
|
||||
}
|
||||
assert_replication_converged(&source_client, &source_bucket, &target_client, &target_bucket).await?;
|
||||
let target_state = list_replication_state(&target_client, &target_bucket).await?;
|
||||
assert!(target_state.is_empty(), "target retained an explicitly purged version: {target_state:?}");
|
||||
|
||||
// The purge state is applied on the source asynchronously after the target
|
||||
// acknowledges the delete; only then does the retained version go away and
|
||||
// the bucket become deletable. A listing that is empty while DeleteBucket
|
||||
// keeps answering BucketNotEmpty is exactly the regression.
|
||||
let deadline = tokio::time::Instant::now() + Duration::from_secs(60);
|
||||
loop {
|
||||
let listing = source_client.list_object_versions().bucket(&source_bucket).send().await?;
|
||||
let listed = listing.versions().len() + listing.delete_markers().len();
|
||||
match source_client.delete_bucket().bucket(&source_bucket).send().await {
|
||||
Ok(_) => break,
|
||||
Err(err) if err.code() == Some("BucketNotEmpty") => {
|
||||
if tokio::time::Instant::now() >= deadline {
|
||||
return Err(format!(
|
||||
"source bucket stayed BucketNotEmpty after the version purge replicated; \
|
||||
ListObjectVersions shows {listed} entries"
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
Err(err) => return Err(err.into()),
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Regression for rustfs/backlog#2340 (not Wasabi specific): a single-part
|
||||
/// object uploaded with `x-amz-checksum-*` must reach the target with the same
|
||||
/// checksum. The outbound options keyed the stored record by algorithm name,
|
||||
/// which the target client sent as `x-amz-meta-*` user metadata, so a replica
|
||||
/// never carried a checksum although the source HEAD returned one.
|
||||
#[tokio::test]
|
||||
async fn test_bucket_replication_forwards_single_part_object_checksums() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let mut source_env = RustFSTestEnvironment::new().await?;
|
||||
let mut source_env_vars = replication_fast_env();
|
||||
source_env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||
source_env.start_rustfs_server_with_env(vec![], &source_env_vars).await?;
|
||||
|
||||
let mut target_env = RustFSTestEnvironment::new().await?;
|
||||
target_env.start_rustfs_server_without_cleanup(vec![]).await?;
|
||||
|
||||
let source_bucket = "replication-checksum-src";
|
||||
let target_bucket = "replication-checksum-dst";
|
||||
let source_client = source_env.create_s3_client();
|
||||
let target_client = target_env.create_s3_client();
|
||||
|
||||
source_client.create_bucket().bucket(source_bucket).send().await?;
|
||||
target_client.create_bucket().bucket(target_bucket).send().await?;
|
||||
enable_bucket_versioning(&source_env, source_bucket).await?;
|
||||
enable_bucket_versioning(&target_env, target_bucket).await?;
|
||||
let target_arn = set_replication_target(&source_env, source_bucket, &target_env, target_bucket).await?;
|
||||
put_bucket_replication(&source_env, source_bucket, &target_arn).await?;
|
||||
|
||||
let body = b"123456789";
|
||||
let crc32_key = "checksum-crc32.txt";
|
||||
let sha256_key = "checksum-sha256.txt";
|
||||
let crc32_put = source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(crc32_key)
|
||||
.body(ByteStream::from_static(body))
|
||||
.checksum_algorithm(aws_sdk_s3::types::ChecksumAlgorithm::Crc32)
|
||||
.send()
|
||||
.await?;
|
||||
let expected_crc32 = crc32_put.checksum_crc32().ok_or("source PUT omitted CRC32")?.to_string();
|
||||
let sha256_put = source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(sha256_key)
|
||||
.body(ByteStream::from_static(body))
|
||||
.checksum_algorithm(aws_sdk_s3::types::ChecksumAlgorithm::Sha256)
|
||||
.send()
|
||||
.await?;
|
||||
let expected_sha256 = sha256_put.checksum_sha256().ok_or("source PUT omitted SHA256")?.to_string();
|
||||
|
||||
for key in [crc32_key, sha256_key] {
|
||||
wait_for_source_replication_status(&source_client, source_bucket, key, "COMPLETED", false).await?;
|
||||
}
|
||||
|
||||
let replica = target_client
|
||||
.head_object()
|
||||
.bucket(target_bucket)
|
||||
.key(crc32_key)
|
||||
.checksum_mode(aws_sdk_s3::types::ChecksumMode::Enabled)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(replica.checksum_crc32(), Some(expected_crc32.as_str()), "replica lost the CRC32 checksum");
|
||||
let replica = target_client
|
||||
.head_object()
|
||||
.bucket(target_bucket)
|
||||
.key(sha256_key)
|
||||
.checksum_mode(aws_sdk_s3::types::ChecksumMode::Enabled)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(
|
||||
replica.checksum_sha256(),
|
||||
Some(expected_sha256.as_str()),
|
||||
"replica lost the SHA256 checksum"
|
||||
);
|
||||
// The bare algorithm name must not leak as user metadata either.
|
||||
assert!(
|
||||
replica
|
||||
.metadata()
|
||||
.is_none_or(|meta| !meta.keys().any(|k| k.eq_ignore_ascii_case("sha256"))),
|
||||
"replica carries the checksum as user metadata: {:?}",
|
||||
replica.metadata()
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_bucket_replication_disabled_delete_marker_does_not_propagate() -> TestResult {
|
||||
init_logging();
|
||||
@@ -8817,9 +9055,11 @@ async fn test_replication_check_flags_multipart_only_version_minting_target() ->
|
||||
.is_some_and(|error| error.contains("CreateMultipartUpload")),
|
||||
"the failure must name the multipart path: {payload}"
|
||||
);
|
||||
// The PutObject leg mirrored, so it is the multipart probe that failed.
|
||||
// The PutObject leg mirrored, so it is the multipart probe that failed;
|
||||
// the mutation phases address the id the PUT reported and still run.
|
||||
assert_eq!(target_report["Phases"]["Put"]["Status"], "OK", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "SKIPPED", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "OK", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["VersionDelete"]["Status"], "OK", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["Cleanup"]["Status"], "OK", "{payload}");
|
||||
|
||||
let probe_key = target
|
||||
@@ -9007,6 +9247,9 @@ async fn test_replication_check_flags_version_minting_target() -> TestResult {
|
||||
let target_bucket = "version-fidelity-dst";
|
||||
target.create_bucket(target_bucket);
|
||||
target.assign_own_version_ids(true);
|
||||
// Wasabi shape: the probe version the VersionDelete phase removed answers
|
||||
// NoSuchVersion to cleanup's second DELETE, which must count as clean.
|
||||
target.reject_unknown_version_deletes(true);
|
||||
|
||||
let mut source_env = RustFSTestEnvironment::new().await?;
|
||||
let mut env_vars = replication_fast_env();
|
||||
@@ -9051,11 +9294,13 @@ async fn test_replication_check_flags_version_minting_target() -> TestResult {
|
||||
fidelity["Code"], "BucketRemoteTargetVersionMismatch",
|
||||
"the failure must carry a machine-readable code: {payload}"
|
||||
);
|
||||
// The probe PUT itself succeeded (fidelity is judged from its response);
|
||||
// the later mutation phases are pointless against a drifting target and
|
||||
// must be skipped, but cleanup still runs.
|
||||
// The probe PUT itself succeeded (fidelity is judged from its response).
|
||||
// The mutation phases address the id the target assigned — the ledger
|
||||
// the worker records per object (rustfs/backlog#2340) — so they run and
|
||||
// pass on a drifting target, and cleanup uses the same id.
|
||||
assert_eq!(target_report["Phases"]["Put"]["Status"], "OK", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "SKIPPED", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "OK", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["VersionDelete"]["Status"], "OK", "{payload}");
|
||||
assert_eq!(target_report["Phases"]["Cleanup"]["Status"], "OK", "{payload}");
|
||||
|
||||
// The probe PUT must carry the source version as `?versionId=` — the
|
||||
@@ -9945,3 +10190,199 @@ async fn test_get_object_tagging_proxies_unreplicated_object_to_replication_targ
|
||||
target.shutdown().await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// backlog#2363
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Wait until the source reports a terminal replication status for `key`.
|
||||
async fn wait_terminal_replication_status(
|
||||
client: &Client,
|
||||
bucket: &str,
|
||||
key: &str,
|
||||
ssec: bool,
|
||||
timeout: Duration,
|
||||
) -> Result<String, Box<dyn Error + Send + Sync>> {
|
||||
let customer_key = BASE64_STANDARD.encode_to_string(REPL17_SSEC_KEY);
|
||||
let customer_key_md5 = sse_customer_key_md5_base64(REPL17_SSEC_KEY);
|
||||
let deadline = tokio::time::Instant::now() + timeout;
|
||||
loop {
|
||||
let request = client.head_object().bucket(bucket).key(key);
|
||||
let head = if ssec {
|
||||
request
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(&customer_key)
|
||||
.sse_customer_key_md5(&customer_key_md5)
|
||||
.send()
|
||||
.await?
|
||||
} else {
|
||||
request.send().await?
|
||||
};
|
||||
let status = head.replication_status().map(|status| status.as_str().to_string());
|
||||
if matches!(status.as_deref(), Some("COMPLETED") | Some("FAILED")) {
|
||||
return Ok(status.unwrap_or_default());
|
||||
}
|
||||
if tokio::time::Instant::now() >= deadline {
|
||||
return Err(format!("{bucket}/{key}: replication never reached a terminal status; last {status:?}").into());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// backlog#2363: SSE-C ciphertext passthrough of objects the source stored
|
||||
/// compressed. The replica on a RustFS target must decrypt to the original
|
||||
/// bytes for a single PUT and for a multipart upload.
|
||||
#[tokio::test]
|
||||
async fn test_bucket_replication_sse_c_compressed_passthrough() -> TestResult {
|
||||
init_logging();
|
||||
const PART_SIZE: usize = 5 * 1024 * 1024;
|
||||
|
||||
let mut source_env = RustFSTestEnvironment::new().await?;
|
||||
let mut target_env = RustFSTestEnvironment::new().await?;
|
||||
let mut source_process_env = replication_fast_env();
|
||||
source_process_env.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||
source_process_env.extend_from_slice(FAST_SCANNER_ENV);
|
||||
source_process_env.extend_from_slice(&[
|
||||
("NO_PROXY", "127.0.0.1,localhost"),
|
||||
("HTTP_PROXY", ""),
|
||||
("HTTPS_PROXY", ""),
|
||||
("RUSTFS_COMPRESSION_ENABLED", "true"),
|
||||
("RUSTFS_COMPRESSION_MULTIPART_ENABLED", "true"),
|
||||
]);
|
||||
source_env.start_rustfs_server_with_env(vec![], &source_process_env).await?;
|
||||
target_env
|
||||
.start_rustfs_server_without_cleanup_with_env(&[
|
||||
("NO_PROXY", "127.0.0.1,localhost"),
|
||||
("HTTP_PROXY", ""),
|
||||
("HTTPS_PROXY", ""),
|
||||
])
|
||||
.await?;
|
||||
|
||||
let source_bucket = "ssec-compressed-src";
|
||||
let target_bucket = "ssec-compressed-dst";
|
||||
let source_client = source_env.create_s3_client();
|
||||
let target_client = target_env.create_s3_client();
|
||||
source_client.create_bucket().bucket(source_bucket).send().await?;
|
||||
target_client.create_bucket().bucket(target_bucket).send().await?;
|
||||
enable_bucket_versioning(&source_env, source_bucket).await?;
|
||||
enable_bucket_versioning(&target_env, target_bucket).await?;
|
||||
let target_arn = set_replication_target(&source_env, source_bucket, &target_env, target_bucket).await?;
|
||||
put_bucket_replication(&source_env, source_bucket, &target_arn).await?;
|
||||
|
||||
let customer_key = BASE64_STANDARD.encode_to_string(REPL17_SSEC_KEY);
|
||||
let customer_key_md5 = sse_customer_key_md5_base64(REPL17_SSEC_KEY);
|
||||
let text = |len: usize, seed: u32| -> Vec<u8> {
|
||||
let mut out = Vec::with_capacity(len + 64);
|
||||
let mut line = 0u64;
|
||||
while out.len() < len {
|
||||
out.extend_from_slice(format!("ssec compressed passthrough seed={seed} line={line} lorem ipsum dolor\n").as_bytes());
|
||||
line += 1;
|
||||
}
|
||||
out.truncate(len);
|
||||
out
|
||||
};
|
||||
|
||||
let single_key = "ssec-compressed-single.txt";
|
||||
let single_body = text(1024 * 1024 + 17, 1);
|
||||
source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(single_key)
|
||||
.content_type("text/plain")
|
||||
.body(ByteStream::from(single_body.clone()))
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(&customer_key)
|
||||
.sse_customer_key_md5(&customer_key_md5)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let multipart_key = "ssec-compressed-multipart.txt";
|
||||
let multipart_parts = [text(PART_SIZE, 2), text(1024 * 1024 + 4096, 3)];
|
||||
let multipart_body: Vec<u8> = multipart_parts.concat();
|
||||
let created = source_client
|
||||
.create_multipart_upload()
|
||||
.bucket(source_bucket)
|
||||
.key(multipart_key)
|
||||
.content_type("text/plain")
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(&customer_key)
|
||||
.sse_customer_key_md5(&customer_key_md5)
|
||||
.send()
|
||||
.await?;
|
||||
let upload_id = created.upload_id().ok_or("missing multipart upload id")?.to_string();
|
||||
let mut completed = Vec::new();
|
||||
for (index, part) in multipart_parts.iter().enumerate() {
|
||||
let part_number = i32::try_from(index + 1)?;
|
||||
let uploaded = source_client
|
||||
.upload_part()
|
||||
.bucket(source_bucket)
|
||||
.key(multipart_key)
|
||||
.upload_id(&upload_id)
|
||||
.part_number(part_number)
|
||||
.body(ByteStream::from(part.clone()))
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(&customer_key)
|
||||
.sse_customer_key_md5(&customer_key_md5)
|
||||
.send()
|
||||
.await?;
|
||||
completed.push(
|
||||
CompletedPart::builder()
|
||||
.part_number(part_number)
|
||||
.set_e_tag(uploaded.e_tag().map(str::to_string))
|
||||
.build(),
|
||||
);
|
||||
}
|
||||
source_client
|
||||
.complete_multipart_upload()
|
||||
.bucket(source_bucket)
|
||||
.key(multipart_key)
|
||||
.upload_id(&upload_id)
|
||||
.multipart_upload(CompletedMultipartUpload::builder().set_parts(Some(completed)).build())
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(&customer_key)
|
||||
.sse_customer_key_md5(&customer_key_md5)
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let mut failures = Vec::new();
|
||||
for (key, body) in [(single_key, &single_body), (multipart_key, &multipart_body)] {
|
||||
let status = wait_terminal_replication_status(&source_client, source_bucket, key, true, Duration::from_secs(120)).await?;
|
||||
if status != "COMPLETED" {
|
||||
failures.push(format!("{key}: source reports {status}"));
|
||||
continue;
|
||||
}
|
||||
let replica = target_client
|
||||
.get_object()
|
||||
.bucket(target_bucket)
|
||||
.key(key)
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(&customer_key)
|
||||
.sse_customer_key_md5(&customer_key_md5)
|
||||
.send()
|
||||
.await;
|
||||
match replica {
|
||||
Ok(replica) => {
|
||||
let content_length = replica.content_length();
|
||||
match replica.body.collect().await {
|
||||
Ok(collected) => {
|
||||
let bytes = collected.into_bytes();
|
||||
if bytes.as_ref() != body.as_slice() {
|
||||
failures.push(format!(
|
||||
"{key}: replica bytes differ (content_length={content_length:?}, got {} bytes, want {})",
|
||||
bytes.len(),
|
||||
body.len()
|
||||
));
|
||||
}
|
||||
}
|
||||
Err(err) => failures.push(format!("{key}: replica body read failed: {err}")),
|
||||
}
|
||||
}
|
||||
Err(err) => failures.push(format!("{key}: replica GET failed: {err}")),
|
||||
}
|
||||
}
|
||||
assert!(
|
||||
failures.is_empty(),
|
||||
"SSE-C compressed passthrough replicas must decrypt to the source bytes: {failures:?}"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -31,17 +31,22 @@
|
||||
//! Adding a target behavior the fleet has shown: add the mode to the fake
|
||||
//! target, add a row here, and record any cell that is red before the fix.
|
||||
|
||||
use crate::common::{RustFSTestEnvironment, init_logging, replication_fast_env};
|
||||
use crate::fake_s3_target::{FAKE_ACCESS_KEY, FAKE_SECRET_KEY};
|
||||
use crate::fake_s3_target::{FakeS3Target, Operation as FakeTargetOperation, RequestRecord};
|
||||
use crate::on_demand_migration::common::fake_source_client;
|
||||
use crate::common::{init_logging, replication_fast_env};
|
||||
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY};
|
||||
use crate::fake_s3_target::{FakeS3Target, FaultAction as FakeTargetFault, Operation as FakeTargetOperation, RequestRecord};
|
||||
use crate::on_demand_migration::common::{OdmEnvOptions, OdmTestEnv, fake_source_client};
|
||||
use crate::replication_extension_test::{
|
||||
LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, enable_bucket_versioning, put_bucket_replication,
|
||||
set_replication_target_with_options,
|
||||
LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, delete_bucket_replication, enable_bucket_versioning,
|
||||
get_replication_reset_status, put_bucket_replication, put_bucket_replication_with_delete_statuses,
|
||||
set_replication_target_with_options, start_bucket_replication_reset,
|
||||
};
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::{ByteStream, DateTime};
|
||||
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart, ObjectLockLegalHoldStatus, ObjectLockMode};
|
||||
use aws_sdk_s3::types::{
|
||||
Checksum, ChecksumAlgorithm, CompletedMultipartUpload, CompletedPart, ObjectAttributes, ObjectLockLegalHold,
|
||||
ObjectLockLegalHoldStatus, ObjectLockMode, ObjectLockRetention, ObjectLockRetentionMode, Tag, Tagging,
|
||||
};
|
||||
use bytes::Bytes;
|
||||
use std::error::Error;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
@@ -63,7 +68,10 @@ enum TargetMode {
|
||||
/// Object Lock parameters must carry `Content-MD5` or `x-amz-checksum-*`.
|
||||
RequireChecksumWithObjectLock,
|
||||
/// AWS S3 / Wasabi / Impossible Cloud: mints its own version ids
|
||||
/// (rustfs/backlog#2085). Data must still land.
|
||||
/// (rustfs/backlog#2085) and, like Wasabi, answers NoSuchVersion to a
|
||||
/// DELETE of an id it never had (rustfs/backlog#2340). Data must still
|
||||
/// land, and every version-addressed mutation must resolve the replica
|
||||
/// through the target-version ledger.
|
||||
MintOwnVersionIds,
|
||||
}
|
||||
|
||||
@@ -80,7 +88,10 @@ impl TargetMode {
|
||||
TargetMode::Baseline => {}
|
||||
TargetMode::RejectAwsChunked => target.reject_aws_chunked_uploads(true),
|
||||
TargetMode::RequireChecksumWithObjectLock => target.require_checksum_for_object_lock(true),
|
||||
TargetMode::MintOwnVersionIds => target.assign_own_version_ids(true),
|
||||
TargetMode::MintOwnVersionIds => {
|
||||
target.assign_own_version_ids(true);
|
||||
target.reject_unknown_version_deletes(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -110,16 +121,23 @@ enum ObjectShape {
|
||||
/// Two-part multipart upload with a GOVERNANCE retention period; the
|
||||
/// lock headers travel on CreateMultipartUpload, which has no body.
|
||||
LockedMultipart,
|
||||
/// ODM stores two local parts while preserving a single-PUT source's MD5 ETag.
|
||||
OdmPreservedMd5Multipart,
|
||||
/// Single-part object uploaded with `x-amz-checksum-sha256`; the replica
|
||||
/// must carry the same header (rustfs/backlog#2340).
|
||||
Checksummed,
|
||||
}
|
||||
|
||||
impl ObjectShape {
|
||||
const ALL: [ObjectShape; 6] = [
|
||||
const ALL: [ObjectShape; 8] = [
|
||||
ObjectShape::Empty,
|
||||
ObjectShape::Plain,
|
||||
ObjectShape::Retention,
|
||||
ObjectShape::LegalHold,
|
||||
ObjectShape::Multipart,
|
||||
ObjectShape::LockedMultipart,
|
||||
ObjectShape::OdmPreservedMd5Multipart,
|
||||
ObjectShape::Checksummed,
|
||||
];
|
||||
|
||||
fn key(self) -> &'static str {
|
||||
@@ -130,6 +148,17 @@ impl ObjectShape {
|
||||
ObjectShape::LegalHold => "matrix/legal-hold.bin",
|
||||
ObjectShape::Multipart => "matrix/multipart.bin",
|
||||
ObjectShape::LockedMultipart => "matrix/locked-multipart.bin",
|
||||
ObjectShape::OdmPreservedMd5Multipart => "matrix/odm-preserved-md5.bin",
|
||||
ObjectShape::Checksummed => "matrix/checksummed.bin",
|
||||
}
|
||||
}
|
||||
|
||||
/// The `x-amz-checksum-*` header the source stored and every upload of
|
||||
/// the replica must repeat.
|
||||
fn forwarded_checksum_header(self) -> Option<&'static str> {
|
||||
match self {
|
||||
ObjectShape::Checksummed => Some("x-amz-checksum-sha256"),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -139,7 +168,8 @@ impl ObjectShape {
|
||||
|
||||
/// Upload the shape to the source and return the bytes the target must
|
||||
/// end up holding.
|
||||
async fn put(self, client: &Client, bucket: &str) -> Result<Bytes, Box<dyn Error + Send + Sync>> {
|
||||
async fn put(self, env: &OdmTestEnv, bucket: &str) -> Result<Bytes, Box<dyn Error + Send + Sync>> {
|
||||
let client = &env.client;
|
||||
let key = self.key();
|
||||
match self {
|
||||
ObjectShape::Empty => {
|
||||
@@ -190,6 +220,19 @@ impl ObjectShape {
|
||||
}
|
||||
ObjectShape::Multipart => multipart_put(client, bucket, key, 0x44, false).await,
|
||||
ObjectShape::LockedMultipart => multipart_put(client, bucket, key, 0x55, true).await,
|
||||
ObjectShape::OdmPreservedMd5Multipart => odm_preserved_md5_multipart(env, bucket, key).await,
|
||||
ObjectShape::Checksummed => {
|
||||
let body = payload(40 * 1024, 0x66);
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(key)
|
||||
.body(ByteStream::from(body.clone()))
|
||||
.checksum_algorithm(ChecksumAlgorithm::Sha256)
|
||||
.send()
|
||||
.await?;
|
||||
Ok(body)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -219,6 +262,595 @@ fn expectation(mode: TargetMode, shape: ObjectShape) -> Expectation {
|
||||
.unwrap_or(Expectation::Completed)
|
||||
}
|
||||
|
||||
/// rustfs/backlog#2340: a target that mints its own version ids (Wasabi,
|
||||
/// AWS S3) answers 404 to a HEAD by the source uuid, which the worker used to
|
||||
/// read as "replica missing" and re-drive the PUT — one more target version
|
||||
/// per heal, MRF retry or resync. Two re-drive shapes, both must converge on
|
||||
/// the single version the first PUT created:
|
||||
/// - the first PUT lands but its response is lost, so the object is FAILED
|
||||
/// and the scanner heal pass re-drives it;
|
||||
/// - an existing-object resync re-drives a COMPLETED object unconditionally.
|
||||
#[tokio::test]
|
||||
async fn matrix_mint_own_version_ids_redrive_does_not_duplicate() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let target = FakeS3Target::start().await?;
|
||||
let target_bucket = "matrix-mint-own-redrive-dst".to_string();
|
||||
target.create_bucket_with_object_lock(target_bucket.clone());
|
||||
target.assign_own_version_ids(true);
|
||||
|
||||
let mut env_vars = replication_fast_env();
|
||||
env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||
env_vars.extend_from_slice(&[
|
||||
("NO_PROXY", "127.0.0.1,localhost"),
|
||||
("HTTP_PROXY", ""),
|
||||
("HTTPS_PROXY", ""),
|
||||
// The scanner heal pass is what re-drives a FAILED object.
|
||||
("RUSTFS_SCANNER_CYCLE", "1"),
|
||||
("RUSTFS_SCANNER_START_DELAY_SECS", "1"),
|
||||
]);
|
||||
let env = OdmTestEnv::start_with(OdmEnvOptions {
|
||||
env: env_vars,
|
||||
..OdmEnvOptions::default()
|
||||
})
|
||||
.await?;
|
||||
let source_env = &env.rustfs;
|
||||
|
||||
let source_bucket = "matrix-mint-own-redrive-src";
|
||||
let source_client = source_env.create_s3_client();
|
||||
source_client
|
||||
.create_bucket()
|
||||
.bucket(source_bucket)
|
||||
.object_lock_enabled_for_bucket(true)
|
||||
.send()
|
||||
.await?;
|
||||
enable_bucket_versioning(source_env, source_bucket).await?;
|
||||
let target_arn = set_replication_target_with_options(
|
||||
source_env,
|
||||
source_bucket,
|
||||
ReplicationTargetOptions {
|
||||
endpoint: &target.address(),
|
||||
access_key: FAKE_ACCESS_KEY,
|
||||
secret_key: FAKE_SECRET_KEY,
|
||||
target_bucket: &target_bucket,
|
||||
secure: false,
|
||||
skip_tls_verify: false,
|
||||
ca_cert_pem: None,
|
||||
},
|
||||
)
|
||||
.await?;
|
||||
put_bucket_replication(source_env, source_bucket, &target_arn).await?;
|
||||
|
||||
// Teach the worker the target's identity contract with one ordinary
|
||||
// write, exactly as production learns it (the PUT response carries the
|
||||
// minted id).
|
||||
let probe_key = "redrive/identity-probe.bin";
|
||||
source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(probe_key)
|
||||
.body(ByteStream::from(payload(4 * 1024, 0x01)))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(
|
||||
wait_for_terminal_replication_status(&source_client, source_bucket, probe_key).await?,
|
||||
"COMPLETED"
|
||||
);
|
||||
|
||||
// Shape 1: the PUT is stored, its response never arrives, heal re-drives.
|
||||
let heal_key = "redrive/heal.bin";
|
||||
target.inject_for_key(FakeTargetOperation::PutObject, heal_key, FakeTargetFault::DisconnectAfterResponse, 1);
|
||||
source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(heal_key)
|
||||
.body(ByteStream::from(payload(8 * 1024, 0x02)))
|
||||
.send()
|
||||
.await?;
|
||||
wait_for_replication_status_and_single_version(&source_client, source_bucket, &target, &target_bucket, heal_key).await?;
|
||||
|
||||
// Shape 2: an existing-object resync re-drives a COMPLETED object.
|
||||
let resync_key = "redrive/resync.bin";
|
||||
source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(resync_key)
|
||||
.body(ByteStream::from(payload(8 * 1024, 0x03)))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(
|
||||
wait_for_terminal_replication_status(&source_client, source_bucket, resync_key).await?,
|
||||
"COMPLETED"
|
||||
);
|
||||
let (reset_arn, _reset_id) = start_bucket_replication_reset(source_env, source_bucket).await?;
|
||||
assert_eq!(reset_arn, target_arn);
|
||||
let resync = async {
|
||||
loop {
|
||||
let status = get_replication_reset_status(source_env, source_bucket, &target_arn).await?;
|
||||
if let Some(entry) = status.targets.iter().find(|entry| entry.arn == target_arn)
|
||||
&& entry.status == "Completed"
|
||||
{
|
||||
return Ok::<_, Box<dyn Error + Send + Sync>>(entry.replicated_count);
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
};
|
||||
let replicated = timeout(Duration::from_secs(90), resync)
|
||||
.await
|
||||
.map_err(|_| "existing-object resync did not complete within 90 seconds")??;
|
||||
assert!(replicated >= 3, "resync must count the located replicas as replicated, got {replicated}");
|
||||
for key in [probe_key, heal_key, resync_key] {
|
||||
let versions = target.stored_versions(&target_bucket, key);
|
||||
assert_eq!(
|
||||
versions.len(),
|
||||
1,
|
||||
"{key}: a re-drive against a target that mints its own version ids must not mint another one: {versions:?}"
|
||||
);
|
||||
}
|
||||
|
||||
target.shutdown().await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// rustfs/backlog#2340 (target-version ledger): on a target that mints its own
|
||||
/// version ids and answers NoSuchVersion to an unknown id (the Wasabi shape),
|
||||
/// every version-addressed mutation must land on the version the target
|
||||
/// assigned, which the replication PUT recorded on the source:
|
||||
/// - a tag update changes the existing target version, no new version;
|
||||
/// - a retention extension and legal hold ON/OFF change that version too;
|
||||
/// - a permanent delete of the older of two same-content generations removes
|
||||
/// exactly that replica and keeps the live one (content identity alone
|
||||
/// could not tell them apart).
|
||||
#[tokio::test]
|
||||
async fn matrix_mint_own_version_ids_addresses_mutations_through_the_ledger() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let target = FakeS3Target::start().await?;
|
||||
let target_bucket = "matrix-mint-own-ledger-dst".to_string();
|
||||
target.create_bucket_with_object_lock(target_bucket.clone());
|
||||
TargetMode::MintOwnVersionIds.apply(&target);
|
||||
|
||||
let mut env_vars = replication_fast_env();
|
||||
env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||
env_vars.extend_from_slice(&[
|
||||
("NO_PROXY", "127.0.0.1,localhost"),
|
||||
("HTTP_PROXY", ""),
|
||||
("HTTPS_PROXY", ""),
|
||||
// The scanner heal pass retries a purge the first attempt lost.
|
||||
("RUSTFS_SCANNER_CYCLE", "1"),
|
||||
("RUSTFS_SCANNER_START_DELAY_SECS", "1"),
|
||||
]);
|
||||
let env = OdmTestEnv::start_with(OdmEnvOptions {
|
||||
env: env_vars,
|
||||
..OdmEnvOptions::default()
|
||||
})
|
||||
.await?;
|
||||
let source_env = &env.rustfs;
|
||||
|
||||
let source_bucket = "matrix-mint-own-ledger-src";
|
||||
let source_client = source_env.create_s3_client();
|
||||
source_client
|
||||
.create_bucket()
|
||||
.bucket(source_bucket)
|
||||
.object_lock_enabled_for_bucket(true)
|
||||
.send()
|
||||
.await?;
|
||||
enable_bucket_versioning(source_env, source_bucket).await?;
|
||||
let target_arn = set_replication_target_with_options(
|
||||
source_env,
|
||||
source_bucket,
|
||||
ReplicationTargetOptions {
|
||||
endpoint: &target.address(),
|
||||
access_key: FAKE_ACCESS_KEY,
|
||||
secret_key: FAKE_SECRET_KEY,
|
||||
target_bucket: &target_bucket,
|
||||
secure: false,
|
||||
skip_tls_verify: false,
|
||||
ca_cert_pem: None,
|
||||
},
|
||||
)
|
||||
.await?;
|
||||
put_bucket_replication_with_delete_statuses(source_env, source_bucket, &target_arn, "Enabled", Some("Enabled")).await?;
|
||||
let target_client = fake_source_client(&target);
|
||||
|
||||
// Tag update on an existing version.
|
||||
let tag_key = "ledger/tags.bin";
|
||||
let tagged = source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(tag_key)
|
||||
.body(ByteStream::from(payload(4 * 1024, 0x01)))
|
||||
.send()
|
||||
.await?;
|
||||
let tag_source_version = tagged.version_id().ok_or("source PUT returned no version id")?.to_string();
|
||||
assert_eq!(
|
||||
wait_for_terminal_replication_status(&source_client, source_bucket, tag_key).await?,
|
||||
"COMPLETED"
|
||||
);
|
||||
let tag_target_version = single_target_version(&target, &target_bucket, tag_key)?;
|
||||
source_client
|
||||
.put_object_tagging()
|
||||
.bucket(source_bucket)
|
||||
.key(tag_key)
|
||||
.version_id(&tag_source_version)
|
||||
.tagging(
|
||||
Tagging::builder()
|
||||
.tag_set(Tag::builder().key("phase").value("after").build()?)
|
||||
.build()?,
|
||||
)
|
||||
.send()
|
||||
.await?;
|
||||
wait_until("tag update on the existing target version", || async {
|
||||
let tags = target_client
|
||||
.get_object_tagging()
|
||||
.bucket(&target_bucket)
|
||||
.key(tag_key)
|
||||
.version_id(&tag_target_version)
|
||||
.send()
|
||||
.await?;
|
||||
Ok(tags
|
||||
.tag_set()
|
||||
.iter()
|
||||
.any(|tag| tag.key() == "phase" && tag.value() == "after"))
|
||||
})
|
||||
.await?;
|
||||
assert_stable_single_version(&target, &target_bucket, tag_key, &tag_target_version).await?;
|
||||
|
||||
// Retention extension and legal hold on an existing version.
|
||||
let lock_key = "ledger/lock.bin";
|
||||
let locked = source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(lock_key)
|
||||
.body(ByteStream::from(payload(4 * 1024, 0x02)))
|
||||
.object_lock_mode(ObjectLockMode::Governance)
|
||||
.object_lock_retain_until_date(retain_until())
|
||||
.send()
|
||||
.await?;
|
||||
let lock_source_version = locked.version_id().ok_or("source PUT returned no version id")?.to_string();
|
||||
assert_eq!(
|
||||
wait_for_terminal_replication_status(&source_client, source_bucket, lock_key).await?,
|
||||
"COMPLETED"
|
||||
);
|
||||
let lock_target_version = single_target_version(&target, &target_bucket, lock_key)?;
|
||||
let extended = DateTime::from_secs(retain_until().secs() + 86_400);
|
||||
source_client
|
||||
.put_object_retention()
|
||||
.bucket(source_bucket)
|
||||
.key(lock_key)
|
||||
.version_id(&lock_source_version)
|
||||
.retention(
|
||||
ObjectLockRetention::builder()
|
||||
.mode(ObjectLockRetentionMode::Governance)
|
||||
.retain_until_date(extended)
|
||||
.build(),
|
||||
)
|
||||
.send()
|
||||
.await?;
|
||||
source_client
|
||||
.put_object_legal_hold()
|
||||
.bucket(source_bucket)
|
||||
.key(lock_key)
|
||||
.version_id(&lock_source_version)
|
||||
.legal_hold(ObjectLockLegalHold::builder().status(ObjectLockLegalHoldStatus::On).build())
|
||||
.send()
|
||||
.await?;
|
||||
wait_until("retention extension and legal hold on the existing target version", || async {
|
||||
let head = target_client
|
||||
.head_object()
|
||||
.bucket(&target_bucket)
|
||||
.key(lock_key)
|
||||
.version_id(&lock_target_version)
|
||||
.send()
|
||||
.await?;
|
||||
Ok(head.object_lock_retain_until_date().map(|date| date.secs()) == Some(extended.secs())
|
||||
&& head.object_lock_legal_hold_status() == Some(&ObjectLockLegalHoldStatus::On))
|
||||
})
|
||||
.await?;
|
||||
source_client
|
||||
.put_object_legal_hold()
|
||||
.bucket(source_bucket)
|
||||
.key(lock_key)
|
||||
.version_id(&lock_source_version)
|
||||
.legal_hold(ObjectLockLegalHold::builder().status(ObjectLockLegalHoldStatus::Off).build())
|
||||
.send()
|
||||
.await?;
|
||||
wait_until("legal hold removal on the existing target version", || async {
|
||||
let head = target_client
|
||||
.head_object()
|
||||
.bucket(&target_bucket)
|
||||
.key(lock_key)
|
||||
.version_id(&lock_target_version)
|
||||
.send()
|
||||
.await?;
|
||||
Ok(head.object_lock_legal_hold_status() == Some(&ObjectLockLegalHoldStatus::Off))
|
||||
})
|
||||
.await?;
|
||||
assert_stable_single_version(&target, &target_bucket, lock_key, &lock_target_version).await?;
|
||||
|
||||
// Permanent delete of the older of two same-content generations.
|
||||
let generations_key = "ledger/generations.bin";
|
||||
let body = payload(4 * 1024, 0x03);
|
||||
let older = source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(generations_key)
|
||||
.body(ByteStream::from(body.clone()))
|
||||
.send()
|
||||
.await?;
|
||||
let older_version = older.version_id().ok_or("source PUT returned no version id")?.to_string();
|
||||
assert_eq!(
|
||||
wait_for_terminal_replication_status(&source_client, source_bucket, generations_key).await?,
|
||||
"COMPLETED"
|
||||
);
|
||||
let older_replica = single_target_version(&target, &target_bucket, generations_key)?;
|
||||
source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(generations_key)
|
||||
.body(ByteStream::from(body))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(
|
||||
wait_for_terminal_replication_status(&source_client, source_bucket, generations_key).await?,
|
||||
"COMPLETED"
|
||||
);
|
||||
wait_until("both generations replicated", || async {
|
||||
Ok(target.stored_versions(&target_bucket, generations_key).len() == 2)
|
||||
})
|
||||
.await?;
|
||||
let newer_replica = target
|
||||
.stored_versions(&target_bucket, generations_key)
|
||||
.into_iter()
|
||||
.map(|(version_id, _)| version_id)
|
||||
.find(|version_id| version_id != &older_replica)
|
||||
.ok_or("the second generation must have its own target version")?;
|
||||
|
||||
source_client
|
||||
.delete_object()
|
||||
.bucket(source_bucket)
|
||||
.key(generations_key)
|
||||
.version_id(&older_version)
|
||||
.send()
|
||||
.await?;
|
||||
wait_until("permanent delete of the older generation's replica", || async {
|
||||
let versions: Vec<String> = target
|
||||
.stored_versions(&target_bucket, generations_key)
|
||||
.into_iter()
|
||||
.map(|(version_id, _)| version_id)
|
||||
.collect();
|
||||
Ok(versions == [newer_replica.clone()])
|
||||
})
|
||||
.await?;
|
||||
assert_stable_single_version(&target, &target_bucket, generations_key, &newer_replica).await?;
|
||||
|
||||
// No mutation above may have gone out as a re-PUT: one upload per key.
|
||||
for key in [tag_key, lock_key] {
|
||||
let puts = target
|
||||
.requests()
|
||||
.iter()
|
||||
.filter(|record| record.key.as_deref() == Some(key) && record.operation == FakeTargetOperation::PutObject)
|
||||
.count();
|
||||
assert_eq!(
|
||||
puts, 1,
|
||||
"{key}: a metadata update must not re-PUT the object on a target that mints its own ids"
|
||||
);
|
||||
}
|
||||
|
||||
target.shutdown().await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// rustfs/backlog#2340 (pending purge lifecycle): a permanent delete whose
|
||||
/// replication keeps failing leaves the version in xl.meta as a PENDING purge,
|
||||
/// hidden from listings. Once the bucket's replication configuration is
|
||||
/// removed nothing can ever confirm that purge remotely, so the delete worker
|
||||
/// must settle it locally (abandoned, with the replica left on the former
|
||||
/// target) — otherwise the bucket stays `BucketNotEmpty` forever with a
|
||||
/// residue the client cannot see.
|
||||
#[tokio::test]
|
||||
async fn matrix_removed_replication_config_abandons_pending_purge() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let target = FakeS3Target::start().await?;
|
||||
let target_bucket = "matrix-abandoned-purge-dst".to_string();
|
||||
target.create_bucket_with_object_lock(target_bucket.clone());
|
||||
TargetMode::MintOwnVersionIds.apply(&target);
|
||||
|
||||
let mut env_vars = replication_fast_env();
|
||||
env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||
env_vars.extend_from_slice(&[
|
||||
("NO_PROXY", "127.0.0.1,localhost"),
|
||||
("HTTP_PROXY", ""),
|
||||
("HTTPS_PROXY", ""),
|
||||
// The scanner heal pass is what revisits a pending purge.
|
||||
("RUSTFS_SCANNER_CYCLE", "1"),
|
||||
("RUSTFS_SCANNER_START_DELAY_SECS", "1"),
|
||||
]);
|
||||
let env = OdmTestEnv::start_with(OdmEnvOptions {
|
||||
env: env_vars,
|
||||
..OdmEnvOptions::default()
|
||||
})
|
||||
.await?;
|
||||
let source_env = &env.rustfs;
|
||||
|
||||
let source_bucket = "matrix-abandoned-purge-src";
|
||||
let source_client = source_env.create_s3_client();
|
||||
source_client
|
||||
.create_bucket()
|
||||
.bucket(source_bucket)
|
||||
.object_lock_enabled_for_bucket(true)
|
||||
.send()
|
||||
.await?;
|
||||
enable_bucket_versioning(source_env, source_bucket).await?;
|
||||
let target_arn = set_replication_target_with_options(
|
||||
source_env,
|
||||
source_bucket,
|
||||
ReplicationTargetOptions {
|
||||
endpoint: &target.address(),
|
||||
access_key: FAKE_ACCESS_KEY,
|
||||
secret_key: FAKE_SECRET_KEY,
|
||||
target_bucket: &target_bucket,
|
||||
secure: false,
|
||||
skip_tls_verify: false,
|
||||
ca_cert_pem: None,
|
||||
},
|
||||
)
|
||||
.await?;
|
||||
put_bucket_replication_with_delete_statuses(source_env, source_bucket, &target_arn, "Enabled", Some("Enabled")).await?;
|
||||
|
||||
let key = "purge/orphaned.bin";
|
||||
let put = source_client
|
||||
.put_object()
|
||||
.bucket(source_bucket)
|
||||
.key(key)
|
||||
.body(ByteStream::from(payload(4 * 1024, 0x07)))
|
||||
.send()
|
||||
.await?;
|
||||
let source_version = put.version_id().ok_or("source PUT returned no version id")?.to_string();
|
||||
assert_eq!(
|
||||
wait_for_terminal_replication_status(&source_client, source_bucket, key).await?,
|
||||
"COMPLETED"
|
||||
);
|
||||
let replica = single_target_version(&target, &target_bucket, key)?;
|
||||
|
||||
// The target refuses every purge: the version stays a pending purge.
|
||||
// More refusals than any scanner cycle can consume within the test.
|
||||
target.inject_for_key(FakeTargetOperation::DeleteObject, key, FakeTargetFault::ResponseStatus(503), 4_000);
|
||||
source_client
|
||||
.delete_object()
|
||||
.bucket(source_bucket)
|
||||
.key(key)
|
||||
.version_id(&source_version)
|
||||
.send()
|
||||
.await?;
|
||||
wait_until("the refused purge to reach the target at least once", || async {
|
||||
Ok(target.count_requests(FakeTargetOperation::DeleteObject, key) >= 1)
|
||||
})
|
||||
.await?;
|
||||
let listed = source_client.list_object_versions().bucket(source_bucket).send().await?;
|
||||
assert!(
|
||||
listed.versions().is_empty() && listed.delete_markers().is_empty(),
|
||||
"a pending purge is hidden from listings: {listed:?}"
|
||||
);
|
||||
let blocked = source_client.delete_bucket().bucket(source_bucket).send().await;
|
||||
assert!(
|
||||
blocked
|
||||
.as_ref()
|
||||
.err()
|
||||
.and_then(|err| err.as_service_error())
|
||||
.is_some_and(|err| err.code() == Some("BucketNotEmpty")),
|
||||
"the hidden pending purge must block DeleteBucket while the target is still configured: {blocked:?}"
|
||||
);
|
||||
|
||||
// Removing the replication configuration orphans the purge; the scanner
|
||||
// heal pass must settle it locally so the bucket becomes deletable.
|
||||
let response = delete_bucket_replication(source_env, source_bucket).await?;
|
||||
assert!(response.status().is_success(), "DeleteBucketReplication: {}", response.status());
|
||||
wait_until("DeleteBucket to succeed once the orphaned purge is abandoned", || async {
|
||||
match source_client.delete_bucket().bucket(source_bucket).send().await {
|
||||
Ok(_) => Ok(true),
|
||||
Err(err) if err.as_service_error().is_some_and(|err| err.code() == Some("BucketNotEmpty")) => Ok(false),
|
||||
Err(err) => Err(err.into()),
|
||||
}
|
||||
})
|
||||
.await?;
|
||||
// Abandoned means abandoned: the replica stays on the former target and,
|
||||
// once the attempts in flight at removal time have drained, no further
|
||||
// purge attempts are sent to it.
|
||||
assert_eq!(
|
||||
single_target_version(&target, &target_bucket, key)?,
|
||||
replica,
|
||||
"an abandoned purge must not touch the replica on the former target"
|
||||
);
|
||||
sleep(Duration::from_secs(3)).await;
|
||||
let settled = target.count_requests(FakeTargetOperation::DeleteObject, key);
|
||||
sleep(Duration::from_secs(3)).await;
|
||||
assert_eq!(
|
||||
target.count_requests(FakeTargetOperation::DeleteObject, key),
|
||||
settled,
|
||||
"purge attempts must stop once the target is no longer configured"
|
||||
);
|
||||
|
||||
target.shutdown().await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn single_target_version(target: &FakeS3Target, target_bucket: &str, key: &str) -> Result<String, Box<dyn Error + Send + Sync>> {
|
||||
let versions = target.stored_versions(target_bucket, key);
|
||||
match versions.as_slice() {
|
||||
[(version_id, false)] => Ok(version_id.clone()),
|
||||
other => Err(format!("{key}: expected exactly one live target version, got {other:?}").into()),
|
||||
}
|
||||
}
|
||||
|
||||
/// The target keeps holding exactly `version_id` for a few scanner cycles: a
|
||||
/// re-driven PUT or a wrong delete would show up here.
|
||||
async fn assert_stable_single_version(target: &FakeS3Target, target_bucket: &str, key: &str, version_id: &str) -> TestResult {
|
||||
for _ in 0..8 {
|
||||
let versions = target.stored_versions(target_bucket, key);
|
||||
if versions.len() != 1 || versions[0].0 != version_id {
|
||||
return Err(
|
||||
format!("{key}: target versions drifted from the single expected replica {version_id}: {versions:?}").into(),
|
||||
);
|
||||
}
|
||||
sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn wait_until<F, Fut>(what: &str, mut probe: F) -> TestResult
|
||||
where
|
||||
F: FnMut() -> Fut,
|
||||
Fut: std::future::Future<Output = Result<bool, Box<dyn Error + Send + Sync>>>,
|
||||
{
|
||||
let wait = async {
|
||||
loop {
|
||||
if probe().await? {
|
||||
return Ok::<_, Box<dyn Error + Send + Sync>>(());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
};
|
||||
timeout(Duration::from_secs(90), wait)
|
||||
.await
|
||||
.map_err(|_| format!("{what} did not happen within 90 seconds"))?
|
||||
}
|
||||
|
||||
/// Wait until `key` is COMPLETED on the source and, for the observation
|
||||
/// window after that, the target still holds exactly one live version of it.
|
||||
async fn wait_for_replication_status_and_single_version(
|
||||
source_client: &Client,
|
||||
source_bucket: &str,
|
||||
target: &FakeS3Target,
|
||||
target_bucket: &str,
|
||||
key: &str,
|
||||
) -> TestResult {
|
||||
// The lost PUT response first settles the object FAILED; only the next
|
||||
// scanner heal pass can turn that into COMPLETED, so FAILED is transient
|
||||
// here and the wait is for COMPLETED alone.
|
||||
let converged = async {
|
||||
loop {
|
||||
let head = source_client.head_object().bucket(source_bucket).key(key).send().await?;
|
||||
if head.replication_status().is_some_and(|status| status.as_str() == "COMPLETED") {
|
||||
return Ok::<_, Box<dyn Error + Send + Sync>>(());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
};
|
||||
timeout(Duration::from_secs(90), converged)
|
||||
.await
|
||||
.map_err(|_| format!("{key}: heal re-drive did not converge to COMPLETED within 90 seconds"))??;
|
||||
// The heal pass keeps visiting the key for a few scanner cycles; a
|
||||
// duplicate would show up here as a second stored version.
|
||||
for _ in 0..12 {
|
||||
let versions = target.stored_versions(target_bucket, key);
|
||||
assert_eq!(versions.len(), 1, "{key}: target minted another version on re-drive: {versions:?}");
|
||||
sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn matrix_baseline_target() -> TestResult {
|
||||
run_row(TargetMode::Baseline).await
|
||||
@@ -270,11 +902,15 @@ async fn run_row(mode: TargetMode) -> TestResult {
|
||||
target.create_bucket_with_object_lock(target_bucket.clone());
|
||||
mode.apply(&target);
|
||||
|
||||
let mut source_env = RustFSTestEnvironment::new().await?;
|
||||
let mut env_vars = replication_fast_env();
|
||||
env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||
env_vars.extend_from_slice(&[("NO_PROXY", "127.0.0.1,localhost"), ("HTTP_PROXY", ""), ("HTTPS_PROXY", "")]);
|
||||
source_env.start_rustfs_server_with_env(vec![], &env_vars).await?;
|
||||
let env = OdmTestEnv::start_with(OdmEnvOptions {
|
||||
env: env_vars,
|
||||
..OdmEnvOptions::default()
|
||||
})
|
||||
.await?;
|
||||
let source_env = &env.rustfs;
|
||||
|
||||
let source_bucket = format!("matrix-{}-src", mode.slug());
|
||||
let source_client = source_env.create_s3_client();
|
||||
@@ -284,9 +920,9 @@ async fn run_row(mode: TargetMode) -> TestResult {
|
||||
.object_lock_enabled_for_bucket(true)
|
||||
.send()
|
||||
.await?;
|
||||
enable_bucket_versioning(&source_env, &source_bucket).await?;
|
||||
enable_bucket_versioning(source_env, &source_bucket).await?;
|
||||
let target_arn = set_replication_target_with_options(
|
||||
&source_env,
|
||||
source_env,
|
||||
&source_bucket,
|
||||
ReplicationTargetOptions {
|
||||
endpoint: &target.address(),
|
||||
@@ -299,14 +935,21 @@ async fn run_row(mode: TargetMode) -> TestResult {
|
||||
},
|
||||
)
|
||||
.await?;
|
||||
put_bucket_replication(&source_env, &source_bucket, &target_arn).await?;
|
||||
put_bucket_replication(source_env, &source_bucket, &target_arn).await?;
|
||||
|
||||
let target_client = fake_source_client(&target);
|
||||
let mut failures = Vec::new();
|
||||
for shape in ObjectShape::ALL {
|
||||
let cell = format!("{}/{:?}", mode.slug(), shape);
|
||||
let expected_body = shape.put(&source_client, &source_bucket).await?;
|
||||
let expected_body = shape.put(&env, &source_bucket).await?;
|
||||
let status = wait_for_terminal_replication_status(&source_client, &source_bucket, shape.key()).await?;
|
||||
if shape == ObjectShape::OdmPreservedMd5Multipart {
|
||||
assert_eq!(
|
||||
env.source.count_requests(FakeTargetOperation::GetObject, shape.key()),
|
||||
2,
|
||||
"one passthrough GET plus one background pull; replication must read the persisted local parts"
|
||||
);
|
||||
}
|
||||
let journal = target.requests();
|
||||
let outcome = match expectation(mode, shape) {
|
||||
Expectation::Completed => {
|
||||
@@ -379,6 +1022,36 @@ async fn check_completed_cell(
|
||||
if uploads.is_empty() {
|
||||
return Err("no upload reached the target although the source reports COMPLETED".into());
|
||||
}
|
||||
if shape == ObjectShape::OdmPreservedMd5Multipart {
|
||||
let key_requests: Vec<_> = journal
|
||||
.iter()
|
||||
.filter(|record| record.key.as_deref() == Some(shape.key()))
|
||||
.collect();
|
||||
for operation in [
|
||||
FakeTargetOperation::CreateMultipartUpload,
|
||||
FakeTargetOperation::CompleteMultipartUpload,
|
||||
] {
|
||||
if !key_requests.iter().any(|record| record.operation == operation) {
|
||||
return Err(format!("preserved-MD5 multipart object did not use {operation:?}").into());
|
||||
}
|
||||
}
|
||||
if key_requests
|
||||
.iter()
|
||||
.any(|record| record.operation == FakeTargetOperation::PutObject)
|
||||
{
|
||||
return Err("preserved-MD5 multipart object used a single PutObject".into());
|
||||
}
|
||||
let mut part_numbers: Vec<_> = key_requests
|
||||
.iter()
|
||||
.filter(|record| record.operation == FakeTargetOperation::UploadPart)
|
||||
.map(|record| record.part_number)
|
||||
.collect();
|
||||
part_numbers.sort_unstable();
|
||||
part_numbers.dedup();
|
||||
if part_numbers != [Some(1), Some(2)] {
|
||||
return Err(format!("preserved-MD5 multipart object uploaded unexpected parts: {part_numbers:?}").into());
|
||||
}
|
||||
}
|
||||
if let Some(framed) = uploads.iter().find(|record| record.transport.aws_chunked) {
|
||||
return Err(format!("{cell}: an upload went out aws-chunked (rustfs#6853 framing): {framed:?}").into());
|
||||
}
|
||||
@@ -401,6 +1074,19 @@ async fn check_completed_cell(
|
||||
}) {
|
||||
return Err(format!("a locked PutObject went out without any integrity header (rustfs#7082): {bare:?}").into());
|
||||
}
|
||||
// rustfs/backlog#2340 contract: a source checksum reaches the target as
|
||||
// the `x-amz-checksum-*` header, not as user metadata; every PutObject of
|
||||
// the shape carries it.
|
||||
if let Some(header) = shape.forwarded_checksum_header()
|
||||
&& let Some(missing) = uploads.iter().find(|record| {
|
||||
record.operation == FakeTargetOperation::PutObject
|
||||
&& !record.transport.checksum_headers.iter().any(|name| name == header)
|
||||
})
|
||||
{
|
||||
return Err(
|
||||
format!("a PutObject went out without the source's {header} header (rustfs/backlog#2340): {missing:?}").into(),
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -455,6 +1141,71 @@ async fn wait_for_terminal_replication_status(
|
||||
}
|
||||
}
|
||||
|
||||
async fn odm_preserved_md5_multipart(env: &OdmTestEnv, bucket: &str, key: &str) -> Result<Bytes, Box<dyn Error + Send + Sync>> {
|
||||
const PART_SIZE: usize = 5 * 1024 * 1024;
|
||||
let origin_bucket = format!("{bucket}-origin");
|
||||
env.source.create_bucket_with_mode(&origin_bucket, BucketMode::Unversioned);
|
||||
let mut spec = env.fake_source_spec(&origin_bucket);
|
||||
// Below the 16 MiB inline default the pull is one tee'd PUT with a single
|
||||
// part; force the passthrough + background multipart write-back instead.
|
||||
spec.policy.inline_max_bytes = 4096;
|
||||
spec.policy.multipart_part_size_bytes = PART_SIZE as u64;
|
||||
spec.policy.preserve_etag = true;
|
||||
env.configure_and_wait(bucket, &spec).await?;
|
||||
|
||||
// A normal source PUT produces the MD5 ETag; only ODM chooses the local parts.
|
||||
let body = payload(PART_SIZE + 4096, 0x66);
|
||||
let source_put = env
|
||||
.source_client()
|
||||
.put_object()
|
||||
.bucket(&origin_bucket)
|
||||
.key(key)
|
||||
.body(ByteStream::from(body.clone()))
|
||||
.send()
|
||||
.await?;
|
||||
let source_etag = source_put.e_tag().ok_or("source PUT omitted its ETag")?.trim_matches('"');
|
||||
assert_eq!(source_etag.len(), 32, "source fixture must have a single-PUT MD5 ETag");
|
||||
assert!(source_etag.bytes().all(|byte| byte.is_ascii_hexdigit()));
|
||||
|
||||
let pulled = env.raw_get(bucket, key).await?;
|
||||
assert_eq!(pulled.status, 200, "{}", String::from_utf8_lossy(&pulled.body));
|
||||
assert_eq!(pulled.body, body);
|
||||
assert!(
|
||||
env.wait_local_listed(bucket, key, Duration::from_secs(30)).await?,
|
||||
"ODM must persist the object"
|
||||
);
|
||||
let attributes = env
|
||||
.client
|
||||
.get_object_attributes()
|
||||
.bucket(bucket)
|
||||
.key(key)
|
||||
.object_attributes(ObjectAttributes::Etag)
|
||||
.object_attributes(ObjectAttributes::ObjectParts)
|
||||
.object_attributes(ObjectAttributes::Checksum)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(attributes.e_tag().map(|etag| etag.trim_matches('"')), Some(source_etag));
|
||||
let parts = attributes
|
||||
.object_parts()
|
||||
.ok_or("the ODM copy must expose its two local parts")?;
|
||||
assert_eq!(parts.total_parts_count(), Some(2));
|
||||
assert_eq!(
|
||||
parts
|
||||
.parts()
|
||||
.iter()
|
||||
.map(|part| (part.part_number(), part.size()))
|
||||
.collect::<Vec<_>>(),
|
||||
[(Some(1), Some(PART_SIZE as i64)), (Some(2), Some(4096))]
|
||||
);
|
||||
assert!(
|
||||
attributes
|
||||
.checksum()
|
||||
.is_none_or(|checksum| checksum == &Checksum::builder().build()),
|
||||
"multipart routing must work without an object checksum record"
|
||||
);
|
||||
Ok(body)
|
||||
}
|
||||
|
||||
async fn multipart_put(
|
||||
client: &Client,
|
||||
bucket: &str,
|
||||
|
||||
@@ -14,9 +14,10 @@
|
||||
|
||||
use crate::common::{
|
||||
RustFSTestClusterEnvironment, RustFSTestEnvironment, admin_request, init_logging, replication_fast_env, rustfs_binary_path,
|
||||
signed_request,
|
||||
};
|
||||
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY, FakeS3Target};
|
||||
use crate::on_demand_migration::common::{ODM_SERVER_ENV, OdmTestEnv, SeedObject};
|
||||
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY, FakeS3Target, Operation as FakeTargetOperation};
|
||||
use crate::on_demand_migration::common::{ODM_SERVER_ENV, OdmTestEnv, SeedObject, fake_source_client};
|
||||
use crate::replication_extension_test::{
|
||||
LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, put_bucket_replication, set_replication_target_with_options,
|
||||
};
|
||||
@@ -25,9 +26,10 @@ use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use aws_sdk_s3::types::{
|
||||
BucketLifecycleConfiguration, BucketVersioningStatus, CompletedMultipartUpload, CompletedPart, DefaultRetention,
|
||||
ExpirationStatus, LifecycleExpiration, LifecycleRule, LifecycleRuleFilter, ObjectLockConfiguration, ObjectLockEnabled,
|
||||
ObjectLockRetentionMode, ObjectLockRule, PublicAccessBlockConfiguration, ServerSideEncryption, ServerSideEncryptionByDefault,
|
||||
ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging, VersioningConfiguration,
|
||||
ExpirationStatus, LifecycleExpiration, LifecycleRule, LifecycleRuleFilter, ObjectAttributes, ObjectLockConfiguration,
|
||||
ObjectLockEnabled, ObjectLockRetentionMode, ObjectLockRule, PublicAccessBlockConfiguration, ServerSideEncryption,
|
||||
ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging,
|
||||
VersioningConfiguration,
|
||||
};
|
||||
use http::{Method, StatusCode};
|
||||
use std::path::{Path, PathBuf};
|
||||
@@ -1205,3 +1207,770 @@ async fn rollback_to_previous_release_reads_current_bucket_metadata() -> TestRes
|
||||
replication_target.shutdown().await;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// rc.5 multipart layouts under the current build (backlog#2147 follow-up to
|
||||
// rustfs#7305)
|
||||
// ---------------------------------------------------------------------------
|
||||
//
|
||||
// rustfs#7305 changed `ObjectInfo::is_multipart` to consult the stored part
|
||||
// list before the ETag shape. Every earlier check of that change used
|
||||
// synthetic metadata; this scenario writes the layouts with the published
|
||||
// rc.5 binary and then reads, describes, and replicates them with the
|
||||
// current build on the same data directory.
|
||||
|
||||
const LAYOUT_PLAIN_BUCKET: &str = "upgrade-layout-plain";
|
||||
const LAYOUT_ENCRYPTED_BUCKET: &str = "upgrade-layout-encrypted";
|
||||
const LAYOUT_REPLICA_BUCKET: &str = "upgrade-layout-replica";
|
||||
const LAYOUT_PART_SIZE: usize = 5 * 1024 * 1024;
|
||||
const LAYOUT_TAIL_SIZE: usize = 1024 * 1024 + 4096;
|
||||
const LAYOUT_SSEC_KEY: &str = "0123456789abcdef0123456789abcdef";
|
||||
const LAYOUT_REPLICATION_TIMEOUT: Duration = Duration::from_secs(180);
|
||||
|
||||
struct LayoutCase {
|
||||
bucket: &'static str,
|
||||
key: &'static str,
|
||||
/// Empty for a single PUT.
|
||||
part_sizes: Vec<usize>,
|
||||
body: Vec<u8>,
|
||||
ssec: bool,
|
||||
/// `false` for layouts whose replication is a known pre-existing failure;
|
||||
/// their outcome is logged, not asserted.
|
||||
assert_replication: bool,
|
||||
/// Recorded from the rc.5 writer.
|
||||
rc5_etag: String,
|
||||
/// Whether rc.5 reported `ObjectParts` for the object.
|
||||
rc5_reported_parts: Option<usize>,
|
||||
}
|
||||
|
||||
impl LayoutCase {
|
||||
fn is_multipart_layout(&self) -> bool {
|
||||
self.part_sizes.len() > 1
|
||||
}
|
||||
|
||||
fn label(&self) -> String {
|
||||
format!("{}/{}", self.bucket, self.key)
|
||||
}
|
||||
}
|
||||
|
||||
fn layout_noise(len: usize, seed: u64) -> Vec<u8> {
|
||||
let mut state = seed ^ 0x9E37_79B9_7F4A_7C15;
|
||||
(0..len)
|
||||
.map(|_| {
|
||||
state ^= state << 13;
|
||||
state ^= state >> 7;
|
||||
state ^= state << 17;
|
||||
(state >> 24) as u8
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn layout_text(len: usize, seed: u64) -> Vec<u8> {
|
||||
let mut out = Vec::with_capacity(len + 64);
|
||||
let mut line = 0u64;
|
||||
while out.len() < len {
|
||||
out.extend_from_slice(format!("rc5 legacy layout seed={seed} line={line} lorem ipsum dolor sit amet\n").as_bytes());
|
||||
line += 1;
|
||||
}
|
||||
out.truncate(len);
|
||||
out
|
||||
}
|
||||
|
||||
fn layout_ssec_key_md5() -> String {
|
||||
use md5::{Digest as _, Md5};
|
||||
let mut hasher = Md5::new();
|
||||
hasher.update(LAYOUT_SSEC_KEY.as_bytes());
|
||||
base64_simd::STANDARD.encode_to_string(hasher.finalize())
|
||||
}
|
||||
|
||||
fn layout_ssec_key() -> String {
|
||||
base64_simd::STANDARD.encode_to_string(LAYOUT_SSEC_KEY)
|
||||
}
|
||||
|
||||
async fn layout_head(
|
||||
client: &Client,
|
||||
case: &LayoutCase,
|
||||
) -> Result<aws_sdk_s3::operation::head_object::HeadObjectOutput, BoxError> {
|
||||
let request = client.head_object().bucket(case.bucket).key(case.key);
|
||||
let request = if case.ssec {
|
||||
request
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(layout_ssec_key())
|
||||
.sse_customer_key_md5(layout_ssec_key_md5())
|
||||
} else {
|
||||
request
|
||||
};
|
||||
Ok(request.send().await?)
|
||||
}
|
||||
|
||||
async fn layout_get(
|
||||
client: &Client,
|
||||
case: &LayoutCase,
|
||||
range: Option<String>,
|
||||
part_number: Option<i32>,
|
||||
) -> Result<aws_sdk_s3::operation::get_object::GetObjectOutput, BoxError> {
|
||||
let request = client.get_object().bucket(case.bucket).key(case.key);
|
||||
let request = if case.ssec {
|
||||
request
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(layout_ssec_key())
|
||||
.sse_customer_key_md5(layout_ssec_key_md5())
|
||||
} else {
|
||||
request
|
||||
};
|
||||
let request = request.set_range(range).set_part_number(part_number);
|
||||
Ok(request.send().await?)
|
||||
}
|
||||
|
||||
async fn layout_attributes(
|
||||
client: &Client,
|
||||
case: &LayoutCase,
|
||||
) -> Result<aws_sdk_s3::operation::get_object_attributes::GetObjectAttributesOutput, BoxError> {
|
||||
let request = client
|
||||
.get_object_attributes()
|
||||
.bucket(case.bucket)
|
||||
.key(case.key)
|
||||
.object_attributes(ObjectAttributes::Etag)
|
||||
.object_attributes(ObjectAttributes::ObjectParts)
|
||||
.object_attributes(ObjectAttributes::ObjectSize)
|
||||
.max_parts(100);
|
||||
let request = if case.ssec {
|
||||
request
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(layout_ssec_key())
|
||||
.sse_customer_key_md5(layout_ssec_key_md5())
|
||||
} else {
|
||||
request
|
||||
};
|
||||
Ok(request.send().await?)
|
||||
}
|
||||
|
||||
/// Write `case` with the rc.5 client; single PUT when `part_sizes` is empty.
|
||||
async fn layout_write(client: &Client, case: &LayoutCase) -> Result<(), BoxError> {
|
||||
let content_type = "text/plain";
|
||||
if case.part_sizes.is_empty() {
|
||||
let request = client
|
||||
.put_object()
|
||||
.bucket(case.bucket)
|
||||
.key(case.key)
|
||||
.content_type(content_type)
|
||||
.body(ByteStream::from(case.body.clone()));
|
||||
let request = if case.ssec {
|
||||
request
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(layout_ssec_key())
|
||||
.sse_customer_key_md5(layout_ssec_key_md5())
|
||||
} else {
|
||||
request
|
||||
};
|
||||
request.send().await?;
|
||||
return Ok(());
|
||||
}
|
||||
let create = client
|
||||
.create_multipart_upload()
|
||||
.bucket(case.bucket)
|
||||
.key(case.key)
|
||||
.content_type(content_type);
|
||||
let create = if case.ssec {
|
||||
create
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(layout_ssec_key())
|
||||
.sse_customer_key_md5(layout_ssec_key_md5())
|
||||
} else {
|
||||
create
|
||||
};
|
||||
let created = create.send().await?;
|
||||
let upload_id = created.upload_id().ok_or("CreateMultipartUpload omitted upload ID")?;
|
||||
let mut completed = Vec::with_capacity(case.part_sizes.len());
|
||||
let mut offset = 0usize;
|
||||
for (index, size) in case.part_sizes.iter().enumerate() {
|
||||
let part_number = i32::try_from(index + 1)?;
|
||||
let chunk = case.body[offset..offset + size].to_vec();
|
||||
offset += size;
|
||||
let upload = client
|
||||
.upload_part()
|
||||
.bucket(case.bucket)
|
||||
.key(case.key)
|
||||
.upload_id(upload_id)
|
||||
.part_number(part_number)
|
||||
.body(ByteStream::from(chunk));
|
||||
let upload = if case.ssec {
|
||||
upload
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(layout_ssec_key())
|
||||
.sse_customer_key_md5(layout_ssec_key_md5())
|
||||
} else {
|
||||
upload
|
||||
};
|
||||
let uploaded = upload.send().await?;
|
||||
completed.push(
|
||||
CompletedPart::builder()
|
||||
.part_number(part_number)
|
||||
.e_tag(uploaded.e_tag().ok_or("UploadPart omitted ETag")?)
|
||||
.build(),
|
||||
);
|
||||
}
|
||||
client
|
||||
.complete_multipart_upload()
|
||||
.bucket(case.bucket)
|
||||
.key(case.key)
|
||||
.upload_id(upload_id)
|
||||
.multipart_upload(CompletedMultipartUpload::builder().set_parts(Some(completed)).build())
|
||||
.send()
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn layout_cases() -> Vec<LayoutCase> {
|
||||
let two = vec![LAYOUT_PART_SIZE, LAYOUT_TAIL_SIZE];
|
||||
let three = vec![LAYOUT_PART_SIZE, LAYOUT_PART_SIZE, 4096];
|
||||
let total = |sizes: &[usize]| sizes.iter().sum::<usize>();
|
||||
let case = |bucket, key, part_sizes: Vec<usize>, body: Vec<u8>, ssec| LayoutCase {
|
||||
bucket,
|
||||
key,
|
||||
part_sizes,
|
||||
body,
|
||||
ssec,
|
||||
assert_replication: true,
|
||||
rc5_etag: String::new(),
|
||||
rc5_reported_parts: None,
|
||||
};
|
||||
vec![
|
||||
case(LAYOUT_PLAIN_BUCKET, "plain/single.bin", vec![], layout_noise(1024 * 1024 + 17, 1), false),
|
||||
case(
|
||||
LAYOUT_PLAIN_BUCKET,
|
||||
"plain/multipart-2.bin",
|
||||
two.clone(),
|
||||
layout_noise(total(&two), 2),
|
||||
false,
|
||||
),
|
||||
case(
|
||||
LAYOUT_PLAIN_BUCKET,
|
||||
"plain/multipart-3.bin",
|
||||
three.clone(),
|
||||
layout_noise(total(&three), 3),
|
||||
false,
|
||||
),
|
||||
case(
|
||||
LAYOUT_PLAIN_BUCKET,
|
||||
"plain/compressed-single.txt",
|
||||
vec![],
|
||||
layout_text(1024 * 1024 + 17, 4),
|
||||
false,
|
||||
),
|
||||
case(
|
||||
LAYOUT_PLAIN_BUCKET,
|
||||
"plain/compressed-multipart-2.txt",
|
||||
two.clone(),
|
||||
layout_text(total(&two), 5),
|
||||
false,
|
||||
),
|
||||
case(
|
||||
LAYOUT_PLAIN_BUCKET,
|
||||
"plain/ssec-multipart-2.bin",
|
||||
two.clone(),
|
||||
layout_noise(total(&two), 6),
|
||||
true,
|
||||
),
|
||||
// SSE-C passthrough replicates the stored ciphertext part by part; a
|
||||
// compressible first part is stored well below 5 MiB, so the sender
|
||||
// declares each part's plaintext length and the target validates the
|
||||
// 5 MiB minimum against it (rustfs/backlog#2363). rc.5 as the sender
|
||||
// still fails this layout (see `rc5_baseline_replicates_multipart_layouts`).
|
||||
case(
|
||||
LAYOUT_PLAIN_BUCKET,
|
||||
"plain/ssec-compressed-multipart-2.txt",
|
||||
two.clone(),
|
||||
layout_text(total(&two), 7),
|
||||
true,
|
||||
),
|
||||
case(
|
||||
LAYOUT_ENCRYPTED_BUCKET,
|
||||
"encrypted/single.bin",
|
||||
vec![],
|
||||
layout_noise(1024 * 1024 + 17, 8),
|
||||
false,
|
||||
),
|
||||
case(
|
||||
LAYOUT_ENCRYPTED_BUCKET,
|
||||
"encrypted/multipart-2.bin",
|
||||
two.clone(),
|
||||
layout_noise(total(&two), 9),
|
||||
false,
|
||||
),
|
||||
case(
|
||||
LAYOUT_ENCRYPTED_BUCKET,
|
||||
"encrypted/multipart-3.bin",
|
||||
three.clone(),
|
||||
layout_noise(total(&three), 10),
|
||||
false,
|
||||
),
|
||||
case(
|
||||
LAYOUT_ENCRYPTED_BUCKET,
|
||||
"encrypted/compressed-multipart-2.txt",
|
||||
two.clone(),
|
||||
layout_text(total(&two), 11),
|
||||
false,
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
fn layout_server_env() -> Vec<(&'static str, &'static str)> {
|
||||
let mut env = bucket_config_server_env();
|
||||
env.push(("RUSTFS_COMPRESSION_ENABLED", "true"));
|
||||
env.push(("RUSTFS_COMPRESSION_MULTIPART_ENABLED", "true"));
|
||||
env
|
||||
}
|
||||
|
||||
fn layout_reported_parts(attributes: &aws_sdk_s3::operation::get_object_attributes::GetObjectAttributesOutput) -> Option<usize> {
|
||||
attributes.object_parts().map(|parts| parts.parts().len())
|
||||
}
|
||||
|
||||
async fn assert_layout_readable(client: &Client, case: &LayoutCase, context: &str) -> TestResult {
|
||||
let label = case.label();
|
||||
let head = layout_head(client, case).await?;
|
||||
assert_eq!(
|
||||
head.e_tag().map(|etag| etag.trim_matches('"')),
|
||||
Some(case.rc5_etag.as_str()),
|
||||
"{context}: {label}: the ETag written by rc.5 must be reported unchanged"
|
||||
);
|
||||
assert_eq!(
|
||||
head.content_length(),
|
||||
Some(i64::try_from(case.body.len())?),
|
||||
"{context}: {label}: HEAD content length"
|
||||
);
|
||||
|
||||
let full = layout_get(client, case, None, None).await?.body.collect().await?.into_bytes();
|
||||
assert_eq!(full.len(), case.body.len(), "{context}: {label}: full GET length");
|
||||
assert!(full == case.body, "{context}: {label}: full GET body must equal the rc.5 upload");
|
||||
|
||||
if case.is_multipart_layout() {
|
||||
let first = case.part_sizes[0];
|
||||
let range = format!("bytes={}-{}", first - 32, first + 31);
|
||||
let crossing = layout_get(client, case, Some(range), None)
|
||||
.await?
|
||||
.body
|
||||
.collect()
|
||||
.await?
|
||||
.into_bytes();
|
||||
assert!(
|
||||
crossing == case.body[first - 32..first + 32],
|
||||
"{context}: {label}: range across the first part boundary"
|
||||
);
|
||||
let tail_start: usize = case.part_sizes[..case.part_sizes.len() - 1].iter().sum();
|
||||
let last_number = i32::try_from(case.part_sizes.len())?;
|
||||
let last = layout_get(client, case, None, Some(last_number)).await?;
|
||||
assert_eq!(
|
||||
last.content_length(),
|
||||
Some(i64::try_from(case.part_sizes[case.part_sizes.len() - 1])?),
|
||||
"{context}: {label}: partNumber={last_number} length"
|
||||
);
|
||||
let last_body = last.body.collect().await?.into_bytes();
|
||||
assert!(
|
||||
last_body == case.body[tail_start..],
|
||||
"{context}: {label}: partNumber={last_number} body must be the stored last part"
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn assert_layout_attributes(client: &Client, case: &LayoutCase, context: &str) -> TestResult {
|
||||
let label = case.label();
|
||||
let attributes = layout_attributes(client, case).await?;
|
||||
assert_eq!(
|
||||
attributes.e_tag().map(|etag| etag.trim_matches('"')),
|
||||
Some(case.rc5_etag.as_str()),
|
||||
"{context}: {label}: attributes ETag"
|
||||
);
|
||||
assert_eq!(
|
||||
attributes.object_size(),
|
||||
Some(i64::try_from(case.body.len())?),
|
||||
"{context}: {label}: attributes ObjectSize"
|
||||
);
|
||||
if case.is_multipart_layout() {
|
||||
let parts = attributes
|
||||
.object_parts()
|
||||
.ok_or_else(|| format!("{context}: {label}: multipart layout must expose ObjectParts"))?;
|
||||
assert_eq!(
|
||||
parts.total_parts_count(),
|
||||
Some(i32::try_from(case.part_sizes.len())?),
|
||||
"{context}: {label}: TotalPartsCount"
|
||||
);
|
||||
let observed: Vec<(Option<i32>, Option<i64>)> =
|
||||
parts.parts().iter().map(|part| (part.part_number(), part.size())).collect();
|
||||
let expected: Vec<(Option<i32>, Option<i64>)> = case
|
||||
.part_sizes
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, size)| (Some(index as i32 + 1), Some(*size as i64)))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
observed, expected,
|
||||
"{context}: {label}: ObjectParts must report the plaintext part layout"
|
||||
);
|
||||
} else {
|
||||
assert!(
|
||||
attributes.object_parts().is_none_or(|parts| parts.parts().is_empty()),
|
||||
"{context}: {label}: a single PUT must not report stored parts"
|
||||
);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn put_layout_replication_rule(env: &RustFSTestEnvironment, bucket: &str, arn: &str) -> TestResult {
|
||||
let body = format!(
|
||||
r#"<ReplicationConfiguration xmlns="http://s3.amazonaws.com/doc/2006-03-01/">
|
||||
<Role></Role>
|
||||
<Rule>
|
||||
<ID>legacy-layouts</ID>
|
||||
<Priority>1</Priority>
|
||||
<Status>Enabled</Status>
|
||||
<Filter><Prefix></Prefix></Filter>
|
||||
<DeleteMarkerReplication><Status>Enabled</Status></DeleteMarkerReplication>
|
||||
<DeleteReplication><Status>Enabled</Status></DeleteReplication>
|
||||
<ExistingObjectReplication><Status>Enabled</Status></ExistingObjectReplication>
|
||||
<Destination><Bucket>{arn}</Bucket></Destination>
|
||||
</Rule>
|
||||
</ReplicationConfiguration>"#
|
||||
);
|
||||
let url = format!("{}/{bucket}?replication", env.url);
|
||||
let response = signed_request(
|
||||
Method::PUT,
|
||||
&url,
|
||||
&env.access_key,
|
||||
&env.secret_key,
|
||||
Some(body.into_bytes()),
|
||||
Some("application/xml"),
|
||||
)
|
||||
.await?;
|
||||
if response.status() != StatusCode::OK {
|
||||
let status = response.status();
|
||||
let body = response.text().await.unwrap_or_default();
|
||||
return Err(format!("put replication rule on {bucket} failed: {status} {body}").into());
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Wait for the existing-object replication of `case` to reach a terminal
|
||||
/// status and return it (`COMPLETED` or `FAILED`).
|
||||
async fn wait_layout_replication_terminal(client: &Client, case: &LayoutCase) -> Result<String, BoxError> {
|
||||
let deadline = Instant::now() + LAYOUT_REPLICATION_TIMEOUT;
|
||||
loop {
|
||||
let head = layout_head(client, case).await?;
|
||||
let status = head.replication_status().map(|status| status.as_str().to_string());
|
||||
if matches!(status.as_deref(), Some("COMPLETED") | Some("FAILED")) {
|
||||
return Ok(status.unwrap_or_default());
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return Err(format!(
|
||||
"{}: existing-object replication never reached a terminal status; last {status:?}",
|
||||
case.label()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct LayoutTransport {
|
||||
status: String,
|
||||
uploaded_parts: Vec<i32>,
|
||||
single_puts: usize,
|
||||
completes: usize,
|
||||
/// Raw per-key journal in target order: (sequence, operation, part number,
|
||||
/// upload id), so duplicate drives can be told apart from retries.
|
||||
journal: Vec<(u64, String, Option<i32>, Option<String>)>,
|
||||
}
|
||||
|
||||
/// Configure every layout bucket to replicate its existing objects to a fresh
|
||||
/// fake target, wait for each case to settle, and report the transport the
|
||||
/// target observed per case.
|
||||
async fn replicate_layouts(
|
||||
env: &RustFSTestEnvironment,
|
||||
client: &Client,
|
||||
cases: &[LayoutCase],
|
||||
) -> Result<(FakeS3Target, Vec<LayoutTransport>), BoxError> {
|
||||
let target = FakeS3Target::start().await?;
|
||||
target.create_bucket(LAYOUT_REPLICA_BUCKET);
|
||||
for bucket in [LAYOUT_PLAIN_BUCKET, LAYOUT_ENCRYPTED_BUCKET] {
|
||||
let arn = set_replication_target_with_options(
|
||||
env,
|
||||
bucket,
|
||||
ReplicationTargetOptions {
|
||||
endpoint: &target.address(),
|
||||
access_key: FAKE_ACCESS_KEY,
|
||||
secret_key: FAKE_SECRET_KEY,
|
||||
target_bucket: LAYOUT_REPLICA_BUCKET,
|
||||
secure: false,
|
||||
skip_tls_verify: false,
|
||||
ca_cert_pem: None,
|
||||
},
|
||||
)
|
||||
.await?;
|
||||
put_layout_replication_rule(env, bucket, &arn).await?;
|
||||
}
|
||||
let mut statuses = Vec::with_capacity(cases.len());
|
||||
for case in cases {
|
||||
statuses.push(wait_layout_replication_terminal(client, case).await?);
|
||||
}
|
||||
let journal = target.requests();
|
||||
let mut transports = Vec::with_capacity(cases.len());
|
||||
for (case, status) in cases.iter().zip(statuses) {
|
||||
let key_requests: Vec<_> = journal
|
||||
.iter()
|
||||
.filter(|record| record.key.as_deref() == Some(case.key))
|
||||
.collect();
|
||||
let mut uploaded_parts: Vec<i32> = key_requests
|
||||
.iter()
|
||||
.filter(|record| record.operation == FakeTargetOperation::UploadPart)
|
||||
.filter_map(|record| record.part_number)
|
||||
.collect();
|
||||
uploaded_parts.sort_unstable();
|
||||
uploaded_parts.dedup();
|
||||
let transport = LayoutTransport {
|
||||
status,
|
||||
uploaded_parts,
|
||||
single_puts: key_requests
|
||||
.iter()
|
||||
.filter(|record| record.operation == FakeTargetOperation::PutObject)
|
||||
.count(),
|
||||
completes: key_requests
|
||||
.iter()
|
||||
.filter(|record| record.operation == FakeTargetOperation::CompleteMultipartUpload)
|
||||
.count(),
|
||||
journal: key_requests
|
||||
.iter()
|
||||
.map(|record| {
|
||||
(
|
||||
record.sequence,
|
||||
format!("{:?}", record.operation),
|
||||
record.part_number,
|
||||
record.upload_id.as_ref().map(|id| id.chars().take(12).collect()),
|
||||
)
|
||||
})
|
||||
.collect(),
|
||||
};
|
||||
tracing::info!(
|
||||
target: "e2e_test::upgrade_compatibility_test",
|
||||
object = %case.label(),
|
||||
?transport,
|
||||
"replication transport observed on the target"
|
||||
);
|
||||
transports.push(transport);
|
||||
}
|
||||
Ok((target, transports))
|
||||
}
|
||||
|
||||
/// rc.5 writes single-PUT, multipart, compressed, SSE-C and SSE-S3 layouts;
|
||||
/// the current build must read every byte, expose the stored part layout
|
||||
/// through GetObjectAttributes and partNumber reads, and replicate the objects
|
||||
/// with the transport that matches their stored parts.
|
||||
#[tokio::test]
|
||||
#[ignore = "requires the pinned 1.0.0-rc.5 release binary"]
|
||||
async fn direct_upgrade_from_rc5_preserves_multipart_layouts() -> TestResult {
|
||||
init_logging();
|
||||
let previous_binary = source_binary()?;
|
||||
let server_env = layout_server_env();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server_from_binary(&previous_binary, vec![], &server_env)
|
||||
.await?;
|
||||
let old_client = env.create_s3_client();
|
||||
env.create_test_bucket(LAYOUT_PLAIN_BUCKET).await?;
|
||||
env.create_test_bucket(LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
enable_versioning(&old_client, LAYOUT_PLAIN_BUCKET).await?;
|
||||
enable_versioning(&old_client, LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
put_default_sse_s3_encryption(&old_client, LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
assert_default_sse_s3_encryption(&old_client, LAYOUT_ENCRYPTED_BUCKET, "rc.5").await?;
|
||||
|
||||
let mut cases = layout_cases();
|
||||
for case in cases.iter_mut() {
|
||||
layout_write(&old_client, case).await?;
|
||||
let head = layout_head(&old_client, case).await?;
|
||||
case.rc5_etag = head
|
||||
.e_tag()
|
||||
.ok_or_else(|| format!("{}: rc.5 HEAD omitted the ETag", case.label()))?
|
||||
.trim_matches('"')
|
||||
.to_string();
|
||||
case.rc5_reported_parts = layout_attributes(&old_client, case)
|
||||
.await
|
||||
.ok()
|
||||
.and_then(|a| layout_reported_parts(&a));
|
||||
tracing::info!(
|
||||
target: "e2e_test::upgrade_compatibility_test",
|
||||
object = %case.label(),
|
||||
parts = case.part_sizes.len(),
|
||||
etag = %case.rc5_etag,
|
||||
rc5_reported_parts = ?case.rc5_reported_parts,
|
||||
"rc.5 wrote a legacy layout"
|
||||
);
|
||||
}
|
||||
// The rc.5 writer must itself still read what it wrote, so a later
|
||||
// failure is attributable to the upgrade rather than to the fixture.
|
||||
for case in &cases {
|
||||
assert_layout_readable(&old_client, case, "rc.5").await?;
|
||||
}
|
||||
|
||||
// Upgrade in place.
|
||||
env.restart_server_preserving_data(vec![], &server_env).await?;
|
||||
let client = env.create_s3_client();
|
||||
for case in &cases {
|
||||
assert_layout_readable(&client, case, "upgraded").await?;
|
||||
assert_layout_attributes(&client, case, "upgraded").await?;
|
||||
}
|
||||
|
||||
// Replicate the pre-existing objects with the current build.
|
||||
let (target, transports) = replicate_layouts(&env, &client, &cases).await?;
|
||||
let replica_client = fake_source_client(&target);
|
||||
for (case, transport) in cases.iter().zip(&transports) {
|
||||
let label = case.label();
|
||||
if !case.assert_replication {
|
||||
continue;
|
||||
}
|
||||
assert_eq!(transport.status, "COMPLETED", "{label}: existing-object replication must complete");
|
||||
if case.is_multipart_layout() {
|
||||
let expected: Vec<i32> = (1..=i32::try_from(case.part_sizes.len())?).collect();
|
||||
assert_eq!(
|
||||
transport.uploaded_parts, expected,
|
||||
"{label}: stored parts must replicate as the same multipart layout"
|
||||
);
|
||||
// An object still PENDING when the next scanner cycle arrives is
|
||||
// not driven a second time (rustfs/backlog#2362); the journal is
|
||||
// logged so a duplicate round is visible if this ever regresses.
|
||||
assert_eq!(
|
||||
transport.completes, 1,
|
||||
"{label}: exactly one CompleteMultipartUpload; journal {:?}",
|
||||
transport.journal
|
||||
);
|
||||
assert_eq!(
|
||||
transport.single_puts, 0,
|
||||
"{label}: a multipart layout must not go out as a single PutObject"
|
||||
);
|
||||
} else {
|
||||
assert_eq!(transport.single_puts, 1, "{label}: a single PUT replicates as exactly one PutObject");
|
||||
assert!(transport.uploaded_parts.is_empty(), "{label}: a single PUT must not go out as multipart");
|
||||
}
|
||||
if !case.ssec {
|
||||
let replica = replica_client
|
||||
.get_object()
|
||||
.bucket(LAYOUT_REPLICA_BUCKET)
|
||||
.key(case.key)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|err| format!("{label}: replica missing on the target: {err}"))?
|
||||
.body
|
||||
.collect()
|
||||
.await?
|
||||
.into_bytes();
|
||||
assert_eq!(replica.len(), case.body.len(), "{label}: replica length");
|
||||
assert!(replica == case.body, "{label}: replica body must equal the rc.5 upload");
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The same layouts replicated by rc.5 itself, without an upgrade. This is the
|
||||
/// baseline that tells a pre-existing transport failure apart from one the
|
||||
/// current build introduced; it records the outcome per layout and only fails
|
||||
/// when the fixture cannot run.
|
||||
#[tokio::test]
|
||||
#[ignore = "requires the pinned 1.0.0-rc.5 release binary"]
|
||||
async fn rc5_baseline_replicates_multipart_layouts() -> TestResult {
|
||||
init_logging();
|
||||
let previous_binary = source_binary()?;
|
||||
let server_env = layout_server_env();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server_from_binary(&previous_binary, vec![], &server_env)
|
||||
.await?;
|
||||
let client = env.create_s3_client();
|
||||
env.create_test_bucket(LAYOUT_PLAIN_BUCKET).await?;
|
||||
env.create_test_bucket(LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
enable_versioning(&client, LAYOUT_PLAIN_BUCKET).await?;
|
||||
enable_versioning(&client, LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
put_default_sse_s3_encryption(&client, LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
|
||||
let mut cases = layout_cases();
|
||||
for case in cases.iter_mut() {
|
||||
layout_write(&client, case).await?;
|
||||
let head = layout_head(&client, case).await?;
|
||||
case.rc5_etag = head
|
||||
.e_tag()
|
||||
.ok_or_else(|| format!("{}: rc.5 HEAD omitted the ETag", case.label()))?
|
||||
.trim_matches('"')
|
||||
.to_string();
|
||||
}
|
||||
let (_target, transports) = replicate_layouts(&env, &client, &cases).await?;
|
||||
let summary: Vec<String> = cases
|
||||
.iter()
|
||||
.zip(&transports)
|
||||
.map(|(case, transport)| {
|
||||
format!(
|
||||
"{}: {} parts={:?} puts={}",
|
||||
case.label(),
|
||||
transport.status,
|
||||
transport.uploaded_parts,
|
||||
transport.single_puts
|
||||
)
|
||||
})
|
||||
.collect();
|
||||
tracing::info!(target: "e2e_test::upgrade_compatibility_test", ?summary, "rc.5 baseline replication outcomes");
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// backlog#2362 under the same conditions that reproduced it with the rc.5
|
||||
/// writer, but with the workspace build on both sides so it runs in the
|
||||
/// ordinary lane: every pre-existing layout is driven through exactly one
|
||||
/// upload round even though the scanner re-scans it every second while the
|
||||
/// first round is still in flight.
|
||||
#[tokio::test]
|
||||
async fn existing_object_replication_drives_each_layout_once() -> TestResult {
|
||||
init_logging();
|
||||
let server_env = layout_server_env();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server_with_env(vec![], &server_env).await?;
|
||||
let writer = env.create_s3_client();
|
||||
env.create_test_bucket(LAYOUT_PLAIN_BUCKET).await?;
|
||||
env.create_test_bucket(LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
enable_versioning(&writer, LAYOUT_PLAIN_BUCKET).await?;
|
||||
enable_versioning(&writer, LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
put_default_sse_s3_encryption(&writer, LAYOUT_ENCRYPTED_BUCKET).await?;
|
||||
|
||||
let mut cases = layout_cases();
|
||||
for case in cases.iter_mut() {
|
||||
layout_write(&writer, case).await?;
|
||||
let head = layout_head(&writer, case).await?;
|
||||
case.rc5_etag = head
|
||||
.e_tag()
|
||||
.ok_or_else(|| format!("{}: HEAD omitted the ETag", case.label()))?
|
||||
.trim_matches('"')
|
||||
.to_string();
|
||||
}
|
||||
|
||||
// The objects come from an earlier process lifetime: the scanner starts
|
||||
// cold and every object is a candidate at once.
|
||||
env.restart_server_preserving_data(vec![], &server_env).await?;
|
||||
let client = env.create_s3_client();
|
||||
let (_target, transports) = replicate_layouts(&env, &client, &cases).await?;
|
||||
let mut duplicates = Vec::new();
|
||||
for (case, transport) in cases.iter().zip(&transports) {
|
||||
assert_eq!(
|
||||
transport.status,
|
||||
"COMPLETED",
|
||||
"{}: existing-object replication must complete",
|
||||
case.label()
|
||||
);
|
||||
let rounds = if case.is_multipart_layout() {
|
||||
transport.completes
|
||||
} else {
|
||||
transport.single_puts
|
||||
};
|
||||
if rounds != 1 {
|
||||
duplicates.push(format!("{}: {rounds} upload rounds; journal {:?}", case.label(), transport.journal));
|
||||
}
|
||||
}
|
||||
assert!(duplicates.is_empty(), "each existing object must be driven exactly once: {duplicates:?}");
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -118,8 +118,6 @@ hotpath-cpu = [
|
||||
# injection, xl.meta transition assertions) via `api::tier::test_util`.
|
||||
# Enable only from `[dev-dependencies]` (rustfs/backlog#1148 ilm-6).
|
||||
test-util = []
|
||||
# Observes real startup CAS only in the dedicated E2E binary.
|
||||
e2e-test-hooks = []
|
||||
|
||||
[dependencies]
|
||||
hotpath.workspace = true
|
||||
@@ -228,7 +226,7 @@ metrics = { workspace = true }
|
||||
# crates.io. The guard scripts/check_no_tokio_io_uring.sh allows an explicit
|
||||
# io-uring integration; only the tokio "io-uring" runtime feature is banned.
|
||||
[target.'cfg(target_os = "linux")'.dependencies]
|
||||
rustfs-uring = "0.2.1"
|
||||
rustfs-uring = "0.2.2"
|
||||
|
||||
[target.'cfg(windows)'.dependencies]
|
||||
winapi-util.workspace = true
|
||||
|
||||
@@ -32,7 +32,8 @@ pub mod bucket {
|
||||
pub mod bucket_target_sys {
|
||||
pub use crate::bucket::bucket_target_sys::{
|
||||
AdvancedPutOptions, BucketTargetError, BucketTargetSys, PutObjectOptions, RemoveObjectOptions, S3ClientError,
|
||||
SsecPassthroughCapability, TargetClient, append_version_id_query,
|
||||
SsecPassthroughCapability, TargetClient, VersionIdentityCapability, append_version_id_query,
|
||||
resolve_delete_api_version_id,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -76,11 +77,26 @@ pub mod bucket {
|
||||
};
|
||||
}
|
||||
|
||||
pub mod recovery_disposition {
|
||||
pub use crate::bucket::lifecycle::recovery_disposition::{
|
||||
IlmRecoveryDispositionExecutionOutcome, IlmRecoveryDispositionReasonCode, IlmRecoveryDispositionState,
|
||||
dry_run_recovery_disposition, execute_recovery_disposition,
|
||||
};
|
||||
}
|
||||
|
||||
pub mod recovery_export {
|
||||
pub use crate::bucket::lifecycle::recovery_export::{
|
||||
IlmRecoveryExportCreated, IlmRecoveryExportObservation, create_recovery_export,
|
||||
inspect_recovery_export_observation, load_recovery_export,
|
||||
};
|
||||
}
|
||||
|
||||
pub mod transition_transaction {
|
||||
pub use crate::bucket::lifecycle::transition_transaction::{
|
||||
TransitionOperatorDeleteResult, TransitionOperatorError, TransitionOperatorProbe, TransitionOperatorStatus,
|
||||
delete_transition_candidate_for_operator, finalize_missing_transition_transaction_for_operator,
|
||||
inspect_transition_transaction_for_operator,
|
||||
TransitionRecoveryRetryResult, TransitionRecoveryRetryStatus, delete_transition_candidate_for_operator,
|
||||
finalize_missing_transition_transaction_for_operator, inspect_transition_recovery_retry_for_operator,
|
||||
inspect_transition_transaction_for_operator, retry_transition_recovery_for_operator,
|
||||
};
|
||||
#[cfg(feature = "test-util")]
|
||||
pub use crate::bucket::lifecycle::transition_transaction::{
|
||||
@@ -293,7 +309,7 @@ pub mod cache {
|
||||
pub mod capacity {
|
||||
pub use crate::core::pools::{
|
||||
DecommissionUnresolvedEntry, PoolDecommissionInfo, PoolStatus, get_total_usable_capacity, get_total_usable_capacity_free,
|
||||
path2_bucket_object, path2_bucket_object_with_base_path,
|
||||
is_pool_activation_fleet_proof_error, path2_bucket_object, path2_bucket_object_with_base_path,
|
||||
};
|
||||
pub use crate::store::utils::is_reserved_or_invalid_bucket;
|
||||
}
|
||||
@@ -368,8 +384,6 @@ pub mod data_usage {
|
||||
pub mod disk {
|
||||
pub use crate::disk::disk_store::get_object_disk_read_timeout;
|
||||
pub use crate::disk::local::ScanGuard;
|
||||
#[cfg(all(feature = "test-util", not(windows)))]
|
||||
pub use crate::disk::os::{LocalPublicationPause, LocalPublicationStage};
|
||||
pub use crate::disk::{
|
||||
BATCH_READ_VERSION_MAX_ITEMS, BUCKET_META_PREFIX, BatchReadVersionItem, BatchReadVersionReq, BatchReadVersionResp,
|
||||
CheckPartsResp, ConditionalFileUpdate, DeleteOptions, Disk, DiskAPI, DiskInfo, DiskInfoOptions, DiskLocation, DiskOption,
|
||||
@@ -402,8 +416,8 @@ pub mod disk {
|
||||
|
||||
pub mod error {
|
||||
pub use crate::error::{
|
||||
Error, Result, StorageError, classify_system_path_failure_reason, is_err_bucket_not_found, is_err_object_not_found,
|
||||
is_err_version_not_found,
|
||||
Error, PoolMetadataError, PoolMetadataFailure, Result, StorageError, classify_system_path_failure_reason,
|
||||
is_err_bucket_not_found, is_err_object_not_found, is_err_version_not_found,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -448,9 +462,12 @@ pub mod notification {
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pub use crate::services::notification_sys::rotate_cross_pool_fence_fleet_proof_for_test;
|
||||
pub use crate::services::notification_sys::{
|
||||
ClusterTierDailyStats, CrossPoolFenceFleetProofToken, LegacyTransitionStateReconcileFleetProofToken, NotificationPeerErr,
|
||||
NotificationSys, ScannerPublicationLeaseGrant, acquire_cross_pool_fence_fleet_proof,
|
||||
ClusterTierDailyStats, CrossPoolFenceFleetProofToken, IlmRecoveryExportFleetProofToken,
|
||||
LegacyTransitionStateReconcileFleetProofToken, NotificationPeerErr, NotificationSys, ScannerPublicationLeaseGrant,
|
||||
acquire_cross_pool_fence_fleet_proof, acquire_ilm_recovery_export_fleet_proof,
|
||||
acquire_legacy_transition_state_reconcile_fleet_proof, cross_pool_fence_fleet_proof_matches, get_global_notification_sys,
|
||||
ilm_recovery_export_fleet_proof_matches, ilm_recovery_export_local_process_epoch,
|
||||
ilm_recovery_export_member_epochs_sha256, ilm_recovery_export_topology_generation,
|
||||
legacy_transition_state_reconcile_fleet_proof_matches, new_global_notification_sys,
|
||||
scanner_peer_transport_error_message_is_retryable, start_remote_version_state_fleet_probe,
|
||||
};
|
||||
@@ -503,7 +520,8 @@ pub mod rpc {
|
||||
pub use crate::cluster::rpc::{
|
||||
AuthenticatedChannel, KMS_SIGNAL_SUBSYSTEM, LocalPeerS3Client, PEER_RESTDRY_RUN, PEER_RESTSIGNAL, PEER_RESTSUB_SYS,
|
||||
PeerRestClient, PeerS3Client, S3PeerSys, SERVICE_SIGNAL_REFRESH_CONFIG, SERVICE_SIGNAL_RELOAD_DYNAMIC,
|
||||
ScannerBucketListing, ScannerPeerActivity, ScannerPeerDirtyUsageSnapshot, ScannerPublicationLease, TONIC_RPC_PREFIX,
|
||||
ScannerBucketListing, ScannerDirtyUsageAcknowledgement, ScannerPeerActivity, ScannerPeerDirtyUsageBucket,
|
||||
ScannerPeerDirtyUsageSnapshot, ScannerPublicationLease, ScannerScopedDirtyUsageAckEntry, TONIC_RPC_PREFIX,
|
||||
TonicInterceptor, build_put_file_auth_trailer, check_and_record_signed_rpc_nonce, decode_heal_bucket_rpc_options,
|
||||
encode_heal_bucket_rpc_options, gen_signature_headers, gen_tonic_replay_scope_headers, gen_tonic_signature_headers,
|
||||
gen_tonic_signature_interceptor, node_service_time_out_client, node_service_time_out_client_no_auth,
|
||||
@@ -546,8 +564,8 @@ pub mod storage {
|
||||
pub use crate::core::pools::HealLifecycleExpiryContext;
|
||||
pub use crate::store::HealWalkVersion;
|
||||
pub use crate::store::{
|
||||
BootstrapLocalTarget, ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, ScannerDataMovementPauseStatus, all_local_disk,
|
||||
all_local_disk_path, find_local_disk_by_ref, init_local_disks, init_local_disks_with_instance_ctx, init_lock_clients,
|
||||
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, ScannerDataMovementPauseStatus, all_local_disk, all_local_disk_path,
|
||||
find_local_disk_by_ref, init_local_disks, init_local_disks_with_instance_ctx, init_lock_clients,
|
||||
prewarm_local_disk_id_map, prewarm_local_disk_id_map_with_instance_ctx,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -18,7 +18,7 @@ use crate::bucket::metadata_sys::get_replication_config;
|
||||
use crate::bucket::remote_s3_client::{
|
||||
PathStyle, REPLICATION_TARGET_RETRY_POLICY, RemoteCredentials, RemoteS3EndpointSpec, build_remote_s3_client,
|
||||
};
|
||||
use crate::bucket::replication::{ObjectLockIntegrity, object_lock_put_integrity};
|
||||
use crate::bucket::replication::{ObjectLockIntegrity, object_lock_put_integrity, replication_etags_match};
|
||||
use crate::bucket::replication::{ReplicationStatusType, ReplicationTargetConfigBridge};
|
||||
use crate::bucket::target::ARN;
|
||||
use crate::bucket::target::BucketTargetType;
|
||||
@@ -33,6 +33,8 @@ use aws_sdk_s3::operation::get_object::{GetObjectError, GetObjectOutput};
|
||||
use aws_sdk_s3::operation::get_object_tagging::{GetObjectTaggingError, GetObjectTaggingOutput};
|
||||
use aws_sdk_s3::operation::head_bucket::HeadBucketError;
|
||||
use aws_sdk_s3::operation::head_object::HeadObjectError;
|
||||
use aws_sdk_s3::operation::put_object_legal_hold::{PutObjectLegalHoldError, PutObjectLegalHoldOutput};
|
||||
use aws_sdk_s3::operation::put_object_retention::{PutObjectRetentionError, PutObjectRetentionOutput};
|
||||
use aws_sdk_s3::operation::put_object_tagging::{PutObjectTaggingError, PutObjectTaggingOutput};
|
||||
use aws_sdk_s3::operation::upload_part::UploadPartOutput;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
@@ -42,6 +44,7 @@ use aws_sdk_s3::types::{
|
||||
ChecksumAlgorithm, ChecksumMode, CompletedMultipartUpload, CompletedPart, ObjectLockLegalHoldStatus, ObjectLockRetentionMode,
|
||||
ServerSideEncryption,
|
||||
};
|
||||
use aws_sdk_s3::types::{ObjectLockLegalHold, ObjectLockRetention};
|
||||
use aws_sdk_s3::{Client as S3Client, operation::head_object::HeadObjectOutput};
|
||||
use aws_smithy_runtime_api::client::orchestrator::HttpRequest;
|
||||
use futures::{StreamExt, stream};
|
||||
@@ -126,6 +129,25 @@ impl From<&BucketTarget> for RemoteS3EndpointSpec {
|
||||
}
|
||||
|
||||
pub type HeadObjectSdkError = Box<SdkError<HeadObjectError>>;
|
||||
|
||||
/// Whether an edited bucket target still addresses the same remote service
|
||||
/// (endpoint, bucket, path style, TLS and identity), so a verdict learned
|
||||
/// about that service stays valid across the edit.
|
||||
fn same_replication_service(edited: &BucketTarget, previous: &BucketTarget) -> bool {
|
||||
let access_key = |target: &BucketTarget| target.credentials.as_ref().map(|credentials| credentials.access_key.clone());
|
||||
edited.endpoint == previous.endpoint
|
||||
&& edited.target_bucket == previous.target_bucket
|
||||
&& edited.secure == previous.secure
|
||||
&& edited.path == previous.path
|
||||
&& access_key(edited) == access_key(previous)
|
||||
}
|
||||
|
||||
/// Page size and page budget for [`TargetClient::locate_replica_by_etag`].
|
||||
const FIND_VERSION_BY_ETAG_PAGE_SIZE: i32 = 1000;
|
||||
const FIND_VERSION_BY_ETAG_MAX_PAGES: usize = 8;
|
||||
/// Candidate cap for [`TargetClient::replica_candidates_by_etag`]: more than
|
||||
/// this many same-content versions of one key is ambiguity by any measure.
|
||||
const FIND_VERSION_BY_ETAG_MAX_MATCHES: usize = 16;
|
||||
pub type GetObjectSdkError = Box<SdkError<GetObjectError>>;
|
||||
pub type GetObjectTaggingSdkError = Box<SdkError<GetObjectTaggingError>>;
|
||||
pub type PutObjectTaggingSdkError = Box<SdkError<PutObjectTaggingError>>;
|
||||
@@ -349,6 +371,13 @@ struct TargetClientBuildProbe {
|
||||
/// their import path while the verdict vocabulary lives with the
|
||||
/// replication decision logic.
|
||||
pub use crate::bucket::replication::SsecPassthroughCapability;
|
||||
/// Version-identity verdicts (see the enum's own docs in
|
||||
/// `rustfs-replication`) are cached here per target ARN and follow the same
|
||||
/// `arn_remotes_map` lifecycle. They carry no TTL: the verdict is refreshed
|
||||
/// by every replication write's response, so it can only go stale on a
|
||||
/// target that receives no writes — and a stale `MintsOwn` costs one extra
|
||||
/// content-identity lookup before a PUT, never a lost replica.
|
||||
pub use crate::bucket::replication::VersionIdentityCapability;
|
||||
|
||||
/// How long an audited SSE-C passthrough verdict stays authoritative.
|
||||
///
|
||||
@@ -375,6 +404,11 @@ pub struct BucketTargetSys {
|
||||
/// SSE-C passthrough capability verdicts keyed by target ARN. See
|
||||
/// [`SsecPassthroughCapability`]; reset alongside `arn_remotes_map`.
|
||||
ssec_passthrough_map: Arc<RwLock<HashMap<String, SsecPassthroughRecord>>>,
|
||||
/// Version-identity verdicts keyed by target ARN. See
|
||||
/// [`VersionIdentityCapability`]; reset alongside `arn_remotes_map`. A std
|
||||
/// lock (never held across an await) so the replication worker can record
|
||||
/// a verdict from inside its synchronous PUT-response audit.
|
||||
version_identity_map: Arc<std::sync::RwLock<HashMap<String, VersionIdentityCapability>>>,
|
||||
pub targets_map: Arc<RwLock<HashMap<String, Vec<BucketTarget>>>>,
|
||||
/// Buckets whose persisted `bucket-targets.json` exists but cannot be
|
||||
/// decoded (rustfs/backlog#2282). Written under the bucket's update mutex
|
||||
@@ -423,6 +457,7 @@ impl BucketTargetSys {
|
||||
Self {
|
||||
arn_remotes_map: Arc::new(RwLock::new(HashMap::new())),
|
||||
ssec_passthrough_map: Arc::new(RwLock::new(HashMap::new())),
|
||||
version_identity_map: Arc::new(std::sync::RwLock::new(HashMap::new())),
|
||||
targets_map: Arc::new(RwLock::new(HashMap::new())),
|
||||
unreadable_targets: Arc::new(RwLock::new(HashSet::new())),
|
||||
h_mutex: Arc::new(RwLock::new(HashMap::new())),
|
||||
@@ -746,10 +781,40 @@ impl BucketTargetSys {
|
||||
arn_remotes_map.remove(&target.arn);
|
||||
health_map.remove(&target.arn);
|
||||
ssec_map.remove(&target.arn);
|
||||
self.forget_version_identity_capability(&target.arn);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Cached version-identity verdict for a target ARN; `Unknown` until a
|
||||
/// replication write or a replication-check VersionFidelity probe judged
|
||||
/// it since the target was built.
|
||||
pub fn version_identity_capability(&self, arn: &str) -> VersionIdentityCapability {
|
||||
self.version_identity_map
|
||||
.read()
|
||||
.unwrap_or_else(|poisoned| poisoned.into_inner())
|
||||
.get(arn)
|
||||
.copied()
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Record a version-identity verdict for a target ARN. Written by the
|
||||
/// replication worker after every PutObject / CompleteMultipartUpload
|
||||
/// response and by the replication-check VersionFidelity phase.
|
||||
pub fn record_version_identity_capability(&self, arn: &str, capability: VersionIdentityCapability) {
|
||||
self.version_identity_map
|
||||
.write()
|
||||
.unwrap_or_else(|poisoned| poisoned.into_inner())
|
||||
.insert(arn.to_string(), capability);
|
||||
}
|
||||
|
||||
fn forget_version_identity_capability(&self, arn: &str) {
|
||||
self.version_identity_map
|
||||
.write()
|
||||
.unwrap_or_else(|poisoned| poisoned.into_inner())
|
||||
.remove(arn);
|
||||
}
|
||||
|
||||
/// Cached SSE-C passthrough capability for a target ARN, plus whether the
|
||||
/// verdict is older than [`SSEC_PASSTHROUGH_CAPABILITY_TTL`]. `(Unknown,
|
||||
/// false)` when no verdict has been recorded since the target was built.
|
||||
@@ -1162,12 +1227,32 @@ impl BucketTargetSys {
|
||||
// Remove existing targets
|
||||
if let Some(existing_targets) = targets_map.remove(bucket) {
|
||||
let mut ssec_map = self.ssec_passthrough_map.write().await;
|
||||
let unchanged_service: HashMap<&str, &BucketTarget> = targets
|
||||
.map(|new_targets| {
|
||||
new_targets
|
||||
.targets
|
||||
.iter()
|
||||
.map(|target| (target.arn.as_str(), target))
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default();
|
||||
for target in existing_targets {
|
||||
arn_remotes_map.remove(&target.arn);
|
||||
health_map.remove(&target.arn);
|
||||
// A rebuilt/edited target may point at a different service:
|
||||
// the SSE-C passthrough verdict must be re-audited from Unknown.
|
||||
ssec_map.remove(&target.arn);
|
||||
// The version-identity verdict survives an edit that keeps the
|
||||
// same remote service (a resync start or a bandwidth change
|
||||
// rewrites the entry in place): forgetting it there would make
|
||||
// the very resync that follows re-drive every object as a
|
||||
// duplicate on a target that mints its own version ids.
|
||||
if unchanged_service
|
||||
.get(target.arn.as_str())
|
||||
.is_none_or(|edited| !same_replication_service(edited, &target))
|
||||
{
|
||||
self.forget_version_identity_capability(&target.arn);
|
||||
}
|
||||
self.update_bandwidth_limit(bucket, &target.arn, 0);
|
||||
}
|
||||
}
|
||||
@@ -1284,6 +1369,7 @@ fn generate_arn(t: &BucketTarget, depl_id: &str) -> String {
|
||||
arn.to_string()
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct RemoveObjectOptions {
|
||||
pub force_delete: bool,
|
||||
pub governance_bypass: bool,
|
||||
@@ -1339,7 +1425,12 @@ fn build_remove_object_headers(version_id: Option<&str>, opts: &RemoveObjectOpti
|
||||
/// and silently creates a delete marker instead of removing the version, while
|
||||
/// the source stamps `VersionPurgeStatus=Complete` (backlog#799 B8 / #857).
|
||||
/// Non-replication callers always pass the version through unchanged.
|
||||
fn resolve_delete_api_version_id(version_id: Option<String>, opts: &RemoveObjectOptions) -> Option<String> {
|
||||
/// The `versionId` a replicated DELETE puts on the wire: none for a
|
||||
/// delete-marker creation (the target mints the marker; the source version
|
||||
/// travels in the internal headers for RustFS peers), the addressed version
|
||||
/// otherwise. A generic S3 target given the version id on a marker-creation
|
||||
/// DELETE would permanently delete that version instead.
|
||||
pub fn resolve_delete_api_version_id(version_id: Option<String>, opts: &RemoveObjectOptions) -> Option<String> {
|
||||
if opts.replication_request && opts.replication_delete_marker {
|
||||
None
|
||||
} else {
|
||||
@@ -1892,6 +1983,125 @@ impl TargetClient {
|
||||
.map_err(Box::new)
|
||||
}
|
||||
|
||||
/// Candidate replicas by content identity on a target that mints its own
|
||||
/// version ids: page `ListObjectVersions` under the exact key and report
|
||||
/// the live versions whose ETag matches `source_etag`, newest first.
|
||||
/// Delete markers and prefix siblings never match. Bounded to
|
||||
/// [`FIND_VERSION_BY_ETAG_MAX_PAGES`] pages and
|
||||
/// [`FIND_VERSION_BY_ETAG_MAX_MATCHES`] candidates so a key with a very
|
||||
/// deep history cannot turn one convergence check into an unbounded scan;
|
||||
/// a replica beyond that window reads as missing, which only costs a
|
||||
/// re-PUT (today's behaviour), never a lost object.
|
||||
///
|
||||
/// Content identity is not version identity: two source generations with
|
||||
/// the same bytes have the same ETag. Callers drop the candidates other
|
||||
/// source versions already claim through their ledgers and refuse an
|
||||
/// [`ReplicaLocation::Ambiguous`] remainder before mutating or deleting.
|
||||
pub async fn replica_candidates_by_etag(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
source_etag: &str,
|
||||
) -> Result<Vec<String>, Box<SdkError<aws_sdk_s3::operation::list_object_versions::ListObjectVersionsError>>> {
|
||||
let mut key_marker: Option<String> = None;
|
||||
let mut version_id_marker: Option<String> = None;
|
||||
let mut matches: Vec<String> = Vec::new();
|
||||
for _ in 0..FIND_VERSION_BY_ETAG_MAX_PAGES {
|
||||
let page = self
|
||||
.client
|
||||
.list_object_versions()
|
||||
.bucket(bucket)
|
||||
.prefix(object)
|
||||
.max_keys(FIND_VERSION_BY_ETAG_PAGE_SIZE)
|
||||
.set_key_marker(key_marker.take())
|
||||
.set_version_id_marker(version_id_marker.take())
|
||||
.send()
|
||||
.await
|
||||
.map_err(Box::new)?;
|
||||
matches.extend(
|
||||
page.versions()
|
||||
.iter()
|
||||
.filter(|version| {
|
||||
version.key() == Some(object)
|
||||
&& version.version_id().is_some_and(|id| !id.is_empty())
|
||||
&& replication_etags_match(Some(source_etag), version.e_tag())
|
||||
})
|
||||
.filter_map(|version| version.version_id().map(str::to_string)),
|
||||
);
|
||||
// A listing that moved past the exact key (every listed key is >=
|
||||
// the prefix), ended, or already filled the candidate cap decides.
|
||||
if matches.len() >= FIND_VERSION_BY_ETAG_MAX_MATCHES
|
||||
|| page
|
||||
.versions()
|
||||
.iter()
|
||||
.any(|version| version.key().is_some_and(|key| key > object))
|
||||
|| !page.is_truncated().unwrap_or(false)
|
||||
{
|
||||
break;
|
||||
}
|
||||
key_marker = page.next_key_marker().map(str::to_string);
|
||||
version_id_marker = page.next_version_id_marker().map(str::to_string);
|
||||
if key_marker.is_none() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
matches.truncate(FIND_VERSION_BY_ETAG_MAX_MATCHES);
|
||||
Ok(matches)
|
||||
}
|
||||
|
||||
/// PutObjectRetention against a replica version on a target that does not
|
||||
/// take retention through the replication PUT's own headers (it mints its
|
||||
/// own version ids, so a re-PUT would create another version instead of
|
||||
/// updating this one). Anti-loop marker always added.
|
||||
pub async fn put_object_retention(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
version_id: Option<String>,
|
||||
mode: ObjectLockRetentionMode,
|
||||
retain_until: aws_sdk_s3::primitives::DateTime,
|
||||
) -> Result<PutObjectRetentionOutput, Box<SdkError<PutObjectRetentionError>>> {
|
||||
let headers = proxy_outbound_headers(HeaderMap::new());
|
||||
self.client
|
||||
.put_object_retention()
|
||||
.bucket(bucket)
|
||||
.key(object)
|
||||
.set_version_id(resolve_read_api_version_id(version_id))
|
||||
.retention(
|
||||
ObjectLockRetention::builder()
|
||||
.mode(mode)
|
||||
.retain_until_date(retain_until)
|
||||
.build(),
|
||||
)
|
||||
.customize()
|
||||
.map_request(move |req| apply_extra_headers(req, &headers))
|
||||
.send()
|
||||
.await
|
||||
.map_err(Box::new)
|
||||
}
|
||||
|
||||
/// PutObjectLegalHold counterpart of [`Self::put_object_retention`].
|
||||
pub async fn put_object_legal_hold(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
version_id: Option<String>,
|
||||
status: ObjectLockLegalHoldStatus,
|
||||
) -> Result<PutObjectLegalHoldOutput, Box<SdkError<PutObjectLegalHoldError>>> {
|
||||
let headers = proxy_outbound_headers(HeaderMap::new());
|
||||
self.client
|
||||
.put_object_legal_hold()
|
||||
.bucket(bucket)
|
||||
.key(object)
|
||||
.set_version_id(resolve_read_api_version_id(version_id))
|
||||
.legal_hold(ObjectLockLegalHold::builder().status(status).build())
|
||||
.customize()
|
||||
.map_request(move |req| apply_extra_headers(req, &headers))
|
||||
.send()
|
||||
.await
|
||||
.map_err(Box::new)
|
||||
}
|
||||
|
||||
/// HEAD used by the read-proxy path (GET/HEAD of an object not yet
|
||||
/// replicated locally, MinIO `proxyHeadToRepTarget`).
|
||||
///
|
||||
@@ -2064,7 +2274,15 @@ impl TargetClient {
|
||||
}
|
||||
}
|
||||
|
||||
match builder
|
||||
// A forwarded source checksum is this PUT's integrity header. In
|
||||
// streaming-checksum mode (`RUSTFS_REPLICATION_STREAMING_CHECKSUMS`)
|
||||
// the SDK would still add its default CRC32 trailer, and a target that
|
||||
// receives both keeps the trailer's algorithm: a forwarded SHA256
|
||||
// vanished from the replica while the source reported COMPLETED. Pin
|
||||
// this request to WhenRequired so nothing is sent beside the source's
|
||||
// own checksum.
|
||||
let forwards_source_checksum = headers.keys().any(|name| name.as_str().starts_with("x-amz-checksum-"));
|
||||
let mut operation = builder
|
||||
.bucket(bucket)
|
||||
.key(object)
|
||||
.content_length(size)
|
||||
@@ -2084,10 +2302,14 @@ impl TargetClient {
|
||||
}
|
||||
|
||||
Result::<_, aws_smithy_types::error::operation::BuildError>::Ok(req)
|
||||
})
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
});
|
||||
if forwards_source_checksum {
|
||||
operation = operation.config_override(
|
||||
aws_sdk_s3::config::Builder::new()
|
||||
.request_checksum_calculation(aws_sdk_s3::config::RequestChecksumCalculation::WhenRequired),
|
||||
);
|
||||
}
|
||||
match operation.send().await {
|
||||
Ok(output) => {
|
||||
// Under SSE-KMS/DSSE or SSE-C the target's ETag is not the MD5
|
||||
// of the stored plaintext, so it cannot be compared against the
|
||||
@@ -2331,6 +2553,45 @@ impl TargetClient {
|
||||
}
|
||||
}
|
||||
|
||||
/// Where a replica stands on a target that mints its own version ids, by
|
||||
/// content identity (exact key + ETag) after the candidates other source
|
||||
/// versions claim were removed. See
|
||||
/// [`TargetClient::replica_candidates_by_etag`].
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum ReplicaLocation {
|
||||
/// No live version under the key carries the source ETag.
|
||||
Missing,
|
||||
/// Exactly one live version carries it: safe to address.
|
||||
Unique(String),
|
||||
/// More than one live version carries it (same bytes replicated for
|
||||
/// several source generations). `newest` is the most recently listed
|
||||
/// one — good enough to prove the replica exists, never good enough to
|
||||
/// pick which one to mutate or delete.
|
||||
Ambiguous { newest: String },
|
||||
}
|
||||
|
||||
impl ReplicaLocation {
|
||||
/// `matches` newest first, as the target listed them.
|
||||
pub fn from_matches(mut matches: Vec<String>) -> Self {
|
||||
match matches.len() {
|
||||
0 => Self::Missing,
|
||||
1 => Self::Unique(matches.remove(0)),
|
||||
_ => Self::Ambiguous {
|
||||
newest: matches.remove(0),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// The version to read for existence/ETag checks, where an ambiguous
|
||||
/// match is still a located replica.
|
||||
pub fn any_version_id(&self) -> Option<&str> {
|
||||
match self {
|
||||
Self::Missing => None,
|
||||
Self::Unique(version_id) | Self::Ambiguous { newest: version_id } => Some(version_id),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub enum BucketTargetError {
|
||||
BucketRemoteTargetNotFound {
|
||||
@@ -2507,13 +2768,21 @@ mod tests {
|
||||
}
|
||||
|
||||
fn header_recording_target_client(response_headers: Vec<(String, String)>) -> (TargetClient, RecordedHeaders) {
|
||||
header_recording_target_client_with_checksums(response_headers, replication_request_checksum_calculation())
|
||||
}
|
||||
|
||||
fn header_recording_target_client_with_checksums(
|
||||
response_headers: Vec<(String, String)>,
|
||||
checksums: RequestChecksumCalculation,
|
||||
) -> (TargetClient, RecordedHeaders) {
|
||||
let request_headers: RecordedHeaders = Arc::new(std::sync::Mutex::new(Vec::new()));
|
||||
let connector = SharedHttpConnector::new(RecordingHeaderConnector {
|
||||
request_headers: Arc::clone(&request_headers),
|
||||
response_headers,
|
||||
});
|
||||
let http_client = http_client_fn(move |_settings, _components| connector.clone());
|
||||
let client = s3_client_for_test(443, Some(http_client));
|
||||
let client =
|
||||
s3_client_for_endpoint_test_with_checksums("https://localhost:443".to_string(), Some(http_client), checksums);
|
||||
(
|
||||
TargetClient {
|
||||
endpoint: "https://localhost:443".to_string(),
|
||||
@@ -2680,6 +2949,47 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// With streaming checksums enabled the SDK adds a CRC32 trailer to every
|
||||
/// upload. A PUT that forwards the source's checksum must not get that
|
||||
/// second algorithm: a target that receives both keeps the trailer's and
|
||||
/// the forwarded SHA256 never reaches the replica (rustfs/backlog#2340).
|
||||
#[tokio::test]
|
||||
async fn streaming_put_object_with_forwarded_checksum_sends_no_sdk_checksum() {
|
||||
let (client, recorded) =
|
||||
header_recording_target_client_with_checksums(Vec::new(), RequestChecksumCalculation::WhenSupported);
|
||||
let mut forwarded = PutObjectOptions::default();
|
||||
forwarded.user_metadata.insert(
|
||||
"x-amz-checksum-sha256".to_string(),
|
||||
"OoJ3yNhRwv3wwtZoGqEIPrPX9xwTnfLl+ka0wStN1g0=".to_string(),
|
||||
);
|
||||
client
|
||||
.put_object("target-bucket", "object", 4, streaming_test_body(b"data"), &forwarded)
|
||||
.await
|
||||
.expect("recorded put_object should succeed");
|
||||
client
|
||||
.put_object("target-bucket", "object", 4, streaming_test_body(b"data"), &PutObjectOptions::default())
|
||||
.await
|
||||
.expect("recorded put_object should succeed");
|
||||
let recorded = recorded.lock().expect("recorded header lock should not be poisoned");
|
||||
let with_forwarded = &recorded[0];
|
||||
assert_eq!(
|
||||
recorded_header(with_forwarded, "x-amz-checksum-sha256"),
|
||||
Some("OoJ3yNhRwv3wwtZoGqEIPrPX9xwTnfLl+ka0wStN1g0=")
|
||||
);
|
||||
assert_eq!(
|
||||
recorded_header(with_forwarded, "x-amz-trailer"),
|
||||
None,
|
||||
"the SDK must not add a trailer checksum"
|
||||
);
|
||||
assert_eq!(recorded_header(with_forwarded, "x-amz-sdk-checksum-algorithm"), None);
|
||||
// Control: the same client still streams a trailer when nothing is forwarded.
|
||||
let without_forwarded = &recorded[1];
|
||||
assert!(
|
||||
recorded_header(without_forwarded, "x-amz-trailer").is_some(),
|
||||
"streaming mode must still apply to uploads without a forwarded checksum: {without_forwarded:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// A forwarded source checksum already satisfies the rule; nothing is added.
|
||||
#[tokio::test]
|
||||
async fn locked_put_object_keeps_a_forwarded_source_checksum() {
|
||||
@@ -3045,6 +3355,14 @@ mod tests {
|
||||
}
|
||||
|
||||
fn s3_client_for_endpoint_test(endpoint: String, http_client: Option<SharedHttpClient>) -> S3Client {
|
||||
s3_client_for_endpoint_test_with_checksums(endpoint, http_client, replication_request_checksum_calculation())
|
||||
}
|
||||
|
||||
fn s3_client_for_endpoint_test_with_checksums(
|
||||
endpoint: String,
|
||||
http_client: Option<SharedHttpClient>,
|
||||
checksums: RequestChecksumCalculation,
|
||||
) -> S3Client {
|
||||
let credentials = SdkCredentials::builder()
|
||||
.access_key_id("test-access")
|
||||
.secret_access_key("test-secret")
|
||||
@@ -3058,7 +3376,7 @@ mod tests {
|
||||
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest())
|
||||
// Mirror the production remote-target builder so recorded requests
|
||||
// exercise the same checksum/framing behavior (#6853).
|
||||
.request_checksum_calculation(replication_request_checksum_calculation());
|
||||
.request_checksum_calculation(checksums);
|
||||
if let Some(http_client) = http_client {
|
||||
config = config.http_client(http_client);
|
||||
}
|
||||
@@ -3221,6 +3539,64 @@ mod tests {
|
||||
assert!(message.contains("connection refused"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn same_replication_service_ignores_resync_and_bandwidth_edits() {
|
||||
let base = BucketTarget {
|
||||
endpoint: "target.example:9000".to_string(),
|
||||
target_bucket: "replica".to_string(),
|
||||
secure: true,
|
||||
path: "on".to_string(),
|
||||
arn: "arn:rustfs:replication:us-east-1:bucket:same".to_string(),
|
||||
credentials: Some(Credentials {
|
||||
access_key: "access".to_string(),
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
let resync_edit = BucketTarget {
|
||||
reset_id: "reset-1".to_string(),
|
||||
bandwidth_limit: 1024,
|
||||
..base.clone()
|
||||
};
|
||||
assert!(same_replication_service(&resync_edit, &base));
|
||||
for moved in [
|
||||
BucketTarget {
|
||||
endpoint: "other.example:9000".to_string(),
|
||||
..base.clone()
|
||||
},
|
||||
BucketTarget {
|
||||
target_bucket: "other".to_string(),
|
||||
..base.clone()
|
||||
},
|
||||
BucketTarget {
|
||||
secure: false,
|
||||
..base.clone()
|
||||
},
|
||||
BucketTarget {
|
||||
credentials: Some(Credentials {
|
||||
access_key: "rotated".to_string(),
|
||||
..Default::default()
|
||||
}),
|
||||
..base.clone()
|
||||
},
|
||||
] {
|
||||
assert!(!same_replication_service(&moved, &base));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn version_identity_verdict_is_per_arn_and_forgotten_with_the_target() {
|
||||
let sys = BucketTargetSys::default();
|
||||
let arn = "arn:rustfs:replication:us-east-1:bucket:identity";
|
||||
assert_eq!(sys.version_identity_capability(arn), VersionIdentityCapability::Unknown);
|
||||
sys.record_version_identity_capability(arn, VersionIdentityCapability::MintsOwn);
|
||||
assert_eq!(sys.version_identity_capability(arn), VersionIdentityCapability::MintsOwn);
|
||||
assert_eq!(sys.version_identity_capability("other"), VersionIdentityCapability::Unknown);
|
||||
// A rebuilt target may point at a different service.
|
||||
sys.forget_version_identity_capability(arn);
|
||||
assert_eq!(sys.version_identity_capability(arn), VersionIdentityCapability::Unknown);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn endpoint_health_key_preserves_explicit_port() {
|
||||
let url = Url::parse("https://remote.example:9443").expect("url should parse");
|
||||
|
||||
@@ -32,6 +32,7 @@ use crate::bucket::lifecycle::manual_transition_job::{
|
||||
record_manual_transition_worker_result_with_reason, renew_manual_transition_job_lease_if_owned,
|
||||
save_manual_transition_job_record_if_current, save_manual_transition_task_if_absent, update_manual_transition_job_record,
|
||||
};
|
||||
use crate::bucket::lifecycle::recovery_disposition_runtime::run_recovery_disposition_maintenance_loop;
|
||||
use crate::bucket::lifecycle::replication_sink;
|
||||
use crate::bucket::lifecycle::replication_sink::{
|
||||
DeleteReplicationConfigSnapshot, ReplicationObjectBridge, ReplicationStatusType, replication_state_to_filemeta,
|
||||
@@ -149,6 +150,7 @@ pub type ExpiryOpType = Box<dyn ExpiryOp + Send + Sync + 'static>;
|
||||
static XXHASH_SEED: u64 = 0;
|
||||
static TIER_FREE_VERSION_RECOVERY_STARTED: OnceLock<()> = OnceLock::new();
|
||||
static MANUAL_TRANSITION_JOB_RECOVERY_STARTED: OnceLock<()> = OnceLock::new();
|
||||
static RECOVERY_DISPOSITION_MAINTENANCE_STARTED: OnceLock<()> = OnceLock::new();
|
||||
|
||||
#[cfg(test)]
|
||||
#[derive(Default)]
|
||||
@@ -2398,9 +2400,20 @@ pub async fn init_background_expiry(api: Arc<ECStore>) {
|
||||
let _ = spawn_tier_free_version_recovery_once(api.clone(), &TIER_FREE_VERSION_RECOVERY_STARTED);
|
||||
spawn_tier_delete_journal_recovery_once(api.clone());
|
||||
spawn_transition_transaction_recovery_once(api.clone());
|
||||
spawn_recovery_disposition_maintenance_once(api.clone());
|
||||
spawn_manual_transition_job_recovery_once(api);
|
||||
}
|
||||
|
||||
fn spawn_recovery_disposition_maintenance_once(api: Arc<ECStore>) -> Option<JoinHandle<()>> {
|
||||
let cancel_token = api.ctx.background_cancel_token()?;
|
||||
if RECOVERY_DISPOSITION_MAINTENANCE_STARTED.set(()).is_err() {
|
||||
return None;
|
||||
}
|
||||
Some(tokio::spawn(async move {
|
||||
run_recovery_disposition_maintenance_loop(api, cancel_token).await;
|
||||
}))
|
||||
}
|
||||
|
||||
fn spawn_manual_transition_job_recovery_once(api: Arc<ECStore>) -> Option<JoinHandle<()>> {
|
||||
if MANUAL_TRANSITION_JOB_RECOVERY_STARTED.set(()).is_err() {
|
||||
return None;
|
||||
|
||||
@@ -41,6 +41,21 @@ where
|
||||
com::read_config(api, file).await
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_limited_preserve_empty<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
|
||||
where
|
||||
S: ObjectIO<
|
||||
Error = Error,
|
||||
RangeSpec = HTTPRangeSpec,
|
||||
HeaderMap = HeaderMap,
|
||||
ObjectOptions = ObjectOptions,
|
||||
ObjectInfo = ObjectInfo,
|
||||
GetObjectReader = GetObjectReader,
|
||||
PutObjectReader = PutObjReader,
|
||||
>,
|
||||
{
|
||||
com::read_config_limited_preserve_empty(api, file, max_bytes).await
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_with_metadata<S>(api: Arc<S>, file: &str, opts: &ObjectOptions) -> Result<(Vec<u8>, ObjectInfo)>
|
||||
where
|
||||
S: ObjectIO<
|
||||
@@ -56,6 +71,26 @@ where
|
||||
com::read_config_with_metadata(api, file, opts).await
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_limited_preserve_empty_with_metadata<S>(
|
||||
api: Arc<S>,
|
||||
file: &str,
|
||||
opts: &ObjectOptions,
|
||||
max_bytes: usize,
|
||||
) -> Result<(Vec<u8>, ObjectInfo)>
|
||||
where
|
||||
S: ObjectIO<
|
||||
Error = Error,
|
||||
RangeSpec = HTTPRangeSpec,
|
||||
HeaderMap = HeaderMap,
|
||||
ObjectOptions = ObjectOptions,
|
||||
ObjectInfo = ObjectInfo,
|
||||
GetObjectReader = GetObjectReader,
|
||||
PutObjectReader = PutObjReader,
|
||||
>,
|
||||
{
|
||||
com::read_config_limited_preserve_empty_with_metadata_opts(api, file, opts, max_bytes).await
|
||||
}
|
||||
|
||||
pub(crate) async fn save_config<S>(api: Arc<S>, file: &str, data: Vec<u8>) -> Result<()>
|
||||
where
|
||||
S: ObjectIO<
|
||||
@@ -126,20 +161,30 @@ where
|
||||
DeletedObject = DeletedObject,
|
||||
>,
|
||||
{
|
||||
match api
|
||||
.delete_object(
|
||||
RUSTFS_META_BUCKET,
|
||||
file,
|
||||
ObjectOptions {
|
||||
http_preconditions: Some(HTTPPreconditions {
|
||||
if_match: Some(etag.to_string()),
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
delete_config_if_match_with_opts(api, file, etag, ObjectOptions::default()).await
|
||||
}
|
||||
|
||||
pub(crate) async fn delete_config_if_match_with_opts<S>(
|
||||
api: Arc<S>,
|
||||
file: &str,
|
||||
etag: &str,
|
||||
mut options: ObjectOptions,
|
||||
) -> Result<()>
|
||||
where
|
||||
S: ObjectOperations<
|
||||
Error = Error,
|
||||
ObjectInfo = ObjectInfo,
|
||||
ObjectOptions = ObjectOptions,
|
||||
FileInfo = FileInfo,
|
||||
ObjectToDelete = ObjectToDelete,
|
||||
DeletedObject = DeletedObject,
|
||||
>,
|
||||
{
|
||||
options.http_preconditions = Some(HTTPPreconditions {
|
||||
if_match: Some(etag.to_string()),
|
||||
..Default::default()
|
||||
});
|
||||
match api.delete_object(RUSTFS_META_BUCKET, file, options).await {
|
||||
Ok(_) => Ok(()),
|
||||
Err(err) => {
|
||||
if err == Error::FileNotFound || matches!(err, Error::ObjectNotFound(_, _)) {
|
||||
|
||||
@@ -22,7 +22,7 @@ use super::{
|
||||
bucket_lifecycle_ops::{
|
||||
ManualTransitionQueueSnapshot, ManualTransitionRunReport, decode_manual_transition_continuation_token,
|
||||
},
|
||||
manual_transition_job, recovery_control, tier_delete_journal, transition_transaction,
|
||||
manual_transition_job, recovery_control, recovery_disposition, recovery_export, tier_delete_journal, transition_transaction,
|
||||
};
|
||||
use crate::error::{Error, Result};
|
||||
use crate::services::tier::tier_probe_intent;
|
||||
@@ -42,6 +42,8 @@ pub(crate) enum DurableIlmRecordKind {
|
||||
ManualTransitionTask,
|
||||
ManualTransitionWorkerResult,
|
||||
RecoveryControl,
|
||||
RecoveryExport,
|
||||
RecoveryDisposition,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
@@ -112,8 +114,20 @@ pub(crate) const RECOVERY_CONTROL_NAMESPACE: DurableIlmNamespace = DurableIlmNam
|
||||
max_record_size: recovery_control::MAX_ILM_RECOVERY_CONTROL_SIZE,
|
||||
kind: DurableIlmRecordKind::RecoveryControl,
|
||||
};
|
||||
pub(crate) const RECOVERY_EXPORT_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "recovery-export",
|
||||
prefix: recovery_export::ILM_RECOVERY_EXPORT_PREFIX,
|
||||
max_record_size: recovery_export::MAX_ILM_RECOVERY_EXPORT_SIZE,
|
||||
kind: DurableIlmRecordKind::RecoveryExport,
|
||||
};
|
||||
pub(crate) const RECOVERY_DISPOSITION_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "recovery-disposition",
|
||||
prefix: recovery_disposition::ILM_RECOVERY_DISPOSITION_PREFIX,
|
||||
max_record_size: recovery_disposition::MAX_ILM_RECOVERY_DISPOSITION_SIZE,
|
||||
kind: DurableIlmRecordKind::RecoveryDisposition,
|
||||
};
|
||||
|
||||
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 10] = [
|
||||
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 12] = [
|
||||
TIER_DELETE_JOURNAL_NAMESPACE,
|
||||
TIER_DELETE_JOURNAL_V6_NAMESPACE,
|
||||
TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE,
|
||||
@@ -124,6 +138,8 @@ pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 10] = [
|
||||
MANUAL_TRANSITION_TASK_NAMESPACE,
|
||||
MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE,
|
||||
RECOVERY_CONTROL_NAMESPACE,
|
||||
RECOVERY_EXPORT_NAMESPACE,
|
||||
RECOVERY_DISPOSITION_NAMESPACE,
|
||||
];
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
@@ -261,6 +277,28 @@ pub(crate) enum DurableIlmRecordCheckpoint {
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
owner_fence_sha256: Option<String>,
|
||||
},
|
||||
RecoveryExport {
|
||||
content_sha256: String,
|
||||
source_generation_sha256: String,
|
||||
topology_generation: String,
|
||||
member_epochs_sha256: String,
|
||||
creator_sha256: String,
|
||||
retain_until_unix_nanos: i64,
|
||||
},
|
||||
RecoveryDisposition {
|
||||
content_sha256: String,
|
||||
identity_sha256: String,
|
||||
copy_manifest_sha256: String,
|
||||
copy_manifest_count: usize,
|
||||
created_at_unix_nanos: i64,
|
||||
revision: u64,
|
||||
state: recovery_disposition::IlmRecoveryDispositionState,
|
||||
owner_fence_sha256: Option<String>,
|
||||
owner_lease_acquired_at_unix_nanos: Option<i64>,
|
||||
owner_lease_expires_at_unix_nanos: Option<i64>,
|
||||
confirmed_absent_sha256: Vec<String>,
|
||||
retain_until_unix_nanos: i64,
|
||||
},
|
||||
}
|
||||
|
||||
impl DurableIlmRecordCheckpoint {
|
||||
@@ -275,7 +313,9 @@ impl DurableIlmRecordCheckpoint {
|
||||
| Self::ManualTransitionScope { content_sha256, .. }
|
||||
| Self::ManualTransitionTask { content_sha256 }
|
||||
| Self::ManualTransitionWorkerResult { content_sha256 }
|
||||
| Self::RecoveryControl { content_sha256, .. } => content_sha256,
|
||||
| Self::RecoveryControl { content_sha256, .. }
|
||||
| Self::RecoveryExport { content_sha256, .. }
|
||||
| Self::RecoveryDisposition { content_sha256, .. } => content_sha256,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -316,6 +356,9 @@ impl DurableIlmRecordCheckpoint {
|
||||
{
|
||||
return Err(Error::other("durable ILM tier delete journal checkpoint is invalid"));
|
||||
}
|
||||
if !recovery_disposition_checkpoint_is_valid(checkpoint) {
|
||||
return Err(Error::other("durable ILM recovery disposition checkpoint is invalid"));
|
||||
}
|
||||
}
|
||||
if self == next {
|
||||
if let Self::ManualTransitionJob {
|
||||
@@ -456,9 +499,7 @@ impl DurableIlmRecordCheckpoint {
|
||||
},
|
||||
) => {
|
||||
previous_identity == next_identity
|
||||
&& transition_state_distance(*previous_state, *next_state)
|
||||
.and_then(|distance| previous_revision.checked_add(distance))
|
||||
.is_some_and(|expected_revision| *next_revision == expected_revision)
|
||||
&& transition_state_revision_is_successor(*previous_state, *previous_revision, *next_state, *next_revision)
|
||||
&& (!previous_remote_version_known || previous_remote_version == next_remote_version)
|
||||
}
|
||||
(
|
||||
@@ -594,6 +635,83 @@ impl DurableIlmRecordCheckpoint {
|
||||
&& previous_attempts == next_attempts;
|
||||
adjacent && (claim || source_refresh || completion)
|
||||
}
|
||||
(
|
||||
Self::RecoveryDisposition {
|
||||
identity_sha256: previous_identity,
|
||||
copy_manifest_sha256: previous_manifest,
|
||||
copy_manifest_count: previous_manifest_count,
|
||||
created_at_unix_nanos: previous_created_at,
|
||||
revision: previous_revision,
|
||||
state: previous_state,
|
||||
owner_fence_sha256: previous_owner,
|
||||
owner_lease_acquired_at_unix_nanos: previous_owner_acquired,
|
||||
owner_lease_expires_at_unix_nanos: previous_owner_expires,
|
||||
confirmed_absent_sha256: previous_confirmed,
|
||||
retain_until_unix_nanos: previous_retain_until,
|
||||
..
|
||||
},
|
||||
Self::RecoveryDisposition {
|
||||
identity_sha256: next_identity,
|
||||
copy_manifest_sha256: next_manifest,
|
||||
copy_manifest_count: next_manifest_count,
|
||||
created_at_unix_nanos: next_created_at,
|
||||
revision: next_revision,
|
||||
state: next_state,
|
||||
owner_fence_sha256: next_owner,
|
||||
owner_lease_acquired_at_unix_nanos: next_owner_acquired,
|
||||
owner_lease_expires_at_unix_nanos: next_owner_expires,
|
||||
confirmed_absent_sha256: next_confirmed,
|
||||
retain_until_unix_nanos: next_retain_until,
|
||||
..
|
||||
},
|
||||
) => {
|
||||
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
|
||||
|
||||
let immutable_identity_matches = previous_identity == next_identity
|
||||
&& previous_manifest == next_manifest
|
||||
&& previous_manifest_count == next_manifest_count
|
||||
&& previous_created_at == next_created_at
|
||||
&& previous_retain_until == next_retain_until;
|
||||
let adjacent = previous_revision.checked_add(1) == Some(*next_revision);
|
||||
let progress_is_monotonic = sorted_sha256_set_is_subset(previous_confirmed, next_confirmed);
|
||||
let legal_edge = match (previous_state, next_state) {
|
||||
(Prepared, Prepared) => {
|
||||
let claim = previous_owner.is_none() && next_owner.is_some();
|
||||
let takeover = previous_owner.is_some()
|
||||
&& previous_owner != next_owner
|
||||
&& previous_owner_expires
|
||||
.zip(*next_owner_acquired)
|
||||
.is_some_and(|(expires, acquired)| acquired >= expires);
|
||||
previous_confirmed == next_confirmed && (claim || takeover)
|
||||
}
|
||||
(Prepared, Applying) => {
|
||||
previous_confirmed == next_confirmed
|
||||
&& previous_owner.is_some()
|
||||
&& previous_owner == next_owner
|
||||
&& previous_owner_acquired == next_owner_acquired
|
||||
&& previous_owner_expires == next_owner_expires
|
||||
}
|
||||
(Applying, Applying) => {
|
||||
let progress = previous_owner == next_owner
|
||||
&& previous_owner_acquired == next_owner_acquired
|
||||
&& previous_owner_expires == next_owner_expires
|
||||
&& previous_confirmed.len().checked_add(1) == Some(next_confirmed.len());
|
||||
let takeover = previous_owner.is_some()
|
||||
&& previous_owner != next_owner
|
||||
&& previous_confirmed == next_confirmed
|
||||
&& previous_owner_expires
|
||||
.zip(*next_owner_acquired)
|
||||
.is_some_and(|(expires, acquired)| acquired >= expires);
|
||||
progress || takeover
|
||||
}
|
||||
(Applying, Completed) => {
|
||||
previous_owner.is_some() && next_owner.is_none() && previous_confirmed == next_confirmed
|
||||
}
|
||||
_ => false,
|
||||
};
|
||||
|
||||
immutable_identity_matches && adjacent && progress_is_monotonic && legal_edge
|
||||
}
|
||||
_ => false,
|
||||
};
|
||||
|
||||
@@ -611,6 +729,11 @@ impl DurableIlmRecordCheckpoint {
|
||||
/// after the exact terminal ETag and terminal receipt were committed, to
|
||||
/// purge older object versions exposed by that deletion.
|
||||
pub(crate) fn is_predecessor_of_terminal(&self, terminal: &Self) -> bool {
|
||||
for checkpoint in [self, terminal] {
|
||||
if !recovery_disposition_checkpoint_is_valid(checkpoint) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if let Self::TierProbeIntent { state, .. } = terminal
|
||||
&& !matches!(
|
||||
state,
|
||||
@@ -627,6 +750,11 @@ impl DurableIlmRecordCheckpoint {
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if let Self::RecoveryDisposition { state, .. } = terminal
|
||||
&& state != &recovery_disposition::IlmRecoveryDispositionState::Completed
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if self == terminal || self.validate_successor(terminal).is_ok() {
|
||||
return true;
|
||||
}
|
||||
@@ -752,11 +880,121 @@ impl DurableIlmRecordCheckpoint {
|
||||
&& terminal_revision > previous_revision
|
||||
&& terminal_attempts >= previous_attempts
|
||||
}
|
||||
(
|
||||
Self::RecoveryDisposition {
|
||||
identity_sha256: previous_identity,
|
||||
copy_manifest_sha256: previous_manifest,
|
||||
copy_manifest_count: previous_manifest_count,
|
||||
created_at_unix_nanos: previous_created_at,
|
||||
revision: previous_revision,
|
||||
state: previous_state,
|
||||
owner_fence_sha256: previous_owner,
|
||||
confirmed_absent_sha256: previous_confirmed,
|
||||
retain_until_unix_nanos: previous_retain_until,
|
||||
..
|
||||
},
|
||||
Self::RecoveryDisposition {
|
||||
identity_sha256: terminal_identity,
|
||||
copy_manifest_sha256: terminal_manifest,
|
||||
copy_manifest_count: terminal_manifest_count,
|
||||
created_at_unix_nanos: terminal_created_at,
|
||||
revision: terminal_revision,
|
||||
state: recovery_disposition::IlmRecoveryDispositionState::Completed,
|
||||
confirmed_absent_sha256: terminal_confirmed,
|
||||
retain_until_unix_nanos: terminal_retain_until,
|
||||
..
|
||||
},
|
||||
) => {
|
||||
matches!(
|
||||
previous_state,
|
||||
recovery_disposition::IlmRecoveryDispositionState::Prepared
|
||||
| recovery_disposition::IlmRecoveryDispositionState::Applying
|
||||
) && previous_identity == terminal_identity
|
||||
&& previous_manifest == terminal_manifest
|
||||
&& previous_manifest_count == terminal_manifest_count
|
||||
&& previous_created_at == terminal_created_at
|
||||
&& previous_retain_until == terminal_retain_until
|
||||
&& terminal_revision.checked_sub(*previous_revision).is_some_and(|distance| {
|
||||
let minimum_distance = match previous_state {
|
||||
recovery_disposition::IlmRecoveryDispositionState::Prepared if previous_owner.is_some() => 3,
|
||||
recovery_disposition::IlmRecoveryDispositionState::Prepared => 4,
|
||||
recovery_disposition::IlmRecoveryDispositionState::Applying
|
||||
if previous_confirmed.len() == *previous_manifest_count =>
|
||||
{
|
||||
1
|
||||
}
|
||||
recovery_disposition::IlmRecoveryDispositionState::Applying => 2,
|
||||
recovery_disposition::IlmRecoveryDispositionState::Completed => u64::MAX,
|
||||
};
|
||||
distance >= minimum_distance
|
||||
})
|
||||
&& sorted_sha256_set_is_subset(previous_confirmed, terminal_confirmed)
|
||||
}
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn recovery_disposition_checkpoint_is_valid(checkpoint: &DurableIlmRecordCheckpoint) -> bool {
|
||||
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
|
||||
|
||||
let DurableIlmRecordCheckpoint::RecoveryDisposition {
|
||||
content_sha256,
|
||||
identity_sha256,
|
||||
copy_manifest_sha256,
|
||||
copy_manifest_count,
|
||||
created_at_unix_nanos,
|
||||
revision,
|
||||
state,
|
||||
owner_fence_sha256,
|
||||
owner_lease_acquired_at_unix_nanos,
|
||||
owner_lease_expires_at_unix_nanos,
|
||||
confirmed_absent_sha256,
|
||||
retain_until_unix_nanos,
|
||||
} = checkpoint
|
||||
else {
|
||||
return true;
|
||||
};
|
||||
let owner_fence_sha256 = owner_fence_sha256.as_deref();
|
||||
|
||||
is_canonical_sha256(content_sha256)
|
||||
&& is_canonical_sha256(identity_sha256)
|
||||
&& is_canonical_sha256(copy_manifest_sha256)
|
||||
&& *copy_manifest_count > 0
|
||||
&& *created_at_unix_nanos > 0
|
||||
&& *revision > 0
|
||||
&& *retain_until_unix_nanos > 0
|
||||
&& owner_fence_sha256.is_none_or(is_canonical_sha256)
|
||||
&& match (
|
||||
owner_fence_sha256,
|
||||
*owner_lease_acquired_at_unix_nanos,
|
||||
*owner_lease_expires_at_unix_nanos,
|
||||
) {
|
||||
(None, None, None) => true,
|
||||
(Some(_), Some(acquired), Some(expires)) => acquired > 0 && expires > acquired,
|
||||
_ => false,
|
||||
}
|
||||
&& confirmed_absent_sha256.len() <= *copy_manifest_count
|
||||
&& confirmed_absent_sha256.iter().all(|digest| is_canonical_sha256(digest))
|
||||
&& confirmed_absent_sha256.windows(2).all(|pair| pair[0] < pair[1])
|
||||
&& match *state {
|
||||
Prepared => confirmed_absent_sha256.is_empty(),
|
||||
Applying => owner_fence_sha256.is_some(),
|
||||
Completed => owner_fence_sha256.is_none() && confirmed_absent_sha256.len() == *copy_manifest_count,
|
||||
}
|
||||
}
|
||||
|
||||
fn is_canonical_sha256(value: &str) -> bool {
|
||||
is_sha256_checksum(value)
|
||||
&& !value
|
||||
.bytes()
|
||||
.any(|byte| byte.is_ascii_hexdigit() && byte.is_ascii_uppercase())
|
||||
}
|
||||
|
||||
fn sorted_sha256_set_is_subset(subset: &[String], superset: &[String]) -> bool {
|
||||
subset.iter().all(|candidate| superset.binary_search(candidate).is_ok())
|
||||
}
|
||||
|
||||
fn tier_delete_dispatch_parent_progress_delta(
|
||||
previous_sequence: u64,
|
||||
previous_completed_journals: u64,
|
||||
@@ -790,6 +1028,22 @@ fn transition_state_distance(
|
||||
}
|
||||
}
|
||||
|
||||
fn transition_state_revision_is_successor(
|
||||
from: transition_transaction::TransitionTransactionState,
|
||||
from_revision: u64,
|
||||
to: transition_transaction::TransitionTransactionState,
|
||||
to_revision: u64,
|
||||
) -> bool {
|
||||
use transition_transaction::TransitionTransactionState::{LocalCommitStarted, UploadOutcomeUnknown};
|
||||
|
||||
if from == UploadOutcomeUnknown && from_revision == 1 && to == LocalCommitStarted {
|
||||
return to_revision == 2;
|
||||
}
|
||||
transition_state_distance(from, to)
|
||||
.and_then(|distance| from_revision.checked_add(distance))
|
||||
.is_some_and(|expected_revision| to_revision == expected_revision)
|
||||
}
|
||||
|
||||
fn tier_probe_state_reaches(
|
||||
from: tier_probe_intent::TierProbeIntentState,
|
||||
to: tier_probe_intent::TierProbeIntentState,
|
||||
@@ -1348,6 +1602,55 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
|
||||
},
|
||||
)
|
||||
}
|
||||
DurableIlmRecordKind::RecoveryExport => {
|
||||
let (protocol, export_id) = recovery_export::recovery_export_id_from_record_object_name(path)?;
|
||||
let export = recovery_export::IlmRecoveryExport::decode(&export_id, data)?;
|
||||
let canonical = recovery_export::recovery_export_record_object_name(protocol, &export_id)?;
|
||||
if canonical != path || export.protocol != protocol {
|
||||
return Err(Error::other("ILM recovery export path is not canonical"));
|
||||
}
|
||||
let source_generation_sha256 = checkpoint_hash(&export.source_generation)?;
|
||||
(
|
||||
"export_id",
|
||||
export_id,
|
||||
DurableIlmRecordCheckpoint::RecoveryExport {
|
||||
content_sha256,
|
||||
source_generation_sha256,
|
||||
topology_generation: export.topology_generation,
|
||||
member_epochs_sha256: export.member_epochs_sha256,
|
||||
creator_sha256: export.creator_sha256,
|
||||
retain_until_unix_nanos: export.retain_until_unix_nanos,
|
||||
},
|
||||
)
|
||||
}
|
||||
DurableIlmRecordKind::RecoveryDisposition => {
|
||||
// The disposition module owns strict schema, checksum, canonical
|
||||
// path, immutable-manifest, and state-specific validation. Keep
|
||||
// this boundary limited to decommission identity/checkpoint
|
||||
// projection so the two readers cannot accept different records.
|
||||
let disposition = recovery_disposition::decode_recovery_disposition_checkpoint(path, data)?;
|
||||
if disposition.content_sha256 != content_sha256 {
|
||||
return Err(Error::other("ILM recovery disposition checkpoint content digest is invalid"));
|
||||
}
|
||||
(
|
||||
"disposition_id",
|
||||
disposition.disposition_id,
|
||||
DurableIlmRecordCheckpoint::RecoveryDisposition {
|
||||
content_sha256: disposition.content_sha256,
|
||||
identity_sha256: disposition.identity_sha256,
|
||||
copy_manifest_sha256: disposition.copy_manifest_sha256,
|
||||
copy_manifest_count: disposition.copy_manifest_count,
|
||||
created_at_unix_nanos: disposition.created_at_unix_nanos,
|
||||
revision: disposition.revision,
|
||||
state: disposition.state,
|
||||
owner_fence_sha256: disposition.owner_fence_sha256,
|
||||
owner_lease_acquired_at_unix_nanos: disposition.owner_lease_acquired_at_unix_nanos,
|
||||
owner_lease_expires_at_unix_nanos: disposition.owner_lease_expires_at_unix_nanos,
|
||||
confirmed_absent_sha256: disposition.confirmed_absent_sha256,
|
||||
retain_until_unix_nanos: disposition.retain_until_unix_nanos,
|
||||
},
|
||||
)
|
||||
}
|
||||
DurableIlmRecordKind::ManualTransitionJob => {
|
||||
let job_id = manual_transition_job::manual_transition_job_id_from_record_object_name(path)
|
||||
.map_err(|err| Error::other(err.to_string()))?;
|
||||
@@ -1503,6 +1806,203 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
fn recovery_disposition_checkpoint(
|
||||
revision: u64,
|
||||
state: recovery_disposition::IlmRecoveryDispositionState,
|
||||
owner_fence: Option<&str>,
|
||||
confirmed_absent_sha256: Vec<String>,
|
||||
) -> DurableIlmRecordCheckpoint {
|
||||
let (owner_lease_acquired_at_unix_nanos, owner_lease_expires_at_unix_nanos) = match owner_fence {
|
||||
Some("f") => (Some(10), Some(20)),
|
||||
Some(_) => (Some(1), Some(10)),
|
||||
None => (None, None),
|
||||
};
|
||||
DurableIlmRecordCheckpoint::RecoveryDisposition {
|
||||
content_sha256: format!("{revision:064x}"),
|
||||
identity_sha256: "a".repeat(64),
|
||||
copy_manifest_sha256: "d".repeat(64),
|
||||
copy_manifest_count: 2,
|
||||
created_at_unix_nanos: 1_700_000_000_000_000_000,
|
||||
revision,
|
||||
state,
|
||||
owner_fence_sha256: owner_fence.map(|digest| digest.repeat(64)),
|
||||
owner_lease_acquired_at_unix_nanos,
|
||||
owner_lease_expires_at_unix_nanos,
|
||||
confirmed_absent_sha256,
|
||||
retain_until_unix_nanos: 1_820_000_000_000_000_000,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_disposition_namespace_is_registered_without_shadowing_its_root() {
|
||||
let disposition_id = "a".repeat(64);
|
||||
let path = format!(
|
||||
"{}/tier_delete_journal/{}/{}/{}.json",
|
||||
recovery_disposition::ILM_RECOVERY_DISPOSITION_PREFIX,
|
||||
&disposition_id[..2],
|
||||
&disposition_id[2..4],
|
||||
disposition_id
|
||||
);
|
||||
let namespace = classify_durable_ilm_record(&path)
|
||||
.expect("recovery disposition path should classify")
|
||||
.expect("recovery disposition should be durable");
|
||||
|
||||
assert_eq!(namespace, &RECOVERY_DISPOSITION_NAMESPACE);
|
||||
assert!(classify_durable_ilm_record(recovery_disposition::ILM_RECOVERY_DISPOSITION_PREFIX).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_disposition_checkpoint_accepts_only_monotonic_progress() {
|
||||
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
|
||||
|
||||
let first_copy = "b".repeat(64);
|
||||
let second_copy = "c".repeat(64);
|
||||
let prepared = recovery_disposition_checkpoint(1, Prepared, None, Vec::new());
|
||||
let claimed = recovery_disposition_checkpoint(2, Prepared, Some("e"), Vec::new());
|
||||
let applying = recovery_disposition_checkpoint(3, Applying, Some("e"), Vec::new());
|
||||
let first_absent = recovery_disposition_checkpoint(4, Applying, Some("e"), vec![first_copy.clone()]);
|
||||
let taken_over = recovery_disposition_checkpoint(5, Applying, Some("f"), vec![first_copy.clone()]);
|
||||
let all_absent = recovery_disposition_checkpoint(6, Applying, Some("f"), vec![first_copy.clone(), second_copy.clone()]);
|
||||
let completed = recovery_disposition_checkpoint(7, Completed, None, vec![first_copy.clone(), second_copy.clone()]);
|
||||
|
||||
prepared
|
||||
.validate_successor(&claimed)
|
||||
.expect("Prepared should record an owner claim without absence progress");
|
||||
claimed
|
||||
.validate_successor(&applying)
|
||||
.expect("Prepared should advance to Applying without folding in deletion progress");
|
||||
applying
|
||||
.validate_successor(&first_absent)
|
||||
.expect("Applying should append newly confirmed absent copies");
|
||||
first_absent
|
||||
.validate_successor(&taken_over)
|
||||
.expect("Applying should record a fenced owner takeover without losing progress");
|
||||
taken_over
|
||||
.validate_successor(&all_absent)
|
||||
.expect("Applying should preserve every earlier confirmation while making progress");
|
||||
all_absent
|
||||
.validate_successor(&completed)
|
||||
.expect("a fully confirmed manifest should advance to Completed");
|
||||
|
||||
assert!(
|
||||
prepared.validate_successor(&completed).is_err(),
|
||||
"adjacent receipt updates must not skip Applying"
|
||||
);
|
||||
assert!(
|
||||
first_absent
|
||||
.validate_successor(&recovery_disposition_checkpoint(5, Applying, Some("e"), Vec::new()))
|
||||
.is_err(),
|
||||
"confirmed-absent progress must not move backwards"
|
||||
);
|
||||
assert!(
|
||||
applying
|
||||
.validate_successor(&recovery_disposition_checkpoint(4, Completed, None, vec![first_copy.clone()]))
|
||||
.is_err(),
|
||||
"Completed must cover the complete immutable copy manifest"
|
||||
);
|
||||
assert!(
|
||||
completed
|
||||
.validate_successor(&recovery_disposition_checkpoint(7, Applying, Some("e"), vec![second_copy]))
|
||||
.is_err(),
|
||||
"Completed is terminal"
|
||||
);
|
||||
assert!(
|
||||
first_absent
|
||||
.validate_successor(&recovery_disposition_checkpoint(5, Applying, Some("e"), vec![first_copy.clone()]))
|
||||
.is_err(),
|
||||
"a same-state revision bump must change the owner fence or absence progress"
|
||||
);
|
||||
assert!(
|
||||
applying
|
||||
.validate_successor(&recovery_disposition_checkpoint(4, Applying, None, vec![first_copy]))
|
||||
.is_err(),
|
||||
"Applying must retain a fenced owner"
|
||||
);
|
||||
|
||||
let mut noncanonical_identity = claimed.clone();
|
||||
if let DurableIlmRecordCheckpoint::RecoveryDisposition { identity_sha256, .. } = &mut noncanonical_identity {
|
||||
*identity_sha256 = "A".repeat(64);
|
||||
}
|
||||
assert!(prepared.validate_successor(&noncanonical_identity).is_err());
|
||||
let mut changed_created_at = claimed;
|
||||
if let DurableIlmRecordCheckpoint::RecoveryDisposition {
|
||||
created_at_unix_nanos, ..
|
||||
} = &mut changed_created_at
|
||||
{
|
||||
*created_at_unix_nanos += 1;
|
||||
}
|
||||
assert!(prepared.validate_successor(&changed_created_at).is_err());
|
||||
|
||||
let mut early_takeover = taken_over;
|
||||
if let DurableIlmRecordCheckpoint::RecoveryDisposition {
|
||||
owner_lease_acquired_at_unix_nanos,
|
||||
..
|
||||
} = &mut early_takeover
|
||||
{
|
||||
*owner_lease_acquired_at_unix_nanos = Some(9);
|
||||
}
|
||||
assert!(first_absent.validate_successor(&early_takeover).is_err());
|
||||
assert!(
|
||||
applying
|
||||
.validate_successor(&recovery_disposition_checkpoint(
|
||||
4,
|
||||
Applying,
|
||||
Some("e"),
|
||||
vec!["c".repeat(64), "b".repeat(64)],
|
||||
))
|
||||
.is_err(),
|
||||
"confirmed-absent entries must be a canonical sorted set"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_disposition_terminal_predecessor_requires_exact_identity_and_full_manifest() {
|
||||
use recovery_disposition::IlmRecoveryDispositionState::{Applying, Completed, Prepared};
|
||||
|
||||
let first_copy = "b".repeat(64);
|
||||
let second_copy = "c".repeat(64);
|
||||
let prepared = recovery_disposition_checkpoint(1, Prepared, None, Vec::new());
|
||||
let applying = recovery_disposition_checkpoint(3, Applying, Some("e"), vec![first_copy.clone()]);
|
||||
let completed = recovery_disposition_checkpoint(5, Completed, None, vec![first_copy.clone(), second_copy]);
|
||||
|
||||
assert!(prepared.is_predecessor_of_terminal(&completed));
|
||||
assert!(applying.is_predecessor_of_terminal(&completed));
|
||||
assert!(
|
||||
!prepared.is_predecessor_of_terminal(&recovery_disposition_checkpoint(2, Applying, Some("e"), Vec::new())),
|
||||
"a nonterminal disposition must not authorize terminal cleanup"
|
||||
);
|
||||
assert!(
|
||||
!prepared.is_predecessor_of_terminal(&recovery_disposition_checkpoint(
|
||||
4,
|
||||
Completed,
|
||||
None,
|
||||
vec![first_copy.clone(), "c".repeat(64)],
|
||||
)),
|
||||
"terminal proof must leave enough revisions for claim, apply, progress, and completion"
|
||||
);
|
||||
assert!(
|
||||
!applying.is_predecessor_of_terminal(&recovery_disposition_checkpoint(
|
||||
4,
|
||||
Completed,
|
||||
None,
|
||||
vec![first_copy.clone(), "c".repeat(64)],
|
||||
)),
|
||||
"an incomplete Applying checkpoint cannot complete without a progress generation"
|
||||
);
|
||||
|
||||
let mut other_identity = completed;
|
||||
if let DurableIlmRecordCheckpoint::RecoveryDisposition { identity_sha256, .. } = &mut other_identity {
|
||||
*identity_sha256 = "e".repeat(64);
|
||||
}
|
||||
assert!(!prepared.is_predecessor_of_terminal(&other_identity));
|
||||
|
||||
let incomplete_terminal = recovery_disposition_checkpoint(4, Completed, None, vec![first_copy]);
|
||||
assert!(
|
||||
!prepared.is_predecessor_of_terminal(&incomplete_terminal),
|
||||
"a partial confirmed-absent set must not become terminal proof"
|
||||
);
|
||||
}
|
||||
|
||||
fn tier_probe_intent_fixture() -> tier_probe_intent::TierProbeIntent {
|
||||
let probe_id = Uuid::parse_str("36e2220e-9ad2-495b-b3bc-c4d2caf70a31").expect("fixture uuid should parse");
|
||||
tier_probe_intent::TierProbeIntent {
|
||||
@@ -1687,6 +2187,64 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn transition_checkpoint_accepts_only_the_distinguishable_compact_edge() {
|
||||
let identity_sha256 = "a".repeat(64);
|
||||
let unknown_remote_sha256 = "b".repeat(64);
|
||||
let known_remote_sha256 = "c".repeat(64);
|
||||
let checkpoint = |revision, state, remote_version_sha256: String, remote_version_known| {
|
||||
DurableIlmRecordCheckpoint::TransitionTransaction {
|
||||
content_sha256: format!("{revision:064x}"),
|
||||
identity_sha256: identity_sha256.clone(),
|
||||
remote_version_sha256,
|
||||
remote_version_known,
|
||||
revision,
|
||||
state,
|
||||
}
|
||||
};
|
||||
let compact_unknown = checkpoint(
|
||||
1,
|
||||
transition_transaction::TransitionTransactionState::UploadOutcomeUnknown,
|
||||
unknown_remote_sha256.clone(),
|
||||
false,
|
||||
);
|
||||
let compact_local_commit = checkpoint(
|
||||
2,
|
||||
transition_transaction::TransitionTransactionState::LocalCommitStarted,
|
||||
known_remote_sha256.clone(),
|
||||
true,
|
||||
);
|
||||
compact_unknown
|
||||
.validate_successor(&compact_local_commit)
|
||||
.expect("compact pre-upload fence should advance directly to the exact local-commit fence");
|
||||
|
||||
let legacy_unknown = checkpoint(
|
||||
2,
|
||||
transition_transaction::TransitionTransactionState::UploadOutcomeUnknown,
|
||||
unknown_remote_sha256,
|
||||
false,
|
||||
);
|
||||
let invalid_legacy_skip = checkpoint(
|
||||
3,
|
||||
transition_transaction::TransitionTransactionState::LocalCommitStarted,
|
||||
known_remote_sha256.clone(),
|
||||
true,
|
||||
);
|
||||
assert!(
|
||||
legacy_unknown.validate_successor(&invalid_legacy_skip).is_err(),
|
||||
"legacy UploadOutcomeUnknown@2 must not masquerade as the compact edge"
|
||||
);
|
||||
let valid_legacy_skip = checkpoint(
|
||||
4,
|
||||
transition_transaction::TransitionTransactionState::LocalCommitStarted,
|
||||
known_remote_sha256,
|
||||
true,
|
||||
);
|
||||
legacy_unknown
|
||||
.validate_successor(&valid_legacy_skip)
|
||||
.expect("legacy receipts may still observe the existing two-edge state advance");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_dispatch_manifest_namespace_validates_monotonic_branches() {
|
||||
use tier_delete_journal::TierDeleteDispatchManifestState::{Aborted, Aborting, Completed, DispatchAuthorized, Preparing};
|
||||
|
||||
@@ -25,6 +25,9 @@ mod object_handlers_common;
|
||||
mod object_lock_boundary;
|
||||
pub use self::core as lifecycle;
|
||||
pub mod recovery_control;
|
||||
pub mod recovery_disposition;
|
||||
pub(crate) mod recovery_disposition_runtime;
|
||||
pub mod recovery_export;
|
||||
mod replication_sink;
|
||||
pub mod rule;
|
||||
mod runtime_boundary;
|
||||
|
||||
@@ -168,7 +168,7 @@ impl IlmRecoverySourceGeneration {
|
||||
Ok(generation)
|
||||
}
|
||||
|
||||
fn validate(&self) -> Result<()> {
|
||||
pub(crate) fn validate(&self) -> Result<()> {
|
||||
if self.source_schema.trim().is_empty() {
|
||||
return Err(IlmRecoveryControlError::Corrupt("source schema is empty"));
|
||||
}
|
||||
@@ -485,6 +485,40 @@ impl IlmRecoveryControl {
|
||||
self.validate()
|
||||
}
|
||||
|
||||
pub fn abandon_for_operator(&mut self, expected_source_generation: &IlmRecoverySourceGeneration) -> Result<()> {
|
||||
if self.owner.is_some()
|
||||
|| self.classification != IlmRecoveryClassification::RetainedAmbiguous
|
||||
|| &self.observed_source_generation != expected_source_generation
|
||||
{
|
||||
return Err(IlmRecoveryControlError::InvalidSuccessor(
|
||||
"operator abandonment requires the exact ownerless retained source generation",
|
||||
));
|
||||
}
|
||||
self.bump_revision()?;
|
||||
self.classification = IlmRecoveryClassification::Abandoned;
|
||||
self.validate()
|
||||
}
|
||||
|
||||
pub fn retry_for_operator(&mut self, expected_source_generation: &IlmRecoverySourceGeneration) -> Result<()> {
|
||||
if self.owner.is_some()
|
||||
|| !matches!(
|
||||
self.classification,
|
||||
IlmRecoveryClassification::RetainedAmbiguous | IlmRecoveryClassification::OperatorRequired
|
||||
)
|
||||
|| self.attempt_count == u64::MAX
|
||||
|| &self.observed_source_generation != expected_source_generation
|
||||
{
|
||||
return Err(IlmRecoveryControlError::InvalidSuccessor(
|
||||
"operator retry requires the exact ownerless retained source generation",
|
||||
));
|
||||
}
|
||||
self.bump_revision()?;
|
||||
self.classification = IlmRecoveryClassification::Retrying;
|
||||
self.consecutive_failure_count = 0;
|
||||
self.next_attempt_at_unix_nanos = None;
|
||||
self.validate()
|
||||
}
|
||||
|
||||
pub fn validate_successor(&self, next: &Self) -> Result<()> {
|
||||
self.validate()?;
|
||||
next.validate()?;
|
||||
@@ -508,12 +542,58 @@ impl IlmRecoveryControl {
|
||||
self.validate_failure_successor(next)
|
||||
}
|
||||
(Some(_), None) => self.validate_finish_successor(next),
|
||||
(None, None)
|
||||
if self.classification == IlmRecoveryClassification::RetainedAmbiguous
|
||||
&& next.classification == IlmRecoveryClassification::Abandoned =>
|
||||
{
|
||||
self.validate_operator_abandon_successor(next)
|
||||
}
|
||||
(None, None)
|
||||
if matches!(
|
||||
self.classification,
|
||||
IlmRecoveryClassification::RetainedAmbiguous | IlmRecoveryClassification::OperatorRequired
|
||||
) && next.classification == IlmRecoveryClassification::Retrying =>
|
||||
{
|
||||
self.validate_operator_retry_successor(next)
|
||||
}
|
||||
(None, None) => Err(IlmRecoveryControlError::InvalidSuccessor(
|
||||
"ownerless control cannot advance without a claim",
|
||||
)),
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_operator_abandon_successor(&self, next: &Self) -> Result<()> {
|
||||
if next.observed_source_generation != self.observed_source_generation
|
||||
|| next.attempt_count != self.attempt_count
|
||||
|| next.consecutive_failure_count != self.consecutive_failure_count
|
||||
|| next.first_failure_at_unix_nanos != self.first_failure_at_unix_nanos
|
||||
|| next.last_failure_at_unix_nanos != self.last_failure_at_unix_nanos
|
||||
|| next.next_attempt_at_unix_nanos != self.next_attempt_at_unix_nanos
|
||||
|| next.last_error_code != self.last_error_code
|
||||
{
|
||||
return Err(IlmRecoveryControlError::InvalidSuccessor(
|
||||
"operator abandonment changed recovery history or source generation",
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn validate_operator_retry_successor(&self, next: &Self) -> Result<()> {
|
||||
if next.observed_source_generation != self.observed_source_generation
|
||||
|| next.attempt_count != self.attempt_count
|
||||
|| next.consecutive_failure_count != 0
|
||||
|| next.first_failure_at_unix_nanos != self.first_failure_at_unix_nanos
|
||||
|| next.last_failure_at_unix_nanos != self.last_failure_at_unix_nanos
|
||||
|| next.next_attempt_at_unix_nanos.is_some()
|
||||
|| next.last_error_code != self.last_error_code
|
||||
{
|
||||
return Err(IlmRecoveryControlError::InvalidSuccessor(
|
||||
"operator retry changed recovery history or source generation",
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn validate_claim_successor(&self, next: &Self) -> Result<()> {
|
||||
if self.classification != IlmRecoveryClassification::Retrying
|
||||
|| next.classification != IlmRecoveryClassification::Retrying
|
||||
@@ -827,6 +907,23 @@ pub async fn observe_recovery_source(
|
||||
api: Arc<ECStore>,
|
||||
canonical_path: &str,
|
||||
source_schema: &str,
|
||||
) -> EcstoreResult<ObservedIlmRecoverySource> {
|
||||
observe_recovery_source_with_options(api, canonical_path, source_schema, false).await
|
||||
}
|
||||
|
||||
pub(crate) async fn observe_recovery_source_no_lock(
|
||||
api: Arc<ECStore>,
|
||||
canonical_path: &str,
|
||||
source_schema: &str,
|
||||
) -> EcstoreResult<ObservedIlmRecoverySource> {
|
||||
observe_recovery_source_with_options(api, canonical_path, source_schema, true).await
|
||||
}
|
||||
|
||||
async fn observe_recovery_source_with_options(
|
||||
api: Arc<ECStore>,
|
||||
canonical_path: &str,
|
||||
source_schema: &str,
|
||||
no_lock: bool,
|
||||
) -> EcstoreResult<ObservedIlmRecoverySource> {
|
||||
validate_canonical_source_path(canonical_path).map_err(recovery_control_store_error)?;
|
||||
if source_schema.trim().is_empty() {
|
||||
@@ -837,7 +934,16 @@ pub async fn observe_recovery_source(
|
||||
let mut observations = Vec::new();
|
||||
for set in api.all_set_disks() {
|
||||
let authority = format!("pool-{}/set-{}", set.pool_index, set.set_index);
|
||||
match config_boundary::read_config_with_metadata(set, canonical_path, &ObjectOptions::default()).await {
|
||||
match config_boundary::read_config_with_metadata(
|
||||
set,
|
||||
canonical_path,
|
||||
&ObjectOptions {
|
||||
no_lock,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok((data, metadata)) => {
|
||||
let etag = metadata
|
||||
.etag
|
||||
@@ -1255,6 +1361,99 @@ mod tests {
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn operator_abandonment_is_an_exact_ownerless_retained_successor() {
|
||||
let mut retained = IlmRecoveryControl::new(
|
||||
control().identity,
|
||||
generation(),
|
||||
IlmRecoveryClassification::RetainedAmbiguous,
|
||||
1_000_000_000,
|
||||
IlmRecoveryErrorCode::OperatorDispositionRequired,
|
||||
)
|
||||
.expect("retained control should build");
|
||||
let previous = retained.clone();
|
||||
retained
|
||||
.abandon_for_operator(&previous.observed_source_generation)
|
||||
.expect("exact retained generation should be abandonable");
|
||||
previous
|
||||
.validate_successor(&retained)
|
||||
.expect("operator abandonment should be a valid successor");
|
||||
assert_eq!(retained.classification, IlmRecoveryClassification::Abandoned);
|
||||
assert_eq!(retained.revision, previous.revision + 1);
|
||||
|
||||
let mut wrong_generation = previous.clone();
|
||||
let mut generation = previous.observed_source_generation.clone();
|
||||
generation.source_etag = "different".to_string();
|
||||
assert!(wrong_generation.abandon_for_operator(&generation).is_err());
|
||||
|
||||
let mut mutated_history = retained.clone();
|
||||
mutated_history.attempt_count += 1;
|
||||
assert!(previous.validate_successor(&mutated_history).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn operator_retry_rearms_exact_retained_generation_without_resetting_history() {
|
||||
for classification in [
|
||||
IlmRecoveryClassification::RetainedAmbiguous,
|
||||
IlmRecoveryClassification::OperatorRequired,
|
||||
] {
|
||||
let mut retained = control();
|
||||
retained
|
||||
.claim("node-a", Uuid::new_v4(), 2_000_000_000, 1)
|
||||
.expect("attempt should claim");
|
||||
retained
|
||||
.record_retryable_failure(2_000_000_001, IlmRecoveryErrorCode::BackendTimeout)
|
||||
.expect("failure should persist");
|
||||
retained.classification = classification;
|
||||
retained.next_attempt_at_unix_nanos = None;
|
||||
if classification == IlmRecoveryClassification::OperatorRequired {
|
||||
retained.attempt_count = u64::from(MAX_RECOVERY_ATTEMPTS);
|
||||
retained.consecutive_failure_count = MAX_RECOVERY_ATTEMPTS;
|
||||
}
|
||||
retained.validate().expect("retained control should remain valid");
|
||||
|
||||
let previous = retained.clone();
|
||||
retained
|
||||
.retry_for_operator(&previous.observed_source_generation)
|
||||
.expect("exact retained generation should be retryable");
|
||||
previous
|
||||
.validate_successor(&retained)
|
||||
.expect("operator retry should be a valid successor");
|
||||
assert_eq!(retained.classification, IlmRecoveryClassification::Retrying);
|
||||
assert_eq!(retained.revision, previous.revision + 1);
|
||||
assert_eq!(retained.attempt_count, previous.attempt_count);
|
||||
assert_eq!(retained.first_failure_at_unix_nanos, previous.first_failure_at_unix_nanos);
|
||||
assert_eq!(retained.last_failure_at_unix_nanos, previous.last_failure_at_unix_nanos);
|
||||
assert_eq!(retained.last_error_code, previous.last_error_code);
|
||||
assert_eq!(retained.consecutive_failure_count, 0);
|
||||
assert_eq!(retained.next_attempt_at_unix_nanos, None);
|
||||
assert!(retained.should_attempt_at(2_000_000_002));
|
||||
|
||||
if classification == IlmRecoveryClassification::OperatorRequired {
|
||||
retained
|
||||
.claim("node-b", Uuid::new_v4(), 2_000_000_002, 1)
|
||||
.expect("operator retry should authorize one new bounded attempt");
|
||||
retained
|
||||
.record_retryable_failure(2_000_000_003, IlmRecoveryErrorCode::BackendTimeout)
|
||||
.expect("the bounded attempt failure should persist");
|
||||
assert_eq!(retained.classification, IlmRecoveryClassification::OperatorRequired);
|
||||
assert_eq!(retained.attempt_count, u64::from(MAX_RECOVERY_ATTEMPTS) + 1);
|
||||
assert_eq!(retained.consecutive_failure_count, 1);
|
||||
}
|
||||
|
||||
let mut wrong_generation = previous;
|
||||
let mut changed_generation = wrong_generation.observed_source_generation.clone();
|
||||
changed_generation.source_etag = "changed".to_string();
|
||||
assert!(wrong_generation.retry_for_operator(&changed_generation).is_err());
|
||||
}
|
||||
|
||||
let mut exhausted = control();
|
||||
exhausted.classification = IlmRecoveryClassification::OperatorRequired;
|
||||
exhausted.attempt_count = u64::MAX;
|
||||
let generation = exhausted.observed_source_generation.clone();
|
||||
assert!(exhausted.retry_for_operator(&generation).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_control_view_redacts_source_and_owner_details() {
|
||||
let mut control = control();
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,840 @@
|
||||
// Copyright 2024 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use std::{collections::HashSet, sync::Arc};
|
||||
|
||||
use rustfs_utils::crypto::{hex_sha256, is_sha256_checksum};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
use super::config_boundary;
|
||||
use super::recovery_control::{
|
||||
IlmRecoveryClassification, IlmRecoveryControl, IlmRecoveryProtocol, IlmRecoverySourceCopy, IlmRecoverySourceGeneration,
|
||||
MAX_ILM_RECOVERY_CONTROL_SIZE, ObservedIlmRecoveryControl, ObservedIlmRecoverySource, recovery_control_record_object_name,
|
||||
};
|
||||
use super::tier_delete_journal::{
|
||||
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA, validate_legacy_tier_delete_recovery_source,
|
||||
};
|
||||
use crate::disk::RUSTFS_META_BUCKET;
|
||||
use crate::error::{Error, Result};
|
||||
use crate::object_api::{ObjectOptions, WriteCompletion};
|
||||
use crate::services::notification_sys::{
|
||||
acquire_ilm_recovery_export_fleet_proof, ilm_recovery_export_fleet_proof_matches, ilm_recovery_export_member_epochs_sha256,
|
||||
ilm_recovery_export_topology_generation,
|
||||
};
|
||||
use crate::storage_api_contracts::{list::ListOperations as _, namespace::NamespaceLocking as _, object::HTTPPreconditions};
|
||||
use crate::store::ECStore;
|
||||
|
||||
pub const ILM_RECOVERY_EXPORT_SCHEMA: &str = "rustfs-ilm-recovery-export-v1";
|
||||
pub const ILM_RECOVERY_EXPORT_PREFIX: &str = "ilm/recovery-exports";
|
||||
pub const MAX_ILM_RECOVERY_EXPORT_SIZE: usize = 128 * 1024;
|
||||
const MAX_ILM_RECOVERY_EXPORTS: usize = 10_000;
|
||||
const MAX_ILM_RECOVERY_EXPORT_BYTES: u64 = 1024 * 1024 * 1024;
|
||||
const MAX_ACTOR_EXPORTS_PER_MINUTE: usize = 10;
|
||||
const MAX_CLUSTER_EXPORTS_PER_MINUTE: usize = 100;
|
||||
const EXPORT_RETENTION_NANOS: i64 = 90 * 24 * 60 * 60 * 1_000_000_000;
|
||||
const EXPORT_ADMISSION_LOCK: &str = "ilm/recovery-admission/export.lock";
|
||||
const MAX_LEGACY_TIER_DELETE_SOURCE_SIZE: usize = 64 * 1024;
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct IlmRecoveryExportObservation {
|
||||
pub control_id: String,
|
||||
pub protocol: IlmRecoveryProtocol,
|
||||
pub control_etag: String,
|
||||
pub control_revision: u64,
|
||||
pub classification: IlmRecoveryClassification,
|
||||
pub canonical_source_path: String,
|
||||
pub source_generation: IlmRecoverySourceGeneration,
|
||||
pub topology_generation: String,
|
||||
pub member_epochs_sha256: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct IlmRecoveryExport {
|
||||
pub export_id: String,
|
||||
pub control_id: String,
|
||||
pub protocol: IlmRecoveryProtocol,
|
||||
pub control_etag: String,
|
||||
pub control_revision: u64,
|
||||
pub classification: IlmRecoveryClassification,
|
||||
pub canonical_source_path: String,
|
||||
pub source_generation: IlmRecoverySourceGeneration,
|
||||
pub topology_generation: String,
|
||||
pub member_epochs_sha256: String,
|
||||
pub creator_sha256: String,
|
||||
pub created_at_unix_nanos: i64,
|
||||
pub retain_until_unix_nanos: i64,
|
||||
pub source_bytes_base64: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
struct PersistedIlmRecoveryExport {
|
||||
schema: String,
|
||||
content_sha256: String,
|
||||
export: IlmRecoveryExport,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
|
||||
pub struct IlmRecoveryExportCreated {
|
||||
pub export_id: String,
|
||||
pub content_sha256: String,
|
||||
pub encoded: Vec<u8>,
|
||||
pub replayed: bool,
|
||||
}
|
||||
|
||||
impl IlmRecoveryExport {
|
||||
fn validate(&self) -> Result<()> {
|
||||
self.source_generation.validate().map_err(Error::other)?;
|
||||
validate_sha256(&self.export_id, "ILM recovery export ID is invalid")?;
|
||||
validate_sha256(&self.control_id, "ILM recovery export control ID is invalid")?;
|
||||
validate_sha256(&self.topology_generation, "ILM recovery export topology generation is invalid")?;
|
||||
validate_sha256(&self.member_epochs_sha256, "ILM recovery export member epoch digest is invalid")?;
|
||||
validate_sha256(&self.creator_sha256, "ILM recovery export creator digest is invalid")?;
|
||||
if self.protocol != IlmRecoveryProtocol::TierDeleteJournal
|
||||
|| self.classification != IlmRecoveryClassification::RetainedAmbiguous
|
||||
|| !is_legacy_export_schema(&self.source_generation.source_schema)
|
||||
{
|
||||
return Err(Error::other("ILM recovery export source is not an exportable legacy journal"));
|
||||
}
|
||||
if self.control_etag.trim().is_empty() || self.control_revision == 0 {
|
||||
return Err(Error::other("ILM recovery export control generation is invalid"));
|
||||
}
|
||||
if self.canonical_source_path.is_empty()
|
||||
|| self.canonical_source_path.starts_with('/')
|
||||
|| self.canonical_source_path.ends_with('/')
|
||||
|| self.canonical_source_path.split('/').any(str::is_empty)
|
||||
{
|
||||
return Err(Error::other("ILM recovery export source path is invalid"));
|
||||
}
|
||||
if self.created_at_unix_nanos <= 0
|
||||
|| self.retain_until_unix_nanos < self.created_at_unix_nanos.saturating_add(EXPORT_RETENTION_NANOS)
|
||||
{
|
||||
return Err(Error::other("ILM recovery export retention is invalid"));
|
||||
}
|
||||
let source = base64_simd::STANDARD
|
||||
.decode_to_vec(self.source_bytes_base64.as_bytes())
|
||||
.map_err(|_| Error::other("ILM recovery export source encoding is invalid"))?;
|
||||
validate_legacy_tier_delete_recovery_source(&self.canonical_source_path, &self.source_generation.source_schema, &source)?;
|
||||
let encoded_len = u64::try_from(source.len()).map_err(|_| Error::other("ILM recovery export source length overflow"))?;
|
||||
if source.is_empty()
|
||||
|| source.len() > MAX_LEGACY_TIER_DELETE_SOURCE_SIZE
|
||||
|| hex_sha256(&source, ToOwned::to_owned) != self.source_generation.content_sha256
|
||||
|| self.source_generation.copies.iter().any(|copy| {
|
||||
copy.canonical_path != self.canonical_source_path
|
||||
|| copy.etag != self.source_generation.source_etag
|
||||
|| copy.content_sha256 != self.source_generation.content_sha256
|
||||
|| copy.encoded_len != encoded_len
|
||||
})
|
||||
{
|
||||
return Err(Error::other("ILM recovery export source bytes do not match the observed generation"));
|
||||
}
|
||||
if recovery_export_id(&self.control_id, &self.source_generation)? != self.export_id {
|
||||
return Err(Error::other("ILM recovery export ID does not match its source generation"));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn encode(&self) -> Result<Vec<u8>> {
|
||||
self.validate()?;
|
||||
let export_bytes = serde_json::to_vec(self).map_err(Error::other)?;
|
||||
let persisted = PersistedIlmRecoveryExport {
|
||||
schema: ILM_RECOVERY_EXPORT_SCHEMA.to_string(),
|
||||
content_sha256: hex_sha256(&export_bytes, ToOwned::to_owned),
|
||||
export: self.clone(),
|
||||
};
|
||||
let encoded = serde_json::to_vec(&persisted).map_err(Error::other)?;
|
||||
if encoded.len() > MAX_ILM_RECOVERY_EXPORT_SIZE {
|
||||
return Err(Error::other("encoded ILM recovery export exceeds maximum size"));
|
||||
}
|
||||
Ok(encoded)
|
||||
}
|
||||
|
||||
pub fn decode(expected_export_id: &str, data: &[u8]) -> Result<Self> {
|
||||
validate_sha256(expected_export_id, "ILM recovery export ID is invalid")?;
|
||||
if data.len() > MAX_ILM_RECOVERY_EXPORT_SIZE {
|
||||
return Err(Error::other("encoded ILM recovery export exceeds maximum size"));
|
||||
}
|
||||
let persisted: PersistedIlmRecoveryExport = serde_json::from_slice(data).map_err(Error::other)?;
|
||||
if persisted.schema != ILM_RECOVERY_EXPORT_SCHEMA {
|
||||
return Err(Error::other("ILM recovery export schema is unsupported"));
|
||||
}
|
||||
validate_sha256(&persisted.content_sha256, "ILM recovery export checksum is invalid")?;
|
||||
let export_bytes = serde_json::to_vec(&persisted.export).map_err(Error::other)?;
|
||||
if hex_sha256(&export_bytes, ToOwned::to_owned) != persisted.content_sha256 {
|
||||
return Err(Error::other("ILM recovery export checksum mismatch"));
|
||||
}
|
||||
persisted.export.validate()?;
|
||||
if persisted.export.export_id != expected_export_id {
|
||||
return Err(Error::other("ILM recovery export ID does not match record key"));
|
||||
}
|
||||
Ok(persisted.export)
|
||||
}
|
||||
}
|
||||
|
||||
pub fn recovery_export_record_object_name(protocol: IlmRecoveryProtocol, export_id: &str) -> Result<String> {
|
||||
validate_sha256(export_id, "ILM recovery export ID is invalid")?;
|
||||
Ok(format!(
|
||||
"{}/{}/{}/{}/{}.json",
|
||||
ILM_RECOVERY_EXPORT_PREFIX,
|
||||
protocol.as_str(),
|
||||
&export_id[..2],
|
||||
&export_id[2..4],
|
||||
export_id
|
||||
))
|
||||
}
|
||||
|
||||
pub fn recovery_export_id_from_record_object_name(object: &str) -> Result<(IlmRecoveryProtocol, String)> {
|
||||
let suffix = object
|
||||
.strip_prefix(ILM_RECOVERY_EXPORT_PREFIX)
|
||||
.and_then(|suffix| suffix.strip_prefix('/'))
|
||||
.ok_or_else(|| Error::other("ILM recovery export path has wrong prefix"))?;
|
||||
let mut parts = suffix.split('/');
|
||||
let protocol = match parts.next() {
|
||||
Some("tier_delete_journal") => IlmRecoveryProtocol::TierDeleteJournal,
|
||||
_ => return Err(Error::other("ILM recovery export protocol is invalid")),
|
||||
};
|
||||
let shard_a = parts
|
||||
.next()
|
||||
.ok_or_else(|| Error::other("ILM recovery export path is incomplete"))?;
|
||||
let shard_b = parts
|
||||
.next()
|
||||
.ok_or_else(|| Error::other("ILM recovery export path is incomplete"))?;
|
||||
let export_id = parts
|
||||
.next()
|
||||
.and_then(|name| name.strip_suffix(".json"))
|
||||
.ok_or_else(|| Error::other("ILM recovery export suffix is invalid"))?;
|
||||
if parts.next().is_some() {
|
||||
return Err(Error::other("ILM recovery export path is not canonical"));
|
||||
}
|
||||
validate_sha256(export_id, "ILM recovery export ID is invalid")?;
|
||||
if shard_a != &export_id[..2] || shard_b != &export_id[2..4] {
|
||||
return Err(Error::other("ILM recovery export shard does not match export ID"));
|
||||
}
|
||||
Ok((protocol, export_id.to_string()))
|
||||
}
|
||||
|
||||
pub async fn inspect_recovery_export_observation(api: Arc<ECStore>, control_id: &str) -> Result<IlmRecoveryExportObservation> {
|
||||
let proof = acquire_ilm_recovery_export_fleet_proof()
|
||||
.await
|
||||
.ok_or_else(|| Error::other("ILM recovery export fleet proof is unavailable"))?;
|
||||
let observed_control = load_exportable_control(api.clone(), control_id).await?;
|
||||
let observed_source = observe_export_source(
|
||||
api,
|
||||
&observed_control.control.identity.canonical_source_path,
|
||||
&observed_control.control.observed_source_generation.source_schema,
|
||||
)
|
||||
.await?;
|
||||
if !observed_source.is_consistent()
|
||||
|| observed_source.generation != observed_control.control.observed_source_generation
|
||||
|| !ilm_recovery_export_fleet_proof_matches(&proof).await
|
||||
{
|
||||
return Err(Error::other("ILM recovery export observation changed or is incomplete"));
|
||||
}
|
||||
Ok(IlmRecoveryExportObservation {
|
||||
control_id: control_id.to_string(),
|
||||
protocol: observed_control.control.identity.protocol,
|
||||
control_etag: observed_control.etag,
|
||||
control_revision: observed_control.control.revision,
|
||||
classification: observed_control.control.classification,
|
||||
canonical_source_path: observed_control.control.identity.canonical_source_path,
|
||||
source_generation: observed_source.generation,
|
||||
topology_generation: ilm_recovery_export_topology_generation(&proof),
|
||||
member_epochs_sha256: ilm_recovery_export_member_epochs_sha256(&proof),
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn create_recovery_export(
|
||||
api: Arc<ECStore>,
|
||||
observation: &IlmRecoveryExportObservation,
|
||||
creator_sha256: &str,
|
||||
) -> Result<IlmRecoveryExportCreated> {
|
||||
validate_sha256(creator_sha256, "ILM recovery export creator digest is invalid")?;
|
||||
let lock = api.new_ns_lock(RUSTFS_META_BUCKET, EXPORT_ADMISSION_LOCK).await?;
|
||||
let admission_guard = lock.get_write_lock(crate::set_disk::get_lock_acquire_timeout()).await?;
|
||||
|
||||
let proof = acquire_ilm_recovery_export_fleet_proof()
|
||||
.await
|
||||
.ok_or_else(|| Error::other("ILM recovery export fleet proof is unavailable"))?;
|
||||
if ilm_recovery_export_topology_generation(&proof) != observation.topology_generation
|
||||
|| ilm_recovery_export_member_epochs_sha256(&proof) != observation.member_epochs_sha256
|
||||
{
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
let control_object = recovery_control_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, &observation.control_id)
|
||||
.map_err(Error::other)?;
|
||||
let control_lock = api.new_ns_lock(RUSTFS_META_BUCKET, &control_object).await?;
|
||||
let control_guard = control_lock
|
||||
.get_read_lock(crate::set_disk::get_lock_acquire_timeout())
|
||||
.await?;
|
||||
let source_lock = api
|
||||
.new_ns_lock(RUSTFS_META_BUCKET, &observation.canonical_source_path)
|
||||
.await?;
|
||||
let source_guard = source_lock.get_read_lock(crate::set_disk::get_lock_acquire_timeout()).await?;
|
||||
let locks_current = || !admission_guard.is_lock_lost() && !control_guard.is_lock_lost() && !source_guard.is_lock_lost();
|
||||
let (current, current_source_bytes) = current_observation_under_proof_no_lock(api.clone(), observation, &proof).await?;
|
||||
if ¤t != observation || !locks_current() {
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
let current_source_base64 = base64_simd::STANDARD.encode_to_string(current_source_bytes);
|
||||
let candidate_export_id = recovery_export_id(¤t.control_id, ¤t.source_generation)?;
|
||||
let object = recovery_export_record_object_name(current.protocol, &candidate_export_id)?;
|
||||
match load_recovery_export_decoded(api.clone(), &candidate_export_id).await {
|
||||
Ok((existing, export)) if export_matches_observation(&export, observation) => {
|
||||
if !locks_current() || !ilm_recovery_export_fleet_proof_matches(&proof).await {
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
api.record_durable_ilm_decommission_progress(&object, &existing.encoded)
|
||||
.await?;
|
||||
if !locks_current() {
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
return Ok(existing.with_replayed());
|
||||
}
|
||||
Ok(_) => return Err(Error::PreconditionFailed),
|
||||
Err(Error::ConfigNotFound) => {}
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
let inventory = collect_export_inventory(api.clone()).await?;
|
||||
if !locks_current() || !ilm_recovery_export_fleet_proof_matches(&proof).await {
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
let created_at_unix_nanos = now_unix_nanos()?;
|
||||
let export = build_export_from_source(¤t, creator_sha256, created_at_unix_nanos, ¤t_source_base64)?;
|
||||
let encoded = export.encode()?;
|
||||
inventory.check(creator_sha256, encoded.len(), created_at_unix_nanos)?;
|
||||
|
||||
let mut write_options = ObjectOptions {
|
||||
max_parity: true,
|
||||
write_completion: WriteCompletion::TailDrained,
|
||||
http_preconditions: Some(HTTPPreconditions {
|
||||
if_none_match: Some("*".to_string()),
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
write_options.add_namespace_lock_guard(&admission_guard);
|
||||
write_options.add_namespace_lock_guard(&control_guard);
|
||||
write_options.add_namespace_lock_guard(&source_guard);
|
||||
if !locks_current() {
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
let write_result = config_boundary::save_config_with_opts(api.clone(), &object, encoded.clone(), &write_options).await;
|
||||
let stored = match load_recovery_export(api.clone(), &export.export_id).await {
|
||||
Ok(stored) if stored.encoded == encoded => stored,
|
||||
Ok(_) => return Err(Error::PreconditionFailed),
|
||||
Err(read_err) => return Err(write_result.err().unwrap_or(read_err)),
|
||||
};
|
||||
if !locks_current() || !ilm_recovery_export_fleet_proof_matches(&proof).await {
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
api.record_durable_ilm_decommission_progress(&object, &encoded).await?;
|
||||
if !locks_current() {
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
Ok(stored)
|
||||
}
|
||||
|
||||
pub async fn load_recovery_export(api: Arc<ECStore>, export_id: &str) -> Result<IlmRecoveryExportCreated> {
|
||||
let (created, _) = load_recovery_export_decoded(api, export_id).await?;
|
||||
Ok(created)
|
||||
}
|
||||
|
||||
async fn load_recovery_export_decoded(
|
||||
api: Arc<ECStore>,
|
||||
export_id: &str,
|
||||
) -> Result<(IlmRecoveryExportCreated, IlmRecoveryExport)> {
|
||||
let object = recovery_export_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, export_id)?;
|
||||
let encoded = config_boundary::read_config_limited_preserve_empty(api, &object, MAX_ILM_RECOVERY_EXPORT_SIZE).await?;
|
||||
let export = IlmRecoveryExport::decode(export_id, &encoded)?;
|
||||
let content_sha256 = hex_sha256(&encoded, ToOwned::to_owned);
|
||||
Ok((
|
||||
IlmRecoveryExportCreated {
|
||||
export_id: export.export_id.clone(),
|
||||
content_sha256,
|
||||
encoded,
|
||||
replayed: false,
|
||||
},
|
||||
export,
|
||||
))
|
||||
}
|
||||
|
||||
impl IlmRecoveryExportCreated {
|
||||
fn with_replayed(mut self) -> Self {
|
||||
self.replayed = true;
|
||||
self
|
||||
}
|
||||
}
|
||||
|
||||
async fn load_exportable_control(api: Arc<ECStore>, control_id: &str) -> Result<ObservedIlmRecoveryControl> {
|
||||
load_exportable_control_with_options(api, control_id, &ObjectOptions::default()).await
|
||||
}
|
||||
|
||||
async fn load_exportable_control_no_lock(api: Arc<ECStore>, control_id: &str) -> Result<ObservedIlmRecoveryControl> {
|
||||
load_exportable_control_with_options(
|
||||
api,
|
||||
control_id,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn load_exportable_control_with_options(
|
||||
api: Arc<ECStore>,
|
||||
control_id: &str,
|
||||
options: &ObjectOptions,
|
||||
) -> Result<ObservedIlmRecoveryControl> {
|
||||
let object = recovery_control_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, control_id).map_err(Error::other)?;
|
||||
let (data, metadata) =
|
||||
config_boundary::read_config_limited_preserve_empty_with_metadata(api, &object, options, MAX_ILM_RECOVERY_CONTROL_SIZE)
|
||||
.await?;
|
||||
let etag = metadata
|
||||
.etag
|
||||
.filter(|etag| !etag.trim().is_empty())
|
||||
.ok_or_else(|| Error::other("ILM recovery control is missing an ETag"))?;
|
||||
let control = IlmRecoveryControl::decode(control_id, &data).map_err(Error::other)?;
|
||||
if control.identity.protocol != IlmRecoveryProtocol::TierDeleteJournal
|
||||
|| control.classification != IlmRecoveryClassification::RetainedAmbiguous
|
||||
|| !is_legacy_export_schema(&control.observed_source_generation.source_schema)
|
||||
{
|
||||
return Err(Error::other("ILM recovery control is not exportable"));
|
||||
}
|
||||
Ok(ObservedIlmRecoveryControl { control, etag })
|
||||
}
|
||||
|
||||
async fn current_observation_under_proof_no_lock(
|
||||
api: Arc<ECStore>,
|
||||
expected: &IlmRecoveryExportObservation,
|
||||
proof: &crate::services::notification_sys::IlmRecoveryExportFleetProofToken,
|
||||
) -> Result<(IlmRecoveryExportObservation, Vec<u8>)> {
|
||||
let observed_control = load_exportable_control_no_lock(api.clone(), &expected.control_id).await?;
|
||||
let observed_source = observe_export_source_no_lock(
|
||||
api,
|
||||
&observed_control.control.identity.canonical_source_path,
|
||||
&observed_control.control.observed_source_generation.source_schema,
|
||||
)
|
||||
.await?;
|
||||
let source_bytes = observed_source
|
||||
.canonical_data
|
||||
.clone()
|
||||
.ok_or_else(|| Error::other("ILM recovery export source copies diverge"))?;
|
||||
if !observed_source.is_consistent()
|
||||
|| observed_source.generation != observed_control.control.observed_source_generation
|
||||
|| !ilm_recovery_export_fleet_proof_matches(proof).await
|
||||
{
|
||||
return Err(Error::PreconditionFailed);
|
||||
}
|
||||
Ok((
|
||||
IlmRecoveryExportObservation {
|
||||
control_id: expected.control_id.clone(),
|
||||
protocol: observed_control.control.identity.protocol,
|
||||
control_etag: observed_control.etag,
|
||||
control_revision: observed_control.control.revision,
|
||||
classification: observed_control.control.classification,
|
||||
canonical_source_path: observed_control.control.identity.canonical_source_path,
|
||||
source_generation: observed_source.generation,
|
||||
topology_generation: ilm_recovery_export_topology_generation(proof),
|
||||
member_epochs_sha256: ilm_recovery_export_member_epochs_sha256(proof),
|
||||
},
|
||||
source_bytes,
|
||||
))
|
||||
}
|
||||
|
||||
async fn observe_export_source(
|
||||
api: Arc<ECStore>,
|
||||
canonical_path: &str,
|
||||
source_schema: &str,
|
||||
) -> Result<ObservedIlmRecoverySource> {
|
||||
if canonical_path.is_empty()
|
||||
|| canonical_path.starts_with('/')
|
||||
|| canonical_path.ends_with('/')
|
||||
|| canonical_path.split('/').any(str::is_empty)
|
||||
|| !is_legacy_export_schema(source_schema)
|
||||
{
|
||||
return Err(Error::other("ILM recovery export source identity is invalid"));
|
||||
}
|
||||
let lock = api.new_ns_lock(RUSTFS_META_BUCKET, canonical_path).await?;
|
||||
let _guard = lock.get_read_lock(crate::set_disk::get_lock_acquire_timeout()).await?;
|
||||
observe_export_source_no_lock(api, canonical_path, source_schema).await
|
||||
}
|
||||
|
||||
async fn observe_export_source_no_lock(
|
||||
api: Arc<ECStore>,
|
||||
canonical_path: &str,
|
||||
source_schema: &str,
|
||||
) -> Result<ObservedIlmRecoverySource> {
|
||||
let mut copies = Vec::new();
|
||||
let mut canonical: Option<(String, String, Vec<u8>)> = None;
|
||||
let mut consistent = true;
|
||||
for set in api.all_set_disks() {
|
||||
let authority = format!("pool-{}/set-{}", set.pool_index, set.set_index);
|
||||
let result = config_boundary::read_config_limited_preserve_empty_with_metadata(
|
||||
set,
|
||||
canonical_path,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
MAX_LEGACY_TIER_DELETE_SOURCE_SIZE,
|
||||
)
|
||||
.await;
|
||||
match result {
|
||||
Ok((data, metadata)) => {
|
||||
if data.is_empty() || data.len() > MAX_LEGACY_TIER_DELETE_SOURCE_SIZE {
|
||||
return Err(Error::other("ILM recovery export source exceeds its protocol size limit"));
|
||||
}
|
||||
validate_legacy_tier_delete_recovery_source(canonical_path, source_schema, &data)?;
|
||||
let etag = metadata
|
||||
.etag
|
||||
.filter(|etag| !etag.trim().is_empty())
|
||||
.ok_or_else(|| Error::other("ILM recovery export source copy is missing an ETag"))?;
|
||||
let content_sha256 = hex_sha256(&data, ToOwned::to_owned);
|
||||
let encoded_len =
|
||||
u64::try_from(data.len()).map_err(|_| Error::other("ILM recovery export source length does not fit u64"))?;
|
||||
copies.push(IlmRecoverySourceCopy {
|
||||
authority,
|
||||
canonical_path: canonical_path.to_string(),
|
||||
etag: etag.clone(),
|
||||
encoded_len,
|
||||
content_sha256: content_sha256.clone(),
|
||||
});
|
||||
match canonical.as_ref() {
|
||||
Some((first_etag, first_digest, first_data)) => {
|
||||
consistent &= first_etag == &etag && first_digest == &content_sha256 && first_data == &data;
|
||||
}
|
||||
None => canonical = Some((etag, content_sha256, data)),
|
||||
}
|
||||
}
|
||||
Err(err) if export_source_is_missing(&err) => {}
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
}
|
||||
let Some((source_etag, content_sha256, source_bytes)) = canonical else {
|
||||
return Err(Error::ConfigNotFound);
|
||||
};
|
||||
let generation =
|
||||
IlmRecoverySourceGeneration::new(source_schema, source_etag, content_sha256, copies).map_err(Error::other)?;
|
||||
Ok(ObservedIlmRecoverySource {
|
||||
generation,
|
||||
canonical_data: consistent.then_some(source_bytes),
|
||||
})
|
||||
}
|
||||
|
||||
fn export_source_is_missing(err: &Error) -> bool {
|
||||
matches!(
|
||||
err,
|
||||
Error::ConfigNotFound | Error::FileNotFound | Error::ObjectNotFound(_, _) | Error::VersionNotFound(_, _, _)
|
||||
)
|
||||
}
|
||||
|
||||
fn build_export_from_source(
|
||||
observation: &IlmRecoveryExportObservation,
|
||||
creator_sha256: &str,
|
||||
created_at_unix_nanos: i64,
|
||||
source_bytes_base64: &str,
|
||||
) -> Result<IlmRecoveryExport> {
|
||||
let retain_until_unix_nanos = created_at_unix_nanos
|
||||
.checked_add(EXPORT_RETENTION_NANOS)
|
||||
.ok_or_else(|| Error::other("ILM recovery export retention timestamp overflow"))?;
|
||||
let export = IlmRecoveryExport {
|
||||
export_id: recovery_export_id(&observation.control_id, &observation.source_generation)?,
|
||||
control_id: observation.control_id.clone(),
|
||||
protocol: observation.protocol,
|
||||
control_etag: observation.control_etag.clone(),
|
||||
control_revision: observation.control_revision,
|
||||
classification: observation.classification,
|
||||
canonical_source_path: observation.canonical_source_path.clone(),
|
||||
source_generation: observation.source_generation.clone(),
|
||||
topology_generation: observation.topology_generation.clone(),
|
||||
member_epochs_sha256: observation.member_epochs_sha256.clone(),
|
||||
creator_sha256: creator_sha256.to_string(),
|
||||
created_at_unix_nanos,
|
||||
retain_until_unix_nanos,
|
||||
source_bytes_base64: source_bytes_base64.to_string(),
|
||||
};
|
||||
export.validate()?;
|
||||
Ok(export)
|
||||
}
|
||||
|
||||
pub(crate) fn recovery_export_id(control_id: &str, generation: &IlmRecoverySourceGeneration) -> Result<String> {
|
||||
validate_sha256(control_id, "ILM recovery export control ID is invalid")?;
|
||||
validate_sha256(&generation.content_sha256, "ILM recovery export source checksum is invalid")?;
|
||||
validate_sha256(&generation.copy_set_sha256, "ILM recovery export copy-set checksum is invalid")?;
|
||||
let mut data = Vec::new();
|
||||
for part in [control_id, &generation.content_sha256, &generation.copy_set_sha256] {
|
||||
data.extend_from_slice(&(part.len() as u64).to_be_bytes());
|
||||
data.extend_from_slice(part.as_bytes());
|
||||
}
|
||||
Ok(hex_sha256(&data, ToOwned::to_owned))
|
||||
}
|
||||
|
||||
fn export_matches_observation(export: &IlmRecoveryExport, observation: &IlmRecoveryExportObservation) -> bool {
|
||||
export.control_id == observation.control_id
|
||||
&& export.protocol == observation.protocol
|
||||
&& export.classification == observation.classification
|
||||
&& export.canonical_source_path == observation.canonical_source_path
|
||||
&& export.source_generation == observation.source_generation
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
struct IlmRecoveryExportInventory {
|
||||
count: usize,
|
||||
bytes: u64,
|
||||
creations: Vec<(i64, String)>,
|
||||
}
|
||||
|
||||
impl IlmRecoveryExportInventory {
|
||||
fn check(&self, creator_sha256: &str, candidate_len: usize, now: i64) -> Result<()> {
|
||||
let recent_after = now.saturating_sub(60 * 1_000_000_000);
|
||||
let cluster_recent = self
|
||||
.creations
|
||||
.iter()
|
||||
.filter(|(created_at, _)| *created_at > recent_after)
|
||||
.count();
|
||||
let actor_recent = self
|
||||
.creations
|
||||
.iter()
|
||||
.filter(|(created_at, creator)| *created_at > recent_after && creator == creator_sha256)
|
||||
.count();
|
||||
check_export_admission(self.count, self.bytes, actor_recent, cluster_recent, candidate_len)
|
||||
}
|
||||
}
|
||||
|
||||
async fn collect_export_inventory(api: Arc<ECStore>) -> Result<IlmRecoveryExportInventory> {
|
||||
let mut marker = None;
|
||||
let mut seen_markers = HashSet::new();
|
||||
let mut inventory = IlmRecoveryExportInventory::default();
|
||||
loop {
|
||||
let page = api
|
||||
.clone()
|
||||
.list_objects_v2(
|
||||
RUSTFS_META_BUCKET,
|
||||
&format!("{ILM_RECOVERY_EXPORT_PREFIX}/"),
|
||||
marker.clone(),
|
||||
None,
|
||||
1_000,
|
||||
false,
|
||||
None,
|
||||
false,
|
||||
)
|
||||
.await?;
|
||||
for object in page.objects {
|
||||
let (_, export_id) = recovery_export_id_from_record_object_name(&object.name)?;
|
||||
let (stored, export) = load_recovery_export_decoded(api.clone(), &export_id).await?;
|
||||
inventory.count = inventory
|
||||
.count
|
||||
.checked_add(1)
|
||||
.ok_or_else(|| Error::other("ILM recovery export count overflow"))?;
|
||||
inventory.bytes = inventory
|
||||
.bytes
|
||||
.checked_add(u64::try_from(stored.encoded.len()).map_err(|_| Error::other("ILM recovery export size overflow"))?)
|
||||
.ok_or_else(|| Error::other("ILM recovery export byte total overflow"))?;
|
||||
inventory
|
||||
.creations
|
||||
.push((export.created_at_unix_nanos, export.creator_sha256));
|
||||
}
|
||||
if !page.is_truncated {
|
||||
break;
|
||||
}
|
||||
let next = page
|
||||
.next_continuation_token
|
||||
.ok_or_else(|| Error::other("ILM recovery export inventory omitted its continuation marker"))?;
|
||||
marker = Some(record_export_inventory_marker(&mut seen_markers, next)?);
|
||||
}
|
||||
Ok(inventory)
|
||||
}
|
||||
|
||||
fn record_export_inventory_marker(seen_markers: &mut HashSet<String>, next: String) -> Result<String> {
|
||||
if !seen_markers.insert(next.clone()) {
|
||||
return Err(Error::other("ILM recovery export inventory repeated its continuation marker"));
|
||||
}
|
||||
Ok(next)
|
||||
}
|
||||
|
||||
fn check_export_admission(
|
||||
count: usize,
|
||||
bytes: u64,
|
||||
actor_recent: usize,
|
||||
cluster_recent: usize,
|
||||
candidate_len: usize,
|
||||
) -> Result<()> {
|
||||
let candidate_len = u64::try_from(candidate_len).map_err(|_| Error::other("ILM recovery export size does not fit u64"))?;
|
||||
if count >= MAX_ILM_RECOVERY_EXPORTS
|
||||
|| bytes
|
||||
.checked_add(candidate_len)
|
||||
.is_none_or(|total| total > MAX_ILM_RECOVERY_EXPORT_BYTES)
|
||||
|| actor_recent >= MAX_ACTOR_EXPORTS_PER_MINUTE
|
||||
|| cluster_recent >= MAX_CLUSTER_EXPORTS_PER_MINUTE
|
||||
{
|
||||
return Err(Error::SlowDown);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn is_legacy_export_schema(schema: &str) -> bool {
|
||||
matches!(schema, TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA | TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA)
|
||||
}
|
||||
|
||||
fn validate_sha256(value: &str, message: &'static str) -> Result<()> {
|
||||
if !is_sha256_checksum(value)
|
||||
|| value
|
||||
.bytes()
|
||||
.any(|byte| byte.is_ascii_hexdigit() && byte.is_ascii_uppercase())
|
||||
{
|
||||
return Err(Error::other(message));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn now_unix_nanos() -> Result<i64> {
|
||||
i64::try_from(time::OffsetDateTime::now_utc().unix_timestamp_nanos())
|
||||
.map_err(|_| Error::other("ILM recovery export timestamp does not fit i64"))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::bucket::lifecycle::recovery_control::IlmRecoverySourceCopy;
|
||||
|
||||
const PINNED_V1_EXPORT: &[u8] = br#"{"schema":"rustfs-ilm-recovery-export-v1","content_sha256":"3dfb3ec3892256e909de1211c1a963ca7008963ff32b3a869f7161a7b9b44028","export":{"export_id":"2b78e7a825bfc2edbf7f773d0b6ed3bf93e360ff1702d73a449109c11bfaa105","control_id":"0fcd568a5cb9bdb4677b69354b11ee415af8f784519cff3da49a26f84eaee7f2","protocol":"tier_delete_journal","control_etag":"control-etag","control_revision":1,"classification":"retained_ambiguous","canonical_source_path":"ilm/tier-delete-journal/872072554f66ab326f10ce7adbae11422b7a4b0663aa7112d6061a8f6ed41b94.json","source_generation":{"source_schema":"rustfs-tier-delete-journal-v1","source_etag":"etag-a","content_sha256":"0e0b010ebdeeb7b41473fe8575e989d6bb1303c0ca551dd984e9400f0ae306bd","copy_set_sha256":"5a7406115b6c3923ffe79dcd1f43ccae7beed786e557163f019dd10ec409a653","copies":[{"authority":"pool-0/set-0","canonical_path":"ilm/tier-delete-journal/872072554f66ab326f10ce7adbae11422b7a4b0663aa7112d6061a8f6ed41b94.json","etag":"etag-a","encoded_len":81,"content_sha256":"0e0b010ebdeeb7b41473fe8575e989d6bb1303c0ca551dd984e9400f0ae306bd"}]},"topology_generation":"e6e2b826e31fca5c36125c48f130dcb6f961e698ff8a8776a1f290cf0892e8e6","member_epochs_sha256":"612dd8a861161819a4ad8f6f3e2a0567602877c043a2353ca933a13c78dc0ed4","creator_sha256":"50c9c4aeb40b5b206b6d98f516f8b8c0efd29ce2e56a76b345fb9240c225a1b7","created_at_unix_nanos":1000000000,"retain_until_unix_nanos":7776001000000000,"source_bytes_base64":"eyJ2ZXJzaW9uIjoxLCJvYmpfbmFtZSI6ImxlZ2FjeS9yZW1vdGUiLCJ2ZXJzaW9uX2lkIjoib3BhcXVlIiwidGllcl9uYW1lIjoiV0FSTSJ9"}}"#;
|
||||
|
||||
fn legacy_source() -> Vec<u8> {
|
||||
br#"{"version":1,"obj_name":"legacy/remote","version_id":"opaque","tier_name":"WARM"}"#.to_vec()
|
||||
}
|
||||
|
||||
fn observation() -> IlmRecoveryExportObservation {
|
||||
let source = legacy_source();
|
||||
let source_path = super::super::tier_delete_journal::tier_delete_journal_object_name(
|
||||
&super::super::tier_delete_journal::decode_tier_delete_journal_entry(&source).expect("legacy fixture should decode"),
|
||||
);
|
||||
let source_sha256 = hex_sha256(&source, ToOwned::to_owned);
|
||||
let generation = IlmRecoverySourceGeneration::new(
|
||||
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA,
|
||||
"etag-a",
|
||||
source_sha256.clone(),
|
||||
vec![IlmRecoverySourceCopy {
|
||||
authority: "pool-0/set-0".to_string(),
|
||||
canonical_path: source_path.clone(),
|
||||
etag: "etag-a".to_string(),
|
||||
encoded_len: source.len() as u64,
|
||||
content_sha256: source_sha256,
|
||||
}],
|
||||
)
|
||||
.expect("generation should be valid");
|
||||
IlmRecoveryExportObservation {
|
||||
control_id: hex_sha256(b"control", ToOwned::to_owned),
|
||||
protocol: IlmRecoveryProtocol::TierDeleteJournal,
|
||||
control_etag: "control-etag".to_string(),
|
||||
control_revision: 1,
|
||||
classification: IlmRecoveryClassification::RetainedAmbiguous,
|
||||
canonical_source_path: source_path,
|
||||
source_generation: generation,
|
||||
topology_generation: hex_sha256(b"topology", ToOwned::to_owned),
|
||||
member_epochs_sha256: hex_sha256(b"epochs", ToOwned::to_owned),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_export_round_trip_is_strict_and_deterministic() {
|
||||
let observed = observation();
|
||||
let creator = hex_sha256(b"actor", ToOwned::to_owned);
|
||||
let export = build_export_from_source(
|
||||
&observed,
|
||||
&creator,
|
||||
1_000_000_000,
|
||||
&base64_simd::STANDARD.encode_to_string(legacy_source()),
|
||||
)
|
||||
.expect("export should be valid");
|
||||
assert_eq!(
|
||||
export.export_id,
|
||||
recovery_export_id(&observed.control_id, &observed.source_generation).unwrap()
|
||||
);
|
||||
let encoded = export.encode().expect("export should encode");
|
||||
assert_eq!(encoded, PINNED_V1_EXPORT, "v1 export wire format must remain pinned");
|
||||
assert_eq!(IlmRecoveryExport::decode(&export.export_id, &encoded).unwrap(), export);
|
||||
assert_eq!(
|
||||
IlmRecoveryExport::decode("2b78e7a825bfc2edbf7f773d0b6ed3bf93e360ff1702d73a449109c11bfaa105", PINNED_V1_EXPORT)
|
||||
.unwrap(),
|
||||
export,
|
||||
);
|
||||
|
||||
let path = recovery_export_record_object_name(export.protocol, &export.export_id).unwrap();
|
||||
let durable = super::super::durable_namespace::validate_durable_ilm_record(&path, &encoded)
|
||||
.expect("export should be registered as a durable ILM record");
|
||||
assert_eq!(durable.namespace, "recovery-export");
|
||||
assert_eq!(durable.id_kind, "export_id");
|
||||
assert_eq!(durable.id, export.export_id);
|
||||
|
||||
let mut wrong_source = export.clone();
|
||||
wrong_source.source_bytes_base64 = base64_simd::STANDARD.encode_to_string(b"changed");
|
||||
assert!(wrong_source.encode().is_err());
|
||||
|
||||
let mut persisted: serde_json::Value = serde_json::from_slice(&encoded).unwrap();
|
||||
persisted["unknown"] = serde_json::json!(true);
|
||||
assert!(IlmRecoveryExport::decode(&export.export_id, &serde_json::to_vec(&persisted).unwrap()).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn export_inventory_rejects_non_adjacent_continuation_cycles() {
|
||||
let mut seen = HashSet::new();
|
||||
assert_eq!(record_export_inventory_marker(&mut seen, "a".to_string()).unwrap(), "a");
|
||||
assert_eq!(record_export_inventory_marker(&mut seen, "b".to_string()).unwrap(), "b");
|
||||
record_export_inventory_marker(&mut seen, "a".to_string())
|
||||
.expect_err("a non-adjacent continuation marker cycle must fail closed");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_export_path_rejects_noncanonical_shards() {
|
||||
let id = hex_sha256(b"export", ToOwned::to_owned);
|
||||
let path = recovery_export_record_object_name(IlmRecoveryProtocol::TierDeleteJournal, &id).unwrap();
|
||||
assert_eq!(recovery_export_id_from_record_object_name(&path).unwrap().1, id);
|
||||
let wrong_shard = path.replacen(&format!("/{}/", &id[..2]), "/zz/", 1);
|
||||
assert!(recovery_export_id_from_record_object_name(&wrong_shard).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn canonical_replay_survives_fleet_rotation_but_not_source_change() {
|
||||
let observed = observation();
|
||||
let creator = hex_sha256(b"actor", ToOwned::to_owned);
|
||||
let export = build_export_from_source(
|
||||
&observed,
|
||||
&creator,
|
||||
1_000_000_000,
|
||||
&base64_simd::STANDARD.encode_to_string(legacy_source()),
|
||||
)
|
||||
.unwrap();
|
||||
let mut rotated = observed;
|
||||
rotated.control_etag = "new-control-etag".to_string();
|
||||
rotated.control_revision += 1;
|
||||
rotated.topology_generation = hex_sha256(b"new-topology", ToOwned::to_owned);
|
||||
rotated.member_epochs_sha256 = hex_sha256(b"new-members", ToOwned::to_owned);
|
||||
assert!(export_matches_observation(&export, &rotated));
|
||||
|
||||
rotated.source_generation.content_sha256 = hex_sha256(b"changed", ToOwned::to_owned);
|
||||
assert!(!export_matches_observation(&export, &rotated));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn export_admission_enforces_exact_count_byte_and_rate_boundaries() {
|
||||
assert!(check_export_admission(9_999, MAX_ILM_RECOVERY_EXPORT_BYTES - 1, 9, 99, 1).is_ok());
|
||||
assert!(check_export_admission(10_000, 0, 0, 0, 1).is_err());
|
||||
assert!(check_export_admission(0, MAX_ILM_RECOVERY_EXPORT_BYTES, 0, 0, 1).is_err());
|
||||
assert!(check_export_admission(0, 0, 10, 0, 1).is_err());
|
||||
assert!(check_export_admission(0, 0, 0, 100, 1).is_err());
|
||||
}
|
||||
}
|
||||
@@ -82,8 +82,8 @@ const TIER_DELETE_DISPATCH_MEMBER_DELETE_CONCURRENCY: usize = 32;
|
||||
const TIER_DELETE_DISPATCH_PREPARE_CONCURRENCY: usize = 16;
|
||||
const TIER_DELETE_DISPATCH_CAS_CONCURRENCY: usize = 32;
|
||||
const TIER_DELETE_JOURNAL_VERSION: u8 = 2;
|
||||
const TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v1";
|
||||
const TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v2";
|
||||
pub(crate) const TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v1";
|
||||
pub(crate) const TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-v2";
|
||||
const TIER_DELETE_JOURNAL_UNKNOWN_RECOVERY_SCHEMA: &str = "rustfs-tier-delete-journal-unknown";
|
||||
const TIER_DELETE_JOURNAL_V1_RECOVERY_CLASS: &str = "tier_delete_journal_v1";
|
||||
const TIER_DELETE_JOURNAL_V2_RECOVERY_CLASS: &str = "tier_delete_journal_v2";
|
||||
@@ -884,6 +884,23 @@ struct PersistedTierDeleteJournalEntry {
|
||||
}
|
||||
|
||||
impl PersistedTierDeleteJournalEntry {
|
||||
fn validate_legacy_recovery_shape(&self) -> Result<()> {
|
||||
let has_later_version_fields = self.version_id_exact.is_some()
|
||||
|| self.version_state.is_some()
|
||||
|| self.state.is_some()
|
||||
|| self.source.is_some()
|
||||
|| self.dispatch.is_some();
|
||||
match self.version {
|
||||
1 if self.backend_identity.is_none() && !has_later_version_fields => Ok(()),
|
||||
TIER_DELETE_JOURNAL_VERSION if self.backend_identity.is_some() && !has_later_version_fields => Ok(()),
|
||||
1 => Err(Error::other("tier delete journal v1 entry contains fields from a later version")),
|
||||
TIER_DELETE_JOURNAL_VERSION => Err(Error::other(
|
||||
"tier delete journal v2 entry is missing its identity or contains fields from a later version",
|
||||
)),
|
||||
_ => Err(Error::other("tier delete journal is not an exportable legacy version")),
|
||||
}
|
||||
}
|
||||
|
||||
fn from_jentry(je: &Jentry) -> Result<Self> {
|
||||
validate_version_state(je.version_state, &je.version_id, je.version_id_exact)?;
|
||||
let legacy_unknown = je.version_state == rustfs_filemeta::TransitionVersionState::Unknown;
|
||||
@@ -5531,6 +5548,12 @@ fn canonical_legacy_tier_delete_journal_identity(object_name: &str) -> Option<&s
|
||||
.then_some(identity)
|
||||
}
|
||||
|
||||
pub(crate) fn validate_legacy_tier_delete_recovery_path(object_name: &str) -> Result<()> {
|
||||
canonical_legacy_tier_delete_journal_identity(object_name)
|
||||
.map(|_| ())
|
||||
.ok_or_else(|| Error::other("legacy tier delete journal path is not canonical"))
|
||||
}
|
||||
|
||||
fn legacy_tier_delete_recovery_descriptor(entry: &Jentry) -> Option<(&'static str, &'static str)> {
|
||||
match entry.persisted_version {
|
||||
1 => Some((TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V1_RECOVERY_CLASS)),
|
||||
@@ -5539,6 +5562,21 @@ fn legacy_tier_delete_recovery_descriptor(entry: &Jentry) -> Option<(&'static st
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn validate_legacy_tier_delete_recovery_source(object_name: &str, source_schema: &str, data: &[u8]) -> Result<()> {
|
||||
validate_legacy_tier_delete_recovery_path(object_name)?;
|
||||
let persisted: PersistedTierDeleteJournalEntry =
|
||||
serde_json::from_slice(data).map_err(|err| Error::other_with_context("decode tier delete journal failed", err))?;
|
||||
persisted.validate_legacy_recovery_shape()?;
|
||||
let entry = persisted.into_jentry()?;
|
||||
let Some((decoded_schema, _)) = legacy_tier_delete_recovery_descriptor(&entry) else {
|
||||
return Err(Error::other("tier delete journal is not an exportable legacy version"));
|
||||
};
|
||||
if decoded_schema != source_schema || tier_delete_journal_object_name(&entry) != object_name {
|
||||
return Err(Error::other("legacy tier delete journal identity does not match its recovery source"));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn legacy_tier_delete_control_matches(
|
||||
control: &IlmRecoveryControl,
|
||||
identity: &IlmRecoveryControlIdentity,
|
||||
@@ -6140,17 +6178,18 @@ where
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{
|
||||
TIER_DELETE_DISPATCH_MANIFEST_VERSION, TIER_DELETE_DISPATCH_PARENT_RECORD_TYPE, TIER_DELETE_DISPATCH_PARENT_VERSION,
|
||||
TIER_DELETE_JOURNAL_EXACT_VERSION, TIER_DELETE_JOURNAL_LEGACY_PREFIX, TIER_DELETE_JOURNAL_SOLE_OWNER_VERSION,
|
||||
TIER_DELETE_JOURNAL_STATE_VERSION, TIER_DELETE_JOURNAL_TRANSACTION_VERSION, TIER_DELETE_JOURNAL_V6_PREFIX,
|
||||
TierDeleteDispatchChunkBinding, TierDeleteDispatchManifest, TierDeleteDispatchManifestState, TierDeleteDispatchParent,
|
||||
TierDeleteDispatchParentState, TierDeleteDispatchRecord, await_tier_delete_journal_recovery,
|
||||
PersistedTierDeleteJournalEntry, TIER_DELETE_DISPATCH_MANIFEST_VERSION, TIER_DELETE_DISPATCH_PARENT_RECORD_TYPE,
|
||||
TIER_DELETE_DISPATCH_PARENT_VERSION, TIER_DELETE_JOURNAL_EXACT_VERSION, TIER_DELETE_JOURNAL_LEGACY_PREFIX,
|
||||
TIER_DELETE_JOURNAL_SOLE_OWNER_VERSION, TIER_DELETE_JOURNAL_STATE_VERSION, TIER_DELETE_JOURNAL_TRANSACTION_VERSION,
|
||||
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA, TIER_DELETE_JOURNAL_V6_PREFIX,
|
||||
TIER_DELETE_JOURNAL_VERSION, TierDeleteDispatchChunkBinding, TierDeleteDispatchManifest, TierDeleteDispatchManifestState,
|
||||
TierDeleteDispatchParent, TierDeleteDispatchParentState, TierDeleteDispatchRecord, await_tier_delete_journal_recovery,
|
||||
decode_tier_delete_dispatch_record, decode_tier_delete_journal_entry, encode_tier_delete_dispatch_manifest,
|
||||
encode_tier_delete_dispatch_parent, encode_tier_delete_journal_entry, object_info_references_tier_delete,
|
||||
record_tier_delete_journal_backend_identity, same_tier_delete_authorization_identity, same_tier_delete_journal_identity,
|
||||
tier_delete_dispatch_child_matches_parent, tier_delete_dispatch_chunk_manifest_object_name,
|
||||
tier_delete_dispatch_journal_set_digest, tier_delete_dispatch_manifest_object_name, tier_delete_journal_object_name,
|
||||
tier_delete_source_matches_dispatch_scope,
|
||||
tier_delete_source_matches_dispatch_scope, validate_legacy_tier_delete_recovery_source,
|
||||
};
|
||||
use crate::bucket::lifecycle::tier_sweeper::{
|
||||
Jentry, TierDeleteDispatchBinding, TierDeleteJournalState, TierDeleteSourceIdentity,
|
||||
@@ -6609,6 +6648,72 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn legacy_recovery_export_rejects_fields_from_later_journal_versions() {
|
||||
let later = bound_v6_journal_entry(TierDeleteJournalState::Prepared);
|
||||
let v1 = PersistedTierDeleteJournalEntry {
|
||||
version: 1,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "opaque".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
backend_identity: None,
|
||||
version_id_exact: None,
|
||||
version_state: None,
|
||||
state: None,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
};
|
||||
let mut v2 = v1.clone();
|
||||
v2.version = TIER_DELETE_JOURNAL_VERSION;
|
||||
v2.backend_identity = Some([7; 32]);
|
||||
|
||||
let assert_rejected = |persisted: PersistedTierDeleteJournalEntry, schema: &str| {
|
||||
let normalized = persisted
|
||||
.clone()
|
||||
.into_jentry()
|
||||
.expect("the generic compatibility decoder should demonstrate the discarded field");
|
||||
let object_name = tier_delete_journal_object_name(&normalized);
|
||||
let encoded = serde_json::to_vec(&persisted).expect("mixed-version journal fixture should encode");
|
||||
let err = validate_legacy_tier_delete_recovery_source(&object_name, schema, &encoded)
|
||||
.expect_err("legacy recovery export must reject fields from later versions");
|
||||
assert!(err.to_string().contains("later version"));
|
||||
};
|
||||
|
||||
let mut invalid_v1 = Vec::new();
|
||||
let mut with_backend = v1.clone();
|
||||
with_backend.backend_identity = Some([7; 32]);
|
||||
invalid_v1.push(with_backend);
|
||||
for persisted in [&v1, &v2] {
|
||||
let schema = if persisted.version == 1 {
|
||||
TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA
|
||||
} else {
|
||||
TIER_DELETE_JOURNAL_V2_RECOVERY_SCHEMA
|
||||
};
|
||||
let mut invalid = Vec::new();
|
||||
let mut with_exact = persisted.clone();
|
||||
with_exact.version_id_exact = Some(false);
|
||||
invalid.push(with_exact);
|
||||
let mut with_version_state = persisted.clone();
|
||||
with_version_state.version_state = Some(rustfs_filemeta::TransitionVersionState::Unknown);
|
||||
invalid.push(with_version_state);
|
||||
let mut with_state = persisted.clone();
|
||||
with_state.state = Some(TierDeleteJournalState::Committed);
|
||||
invalid.push(with_state);
|
||||
let mut with_source = persisted.clone();
|
||||
with_source.source = later.source.clone();
|
||||
invalid.push(with_source);
|
||||
let mut with_dispatch = persisted.clone();
|
||||
with_dispatch.dispatch = later.dispatch.clone();
|
||||
invalid.push(with_dispatch);
|
||||
for record in invalid {
|
||||
assert_rejected(record, schema);
|
||||
}
|
||||
}
|
||||
for record in invalid_v1 {
|
||||
assert_rejected(record, TIER_DELETE_JOURNAL_V1_RECOVERY_SCHEMA);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_path_is_stable_and_sanitized() {
|
||||
let je = journal_entry();
|
||||
|
||||
@@ -34,7 +34,7 @@ use crate::bucket::lifecycle::tier_sweeper::{
|
||||
};
|
||||
use crate::disk::RUSTFS_META_BUCKET;
|
||||
use crate::error::{Error, Result as EcstoreResult};
|
||||
use crate::object_api::ObjectOptions;
|
||||
use crate::object_api::{ObjectInfo, ObjectOptions};
|
||||
use crate::services::tier::{tier::TierConfigMgr, warm_backend::TransitionCandidateProbe};
|
||||
use crate::storage_api_contracts::{
|
||||
list::ListOperations as _,
|
||||
@@ -273,6 +273,14 @@ pub struct TransitionTransactionInit {
|
||||
|
||||
impl TransitionTransaction {
|
||||
pub fn new(init: TransitionTransactionInit) -> Result<Self> {
|
||||
Self::new_with_initial_state(init, TransitionTransactionState::UploadStarted)
|
||||
}
|
||||
|
||||
pub(crate) fn new_compact(init: TransitionTransactionInit) -> Result<Self> {
|
||||
Self::new_with_initial_state(init, TransitionTransactionState::UploadOutcomeUnknown)
|
||||
}
|
||||
|
||||
fn new_with_initial_state(init: TransitionTransactionInit, state: TransitionTransactionState) -> Result<Self> {
|
||||
let remote_object =
|
||||
canonical_transition_remote_object(init.deployment_id, &init.source.bucket, init.transaction_id, init.write_id)?;
|
||||
let transaction = Self {
|
||||
@@ -286,7 +294,7 @@ impl TransitionTransaction {
|
||||
backend_fingerprint: init.backend_fingerprint,
|
||||
remote_object,
|
||||
remote_version: TransitionRemoteVersion::unknown(),
|
||||
state: TransitionTransactionState::UploadStarted,
|
||||
state,
|
||||
not_after_unix_nanos: init.not_after_unix_nanos,
|
||||
};
|
||||
transaction.validate()?;
|
||||
@@ -356,7 +364,7 @@ impl TransitionTransaction {
|
||||
remote_version: Option<TransitionRemoteVersion>,
|
||||
) -> Result<TransitionTransactionFence> {
|
||||
self.check_fence(fence)?;
|
||||
if !state_change_allowed(self.state, next) {
|
||||
if !state_change_allowed_at(self.state, next, self.revision) {
|
||||
return Err(TransitionTransactionError::InvalidStateChange {
|
||||
from: self.state,
|
||||
to: next,
|
||||
@@ -388,6 +396,14 @@ impl TransitionTransaction {
|
||||
}
|
||||
self.remote_version = TransitionRemoteVersion::unknown();
|
||||
}
|
||||
TransitionTransactionState::LocalCommitStarted if self.state == TransitionTransactionState::UploadOutcomeUnknown => {
|
||||
let remote_version =
|
||||
remote_version.ok_or(TransitionTransactionError::Corrupt("compact local commit requires remote version"))?;
|
||||
if remote_version.is_unknown() {
|
||||
return Err(TransitionTransactionError::Corrupt("compact local commit requires known remote version"));
|
||||
}
|
||||
self.remote_version = remote_version;
|
||||
}
|
||||
TransitionTransactionState::LocalCommitStarted | TransitionTransactionState::Committed => {
|
||||
if let Some(remote_version) = remote_version
|
||||
&& remote_version != self.remote_version
|
||||
@@ -640,7 +656,7 @@ pub(crate) async fn save_transition_transaction_record_if_current(
|
||||
) -> EcstoreResult<()> {
|
||||
let object = transition_transaction_record_object_name(next.transaction_id).map_err(transition_transaction_store_error)?;
|
||||
let revision_is_next = expected.revision.checked_add(1) == Some(next.revision);
|
||||
let state_is_next = state_change_allowed(expected.state, next.state)
|
||||
let state_is_next = state_change_allowed_at(expected.state, next.state, expected.revision)
|
||||
|| matches!(
|
||||
(expected.state, next.state),
|
||||
(
|
||||
@@ -954,6 +970,10 @@ pub enum TransitionOperatorError {
|
||||
expected: String,
|
||||
actual: TransitionOperatorProbe,
|
||||
},
|
||||
#[error("transition recovery control is stale")]
|
||||
StaleRecoveryControl,
|
||||
#[error("transition recovery control is not eligible for operator retry")]
|
||||
RetryNotAllowed,
|
||||
#[error("transition transaction store failed: {0}")]
|
||||
Store(#[source] Error),
|
||||
#[error("remote tier reconciliation failed: {0}")]
|
||||
@@ -962,6 +982,179 @@ pub enum TransitionOperatorError {
|
||||
|
||||
type TransitionOperatorResult<T> = std::result::Result<T, TransitionOperatorError>;
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
|
||||
pub struct TransitionRecoveryRetryStatus {
|
||||
pub control_id: String,
|
||||
pub transaction_id: Uuid,
|
||||
pub state: TransitionTransactionState,
|
||||
pub classification: IlmRecoveryClassification,
|
||||
pub control_revision: u64,
|
||||
pub attempt_count: u64,
|
||||
pub consecutive_failure_count: u32,
|
||||
pub last_error_code: IlmRecoveryErrorCode,
|
||||
pub source_generation_sha256: String,
|
||||
pub copy_set_sha256: String,
|
||||
pub retry_ready: bool,
|
||||
#[serde(skip_serializing_if = "Option::is_none")]
|
||||
pub retry_not_ready_reason: Option<&'static str>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
|
||||
pub struct TransitionRecoveryRetryResult {
|
||||
pub control_id: String,
|
||||
pub transaction_id: Uuid,
|
||||
pub previous_revision: u64,
|
||||
pub revision: u64,
|
||||
pub classification: IlmRecoveryClassification,
|
||||
pub attempt_count: u64,
|
||||
pub source_generation_sha256: String,
|
||||
}
|
||||
|
||||
struct TransitionRecoveryRetryContext {
|
||||
observed: ObservedIlmRecoveryControl,
|
||||
transaction: TransitionTransaction,
|
||||
source_generation_sha256: String,
|
||||
}
|
||||
|
||||
fn transition_recovery_retry_readiness(control: &IlmRecoveryControl) -> (bool, Option<&'static str>) {
|
||||
if control.owner.is_some() {
|
||||
return (false, Some("attempt_owned"));
|
||||
}
|
||||
match control.classification {
|
||||
IlmRecoveryClassification::RetainedAmbiguous | IlmRecoveryClassification::OperatorRequired => (true, None),
|
||||
IlmRecoveryClassification::Retrying => (false, Some("already_retrying")),
|
||||
IlmRecoveryClassification::Corrupt => (false, Some("source_corrupt")),
|
||||
IlmRecoveryClassification::Abandoned => (false, Some("source_abandoned")),
|
||||
IlmRecoveryClassification::Terminal => (false, Some("source_terminal")),
|
||||
}
|
||||
}
|
||||
|
||||
async fn load_transition_recovery_retry_context(
|
||||
api: Arc<ECStore>,
|
||||
control_id: &str,
|
||||
) -> TransitionOperatorResult<TransitionRecoveryRetryContext> {
|
||||
let observed = match load_recovery_control(api.clone(), IlmRecoveryProtocol::TransitionTransaction, control_id).await {
|
||||
Ok(observed) => observed,
|
||||
Err(Error::ConfigNotFound) => return Err(TransitionOperatorError::NotFound),
|
||||
Err(err) => return Err(TransitionOperatorError::Store(err)),
|
||||
};
|
||||
let transaction_id = Uuid::parse_str(&observed.control.identity.stable_operation_identity)
|
||||
.ok()
|
||||
.filter(|transaction_id| !transaction_id.is_nil())
|
||||
.ok_or(TransitionOperatorError::StaleRecoveryControl)?;
|
||||
let canonical_path = transition_transaction_record_object_name(transaction_id)
|
||||
.map_err(|err| TransitionOperatorError::Store(Error::other(err)))?;
|
||||
if observed.control.identity.canonical_source_path != canonical_path
|
||||
|| observed.control.identity.record_class != "transition_transaction_v1"
|
||||
{
|
||||
return Err(TransitionOperatorError::StaleRecoveryControl);
|
||||
}
|
||||
let transaction = match load_transition_transaction_record(api.clone(), transaction_id).await {
|
||||
Ok(transaction) => transaction,
|
||||
Err(Error::ConfigNotFound) => return Err(TransitionOperatorError::NotFound),
|
||||
Err(err) => return Err(TransitionOperatorError::Store(err)),
|
||||
};
|
||||
let source = observe_recovery_source(api, &canonical_path, TRANSITION_TRANSACTION_SCHEMA)
|
||||
.await
|
||||
.map_err(TransitionOperatorError::Store)?;
|
||||
let exact_source = source.is_consistent()
|
||||
&& source.generation == observed.control.observed_source_generation
|
||||
&& source
|
||||
.canonical_data
|
||||
.as_deref()
|
||||
.is_some_and(|data| TransitionTransaction::decode(transaction_id, data).is_ok_and(|decoded| decoded == transaction));
|
||||
if !exact_source {
|
||||
return Err(TransitionOperatorError::StaleRecoveryControl);
|
||||
}
|
||||
let generation = serde_json::to_vec(&observed.control.observed_source_generation)
|
||||
.map_err(|err| TransitionOperatorError::Store(Error::other(err)))?;
|
||||
Ok(TransitionRecoveryRetryContext {
|
||||
observed,
|
||||
transaction,
|
||||
source_generation_sha256: hex_sha256(&generation, ToOwned::to_owned),
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn inspect_transition_recovery_retry_for_operator(
|
||||
api: Arc<ECStore>,
|
||||
control_id: &str,
|
||||
) -> TransitionOperatorResult<TransitionRecoveryRetryStatus> {
|
||||
let context = load_transition_recovery_retry_context(api, control_id).await?;
|
||||
let (retry_ready, retry_not_ready_reason) = transition_recovery_retry_readiness(&context.observed.control);
|
||||
Ok(TransitionRecoveryRetryStatus {
|
||||
control_id: control_id.to_string(),
|
||||
transaction_id: context.transaction.transaction_id,
|
||||
state: context.transaction.state,
|
||||
classification: context.observed.control.classification,
|
||||
control_revision: context.observed.control.revision,
|
||||
attempt_count: context.observed.control.attempt_count,
|
||||
consecutive_failure_count: context.observed.control.consecutive_failure_count,
|
||||
last_error_code: context.observed.control.last_error_code,
|
||||
source_generation_sha256: context.source_generation_sha256,
|
||||
copy_set_sha256: context.observed.control.observed_source_generation.copy_set_sha256.clone(),
|
||||
retry_ready,
|
||||
retry_not_ready_reason,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn retry_transition_recovery_for_operator(
|
||||
api: Arc<ECStore>,
|
||||
control_id: &str,
|
||||
expected_control_revision: u64,
|
||||
expected_source_generation_sha256: &str,
|
||||
) -> TransitionOperatorResult<TransitionRecoveryRetryResult> {
|
||||
let control_object = recovery_control_record_object_name(IlmRecoveryProtocol::TransitionTransaction, control_id)
|
||||
.map_err(|err| TransitionOperatorError::Store(Error::other(err)))?;
|
||||
let retry_lock = api
|
||||
.new_ns_lock(RUSTFS_META_BUCKET, &format!("{control_object}.recovery-lock"))
|
||||
.await
|
||||
.map_err(TransitionOperatorError::Store)?;
|
||||
let retry_guard = retry_lock
|
||||
.get_write_lock(crate::set_disk::get_lock_acquire_timeout())
|
||||
.await
|
||||
.map_err(|err| TransitionOperatorError::Store(Error::other(err)))?;
|
||||
let context = load_transition_recovery_retry_context(api.clone(), control_id).await?;
|
||||
let (retry_ready, _) = transition_recovery_retry_readiness(&context.observed.control);
|
||||
if !retry_ready {
|
||||
return Err(TransitionOperatorError::RetryNotAllowed);
|
||||
}
|
||||
if retry_guard.is_lock_lost()
|
||||
|| expected_control_revision == 0
|
||||
|| context.observed.control.revision != expected_control_revision
|
||||
|| context.source_generation_sha256 != expected_source_generation_sha256
|
||||
{
|
||||
return Err(TransitionOperatorError::StaleRecoveryControl);
|
||||
}
|
||||
let previous_revision = context.observed.control.revision;
|
||||
let mut next = context.observed.control.clone();
|
||||
next.retry_for_operator(&context.observed.control.observed_source_generation)
|
||||
.map_err(|_| TransitionOperatorError::RetryNotAllowed)?;
|
||||
if retry_guard.is_lock_lost() {
|
||||
return Err(TransitionOperatorError::StaleRecoveryControl);
|
||||
}
|
||||
save_recovery_control_if_current(api.clone(), &context.observed, &next)
|
||||
.await
|
||||
.map_err(|err| match err {
|
||||
Error::PreconditionFailed => TransitionOperatorError::StaleRecoveryControl,
|
||||
err => TransitionOperatorError::Store(err),
|
||||
})?;
|
||||
let persisted = load_recovery_control(api, IlmRecoveryProtocol::TransitionTransaction, control_id)
|
||||
.await
|
||||
.map_err(TransitionOperatorError::Store)?;
|
||||
if retry_guard.is_lock_lost() || persisted.control != next {
|
||||
return Err(TransitionOperatorError::StaleRecoveryControl);
|
||||
}
|
||||
Ok(TransitionRecoveryRetryResult {
|
||||
control_id: control_id.to_string(),
|
||||
transaction_id: context.transaction.transaction_id,
|
||||
previous_revision,
|
||||
revision: persisted.control.revision,
|
||||
classification: persisted.control.classification,
|
||||
attempt_count: persisted.control.attempt_count,
|
||||
source_generation_sha256: context.source_generation_sha256,
|
||||
})
|
||||
}
|
||||
|
||||
fn validate_operator_reconcile_transaction(
|
||||
transaction: &TransitionTransaction,
|
||||
now_unix_nanos: i128,
|
||||
@@ -1706,12 +1899,27 @@ async fn local_commit_matches_transaction(api: Arc<ECStore>, transaction: &Trans
|
||||
.get_object_info(&transaction.source.bucket, &transaction.source.object, &opts)
|
||||
.await?;
|
||||
let transitioned = &object.transitioned_object;
|
||||
Ok(transitioned.status == TRANSITION_COMPLETE
|
||||
Ok(local_object_matches_transition_source(&object, &transaction.source)
|
||||
&& transitioned.status == TRANSITION_COMPLETE
|
||||
&& transitioned.name == transaction.remote_object
|
||||
&& transitioned.tier == transaction.tier_name
|
||||
&& transitioned.version_id == transaction.remote_version.tier_delete_version_id().unwrap_or_default())
|
||||
}
|
||||
|
||||
fn local_object_matches_transition_source(object: &ObjectInfo, source: &TransitionSourceIdentity) -> bool {
|
||||
let observed_version_id = object.version_id.filter(|version_id| !version_id.is_nil());
|
||||
let observed_mod_time = object
|
||||
.mod_time
|
||||
.and_then(|mod_time| i64::try_from(mod_time.unix_timestamp_nanos()).ok());
|
||||
object.bucket == source.bucket
|
||||
&& object.name == source.object
|
||||
&& observed_version_id == source.version_id
|
||||
&& object.data_dir == Some(source.data_dir)
|
||||
&& observed_mod_time == Some(source.mod_time_unix_nanos)
|
||||
&& object.size == source.size
|
||||
&& object.etag.as_deref() == Some(source.etag.as_str())
|
||||
}
|
||||
|
||||
fn transition_source_lookup_options(transaction: &TransitionTransaction) -> ObjectOptions {
|
||||
ObjectOptions {
|
||||
version_id: match transaction.source.version_mode {
|
||||
@@ -1954,7 +2162,7 @@ where
|
||||
}
|
||||
}
|
||||
|
||||
fn state_change_allowed(from: TransitionTransactionState, to: TransitionTransactionState) -> bool {
|
||||
fn state_change_allowed_at(from: TransitionTransactionState, to: TransitionTransactionState, revision: u64) -> bool {
|
||||
matches!(
|
||||
(from, to),
|
||||
(TransitionTransactionState::UploadStarted, TransitionTransactionState::Uploaded)
|
||||
@@ -1966,7 +2174,9 @@ fn state_change_allowed(from: TransitionTransactionState, to: TransitionTransact
|
||||
| (TransitionTransactionState::UploadOutcomeUnknown, TransitionTransactionState::Uploaded)
|
||||
| (TransitionTransactionState::Uploaded, TransitionTransactionState::LocalCommitStarted)
|
||||
| (TransitionTransactionState::LocalCommitStarted, TransitionTransactionState::Committed)
|
||||
)
|
||||
) || (revision == 1
|
||||
&& from == TransitionTransactionState::UploadOutcomeUnknown
|
||||
&& to == TransitionTransactionState::LocalCommitStarted)
|
||||
}
|
||||
|
||||
fn state_requires_known_remote_version(state: TransitionTransactionState) -> bool {
|
||||
@@ -2183,6 +2393,41 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn local_commit_proof_requires_the_complete_source_identity() {
|
||||
let source = source_identity(TransitionSourceVersionMode::Versioned);
|
||||
let exact = ObjectInfo {
|
||||
bucket: source.bucket.clone(),
|
||||
name: source.object.clone(),
|
||||
version_id: source.version_id,
|
||||
data_dir: Some(source.data_dir),
|
||||
mod_time: Some(
|
||||
time::OffsetDateTime::from_unix_timestamp_nanos(i128::from(source.mod_time_unix_nanos))
|
||||
.expect("source timestamp should be valid"),
|
||||
),
|
||||
size: source.size,
|
||||
etag: Some(source.etag.clone()),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(local_object_matches_transition_source(&exact, &source));
|
||||
|
||||
let mut changed = exact.clone();
|
||||
changed.version_id = Some(Uuid::new_v4());
|
||||
assert!(!local_object_matches_transition_source(&changed, &source));
|
||||
changed = exact.clone();
|
||||
changed.data_dir = Some(Uuid::new_v4());
|
||||
assert!(!local_object_matches_transition_source(&changed, &source));
|
||||
changed = exact.clone();
|
||||
changed.mod_time = changed.mod_time.map(|value| value + Duration::from_nanos(1));
|
||||
assert!(!local_object_matches_transition_source(&changed, &source));
|
||||
changed = exact.clone();
|
||||
changed.size += 1;
|
||||
assert!(!local_object_matches_transition_source(&changed, &source));
|
||||
changed = exact;
|
||||
changed.etag = Some("different-etag".to_string());
|
||||
assert!(!local_object_matches_transition_source(&changed, &source));
|
||||
}
|
||||
|
||||
fn cleanup_proof(transaction: &TransitionTransaction, decision: TransitionCleanupDecision) -> TransitionCleanupProof {
|
||||
TransitionCleanupProof {
|
||||
transaction_id: transaction.transaction_id,
|
||||
@@ -2384,6 +2629,57 @@ mod tests {
|
||||
assert_eq!(transaction.state, TransitionTransactionState::Committed);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compact_state_sequence_is_distinguishable_and_keeps_legacy_edges_strict() {
|
||||
let init = TransitionTransactionInit {
|
||||
deployment_id: Uuid::new_v4(),
|
||||
transaction_id: Uuid::new_v4(),
|
||||
owner_epoch: Uuid::new_v4(),
|
||||
write_id: Uuid::new_v4(),
|
||||
source: source_identity(TransitionSourceVersionMode::Versioned),
|
||||
tier_name: "warm-tier".to_string(),
|
||||
backend_fingerprint: BACKEND_FINGERPRINT,
|
||||
not_after_unix_nanos: 1_780_000_000_000_000_000,
|
||||
};
|
||||
let mut compact = TransitionTransaction::new_compact(init).expect("compact transaction should be created");
|
||||
assert_eq!(compact.state, TransitionTransactionState::UploadOutcomeUnknown);
|
||||
assert_eq!(compact.revision, 1);
|
||||
let remote_version = TransitionRemoteVersion::versioned(Uuid::new_v4().to_string());
|
||||
let fence = compact
|
||||
.advance(
|
||||
compact.fence(),
|
||||
TransitionTransactionState::LocalCommitStarted,
|
||||
Some(remote_version.clone()),
|
||||
)
|
||||
.expect("compact upload should persist its exact candidate at the local commit fence");
|
||||
assert_eq!(fence.revision, 2);
|
||||
assert_eq!(compact.remote_version, remote_version);
|
||||
assert_eq!(compact.state, TransitionTransactionState::LocalCommitStarted);
|
||||
let encoded = compact
|
||||
.encode()
|
||||
.expect("compact transaction should encode as v1-compatible bytes");
|
||||
assert_eq!(
|
||||
TransitionTransaction::decode(compact.transaction_id, &encoded).expect("compact transaction should decode"),
|
||||
compact
|
||||
);
|
||||
|
||||
let mut legacy_unknown = new_transaction();
|
||||
legacy_unknown
|
||||
.advance(legacy_unknown.fence(), TransitionTransactionState::UploadOutcomeUnknown, None)
|
||||
.expect("legacy transaction should persist its pre-upload fence");
|
||||
assert!(matches!(
|
||||
legacy_unknown.advance(
|
||||
legacy_unknown.fence(),
|
||||
TransitionTransactionState::LocalCommitStarted,
|
||||
Some(TransitionRemoteVersion::unversioned()),
|
||||
),
|
||||
Err(TransitionTransactionError::InvalidStateChange {
|
||||
from: TransitionTransactionState::UploadOutcomeUnknown,
|
||||
to: TransitionTransactionState::LocalCommitStarted,
|
||||
})
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cleanup_pending_requires_exact_proof_and_state_specific_decision() {
|
||||
let mut transaction = new_transaction();
|
||||
|
||||
@@ -54,6 +54,9 @@ pub type BucketConfigPublishHook = Box<dyn Fn(&str, &str, Option<(&[u8], OffsetD
|
||||
pub static BUCKET_CONFIG_PUBLISH_HOOK: std::sync::OnceLock<BucketConfigPublishHook> = std::sync::OnceLock::new();
|
||||
|
||||
const BUCKET_METADATA_REFRESH_INTERVAL: Duration = Duration::from_secs(15 * 60);
|
||||
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
||||
const LOG_SUBSYSTEM_BUCKET_METADATA: &str = "bucket_metadata";
|
||||
const EVENT_BUCKET_METADATA_LOAD_FAILED: &str = "bucket_metadata_load_failed";
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
struct ConfigWriteLockProbeState {
|
||||
@@ -1614,13 +1617,20 @@ impl BucketMetadataSys {
|
||||
|
||||
let results = join_all(futures).await;
|
||||
|
||||
for (idx, res) in results.into_iter().enumerate() {
|
||||
for (bucket, res) in buckets.iter().zip(results) {
|
||||
match res {
|
||||
Ok(()) => {}
|
||||
Err(e) => {
|
||||
error!("Unable to load bucket metadata, will be retried: {:?}", e);
|
||||
if let Some(bucket) = buckets.get(idx) {
|
||||
failed_buckets.insert(bucket.clone());
|
||||
if failed_buckets.insert(bucket.clone()) {
|
||||
error!(
|
||||
event = EVENT_BUCKET_METADATA_LOAD_FAILED,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_BUCKET_METADATA,
|
||||
result = "retry_pending",
|
||||
bucket = %bucket,
|
||||
error_code = ?e.code(),
|
||||
"Unable to load bucket metadata; retry scheduled"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1647,12 +1657,19 @@ impl BucketMetadataSys {
|
||||
});
|
||||
}
|
||||
let results = join_all(futures).await;
|
||||
for (idx, result) in results.into_iter().enumerate() {
|
||||
if let Err(err) = result {
|
||||
error!("Unable to load bucket metadata, will be retried: {:?}", err);
|
||||
if let Some(bucket) = buckets.get(idx) {
|
||||
failed_buckets.insert(bucket.clone());
|
||||
}
|
||||
for (bucket, result) in buckets.iter().zip(results) {
|
||||
if let Err(err) = result
|
||||
&& failed_buckets.insert(bucket.clone())
|
||||
{
|
||||
error!(
|
||||
event = EVENT_BUCKET_METADATA_LOAD_FAILED,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_BUCKET_METADATA,
|
||||
result = "retry_pending",
|
||||
bucket = %bucket,
|
||||
error_code = ?err.code(),
|
||||
"Unable to load bucket metadata; retry scheduled"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -66,6 +66,7 @@ pub(crate) use replication_lifecycle_bridge::ReplicationLifecycleBridge;
|
||||
pub(crate) use replication_migration_bridge::ReplicationMigrationBridge;
|
||||
pub use replication_object_bridge::ReplicationObjectBridge;
|
||||
pub use replication_object_config::{DeleteReplicationConfigSnapshot, ReplicationConfig};
|
||||
pub(crate) use replication_object_decision_boundary::replication_etags_match;
|
||||
pub use replication_object_decision_boundary::{
|
||||
MustReplicateOptions, ReplicationDeleteScheduleInput, ReplicationDeleteStateSource, delete_replication_state_from_config,
|
||||
delete_replication_version_id, should_schedule_delete_replication, should_use_existing_delete_replication_info,
|
||||
@@ -88,5 +89,6 @@ pub use replication_state::{ReplicationStats, RuntimeReplicationTargetBacklog};
|
||||
pub use replication_stats_boundary::{BucketReplicationStat, BucketReplicationStats, BucketStats, InQueueMetric, XferStats};
|
||||
pub use replication_storage_boundary::{ReplicationObjectIO, ReplicationStorage};
|
||||
pub use replication_target_boundary::SsecPassthroughCapability;
|
||||
pub use replication_target_boundary::VersionIdentityCapability;
|
||||
pub use replication_target_boundary::{ObjectLockIntegrity, object_lock_put_integrity};
|
||||
pub(crate) use replication_target_config_bridge::ReplicationTargetConfigBridge;
|
||||
|
||||
@@ -12,6 +12,8 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) use rustfs_filemeta::ObjectPartInfo;
|
||||
pub use rustfs_replication::{MrfOpKind, MrfReplicateEntry};
|
||||
pub(crate) use rustfs_replication::{
|
||||
REPLICATE_EXISTING, REPLICATE_HEAL_DELETE, ReplicateTargetDecision, ReplicatedInfos, ReplicatedTargetInfo, ReplicationAction,
|
||||
|
||||
@@ -12,6 +12,8 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) use rustfs_replication::ReplicationMultipartPlanError;
|
||||
pub use rustfs_replication::{
|
||||
MustReplicateOptions, ReplicationDeleteScheduleInput, ReplicationDeleteStateSource, delete_replication_state_from_config,
|
||||
delete_replication_version_id, should_schedule_delete_replication, should_use_existing_delete_replication_info,
|
||||
|
||||
@@ -75,6 +75,7 @@ use tracing::{debug, info, instrument, warn};
|
||||
const EVENT_REPLICATION_WORKER_RESIZE_SKIPPED: &str = "replication_worker_resize_skipped";
|
||||
const EVENT_REPLICATION_WORKER_RESIZED: &str = "replication_worker_resized";
|
||||
const EVENT_REPLICATION_BACKPRESSURE: &str = "replication_backpressure";
|
||||
const EVENT_REPLICATION_IN_FLIGHT_SKIPPED: &str = "replication_in_flight_skipped";
|
||||
const EVENT_REPLICATION_RESYNC_LOAD_SKIPPED: &str = "replication_resync_load_skipped";
|
||||
const EVENT_REPLICATION_RESYNC_RECOVERED: &str = "replication_resync_recovered";
|
||||
const EVENT_REPLICATION_MRF_QUEUE_UNAVAILABLE: &str = "replication_mrf_queue_unavailable";
|
||||
@@ -1089,6 +1090,9 @@ pub struct ReplicationPool<S: ReplicationStorage> {
|
||||
workers: RwLock<Vec<Sender<ReplicationOperation>>>,
|
||||
lrg_workers: RwLock<Vec<Sender<ReplicationOperation>>>,
|
||||
|
||||
/// Object versions queued or being replicated right now (backlog#2362).
|
||||
in_flight: Arc<ReplicationInFlight>,
|
||||
|
||||
// MRF (Most Recent Failures) channels
|
||||
mrf_replica_tx: Sender<ReplicationOperation>,
|
||||
// Shared among N MRF workers; Arc allows spawning more than one worker.
|
||||
@@ -1147,6 +1151,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
storage,
|
||||
workers: RwLock::new(Vec::new()),
|
||||
lrg_workers: RwLock::new(Vec::new()),
|
||||
in_flight: Arc::new(ReplicationInFlight::default()),
|
||||
mrf_replica_tx,
|
||||
mrf_replica_rx: Arc::new(Mutex::new(mrf_replica_rx)),
|
||||
mrf_save_tx,
|
||||
@@ -1202,12 +1207,13 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
let active_counter = self.active_lrg_workers.clone();
|
||||
let storage = self.storage.clone();
|
||||
let stats = self.stats.clone();
|
||||
let in_flight = self.in_flight.clone();
|
||||
|
||||
let handle = tokio::spawn(async move {
|
||||
let mut rx = rx;
|
||||
while let Some(operation) = rx.recv().await {
|
||||
let _active = ActiveWorkerGuard::new(active_counter.clone());
|
||||
process_replication_operation(operation, stats.clone(), storage.clone()).await;
|
||||
process_replication_operation(operation, stats.clone(), storage.clone(), in_flight.clone()).await;
|
||||
}
|
||||
});
|
||||
|
||||
@@ -1261,12 +1267,13 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
let active_counter = self.active_workers.clone();
|
||||
let stats = self.stats.clone();
|
||||
let storage = self.storage.clone();
|
||||
let in_flight = self.in_flight.clone();
|
||||
|
||||
let handle = tokio::spawn(async move {
|
||||
let mut rx = rx;
|
||||
while let Some(operation) = rx.recv().await {
|
||||
let _active = ActiveWorkerGuard::new(active_counter.clone());
|
||||
process_replication_operation(operation, stats.clone(), storage.clone()).await;
|
||||
process_replication_operation(operation, stats.clone(), storage.clone(), in_flight.clone()).await;
|
||||
}
|
||||
});
|
||||
|
||||
@@ -1305,6 +1312,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
let active_counter = self.active_mrf_workers.clone();
|
||||
let stats = self.stats.clone();
|
||||
let storage = self.storage.clone();
|
||||
let in_flight = self.in_flight.clone();
|
||||
let mrf_rx = Arc::clone(&self.mrf_replica_rx);
|
||||
|
||||
let handle = tokio::spawn(async move {
|
||||
@@ -1324,7 +1332,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
let Some(operation) = operation else { break };
|
||||
|
||||
let _active = ActiveWorkerGuard::new(active_counter.clone());
|
||||
process_replication_operation(operation, stats.clone(), storage.clone()).await;
|
||||
process_replication_operation(operation, stats.clone(), storage.clone(), in_flight.clone()).await;
|
||||
}
|
||||
});
|
||||
self.task_handles.lock().await.push(handle);
|
||||
@@ -1454,6 +1462,24 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
|
||||
/// Queues a replica task
|
||||
pub async fn queue_replica_task(&self, ri: ReplicateObjectInfo) -> ReplicationQueueAdmission {
|
||||
// A version that is already queued or being uploaded is not driven a
|
||||
// second time: the scanner heal pass sees it as PENDING until the
|
||||
// first upload lands and would otherwise re-queue it every cycle
|
||||
// (backlog#2362). The key is released when the worker finishes, or
|
||||
// below when no worker accepts the task.
|
||||
if !self.in_flight.try_begin(&ri) {
|
||||
debug!(
|
||||
event = EVENT_REPLICATION_IN_FLIGHT_SKIPPED,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_REPLICATION,
|
||||
bucket = %ri.bucket,
|
||||
object = %ri.name,
|
||||
version_id = ?ri.version_id,
|
||||
op_type = ?ri.op_type,
|
||||
"Replication task already in flight; not queued again"
|
||||
);
|
||||
return ReplicationQueueAdmission::Skipped;
|
||||
}
|
||||
let target_arns = ri.dsc.replicate_target_arns();
|
||||
// If object is large, queue it to a static set of large workers
|
||||
if should_queue_large_object(ri.size) {
|
||||
@@ -1484,7 +1510,9 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
let resize = large_worker_backpressure_resize(existing, self.active_lrg_workers(), max_l_workers);
|
||||
drop(lrg_workers);
|
||||
|
||||
// Queue to MRF if worker is busy.
|
||||
// Queue to MRF if worker is busy. The MRF replay re-enters
|
||||
// this function, so the version is no longer in flight.
|
||||
self.in_flight.finish(&ri);
|
||||
let admission = self.queue_mrf_save_admission(ri.to_mrf_entry(), "large_object").await;
|
||||
|
||||
if let Some(resize) = resize {
|
||||
@@ -1493,6 +1521,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
return admission;
|
||||
}
|
||||
}
|
||||
self.in_flight.finish(&ri);
|
||||
return ReplicationQueueAdmission::Missed;
|
||||
}
|
||||
|
||||
@@ -1501,6 +1530,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
let ch = self.worker_queue_channel(&ri.op_type, &ri.bucket, &ri.name, ri.size).await;
|
||||
|
||||
let Some(channel) = ch else {
|
||||
self.in_flight.finish(&ri);
|
||||
return ReplicationQueueAdmission::Missed;
|
||||
};
|
||||
|
||||
@@ -1512,7 +1542,9 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
self.stats.dec_q(&ri.bucket, ri.size, ri.delete_marker, ri.op_type);
|
||||
self.stats.dec_target_q(&ri.bucket, &target_arns, ri.size);
|
||||
|
||||
// Queue to MRF if all workers are busy.
|
||||
// Queue to MRF if all workers are busy. The MRF replay re-enters this
|
||||
// function, so the version is no longer in flight.
|
||||
self.in_flight.finish(&ri);
|
||||
let admission = self.queue_mrf_save_admission(ri.to_mrf_entry(), "object").await;
|
||||
|
||||
// Try to scale up workers based on priority
|
||||
@@ -1811,7 +1843,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
) {
|
||||
while let Some(operation) = rx.recv().await {
|
||||
let _active = ActiveWorkerGuard::new(active_counter.clone());
|
||||
process_replication_operation(operation, stats.clone(), self.storage.clone()).await;
|
||||
process_replication_operation(operation, stats.clone(), self.storage.clone(), self.in_flight.clone()).await;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1829,7 +1861,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
) {
|
||||
while let Some(operation) = rx.recv().await {
|
||||
let _active = ActiveWorkerGuard::new(active_counter.clone());
|
||||
process_replication_operation(operation, stats.clone(), storage.clone()).await;
|
||||
process_replication_operation(operation, stats.clone(), storage.clone(), self.in_flight.clone()).await;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1846,7 +1878,7 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
) {
|
||||
while let Some(operation) = rx.recv().await {
|
||||
let _active = ActiveWorkerGuard::new(active_counter.clone());
|
||||
process_replication_operation(operation, stats.clone(), self.storage.clone()).await;
|
||||
process_replication_operation(operation, stats.clone(), self.storage.clone(), self.in_flight.clone()).await;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2281,6 +2313,64 @@ impl<S: ReplicationStorage> ReplicationPool<S> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Object versions currently queued or being uploaded, keyed by bucket,
|
||||
/// object name and version. `queue_replica_task` admits a version only once
|
||||
/// while it is in flight; the scanner heal pass and MRF replays that arrive
|
||||
/// in the meantime are `Skipped` instead of driving a second complete upload
|
||||
/// (backlog#2362). Entries are removed when the worker finishes the task or
|
||||
/// when no worker accepted it.
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct ReplicationInFlight {
|
||||
keys: std::sync::Mutex<std::collections::HashSet<(String, String, Option<uuid::Uuid>)>>,
|
||||
}
|
||||
|
||||
impl ReplicationInFlight {
|
||||
fn lock(&self) -> std::sync::MutexGuard<'_, std::collections::HashSet<(String, String, Option<uuid::Uuid>)>> {
|
||||
self.keys.lock().unwrap_or_else(|poisoned| poisoned.into_inner())
|
||||
}
|
||||
|
||||
/// Claim `ri`; `false` when the same version is already in flight.
|
||||
fn try_begin(&self, ri: &ReplicateObjectInfo) -> bool {
|
||||
self.lock().insert((ri.bucket.clone(), ri.name.clone(), ri.version_id))
|
||||
}
|
||||
|
||||
fn finish(&self, ri: &ReplicateObjectInfo) {
|
||||
self.lock().remove(&(ri.bucket.clone(), ri.name.clone(), ri.version_id));
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
fn len(&self) -> usize {
|
||||
self.lock().len()
|
||||
}
|
||||
}
|
||||
|
||||
/// Releases the in-flight claim when the worker is done with the task,
|
||||
/// including when replication panics.
|
||||
struct ReplicationInFlightGuard {
|
||||
in_flight: Arc<ReplicationInFlight>,
|
||||
key: ReplicateObjectInfo,
|
||||
}
|
||||
|
||||
impl ReplicationInFlightGuard {
|
||||
fn new(in_flight: Arc<ReplicationInFlight>, ri: &ReplicateObjectInfo) -> Self {
|
||||
Self {
|
||||
in_flight,
|
||||
key: ReplicateObjectInfo {
|
||||
bucket: ri.bucket.clone(),
|
||||
name: ri.name.clone(),
|
||||
version_id: ri.version_id,
|
||||
..Default::default()
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for ReplicationInFlightGuard {
|
||||
fn drop(&mut self) {
|
||||
self.in_flight.finish(&self.key);
|
||||
}
|
||||
}
|
||||
|
||||
struct ActiveWorkerGuard {
|
||||
counter: Arc<AtomicI32>,
|
||||
}
|
||||
@@ -2342,10 +2432,12 @@ async fn process_replication_operation<S: ReplicationStorage>(
|
||||
operation: ReplicationOperation,
|
||||
stats: Arc<ReplicationStats>,
|
||||
storage: Arc<S>,
|
||||
in_flight: Arc<ReplicationInFlight>,
|
||||
) {
|
||||
match operation {
|
||||
ReplicationOperation::Object(obj_info) => {
|
||||
let _backlog = ReplicationBacklogGuard::for_object(stats, obj_info.as_ref());
|
||||
let _in_flight = ReplicationInFlightGuard::new(in_flight, obj_info.as_ref());
|
||||
replicate_object(*obj_info, storage).await;
|
||||
}
|
||||
ReplicationOperation::Delete(del_info) => {
|
||||
@@ -3079,7 +3171,11 @@ pub async fn queue_replication_heal(bucket: &str, oi: ObjectInfo, retry_count: u
|
||||
}
|
||||
|
||||
let rcfg = match ReplicationMetadataStore::optional_replication_config(bucket).await {
|
||||
Ok(Some(config)) => config,
|
||||
Ok(Some(config)) => Some(config),
|
||||
// A bucket without a configuration still owes its pending purges an
|
||||
// answer: the delete worker finishes them locally as abandoned, which
|
||||
// is what makes the bucket deletable again (rustfs/backlog#2340).
|
||||
Ok(None) if owes_version_purge(&oi) => None,
|
||||
Ok(None) => return ReplicationQueueAdmission::Skipped,
|
||||
Err(err) => {
|
||||
debug!(
|
||||
@@ -3129,7 +3225,7 @@ pub async fn queue_replication_heal(bucket: &str, oi: ObjectInfo, retry_count: u
|
||||
}
|
||||
};
|
||||
|
||||
let rcfg_wrapper = ReplicationConfig::new(Some(rcfg), tgts);
|
||||
let rcfg_wrapper = ReplicationConfig::new(rcfg, tgts);
|
||||
queue_replication_heal_internal(bucket, oi, rcfg_wrapper, retry_count)
|
||||
.await
|
||||
.admission
|
||||
@@ -3157,6 +3253,17 @@ pub async fn queue_replication_metadata(bucket: &str, oi: ObjectInfo, retry_coun
|
||||
}
|
||||
}
|
||||
|
||||
/// A version purge the persisted state still owes to named targets. Without
|
||||
/// the target list nothing can be settled, so such a version keeps the
|
||||
/// ordinary "no configuration, nothing to heal" skip.
|
||||
fn owes_version_purge(oi: &ObjectInfo) -> bool {
|
||||
!oi.version_purge_status.is_empty()
|
||||
&& oi
|
||||
.version_purge_status_internal
|
||||
.as_deref()
|
||||
.is_some_and(|statuses| !statuses.trim().is_empty())
|
||||
}
|
||||
|
||||
/// queue_replication_heal_internal enqueues objects that failed replication OR eligible for resyncing through
|
||||
/// an ongoing resync operation or via existing objects replication configuration setting.
|
||||
pub(crate) async fn queue_replication_heal_internal(
|
||||
@@ -3175,7 +3282,11 @@ pub(crate) async fn queue_replication_heal_internal(
|
||||
};
|
||||
}
|
||||
|
||||
if rcfg.config.is_none() || rcfg.remotes.is_none() {
|
||||
// Without a configuration or targets there is nothing to replicate —
|
||||
// except a version purge the bucket still owes: its stored decision names
|
||||
// the targets, and the delete worker settles the ones no longer
|
||||
// configured as abandoned (rustfs/backlog#2340).
|
||||
if (rcfg.config.is_none() || rcfg.remotes.is_none()) && !owes_version_purge(&oi) {
|
||||
return ReplicationHealQueueResult {
|
||||
object_info: roi,
|
||||
admission: ReplicationQueueAdmission::Skipped,
|
||||
@@ -3220,12 +3331,15 @@ pub(crate) async fn queue_replication_heal_internal(
|
||||
}
|
||||
ReplicationHealQueueAction::QueueDelete(dv) => {
|
||||
// A purge the peer denied under object lock cannot succeed until
|
||||
// the lock lapses (#6850); requeuing it every heal cycle only
|
||||
// the lock lapses (#6850), and one whose replica cannot be told
|
||||
// apart on a target that mints its own version ids cannot
|
||||
// succeed until the ledger or an operator resolves it
|
||||
// (rustfs/backlog#2340); requeuing either every heal cycle only
|
||||
// burns bandwidth and failure counters. The backoff expires on
|
||||
// its own, so the purge is probed again — and converges — once
|
||||
// the retention window has a chance of being over.
|
||||
// the condition has a chance of being over.
|
||||
if super::replication_object_decision_boundary::is_version_delete_replication(&dv.delete_object)
|
||||
&& super::replication_resyncer::object_lock_denied_purge_backoff_active(&dv)
|
||||
&& super::replication_resyncer::purge_backoff_active(&dv)
|
||||
{
|
||||
return ReplicationHealQueueResult {
|
||||
object_info: roi,
|
||||
@@ -3707,6 +3821,7 @@ mod tests {
|
||||
stats: Arc::new(ReplicationStats::new()),
|
||||
workers: RwLock::new(Vec::new()),
|
||||
lrg_workers: RwLock::new(Vec::new()),
|
||||
in_flight: Arc::new(ReplicationInFlight::default()),
|
||||
mrf_replica_tx,
|
||||
mrf_replica_rx: Arc::new(Mutex::new(mrf_replica_rx)),
|
||||
mrf_save_tx,
|
||||
@@ -3773,6 +3888,90 @@ mod tests {
|
||||
assert_eq!(current_queue(&pool, "admission-bucket").await, (1, 4096));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn queue_replica_task_admits_a_version_once_while_it_is_in_flight() {
|
||||
let pool = new_test_replication_pool(Arc::new(LoadResyncNodeStore::new("node-a", empty_resync_shared_state()))).await;
|
||||
let (tx, _rx) = mpsc::channel(4);
|
||||
pool.workers.write().await.push(tx);
|
||||
let ri = ReplicateObjectInfo {
|
||||
bucket: "in-flight-bucket".to_string(),
|
||||
name: "object".to_string(),
|
||||
version_id: Some(uuid::Uuid::new_v4()),
|
||||
size: 4096,
|
||||
op_type: ReplicationType::Object,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
assert_eq!(pool.queue_replica_task(ri.clone()).await, ReplicationQueueAdmission::Queued);
|
||||
// backlog#2362: the scanner heal pass sees the version as PENDING
|
||||
// until the worker lands it; a second request must not drive it again.
|
||||
assert_eq!(pool.queue_replica_task(ri.clone()).await, ReplicationQueueAdmission::Skipped);
|
||||
assert_eq!(current_queue(&pool, "in-flight-bucket").await, (1, 4096));
|
||||
|
||||
// Another version of the same key is independent work.
|
||||
let newer = ReplicateObjectInfo {
|
||||
version_id: Some(uuid::Uuid::new_v4()),
|
||||
..ri.clone()
|
||||
};
|
||||
assert_eq!(pool.queue_replica_task(newer).await, ReplicationQueueAdmission::Queued);
|
||||
assert_eq!(pool.in_flight.len(), 2);
|
||||
|
||||
// Once the worker finishes, the same version may be queued again
|
||||
// (for example after a FAILED status).
|
||||
pool.in_flight.finish(&ri);
|
||||
assert_eq!(pool.queue_replica_task(ri).await, ReplicationQueueAdmission::Queued);
|
||||
assert_eq!(current_queue(&pool, "in-flight-bucket").await, (3, 3 * 4096));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn queue_replica_task_releases_the_version_when_no_worker_accepts_it() {
|
||||
let pool = new_test_replication_pool(Arc::new(LoadResyncNodeStore::new("node-a", empty_resync_shared_state()))).await;
|
||||
let ri = ReplicateObjectInfo {
|
||||
bucket: "no-worker-bucket".to_string(),
|
||||
name: "object".to_string(),
|
||||
version_id: Some(uuid::Uuid::new_v4()),
|
||||
size: 4096,
|
||||
op_type: ReplicationType::Object,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
// No worker channel: the task is missed and must not stay claimed.
|
||||
assert_eq!(pool.queue_replica_task(ri.clone()).await, ReplicationQueueAdmission::Missed);
|
||||
assert_eq!(pool.in_flight.len(), 0);
|
||||
assert_eq!(pool.queue_replica_task(ri.clone()).await, ReplicationQueueAdmission::Missed);
|
||||
|
||||
// A full worker channel hands the task to the MRF save path; the MRF
|
||||
// replay re-enters the queue, so the claim is released here too.
|
||||
let (tx, _rx) = mpsc::channel(1);
|
||||
pool.workers.write().await.push(tx);
|
||||
assert_eq!(pool.queue_replica_task(ri.clone()).await, ReplicationQueueAdmission::Queued);
|
||||
let overflow = ReplicateObjectInfo {
|
||||
version_id: Some(uuid::Uuid::new_v4()),
|
||||
..ri
|
||||
};
|
||||
assert_eq!(pool.queue_replica_task(overflow).await, ReplicationQueueAdmission::Queued);
|
||||
assert_eq!(pool.in_flight.len(), 1, "only the version held by the worker channel stays in flight");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn in_flight_guard_releases_the_version_on_drop() {
|
||||
let in_flight = Arc::new(ReplicationInFlight::default());
|
||||
let ri = ReplicateObjectInfo {
|
||||
bucket: "guard-bucket".to_string(),
|
||||
name: "object".to_string(),
|
||||
version_id: Some(uuid::Uuid::new_v4()),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(in_flight.try_begin(&ri));
|
||||
assert!(!in_flight.try_begin(&ri));
|
||||
{
|
||||
let _guard = ReplicationInFlightGuard::new(in_flight.clone(), &ri);
|
||||
assert_eq!(in_flight.len(), 1);
|
||||
}
|
||||
assert_eq!(in_flight.len(), 0);
|
||||
assert!(in_flight.try_begin(&ri));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn regular_worker_admission_counts_target_backlog_before_receive() {
|
||||
let pool = new_test_replication_pool(Arc::new(LoadResyncNodeStore::new("node-a", empty_resync_shared_state()))).await;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -38,7 +38,7 @@ use time::format_description::well_known::Rfc3339;
|
||||
|
||||
pub(crate) use crate::bucket::bucket_target_sys::{
|
||||
AdvancedPutOptions, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, RemotePutObjectResponse, RemoveObjectOptions,
|
||||
S3ClientError, TargetClient, resolve_read_api_version_id,
|
||||
ReplicaLocation, S3ClientError, TargetClient, resolve_read_api_version_id,
|
||||
};
|
||||
#[cfg(test)]
|
||||
pub(crate) use crate::bucket::target::BucketTarget;
|
||||
@@ -48,6 +48,7 @@ pub use rustfs_replication::{ObjectLockIntegrity, object_lock_put_integrity};
|
||||
pub(crate) use rustfs_replication::{
|
||||
SsecPassthroughGate, is_replication_target_offline_error, ssec_passthrough_gate, version_identity_drifted,
|
||||
};
|
||||
pub use rustfs_replication::{VersionIdentityCapability, version_identity_capability_from_put};
|
||||
|
||||
use super::replication_config_store::ReplicationConfigStore;
|
||||
use super::replication_error_boundary::{Error, Result};
|
||||
@@ -192,6 +193,14 @@ impl ReplicationTargetStore {
|
||||
.await
|
||||
}
|
||||
|
||||
pub(crate) fn version_identity_capability(arn: &str) -> VersionIdentityCapability {
|
||||
BucketTargetSys::get().version_identity_capability(arn)
|
||||
}
|
||||
|
||||
pub(crate) fn record_version_identity_capability(arn: &str, capability: VersionIdentityCapability) {
|
||||
BucketTargetSys::get().record_version_identity_capability(arn, capability)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) async fn register_test_target(target_client: &Arc<TargetClient>) {
|
||||
BucketTargetSys::get().arn_remotes_map.write().await.insert(
|
||||
@@ -238,6 +247,23 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
|
||||
meta.insert(key.to_string(), value.to_string());
|
||||
}
|
||||
|
||||
// A compressed SSE-C object passes through as its stored bytes. The target
|
||||
// cannot infer the compression layout from ciphertext, so the scheme and
|
||||
// the plaintext size travel as transport headers; each UploadPart carries
|
||||
// its own plaintext length (backlog#2363).
|
||||
if is_ssec && let Some(scheme) = get_str(&object_info.user_defined, rustfs_utils::http::SUFFIX_COMPRESSION) {
|
||||
insert_header_map(&mut meta, rustfs_utils::http::SUFFIX_REPLICATION_COMPRESSION, scheme);
|
||||
if let Ok(actual_size) = object_info.get_actual_size()
|
||||
&& actual_size >= 0
|
||||
{
|
||||
insert_header_map(
|
||||
&mut meta,
|
||||
rustfs_utils::http::SUFFIX_REPLICATION_COMPRESSION_ACTUAL_SIZE,
|
||||
actual_size.to_string(),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Managed SSE replicates as plaintext (the replication reader decrypts via
|
||||
// the object-encryption resolver) and re-encrypts on the target with the
|
||||
// target's own KMS. Send only the encryption intent — never the source
|
||||
@@ -248,7 +274,16 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
|
||||
meta.insert(AMZ_SERVER_SIDE_ENCRYPTION.to_string(), "aws:kms".to_string());
|
||||
}
|
||||
|
||||
let mut is_multipart = object_info.is_multipart();
|
||||
// Older transformed objects can have physical parts without logical part
|
||||
// lengths. Keep their existing whole-object transport: physical sizes are
|
||||
// not plaintext boundaries for a multipart replication read.
|
||||
let legacy_single_put = object_info.etag.as_deref().is_none_or(|etag| etag.len() == 32);
|
||||
let base_is_multipart = object_info.is_multipart()
|
||||
&& !(legacy_single_put
|
||||
&& object_info.parts.len() > 1
|
||||
&& (object_info.is_compressed() || object_info.is_encrypted())
|
||||
&& object_info.parts.iter().any(|part| part.actual_size <= 0));
|
||||
let mut is_multipart = base_is_multipart;
|
||||
|
||||
if let Some(checksum_data) = &object_info.checksum
|
||||
&& !checksum_data.is_empty()
|
||||
@@ -259,8 +294,8 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
|
||||
} else if object_info.is_encrypted() {
|
||||
// Encrypted checksums cannot be exposed as plaintext headers, and
|
||||
// decrypt_checksums reports is_multipart=false for them (a value
|
||||
// the response path relies on). Keep the object's own multipart
|
||||
// flag so encrypted objects stay on the multipart route.
|
||||
// the response path relies on). Keep the transport selected from
|
||||
// the object's layout and readable part boundaries.
|
||||
} else {
|
||||
let (checksum_meta, checksum_record_is_multipart) = object_info.decrypt_checksums(0, &HeaderMap::new())?;
|
||||
// The checksum record describes how the *checksum* is composed,
|
||||
@@ -268,23 +303,37 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
|
||||
// MULTIPART flag even on a multipart upload, so trusting it here
|
||||
// routed a 768-part object through a single PutObject and the
|
||||
// target rejected the 6 GiB body with EntityTooLarge
|
||||
// (rustfs#6825). The object's own shape is the authority: the
|
||||
// (rustfs#6825). The usable part layout is the authority: the
|
||||
// record may only add multipart-ness, never take it away.
|
||||
is_multipart = object_info.is_multipart() || checksum_record_is_multipart;
|
||||
is_multipart = base_is_multipart || checksum_record_is_multipart;
|
||||
|
||||
for (key, value) in checksum_meta.iter() {
|
||||
if key != AMZ_CHECKSUM_TYPE {
|
||||
meta.insert(key.clone(), value.clone());
|
||||
}
|
||||
}
|
||||
|
||||
if !object_info.is_multipart()
|
||||
if !base_is_multipart
|
||||
&& checksum_meta
|
||||
.get(AMZ_CHECKSUM_TYPE)
|
||||
.is_some_and(|value| value == AMZ_CHECKSUM_TYPE_FULL_OBJECT)
|
||||
{
|
||||
is_multipart = false;
|
||||
}
|
||||
|
||||
// The record keys each checksum by algorithm name ("CRC32"); the
|
||||
// target only reads `x-amz-checksum-<algorithm>`. Inserting the bare
|
||||
// name here made `PutObjectOptions::header()` send it as user
|
||||
// metadata (`x-amz-meta-crc32`), so no replica ever carried the
|
||||
// source checksum (rustfs/backlog#2340). The object-level record
|
||||
// describes one PUT body: a multipart replica is rebuilt part by
|
||||
// part, and its CreateMultipartUpload must not announce a checksum
|
||||
// the parts do not carry, so the record is forwarded on the
|
||||
// single-PUT route only (MinIO `getCRCMeta` parity).
|
||||
if !is_multipart {
|
||||
for (key, value) in checksum_meta.iter() {
|
||||
if key == AMZ_CHECKSUM_TYPE {
|
||||
continue;
|
||||
}
|
||||
if let Some(header) = rustfs_rio::ChecksumType::from_string(key).key() {
|
||||
meta.insert(header.to_string(), value.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -516,6 +565,7 @@ fn is_standard_header(key: &str) -> bool {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::super::replication_filemeta_boundary::ObjectPartInfo;
|
||||
use super::*;
|
||||
use aws_smithy_types::DateTime;
|
||||
use rustfs_replication::content_matches_by_etag;
|
||||
@@ -550,6 +600,162 @@ mod tests {
|
||||
checksum.to_bytes(&combined)
|
||||
}
|
||||
|
||||
fn replication_route_metadata() -> [(&'static str, Arc<HashMap<String, String>>); 4] {
|
||||
let mut compressed = HashMap::new();
|
||||
rustfs_utils::http::insert_str(&mut compressed, rustfs_utils::http::SUFFIX_COMPRESSION, "zstd".to_string());
|
||||
[
|
||||
("plain", Arc::new(HashMap::new())),
|
||||
("compressed", Arc::new(compressed)),
|
||||
(
|
||||
"encrypted",
|
||||
Arc::new(HashMap::from([(AMZ_SERVER_SIDE_ENCRYPTION.to_string(), "AES256".to_string())])),
|
||||
),
|
||||
(
|
||||
"ssec",
|
||||
Arc::new(HashMap::from([(SSEC_ALGORITHM_HEADER.to_string(), "AES256".to_string())])),
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
fn replication_route_object(
|
||||
etag: Option<&str>,
|
||||
actual_sizes: [i64; 3],
|
||||
metadata: Arc<HashMap<String, String>>,
|
||||
) -> ObjectInfo {
|
||||
ObjectInfo {
|
||||
etag: etag.map(str::to_string),
|
||||
size: 48,
|
||||
actual_size: 12,
|
||||
user_defined: metadata,
|
||||
parts: Arc::new(
|
||||
actual_sizes
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
.map(|(index, actual_size)| ObjectPartInfo {
|
||||
number: index + 1,
|
||||
size: 16,
|
||||
actual_size,
|
||||
..Default::default()
|
||||
})
|
||||
.collect(),
|
||||
),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn compressed_ssec_objects_declare_their_compression_layout_on_the_wire() {
|
||||
use rustfs_utils::http::{
|
||||
SUFFIX_ACTUAL_SIZE, SUFFIX_COMPRESSION, SUFFIX_REPLICATION_COMPRESSION, SUFFIX_REPLICATION_COMPRESSION_ACTUAL_SIZE,
|
||||
insert_str,
|
||||
};
|
||||
|
||||
let mut ssec_compressed = HashMap::from([(SSEC_ALGORITHM_HEADER.to_string(), "AES256".to_string())]);
|
||||
insert_str(&mut ssec_compressed, SUFFIX_COMPRESSION, "klauspost/compress/s2".to_string());
|
||||
insert_str(&mut ssec_compressed, SUFFIX_ACTUAL_SIZE, "6295552".to_string());
|
||||
let object_info = ObjectInfo {
|
||||
etag: Some("0123456789abcdef0123456789abcdef-2".to_string()),
|
||||
size: 4321,
|
||||
actual_size: 6295552,
|
||||
user_defined: Arc::new(ssec_compressed),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
// SSE-C passthrough sends stored bytes: the scheme and the plaintext
|
||||
// size travel as transport headers, never as the internal key
|
||||
// (backlog#2363).
|
||||
let (options, _) = replication_put_object_options("STANDARD", &object_info).expect("ssec put options");
|
||||
assert_eq!(
|
||||
get_header_map(&options.user_metadata, SUFFIX_REPLICATION_COMPRESSION).as_deref(),
|
||||
Some("klauspost/compress/s2")
|
||||
);
|
||||
assert_eq!(
|
||||
get_header_map(&options.user_metadata, SUFFIX_REPLICATION_COMPRESSION_ACTUAL_SIZE).as_deref(),
|
||||
Some("6295552")
|
||||
);
|
||||
assert!(
|
||||
!options
|
||||
.user_metadata
|
||||
.keys()
|
||||
.any(|key| rustfs_utils::http::is_internal_key(key)),
|
||||
"internal metadata never leaves the source as plain metadata: {:?}",
|
||||
options.user_metadata
|
||||
);
|
||||
|
||||
// A compressed object that is not SSE-C is decompressed by the
|
||||
// replication reader and travels as plaintext: no layout headers.
|
||||
let mut plain_compressed = HashMap::new();
|
||||
insert_str(&mut plain_compressed, SUFFIX_COMPRESSION, "klauspost/compress/s2".to_string());
|
||||
insert_str(&mut plain_compressed, SUFFIX_ACTUAL_SIZE, "6295552".to_string());
|
||||
let plain = ObjectInfo {
|
||||
user_defined: Arc::new(plain_compressed),
|
||||
..object_info
|
||||
};
|
||||
let (options, _) = replication_put_object_options("STANDARD", &plain).expect("plain put options");
|
||||
assert!(get_header_map(&options.user_metadata, SUFFIX_REPLICATION_COMPRESSION).is_none());
|
||||
assert!(get_header_map(&options.user_metadata, SUFFIX_REPLICATION_COMPRESSION_ACTUAL_SIZE).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn legacy_transformed_single_put_parts_keep_the_previous_replication_route() {
|
||||
let [_, (_, compressed), (_, encrypted), (_, ssec)] = replication_route_metadata();
|
||||
let cases = [
|
||||
(
|
||||
"compressed middle zero",
|
||||
compressed.clone(),
|
||||
Some("0123456789abcdef0123456789abcdef"),
|
||||
[4, 0, 4],
|
||||
),
|
||||
("compressed tail unknown", compressed, None, [4, 4, -1]),
|
||||
(
|
||||
"encrypted middle unknown",
|
||||
encrypted,
|
||||
Some("gggggggggggggggggggggggggggggggg"),
|
||||
[4, -1, 4],
|
||||
),
|
||||
("ssec tail zero", ssec.clone(), None, [4, 4, 0]),
|
||||
("ssec middle unknown", ssec, Some("gggggggggggggggggggggggggggggggg"), [4, -1, 4]),
|
||||
];
|
||||
for (name, metadata, etag, actual_sizes) in cases {
|
||||
for checksum in [None, Some(full_object_multipart_checksum_record())] {
|
||||
let mut object_info = replication_route_object(etag, actual_sizes, metadata.clone());
|
||||
object_info.checksum = checksum;
|
||||
assert!(object_info.is_multipart(), "{name}: physical parts remain visible to metadata APIs");
|
||||
assert!(object_info.is_compressed() || object_info.is_encrypted());
|
||||
|
||||
let (options, is_multipart) =
|
||||
replication_put_object_options("STANDARD", &object_info).expect("legacy transformed put options");
|
||||
assert!(
|
||||
!is_multipart,
|
||||
"{name}: unknown logical part sizes must preserve the old whole-object route"
|
||||
);
|
||||
assert_eq!(options.internal.source_etag, etag.unwrap_or_default());
|
||||
if metadata.contains_key(SSEC_ALGORITHM_HEADER) {
|
||||
assert_eq!(
|
||||
get_header_map(&options.user_metadata, SUFFIX_REPLICATION_SSEC_CRC).is_some(),
|
||||
object_info.checksum.is_some(),
|
||||
"SSE-C checksums retain their raw passthrough transport"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn positive_part_sizes_and_legacy_multipart_etags_keep_the_replication_route() {
|
||||
for (name, metadata) in replication_route_metadata() {
|
||||
for (etag, actual_sizes) in [
|
||||
("0123456789abcdef0123456789abcdef", [4, 4, 4]),
|
||||
("0123456789abcdef0123456789abcdef-3", [4, 0, -1]),
|
||||
] {
|
||||
let mut object_info = replication_route_object(Some(etag), actual_sizes, metadata.clone());
|
||||
object_info.checksum = Some(full_object_multipart_checksum_record());
|
||||
let (_, is_multipart) = replication_put_object_options("STANDARD", &object_info).expect("multipart put options");
|
||||
assert!(is_multipart, "{name}/{etag}: usable sizes and old multipart ETags must retain MPU");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn multipart_object_with_full_object_checksum_keeps_the_multipart_route() {
|
||||
// rustfs#6825: a 768-part upload was replicated with a single
|
||||
@@ -582,6 +788,36 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stored_multipart_parts_keep_the_replication_route_without_a_multipart_etag() {
|
||||
for etag in [Some("0123456789abcdef0123456789abcdef"), None] {
|
||||
for checksum in [None, Some(full_object_multipart_checksum_record())] {
|
||||
let object_info = ObjectInfo {
|
||||
etag: etag.map(str::to_string),
|
||||
checksum,
|
||||
parts: Arc::new(
|
||||
(1..=2)
|
||||
.map(|number| ObjectPartInfo {
|
||||
number,
|
||||
..Default::default()
|
||||
})
|
||||
.collect(),
|
||||
),
|
||||
..Default::default()
|
||||
};
|
||||
let (options, is_multipart) =
|
||||
replication_put_object_options("STANDARD", &object_info).expect("build put options");
|
||||
|
||||
assert!(
|
||||
is_multipart,
|
||||
"stored parts must retain multipart routing: etag={etag:?}, checksum={:?}",
|
||||
object_info.checksum
|
||||
);
|
||||
assert_eq!(options.internal.source_etag, etag.unwrap_or_default());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn checksum_record_never_changes_the_transport_a_single_part_object_needs() {
|
||||
// The mirror of the rustfs#6825 guard: an object stored as one PUT
|
||||
@@ -592,6 +828,10 @@ mod tests {
|
||||
let object_info = ObjectInfo {
|
||||
etag: Some("0123456789abcdef0123456789abcdef".to_string()),
|
||||
checksum: Some(checksum.to_bytes(&[])),
|
||||
parts: Arc::new(vec![ObjectPartInfo {
|
||||
number: 1,
|
||||
..Default::default()
|
||||
}]),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
@@ -628,6 +868,19 @@ mod tests {
|
||||
let (_, is_multipart) = replication_put_object_options("STANDARD", &object_info).expect("build put options");
|
||||
|
||||
assert!(is_multipart, "a composite-checksum multipart object must stay on the multipart transport");
|
||||
|
||||
for (name, metadata) in replication_route_metadata() {
|
||||
let mut legacy = replication_route_object(Some("0123456789abcdef0123456789abcdef"), [4, 0, 4], metadata);
|
||||
legacy.checksum = Some(checksum.to_bytes(&combined));
|
||||
let (_, record_is_multipart) = legacy.decrypt_checksums(0, &HeaderMap::new()).expect("decode checksum");
|
||||
let (_, is_multipart) = replication_put_object_options("STANDARD", &legacy).expect("legacy checksum put options");
|
||||
if legacy.is_encrypted() {
|
||||
assert!(!is_multipart, "{name}: encrypted checksum records must not change the old transport");
|
||||
} else {
|
||||
assert!(record_is_multipart, "the composite checksum must carry its own multipart signal");
|
||||
assert!(is_multipart, "{name}: a composite record can still promote the legacy route to MPU");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1308,12 +1561,63 @@ mod tests {
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let (opts, _is_multipart) = replication_put_object_options("", &object_info).expect("build replication put options");
|
||||
let (opts, is_multipart) = replication_put_object_options("", &object_info).expect("build replication put options");
|
||||
|
||||
assert!(!is_multipart, "{name}: a single-part checksum record must keep the single-PUT route");
|
||||
let header = ty.key().expect("every forwarded algorithm has an x-amz-checksum header");
|
||||
assert_eq!(
|
||||
opts.user_metadata.get(name),
|
||||
opts.user_metadata.get(header),
|
||||
Some(&checksum.encoded),
|
||||
"replication must forward the {name} checksum into user_metadata identically to the classic algorithms"
|
||||
"replication must forward the {name} checksum as the {header} header"
|
||||
);
|
||||
assert!(
|
||||
!opts.user_metadata.contains_key(name),
|
||||
"{name}: the bare algorithm name would leave as x-amz-meta user metadata"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The object-level record of a multipart upload (composite or full-object)
|
||||
/// must not become a PutObject checksum header: the replica is rebuilt
|
||||
/// through CreateMultipartUpload/UploadPart, and a checksum announced there
|
||||
/// that the parts do not carry would be rejected by the target.
|
||||
#[test]
|
||||
fn replication_put_object_options_keeps_multipart_checksum_records_off_the_wire() {
|
||||
let mut composite_type = rustfs_rio::ChecksumType::from_string("crc32");
|
||||
composite_type
|
||||
.merge(rustfs_rio::ChecksumType::MULTIPART)
|
||||
.merge(rustfs_rio::ChecksumType::INCLUDES_MULTIPART);
|
||||
let mut combined = Vec::new();
|
||||
for part in [b"part-one".as_slice(), b"part-two".as_slice()] {
|
||||
let part_checksum =
|
||||
rustfs_rio::Checksum::new_from_data(rustfs_rio::ChecksumType::from_string("crc32"), part).expect("part checksum");
|
||||
combined.extend_from_slice(part_checksum.raw.as_slice());
|
||||
}
|
||||
let composite = rustfs_rio::Checksum::new_from_data(composite_type, &combined)
|
||||
.expect("composite checksum")
|
||||
.to_bytes(&combined);
|
||||
|
||||
for (label, checksum, etag) in [
|
||||
("composite", composite, "0123456789abcdef0123456789abcdef-2"),
|
||||
(
|
||||
"full-object",
|
||||
full_object_multipart_checksum_record(),
|
||||
"0123456789abcdef0123456789abcdef-3",
|
||||
),
|
||||
] {
|
||||
let object_info = ObjectInfo {
|
||||
etag: Some(etag.to_string()),
|
||||
checksum: Some(checksum),
|
||||
..Default::default()
|
||||
};
|
||||
let (opts, is_multipart) = replication_put_object_options("", &object_info).expect("build replication put options");
|
||||
assert!(is_multipart, "{label}: a multipart object must keep the multipart route");
|
||||
assert!(
|
||||
opts.user_metadata
|
||||
.keys()
|
||||
.all(|key| !key.starts_with("x-amz-checksum-") && key != "CRC32"),
|
||||
"{label}: no object-level checksum may reach the target's CreateMultipartUpload: {:?}",
|
||||
opts.user_metadata
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +48,8 @@ pub use internode_data_transport::build_internode_data_transport_from_env;
|
||||
pub(crate) use peer_rest_client::TierConfigReloadOutcome;
|
||||
pub use peer_rest_client::{
|
||||
KMS_SIGNAL_SUBSYSTEM, PEER_RESTDRY_RUN, PEER_RESTSIGNAL, PEER_RESTSUB_SYS, PeerRestClient, SERVICE_SIGNAL_REFRESH_CONFIG,
|
||||
SERVICE_SIGNAL_RELOAD_DYNAMIC, ScannerPeerActivity, ScannerPeerDirtyUsageSnapshot, ScannerPublicationLease,
|
||||
SERVICE_SIGNAL_RELOAD_DYNAMIC, ScannerDirtyUsageAcknowledgement, ScannerPeerActivity, ScannerPeerDirtyUsageBucket,
|
||||
ScannerPeerDirtyUsageSnapshot, ScannerPublicationLease, ScannerScopedDirtyUsageAckEntry,
|
||||
};
|
||||
pub(crate) use peer_s3_client::heal_bucket_local_on_disks;
|
||||
pub use peer_s3_client::{
|
||||
|
||||
@@ -49,10 +49,11 @@ use rustfs_protos::proto_gen::node_service::{
|
||||
LoadRebalanceMetaRequest, LoadServiceAccountRequest, LoadTransitionTierConfigRequest, LoadUserRequest,
|
||||
LocalStorageInfoRequest, Mss, ReloadPoolMetaRequest, ReloadSiteReplicationConfigRequest, ReplacementRecoveryStatusRequest,
|
||||
ScannerActivityRequest, ScannerActivityResponse, ScannerDirtyUsageSnapshotRequest, ScannerDirtyUsageSnapshotResponse,
|
||||
ScannerPublicationLeaseReleaseRequest, ScannerPublicationLeaseRequest, ScannerPublicationLeaseResponse, ServerInfoRequest,
|
||||
SignalServiceRequest, SignalServiceResponse, StartDecommissionRequest, StartProfilingRequest, StopRebalanceRequest,
|
||||
TierDailyStatsRequest, TierMutationAbortRequest, TierMutationCommitRequest, TierMutationControlResponse,
|
||||
TierMutationFailureClass, TierMutationPeerState, TierMutationPrepareRequest, node_service_client::NodeServiceClient,
|
||||
ScannerPublicationLeaseReleaseRequest, ScannerPublicationLeaseRequest, ScannerPublicationLeaseResponse,
|
||||
ScannerScopedDirtyUsageAckRequest, ScannerScopedDirtyUsageEntry, ServerInfoRequest, SignalServiceRequest,
|
||||
SignalServiceResponse, StartDecommissionRequest, StartProfilingRequest, StopRebalanceRequest, TierDailyStatsRequest,
|
||||
TierMutationAbortRequest, TierMutationCommitRequest, TierMutationControlResponse, TierMutationFailureClass,
|
||||
TierMutationPeerState, TierMutationPrepareRequest, node_service_client::NodeServiceClient,
|
||||
tier_mutation_control_service_client::TierMutationControlServiceClient,
|
||||
};
|
||||
pub use rustfs_protos::{PEER_RESTDRY_RUN, PEER_RESTSIGNAL, PEER_RESTSUB_SYS};
|
||||
@@ -92,6 +93,7 @@ const HEAL_CONTROL_PAYLOAD_MAX_SIZE: usize = 64 * 1024;
|
||||
const PEER_REST_RECOVERY_MAX_ATTEMPTS: u32 = 60;
|
||||
const PEER_REST_RECOVERY_MAX_BACKOFF: Duration = Duration::from_secs(30);
|
||||
const SCANNER_ACTIVITY_MAX_MESSAGE_SIZE: usize = 1024;
|
||||
const SCANNER_SCOPED_DIRTY_USAGE_STAGE_TIMEOUT: Duration = Duration::from_secs(5);
|
||||
/// Reserve time for the acquire response's network/clock uncertainty. The
|
||||
/// server owns the real expiry; this local deadline is intentionally earlier
|
||||
/// so a coordinator never starts a bounded persistence operation at the edge
|
||||
@@ -192,12 +194,102 @@ pub struct ScannerPeerActivity {
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct ScannerPeerDirtyUsageSnapshot {
|
||||
pub owner_id: String,
|
||||
pub instance_id: String,
|
||||
pub generation: u64,
|
||||
pub pending_bucket_count: u64,
|
||||
pub protocol_version: u32,
|
||||
pub complete: bool,
|
||||
pub buckets: BTreeMap<String, u64>,
|
||||
pub buckets: BTreeMap<String, ScannerPeerDirtyUsageBucket>,
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct ScannerPeerDirtyUsageBucket {
|
||||
pub bucket_incarnation: Uuid,
|
||||
pub generation: u64,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub struct ScannerScopedDirtyUsageAckEntry {
|
||||
pub bucket: String,
|
||||
pub bucket_incarnation: Uuid,
|
||||
pub generation: u64,
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq)]
|
||||
pub enum ScannerDirtyUsageAcknowledgement {
|
||||
Generation {
|
||||
host: String,
|
||||
instance_id: String,
|
||||
generation: u64,
|
||||
},
|
||||
Scoped {
|
||||
host: String,
|
||||
owner_id: String,
|
||||
instance_id: String,
|
||||
entries: Vec<ScannerScopedDirtyUsageAckEntry>,
|
||||
},
|
||||
}
|
||||
|
||||
fn scanner_scoped_dirty_usage_ack_payloads(
|
||||
owner_id: String,
|
||||
instance_id: String,
|
||||
probe_only: bool,
|
||||
entries: Vec<ScannerScopedDirtyUsageAckEntry>,
|
||||
) -> Result<Vec<ScannerScopedDirtyUsageAckRequest>> {
|
||||
use rustfs_protos::scoped_dirty_usage::*;
|
||||
|
||||
if entries.is_empty() {
|
||||
return Err(Error::other("scoped dirty usage acknowledgement entries must be nonempty"));
|
||||
}
|
||||
|
||||
let mut payloads = Vec::with_capacity(entries.len().div_ceil(SCOPED_DIRTY_USAGE_MAX_ENTRIES as usize));
|
||||
let mut batch = Vec::with_capacity(SCOPED_DIRTY_USAGE_MAX_ENTRIES as usize);
|
||||
for entry in entries {
|
||||
batch.push(ScannerScopedDirtyUsageEntry {
|
||||
bucket: entry.bucket,
|
||||
bucket_incarnation: entry.bucket_incarnation.as_bytes().to_vec().into(),
|
||||
generation: entry.generation,
|
||||
});
|
||||
if batch.len() == SCOPED_DIRTY_USAGE_MAX_ENTRIES as usize {
|
||||
payloads.push(scanner_scoped_dirty_usage_ack_payload(
|
||||
&owner_id,
|
||||
&instance_id,
|
||||
probe_only,
|
||||
std::mem::take(&mut batch),
|
||||
)?);
|
||||
}
|
||||
}
|
||||
if !batch.is_empty() {
|
||||
payloads.push(scanner_scoped_dirty_usage_ack_payload(&owner_id, &instance_id, probe_only, batch)?);
|
||||
}
|
||||
|
||||
Ok(payloads)
|
||||
}
|
||||
|
||||
fn scanner_scoped_dirty_usage_ack_payload(
|
||||
owner_id: &str,
|
||||
instance_id: &str,
|
||||
probe_only: bool,
|
||||
entries: Vec<ScannerScopedDirtyUsageEntry>,
|
||||
) -> Result<ScannerScopedDirtyUsageAckRequest> {
|
||||
use rustfs_protos::scoped_dirty_usage::*;
|
||||
|
||||
let payload = ScannerScopedDirtyUsageAckRequest {
|
||||
challenge: Uuid::new_v4().as_bytes().to_vec().into(),
|
||||
protocol_version: SCOPED_DIRTY_USAGE_PROTOCOL_VERSION,
|
||||
owner_id: owner_id.to_string(),
|
||||
instance_id: instance_id.to_string(),
|
||||
scope: SCOPED_DIRTY_USAGE_BUCKET_SCOPE,
|
||||
probe_only,
|
||||
entries,
|
||||
};
|
||||
canonical_scoped_dirty_usage_request(&payload).map_err(|err| Error::other(err.to_string()))?;
|
||||
Ok(payload)
|
||||
}
|
||||
|
||||
fn scanner_scoped_dirty_usage_ack_reconciled(activity: &ScannerPeerActivity, expected_instance_id: &str) -> bool {
|
||||
activity.instance_id == expected_instance_id && activity.dirty_usage_pending == Some(false)
|
||||
}
|
||||
|
||||
fn scanner_instance_id_is_valid(instance_id: &str) -> bool {
|
||||
@@ -351,6 +443,11 @@ fn decode_scanner_dirty_usage_snapshot_with_verifier(
|
||||
if !scanner_instance_id_is_valid(&response.instance_id) {
|
||||
return Err(Error::other("peer returned an invalid scanner dirty usage snapshot instance ID"));
|
||||
}
|
||||
let owner_id = Uuid::parse_str(&response.owner_id)
|
||||
.ok()
|
||||
.filter(|owner_id| !owner_id.is_nil())
|
||||
.map(|owner_id| owner_id.to_string())
|
||||
.ok_or_else(|| Error::other("peer returned an invalid scanner dirty usage snapshot owner"))?;
|
||||
if response.generation == u64::MAX {
|
||||
return Err(Error::other("peer scanner dirty usage snapshot exhausted its generation"));
|
||||
}
|
||||
@@ -386,9 +483,14 @@ fn decode_scanner_dirty_usage_snapshot_with_verifier(
|
||||
if bucket.generation == 0 || bucket.generation > response.generation {
|
||||
return Err(Error::other("peer scanner dirty usage snapshot contains an invalid bucket generation"));
|
||||
}
|
||||
Uuid::from_slice(bucket.bucket_incarnation.as_ref())
|
||||
.ok()
|
||||
.filter(|bucket_incarnation| !bucket_incarnation.is_nil())
|
||||
.ok_or_else(|| Error::other("peer scanner dirty usage snapshot contains an invalid bucket incarnation"))?;
|
||||
}
|
||||
|
||||
Ok(ScannerPeerDirtyUsageSnapshot {
|
||||
owner_id,
|
||||
instance_id: response.instance_id,
|
||||
generation: response.generation,
|
||||
pending_bucket_count: response.pending_bucket_count,
|
||||
@@ -397,7 +499,16 @@ fn decode_scanner_dirty_usage_snapshot_with_verifier(
|
||||
buckets: response
|
||||
.buckets
|
||||
.into_iter()
|
||||
.map(|bucket| (bucket.bucket, bucket.generation))
|
||||
.map(|bucket| {
|
||||
(
|
||||
bucket.bucket,
|
||||
ScannerPeerDirtyUsageBucket {
|
||||
bucket_incarnation: Uuid::from_slice(bucket.bucket_incarnation.as_ref())
|
||||
.expect("bucket incarnation was validated"),
|
||||
generation: bucket.generation,
|
||||
},
|
||||
)
|
||||
})
|
||||
.collect(),
|
||||
})
|
||||
}
|
||||
@@ -1688,6 +1799,24 @@ impl PeerRestClient {
|
||||
Ok((self.topology_member.clone(), supported_version, epoch))
|
||||
}
|
||||
|
||||
pub async fn probe_ilm_recovery_export(&self, topology_fingerprint: String) -> Result<(String, Uuid)> {
|
||||
let probe = rustfs_protos::ilm_recovery_export_capability_probe(Uuid::new_v4().as_bytes());
|
||||
let result = self
|
||||
.heal_control(rustfs_protos::HEAL_CONTROL_PROTOCOL_VERSION, topology_fingerprint, probe)
|
||||
.await?;
|
||||
let epoch = decode_remote_version_state_capability(&self.topology_member, &result)?;
|
||||
Ok((self.topology_member.clone(), epoch))
|
||||
}
|
||||
|
||||
pub async fn probe_transition_transaction_compaction(&self, topology_fingerprint: String) -> Result<(String, Uuid)> {
|
||||
let probe = rustfs_protos::transition_transaction_compaction_capability_probe(Uuid::new_v4().as_bytes());
|
||||
let result = self
|
||||
.heal_control(rustfs_protos::HEAL_CONTROL_PROTOCOL_VERSION, topology_fingerprint, probe)
|
||||
.await?;
|
||||
let epoch = decode_remote_version_state_capability(&self.topology_member, &result)?;
|
||||
Ok((self.topology_member.clone(), epoch))
|
||||
}
|
||||
|
||||
pub async fn load_bucket_metadata(&self, bucket: &str, scanner_maintenance_change: bool) -> Result<()> {
|
||||
let result = tokio::time::timeout(BUCKET_METADATA_RELOAD_TIMEOUT, async {
|
||||
let result = self.load_bucket_metadata_once(bucket, scanner_maintenance_change).await;
|
||||
@@ -2068,19 +2197,10 @@ impl PeerRestClient {
|
||||
&self,
|
||||
owner_id: String,
|
||||
instance_id: String,
|
||||
entries: Vec<rustfs_protos::proto_gen::node_service::ScannerScopedDirtyUsageEntry>,
|
||||
entries: Vec<ScannerScopedDirtyUsageAckEntry>,
|
||||
) -> Result<bool> {
|
||||
use rustfs_protos::scoped_dirty_usage::*;
|
||||
let payload = rustfs_protos::proto_gen::node_service::ScannerScopedDirtyUsageAckRequest {
|
||||
challenge: Uuid::new_v4().as_bytes().to_vec().into(),
|
||||
protocol_version: SCOPED_DIRTY_USAGE_PROTOCOL_VERSION,
|
||||
owner_id,
|
||||
instance_id,
|
||||
scope: SCOPED_DIRTY_USAGE_BUCKET_SCOPE,
|
||||
probe_only: true,
|
||||
entries,
|
||||
};
|
||||
let canonical = canonical_scoped_dirty_usage_request(&payload).map_err(|err| Error::other(err.to_string()))?;
|
||||
let payloads = scanner_scoped_dirty_usage_ack_payloads(owner_id, instance_id, true, entries)?;
|
||||
self.finalize_result(
|
||||
async {
|
||||
let mut client = super::client::scanner_control_time_out_client(
|
||||
@@ -2088,26 +2208,106 @@ impl PeerRestClient {
|
||||
TonicInterceptor::Signature(gen_tonic_signature_interceptor()),
|
||||
)
|
||||
.await?;
|
||||
for payload in payloads {
|
||||
let canonical =
|
||||
canonical_scoped_dirty_usage_request(&payload).map_err(|err| Error::other(err.to_string()))?;
|
||||
let mut request = Request::new(payload.clone());
|
||||
set_tonic_canonical_body_digest(&mut request, &canonical)?;
|
||||
let response = client.scanner_scoped_dirty_usage_ack(request).await?.into_inner();
|
||||
let body = canonical_scoped_dirty_usage_response(&canonical, &response)
|
||||
.map_err(|_| Error::other("scoped dirty usage capability response is too large"))?;
|
||||
verify_tonic_rpc_response_proof(&body, response.response_proof.as_ref())?;
|
||||
if response.protocol_version != SCOPED_DIRTY_USAGE_PROTOCOL_VERSION
|
||||
|| response.owner_id != payload.owner_id
|
||||
|| response.instance_id != payload.instance_id
|
||||
|| response.max_entries != SCOPED_DIRTY_USAGE_MAX_ENTRIES
|
||||
|| response.max_request_bytes != SCOPED_DIRTY_USAGE_MAX_REQUEST_BYTES
|
||||
|| response.cleared != 0
|
||||
{
|
||||
return Err(Error::other("scoped dirty usage capability response does not match request"));
|
||||
}
|
||||
if !response.supported {
|
||||
return Ok(false);
|
||||
}
|
||||
}
|
||||
Ok(true)
|
||||
}
|
||||
.await,
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn acknowledge_scanner_scoped_dirty_usage(
|
||||
&self,
|
||||
owner_id: String,
|
||||
instance_id: String,
|
||||
entries: Vec<ScannerScopedDirtyUsageAckEntry>,
|
||||
) -> Result<ScannerPeerActivity> {
|
||||
use rustfs_protos::scoped_dirty_usage::*;
|
||||
let payloads = scanner_scoped_dirty_usage_ack_payloads(owner_id, instance_id.clone(), false, entries)?;
|
||||
let ack_attempt = async {
|
||||
let mut client = super::client::scanner_control_time_out_client(
|
||||
&self.grid_host,
|
||||
TonicInterceptor::Signature(gen_tonic_signature_interceptor()),
|
||||
)
|
||||
.await?;
|
||||
for payload in payloads {
|
||||
let canonical = canonical_scoped_dirty_usage_request(&payload).map_err(|err| Error::other(err.to_string()))?;
|
||||
let mut request = Request::new(payload.clone());
|
||||
set_tonic_canonical_body_digest(&mut request, &canonical)?;
|
||||
let response = client.scanner_scoped_dirty_usage_ack(request).await?.into_inner();
|
||||
let body = canonical_scoped_dirty_usage_response(&canonical, &response)
|
||||
.map_err(|_| Error::other("scoped dirty usage capability response is too large"))?;
|
||||
.map_err(|_| Error::other("scoped dirty usage acknowledgement response is too large"))?;
|
||||
verify_tonic_rpc_response_proof(&body, response.response_proof.as_ref())?;
|
||||
if response.protocol_version != SCOPED_DIRTY_USAGE_PROTOCOL_VERSION
|
||||
|| response.owner_id != payload.owner_id
|
||||
|| response.instance_id != payload.instance_id
|
||||
|| response.max_entries != SCOPED_DIRTY_USAGE_MAX_ENTRIES
|
||||
|| response.max_request_bytes != SCOPED_DIRTY_USAGE_MAX_REQUEST_BYTES
|
||||
|| response.cleared != 0
|
||||
|| !response.supported
|
||||
{
|
||||
return Err(Error::other("scoped dirty usage capability response does not match request"));
|
||||
return Err(Error::other("scoped dirty usage acknowledgement response does not match request"));
|
||||
}
|
||||
Ok(response.supported)
|
||||
}
|
||||
.await,
|
||||
)
|
||||
.await
|
||||
Ok(())
|
||||
};
|
||||
let result = match timeout(SCANNER_SCOPED_DIRTY_USAGE_STAGE_TIMEOUT, ack_attempt).await {
|
||||
Ok(result) => self.finalize_result(result).await,
|
||||
Err(_) => {
|
||||
self.prepare_retry_with_timeout(SCANNER_SCOPED_DIRTY_USAGE_STAGE_TIMEOUT)
|
||||
.await;
|
||||
Err(Error::other("scoped dirty usage acknowledgement deadline elapsed"))
|
||||
}
|
||||
};
|
||||
|
||||
match result {
|
||||
Ok(()) => {
|
||||
let activity = self.scanner_scoped_dirty_usage_activity_confirmation().await?;
|
||||
if activity.instance_id == instance_id {
|
||||
Ok(activity)
|
||||
} else {
|
||||
Err(Error::other(
|
||||
"scoped dirty usage acknowledgement peer restarted before activity confirmation",
|
||||
))
|
||||
}
|
||||
}
|
||||
Err(err) => {
|
||||
if Self::is_network_like_error(&err) {
|
||||
self.prepare_retry_with_timeout(SCANNER_SCOPED_DIRTY_USAGE_STAGE_TIMEOUT)
|
||||
.await;
|
||||
}
|
||||
match self.scanner_scoped_dirty_usage_activity_confirmation().await {
|
||||
Ok(activity) if scanner_scoped_dirty_usage_ack_reconciled(&activity, &instance_id) => Ok(activity),
|
||||
_ => Err(err),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn scanner_scoped_dirty_usage_activity_confirmation(&self) -> Result<ScannerPeerActivity> {
|
||||
timeout(SCANNER_SCOPED_DIRTY_USAGE_STAGE_TIMEOUT, self.scanner_activity())
|
||||
.await
|
||||
.map_err(|_| Error::other("scoped dirty usage activity confirmation timed out"))?
|
||||
}
|
||||
|
||||
pub async fn acknowledge_scanner_dirty_usage(&self, instance_id: String, generation: u64) -> Result<ScannerPeerActivity> {
|
||||
@@ -2836,16 +3036,79 @@ mod tests {
|
||||
rustfs_protos::proto_gen::node_service::ScannerDirtyUsageBucket {
|
||||
bucket: "archive".to_string(),
|
||||
generation: 3,
|
||||
bucket_incarnation: Uuid::from_u128(0x11111111111111111111111111111111).as_bytes().to_vec().into(),
|
||||
},
|
||||
rustfs_protos::proto_gen::node_service::ScannerDirtyUsageBucket {
|
||||
bucket: "photos".to_string(),
|
||||
generation: 7,
|
||||
bucket_incarnation: Uuid::from_u128(0x22222222222222222222222222222222).as_bytes().to_vec().into(),
|
||||
},
|
||||
],
|
||||
response_proof: b"proof".to_vec().into(),
|
||||
owner_id: "33333333-3333-3333-3333-333333333333".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn scanner_scoped_dirty_usage_ack_payloads_split_at_protocol_limit() {
|
||||
use rustfs_protos::scoped_dirty_usage::{SCOPED_DIRTY_USAGE_MAX_ENTRIES, canonical_scoped_dirty_usage_request};
|
||||
|
||||
let entries = (0..=SCOPED_DIRTY_USAGE_MAX_ENTRIES)
|
||||
.map(|index| ScannerScopedDirtyUsageAckEntry {
|
||||
bucket: format!("bucket-{index:02}"),
|
||||
bucket_incarnation: Uuid::from_u128(0x11111111111111111111111111111111),
|
||||
generation: 9,
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
|
||||
let payloads = scanner_scoped_dirty_usage_ack_payloads(
|
||||
"33333333-3333-3333-3333-333333333333".to_string(),
|
||||
"0123456789abcdef0123456789abcdef".to_string(),
|
||||
false,
|
||||
entries,
|
||||
)
|
||||
.expect("33 entries should split into valid scoped dirty usage requests");
|
||||
|
||||
assert_eq!(payloads.len(), 2);
|
||||
assert_eq!(payloads[0].entries.len(), SCOPED_DIRTY_USAGE_MAX_ENTRIES as usize);
|
||||
assert_eq!(payloads[1].entries.len(), 1);
|
||||
assert_eq!(payloads[0].entries.first().map(|entry| entry.bucket.as_str()), Some("bucket-00"));
|
||||
assert_eq!(payloads[0].entries.last().map(|entry| entry.bucket.as_str()), Some("bucket-31"));
|
||||
assert_eq!(payloads[1].entries.first().map(|entry| entry.bucket.as_str()), Some("bucket-32"));
|
||||
for payload in payloads {
|
||||
canonical_scoped_dirty_usage_request(&payload).expect("each split scoped ACK payload should be canonical");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn scanner_scoped_dirty_usage_ack_reconciliation_requires_same_clean_instance() {
|
||||
let activity = |instance_id: &str, pending| ScannerPeerActivity {
|
||||
instance_id: instance_id.to_string(),
|
||||
namespace_generation: 1,
|
||||
maintenance_generation: 1,
|
||||
protocol_version: SCANNER_ACTIVITY_PROTOCOL_VERSION,
|
||||
topology_digest: Some([1; 32]),
|
||||
data_movement_active: Some(false),
|
||||
dirty_usage_generation: Some(9),
|
||||
dirty_usage_pending: pending,
|
||||
movement_generation: Some(1),
|
||||
publication_blocked: Some(false),
|
||||
};
|
||||
|
||||
assert!(scanner_scoped_dirty_usage_ack_reconciled(
|
||||
&activity("0123456789abcdef0123456789abcdef", Some(false)),
|
||||
"0123456789abcdef0123456789abcdef"
|
||||
));
|
||||
assert!(!scanner_scoped_dirty_usage_ack_reconciled(
|
||||
&activity("0123456789abcdef0123456789abcdef", Some(true)),
|
||||
"0123456789abcdef0123456789abcdef"
|
||||
));
|
||||
assert!(!scanner_scoped_dirty_usage_ack_reconciled(
|
||||
&activity("fedcba9876543210fedcba9876543210", Some(false)),
|
||||
"0123456789abcdef0123456789abcdef"
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn scanner_dirty_usage_snapshot_requires_a_complete_authenticated_ordered_view() {
|
||||
let decoded = decode_test_scanner_dirty_usage_snapshot(test_scanner_dirty_usage_snapshot_response())
|
||||
@@ -2854,9 +3117,18 @@ mod tests {
|
||||
assert_eq!(decoded.generation, 7);
|
||||
assert_eq!(decoded.pending_bucket_count, 2);
|
||||
assert_eq!(decoded.protocol_version, SCANNER_DIRTY_USAGE_SNAPSHOT_PROTOCOL_VERSION);
|
||||
assert_eq!(decoded.owner_id, "33333333-3333-3333-3333-333333333333");
|
||||
assert!(decoded.complete);
|
||||
assert_eq!(decoded.buckets.get("archive"), Some(&3));
|
||||
assert_eq!(decoded.buckets.get("photos"), Some(&7));
|
||||
assert_eq!(
|
||||
decoded.buckets.get("archive").map(|bucket| bucket.bucket_incarnation),
|
||||
Some(Uuid::from_u128(0x11111111111111111111111111111111))
|
||||
);
|
||||
assert_eq!(decoded.buckets.get("archive").map(|bucket| bucket.generation), Some(3));
|
||||
assert_eq!(
|
||||
decoded.buckets.get("photos").map(|bucket| bucket.bucket_incarnation),
|
||||
Some(Uuid::from_u128(0x22222222222222222222222222222222))
|
||||
);
|
||||
assert_eq!(decoded.buckets.get("photos").map(|bucket| bucket.generation), Some(7));
|
||||
|
||||
let overflow_count =
|
||||
u64::try_from(SCANNER_DIRTY_USAGE_SNAPSHOT_MAX_ENTRIES + 1).expect("the test snapshot entry limit should fit in u64");
|
||||
@@ -2907,6 +3179,14 @@ mod tests {
|
||||
empty_bucket.buckets[0].bucket.clear();
|
||||
cases.push((empty_bucket, "empty bucket name"));
|
||||
|
||||
let mut invalid_owner = test_scanner_dirty_usage_snapshot_response();
|
||||
invalid_owner.owner_id.clear();
|
||||
cases.push((invalid_owner, "snapshot owner"));
|
||||
|
||||
let mut invalid_incarnation = test_scanner_dirty_usage_snapshot_response();
|
||||
invalid_incarnation.buckets[0].bucket_incarnation = Uuid::nil().as_bytes().to_vec().into();
|
||||
cases.push((invalid_incarnation, "bucket incarnation"));
|
||||
|
||||
let mut partial = test_scanner_dirty_usage_snapshot_response();
|
||||
partial.complete = false;
|
||||
cases.push((partial, "entry-limit overflow"));
|
||||
@@ -2919,6 +3199,7 @@ mod tests {
|
||||
.map(|index| rustfs_protos::proto_gen::node_service::ScannerDirtyUsageBucket {
|
||||
bucket: format!("bucket-{index:04}"),
|
||||
generation: 1,
|
||||
bucket_incarnation: Uuid::from_u128(0x11111111111111111111111111111111).as_bytes().to_vec().into(),
|
||||
})
|
||||
.collect(),
|
||||
..test_scanner_dirty_usage_snapshot_response()
|
||||
|
||||
@@ -449,14 +449,6 @@ where
|
||||
Ok(data)
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_limited<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
|
||||
where
|
||||
S: EcstoreObjectIO,
|
||||
{
|
||||
let (data, _obj) = read_config_with_metadata_inner(api, file, &ObjectOptions::default(), false, Some(max_bytes)).await?;
|
||||
Ok(data)
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_limited_preserve_empty<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
|
||||
where
|
||||
S: EcstoreObjectIO,
|
||||
@@ -465,6 +457,14 @@ where
|
||||
Ok(data)
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_limited<S>(api: Arc<S>, file: &str, max_bytes: usize) -> Result<Vec<u8>>
|
||||
where
|
||||
S: EcstoreObjectIO,
|
||||
{
|
||||
let (data, _obj) = read_config_with_metadata_inner(api, file, &ObjectOptions::default(), false, Some(max_bytes)).await?;
|
||||
Ok(data)
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_limited_preserve_empty_with_metadata<S>(
|
||||
api: Arc<S>,
|
||||
file: &str,
|
||||
@@ -476,6 +476,18 @@ where
|
||||
read_config_with_metadata_inner(api, file, &ObjectOptions::default(), true, Some(max_bytes)).await
|
||||
}
|
||||
|
||||
pub(crate) async fn read_config_limited_preserve_empty_with_metadata_opts<S>(
|
||||
api: Arc<S>,
|
||||
file: &str,
|
||||
opts: &ObjectOptions,
|
||||
max_bytes: usize,
|
||||
) -> Result<(Vec<u8>, ObjectInfo)>
|
||||
where
|
||||
S: EcstoreObjectIO,
|
||||
{
|
||||
read_config_with_metadata_inner(api, file, opts, true, Some(max_bytes)).await
|
||||
}
|
||||
|
||||
/// Read an existing config object without treating an empty payload as absent.
|
||||
/// Callers that validate their own payload format need to distinguish corruption
|
||||
/// from `ConfigNotFound`.
|
||||
|
||||
@@ -747,18 +747,27 @@ mod tests {
|
||||
let mut kvs = KVS::new();
|
||||
kvs.insert(CLASS_STANDARD.to_string(), "EC:2".to_string());
|
||||
|
||||
let err = lookup_config_for_pools_with_env(&kvs, &[4, 2], no_env_overrides())
|
||||
.expect_err("EC:2 must be rejected by the two-drive pool");
|
||||
assert!(
|
||||
err.to_string().contains("pool 1") && err.to_string().contains("2 drives"),
|
||||
"error must identify the rejecting pool: {err}"
|
||||
);
|
||||
for drives in [2, 3] {
|
||||
let err = lookup_config_for_pools_with_env(&kvs, &[4, drives], no_env_overrides())
|
||||
.expect_err("EC:2 must be rejected by a pool with fewer than four drives per set");
|
||||
assert!(
|
||||
err.to_string().contains("pool 1") && err.to_string().contains(&format!("{drives} drives")),
|
||||
"error must identify the rejecting pool: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
let cfg =
|
||||
lookup_config_for_pools_with_env(&kvs, &[4, 4], no_env_overrides()).expect("EC:2 is valid for both four-drive pools");
|
||||
assert_eq!(cfg.parities_for_sc(STANDARD), Some(vec![2, 2]));
|
||||
|
||||
kvs.insert(CLASS_STANDARD.to_string(), "EC:1".to_string());
|
||||
let cfg = lookup_config_for_pools_with_env(&kvs, &[4, 2], no_env_overrides()).expect("EC:1 is valid for both pools");
|
||||
assert_eq!(cfg.parity_for_sc(STANDARD, 4), Some(1));
|
||||
assert_eq!(cfg.parity_for_sc(STANDARD, 2), Some(1));
|
||||
assert_eq!(cfg.get_parity_for_sc(STANDARD), Some(1));
|
||||
for drives in [2, 3, 4] {
|
||||
let cfg =
|
||||
lookup_config_for_pools_with_env(&kvs, &[4, drives], no_env_overrides()).expect("EC:1 is valid for both pools");
|
||||
assert_eq!(cfg.parity_for_sc(STANDARD, 4), Some(1));
|
||||
assert_eq!(cfg.parity_for_sc(STANDARD, drives), Some(1));
|
||||
assert_eq!(cfg.get_parity_for_sc(STANDARD), Some(1));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
+1807
-359
File diff suppressed because it is too large
Load Diff
@@ -39,6 +39,7 @@ mod capacity_dedup_tests {
|
||||
..Default::default()
|
||||
},
|
||||
disks: disks.clone(),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let total = get_total_usable_capacity(&disks, &info);
|
||||
@@ -73,6 +74,7 @@ mod capacity_dedup_tests {
|
||||
..Default::default()
|
||||
},
|
||||
disks: disks.clone(),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let total = get_total_usable_capacity(&disks, &info);
|
||||
@@ -150,6 +152,7 @@ mod capacity_dedup_tests {
|
||||
..Default::default()
|
||||
},
|
||||
disks: disks.clone(),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let total = get_total_usable_capacity(&disks, &info);
|
||||
|
||||
@@ -2673,7 +2673,7 @@ mod tests {
|
||||
]),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!object_info.is_multipart());
|
||||
assert!(object_info.is_multipart());
|
||||
assert!(should_use_multipart_data_movement(&object_info, false));
|
||||
|
||||
let single_nonstandard_part = ObjectInfo {
|
||||
@@ -3050,7 +3050,7 @@ mod tests {
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
assert!(!object_info.is_multipart());
|
||||
assert!(object_info.is_multipart());
|
||||
assert!(object_info.parts.iter().any(|part| part.checksums.is_some()));
|
||||
let opts = data_movement_put_object_opts(&object_info, 0);
|
||||
assert!(!rustfs_utils::http::contains_key_str(&opts.user_defined, SUFFIX_PART_CHECKSUMS));
|
||||
|
||||
@@ -195,6 +195,13 @@ fn resolve_drive_timeout_profile_from_env() -> DriveTimeoutProfile {
|
||||
DriveTimeoutProfile::parse(rustfs_config::DEFAULT_DRIVE_TIMEOUT_PROFILE).unwrap_or(DriveTimeoutProfile::Default)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
tokio::task_local! {
|
||||
/// Artificial `disk_info` latency for tests that pin how the admin storage
|
||||
/// walk composes per-drive probe time.
|
||||
pub(crate) static DISK_INFO_PROBE_DELAY_FOR_TEST: Duration;
|
||||
}
|
||||
|
||||
fn get_drive_timeout_profile() -> DriveTimeoutProfile {
|
||||
#[cfg(test)]
|
||||
{
|
||||
@@ -2036,6 +2043,10 @@ impl DiskAPI for LocalDiskWrapper {
|
||||
.track_disk_health_with_op_and_timeout_action(
|
||||
"disk_info",
|
||||
|| async {
|
||||
#[cfg(test)]
|
||||
if let Ok(delay) = DISK_INFO_PROBE_DELAY_FOR_TEST.try_with(|delay| *delay) {
|
||||
tokio::time::sleep(delay).await;
|
||||
}
|
||||
let result = self.disk.disk_info(opts).await?;
|
||||
|
||||
if let Some(current_disk_id) = *self.disk_id.read().await
|
||||
|
||||
@@ -425,6 +425,7 @@ impl From<rustfs_filemeta::Error> for DiskError {
|
||||
rustfs_filemeta::Error::FileVersionNotFound => DiskError::FileVersionNotFound,
|
||||
rustfs_filemeta::Error::FileCorrupt => DiskError::FileCorrupt,
|
||||
rustfs_filemeta::Error::MethodNotAllowed => DiskError::MethodNotAllowed,
|
||||
rustfs_filemeta::Error::MaxVersionsExceeded => DiskError::MaxVersionsExceeded,
|
||||
e => DiskError::other(e),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -98,6 +98,35 @@ const INLINE_METADATA_ROLLBACK_DIR_XOR: u128 = 0x7275737466735f696e6c696e655f726
|
||||
const DELETE_MARKER_ROLLBACK_FILE: &str = "xl.meta.delete-marker.rollback";
|
||||
pub(crate) const DELETE_DATA_DIR_MARKER_PREFIX: &str = "delete-data.";
|
||||
pub(crate) const RESERVED_DELETE_DATA_DIR_MARKER_PREFIX: &str = "reserve-delete-data.";
|
||||
/// Largest directory read the delete-residue probe issues before it must
|
||||
/// fall back to a complete read. Residue holds one or two data dirs, so an
|
||||
/// under-filled batch settles the common case without materializing large
|
||||
/// child sets; a full batch cannot prove no listable child hides behind it.
|
||||
const DELETE_RESIDUE_PROBE_LIMIT: i32 = 8;
|
||||
|
||||
/// A `part.N` file with a positive part number, the shape erasure data takes
|
||||
/// inside a version data dir.
|
||||
pub(crate) fn metadata_less_part_file(entry: &str) -> bool {
|
||||
entry
|
||||
.strip_prefix("part.")
|
||||
.is_some_and(|part_number| part_number.parse::<usize>().is_ok_and(|part_number| part_number > 0))
|
||||
}
|
||||
|
||||
fn is_delete_transaction_marker(entry: &str, prefix: &str) -> bool {
|
||||
entry
|
||||
.strip_prefix(prefix)
|
||||
.is_some_and(|transaction| Uuid::parse_str(transaction).is_ok_and(|uuid| !uuid.is_nil()))
|
||||
}
|
||||
|
||||
/// Whether a `list_dir` entry inside a UUID data dir is erasure data or a
|
||||
/// delete-transaction marker. Anything else (a subdirectory, an `xl.meta`, an
|
||||
/// unknown file) means the directory is not plain delete residue.
|
||||
fn is_metadata_less_data_dir_entry(entry: &str) -> bool {
|
||||
!entry.ends_with(SLASH_SEPARATOR)
|
||||
&& (metadata_less_part_file(entry)
|
||||
|| is_delete_transaction_marker(entry, DELETE_DATA_DIR_MARKER_PREFIX)
|
||||
|| is_delete_transaction_marker(entry, RESERVED_DELETE_DATA_DIR_MARKER_PREFIX))
|
||||
}
|
||||
const STARTUP_CLEANUP_WAIT_TIMEOUT: Duration = Duration::from_secs(2);
|
||||
const ENV_BITROT_SIZE_MISMATCH_RETRY_COUNT: &str = "RUSTFS_BITROT_SIZE_MISMATCH_RETRY_COUNT";
|
||||
const ENV_BITROT_SIZE_MISMATCH_RETRY_DELAY_MS: &str = "RUSTFS_BITROT_SIZE_MISMATCH_RETRY_DELAY_MS";
|
||||
@@ -7644,15 +7673,18 @@ impl LocalDisk {
|
||||
{
|
||||
meta.name.push_str(SLASH_SEPARATOR);
|
||||
// Conservative listings verify physical prefixes. Never-versioned
|
||||
// buckets use the bounded fast path and reclaim residue after an
|
||||
// exact recursive listing proves that prefix empty.
|
||||
if opts.recursive
|
||||
|| opts.incl_deleted
|
||||
|| opts.skip_hidden_prefix_check
|
||||
|| self
|
||||
.directory_has_listing_entry(&opts.bucket, &meta.name, opts.incl_deleted, stall)
|
||||
// buckets use the bounded fast path, which only has to rule out
|
||||
// the data dirs a deleted version leaves behind; an empty listing
|
||||
// of such a prefix then reclaims committed residue.
|
||||
let listable = if opts.recursive || opts.incl_deleted {
|
||||
true
|
||||
} else if opts.skip_hidden_prefix_check {
|
||||
!self.directory_is_delete_residue(&opts.bucket, &meta.name, stall).await?
|
||||
} else {
|
||||
self.directory_has_listing_entry(&opts.bucket, &meta.name, opts.incl_deleted, stall)
|
||||
.await?
|
||||
{
|
||||
};
|
||||
if listable {
|
||||
schedule_dir(&mut dir_stack, meta.name, false, None, true);
|
||||
}
|
||||
}
|
||||
@@ -7776,6 +7808,74 @@ impl LocalDisk {
|
||||
Ok(false)
|
||||
}
|
||||
|
||||
/// Whether the metadata-less directory `dir_name` holds nothing but the
|
||||
/// data dirs of deleted versions: it is itself a non-nil UUID directory of
|
||||
/// `part.N` files and delete-transaction markers, or every child is one.
|
||||
/// That is what an interrupted or deferred version delete leaves behind
|
||||
/// once the `xl.meta` is gone, and it must not surface as a prefix. Real
|
||||
/// object children are directories carrying their own `xl.meta`, so the
|
||||
/// first non-UUID child, stray file, or subdirectory inside a UUID child
|
||||
/// proves the directory is a genuine prefix. Reads are bounded: a
|
||||
/// directory that vanishes mid-probe holds nothing listable.
|
||||
async fn directory_is_delete_residue(&self, bucket: &str, dir_name: &str, stall: Option<Duration>) -> Result<bool> {
|
||||
let dir_name = dir_name.trim_end_matches(SLASH_SEPARATOR);
|
||||
let Some(entries) = self.read_dir_for_residue_probe(bucket, dir_name, stall).await? else {
|
||||
return Ok(false);
|
||||
};
|
||||
if entries.is_empty() {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
let is_data_dir = dir_name
|
||||
.rsplit(SLASH_SEPARATOR)
|
||||
.next()
|
||||
.is_some_and(|name| Uuid::parse_str(name).is_ok_and(|uuid| !uuid.is_nil()));
|
||||
if is_data_dir && entries.iter().all(|entry| is_metadata_less_data_dir_entry(entry)) {
|
||||
return Ok(true);
|
||||
}
|
||||
|
||||
for entry in entries {
|
||||
let Some(child) = entry.strip_suffix(SLASH_SEPARATOR) else {
|
||||
return Ok(false);
|
||||
};
|
||||
if !Uuid::parse_str(child).is_ok_and(|uuid| !uuid.is_nil()) {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
let child_path = path_join_buf(&[dir_name, child]);
|
||||
let Some(child_entries) = self.read_dir_for_residue_probe(bucket, &child_path, stall).await? else {
|
||||
continue;
|
||||
};
|
||||
if !child_entries.iter().all(|entry| is_metadata_less_data_dir_entry(entry)) {
|
||||
return Ok(false);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
/// Read `dir` with a bounded batch first and a complete read only when the
|
||||
/// batch was full. `None` when the directory does not exist any more.
|
||||
async fn read_dir_for_residue_probe(&self, bucket: &str, dir: &str, stall: Option<Duration>) -> Result<Option<Vec<String>>> {
|
||||
for count in [DELETE_RESIDUE_PROBE_LIMIT, -1] {
|
||||
let entries = match with_walk_stall_timeout(stall, self.list_dir("", bucket, dir, count)).await {
|
||||
Ok(entries) => entries,
|
||||
Err(err) => {
|
||||
if err == DiskError::VolumeNotFound || err == Error::FileNotFound {
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
return Err(err);
|
||||
}
|
||||
};
|
||||
if count < 0 || entries.len() < count as usize {
|
||||
return Ok(Some(entries));
|
||||
}
|
||||
}
|
||||
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
/// Whether anything under `dir_name` would appear in a listing. With
|
||||
/// `incl_deleted`, any `xl.meta` counts (versioned listings surface
|
||||
/// delete-marker-only objects too); otherwise the metadata must hold a
|
||||
@@ -8708,6 +8808,34 @@ impl DiskAPI for LocalDisk {
|
||||
});
|
||||
}
|
||||
|
||||
let rollback_after_fsync_failure = |parent: &Path, current: Option<&[u8]>| -> std::io::Result<()> {
|
||||
match current {
|
||||
Some(previous) => {
|
||||
let rollback_temporary =
|
||||
parent.join(format!(".{}.{}.rollback.tmp", path.replace('/', "_"), Uuid::new_v4()));
|
||||
let rollback_result = (|| -> std::io::Result<()> {
|
||||
let mut staged = std::fs::OpenOptions::new()
|
||||
.create_new(true)
|
||||
.write(true)
|
||||
.open(&rollback_temporary)?;
|
||||
staged.write_all(previous)?;
|
||||
staged.sync_all()?;
|
||||
std::fs::rename(&rollback_temporary, &file_path)
|
||||
})();
|
||||
if let Err(err) = rollback_result {
|
||||
let _ = std::fs::remove_file(&rollback_temporary);
|
||||
return Err(err);
|
||||
}
|
||||
}
|
||||
None => match std::fs::remove_file(&file_path) {
|
||||
Ok(()) => {}
|
||||
Err(err) if err.kind() == ErrorKind::NotFound => {}
|
||||
Err(err) => return Err(err),
|
||||
},
|
||||
}
|
||||
os::fsync_dir_std(parent)
|
||||
};
|
||||
|
||||
match replacement {
|
||||
Some(replacement) => {
|
||||
let parent = file_path
|
||||
@@ -8727,14 +8855,19 @@ impl DiskAPI for LocalDisk {
|
||||
let _ = std::fs::remove_file(&temporary);
|
||||
return Err(err);
|
||||
}
|
||||
if sync_metadata {
|
||||
os::fsync_dir_std(parent)?;
|
||||
if sync_metadata && let Err(err) = os::fsync_dir_std(parent) {
|
||||
rollback_after_fsync_failure(parent, current.as_deref())?;
|
||||
return Err(err);
|
||||
}
|
||||
}
|
||||
None => {
|
||||
std::fs::remove_file(&file_path)?;
|
||||
if sync_metadata && let Some(parent) = file_path.parent() {
|
||||
os::fsync_dir_std(parent)?;
|
||||
if sync_metadata
|
||||
&& let Some(parent) = file_path.parent()
|
||||
&& let Err(err) = os::fsync_dir_std(parent)
|
||||
{
|
||||
rollback_after_fsync_failure(parent, current.as_deref())?;
|
||||
return Err(err);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -17755,6 +17888,133 @@ mod test {
|
||||
assert_eq!(fast_path_probes, 0);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_scan_dir_nonrecursive_fast_path_hides_delete_residue() {
|
||||
use rustfs_filemeta::MetacacheReader;
|
||||
use tempfile::tempdir;
|
||||
|
||||
let dir = tempdir().expect("tempdir should be created");
|
||||
let bucket = "test-bucket";
|
||||
let bucket_dir = dir.path().join(bucket);
|
||||
|
||||
async fn write_object(object_dir: &Path, object_name: &str) {
|
||||
fs::create_dir_all(object_dir)
|
||||
.await
|
||||
.expect("object directory should be created");
|
||||
let mut metadata = FileMeta::default();
|
||||
let mut file_info = FileInfo::new(object_name, 1, 1);
|
||||
file_info.mod_time = Some(OffsetDateTime::now_utc());
|
||||
metadata.add_version(file_info).expect("metadata should be valid");
|
||||
fs::write(
|
||||
object_dir.join(STORAGE_FORMAT_FILE),
|
||||
metadata.marshal_msg().expect("metadata should encode"),
|
||||
)
|
||||
.await
|
||||
.expect("object metadata should be written");
|
||||
}
|
||||
|
||||
// A deleted version whose data dir survived: part files only.
|
||||
let residue = bucket_dir
|
||||
.join("residue/2026/object.parquet")
|
||||
.join(Uuid::new_v4().to_string());
|
||||
fs::create_dir_all(&residue).await.expect("residue should be created");
|
||||
fs::write(residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
|
||||
// The same shape after a committed delete transaction.
|
||||
let committed = bucket_dir.join("committed/object").join(Uuid::new_v4().to_string());
|
||||
fs::create_dir_all(&committed)
|
||||
.await
|
||||
.expect("committed residue should be created");
|
||||
fs::write(committed.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
fs::write(committed.join(format!("{DELETE_DATA_DIR_MARKER_PREFIX}{}", Uuid::new_v4())), [])
|
||||
.await
|
||||
.expect("delete marker should be written");
|
||||
|
||||
// A user prefix made of UUID-named directories holding real objects.
|
||||
let upload = Uuid::new_v4().to_string();
|
||||
write_object(&bucket_dir.join("uploads").join(&upload).join("file"), &format!("uploads/{upload}/file")).await;
|
||||
|
||||
// An object whose key is itself a UUID.
|
||||
let named = Uuid::new_v4().to_string();
|
||||
write_object(&bucket_dir.join("named").join(&named), &format!("named/{named}")).await;
|
||||
|
||||
// Residue next to a live child object.
|
||||
let mixed_residue = bucket_dir.join("mixed").join(Uuid::new_v4().to_string());
|
||||
fs::create_dir_all(&mixed_residue)
|
||||
.await
|
||||
.expect("mixed residue should be created");
|
||||
fs::write(mixed_residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
write_object(&bucket_dir.join("mixed/child"), "mixed/child").await;
|
||||
|
||||
let endpoint =
|
||||
Endpoint::try_from(dir.path().to_str().expect("tempdir path should be UTF-8")).expect("endpoint should parse");
|
||||
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should initialize");
|
||||
|
||||
async fn scan_names(disk: &LocalDisk, bucket: &str, current: &str) -> Vec<String> {
|
||||
let (reader, mut writer) = tokio::io::duplex(64 * 1024);
|
||||
let mut output = MetacacheWriter::new(&mut writer);
|
||||
let opts = WalkDirOptions {
|
||||
bucket: bucket.to_string(),
|
||||
base_dir: current.to_string(),
|
||||
skip_hidden_prefix_check: true,
|
||||
..Default::default()
|
||||
};
|
||||
let mut objects_returned = 0;
|
||||
disk.scan_dir(
|
||||
current.to_string(),
|
||||
"".to_string(),
|
||||
&opts,
|
||||
&mut output,
|
||||
&mut objects_returned,
|
||||
false,
|
||||
None,
|
||||
)
|
||||
.await
|
||||
.expect("scan_dir should succeed");
|
||||
output.close().await.expect("metacache writer should close");
|
||||
drop(output);
|
||||
drop(writer);
|
||||
|
||||
let mut names = MetacacheReader::new(reader)
|
||||
.read_all()
|
||||
.await
|
||||
.expect("scan output should decode")
|
||||
.into_iter()
|
||||
.map(|entry| entry.name)
|
||||
.collect::<Vec<_>>();
|
||||
names.sort();
|
||||
names
|
||||
}
|
||||
|
||||
// Directories whose only content is a deleted version's data dir are
|
||||
// not prefixes; their ancestors stay ordinary directories until an
|
||||
// empty listing reclaims them.
|
||||
assert_eq!(scan_names(&disk, bucket, "residue/2026/").await, Vec::<String>::new());
|
||||
assert_eq!(scan_names(&disk, bucket, "committed/").await, Vec::<String>::new());
|
||||
|
||||
// UUID-named directories holding real objects, an object keyed by a
|
||||
// UUID, and residue beside a live child all remain visible.
|
||||
assert_eq!(scan_names(&disk, bucket, "uploads/").await, vec![format!("uploads/{upload}/")]);
|
||||
assert_eq!(scan_names(&disk, bucket, "named/").await, vec![format!("named/{named}")]);
|
||||
assert_eq!(scan_names(&disk, bucket, "mixed/").await, vec!["mixed/child".to_owned()]);
|
||||
assert_eq!(
|
||||
scan_names(&disk, bucket, "").await,
|
||||
vec![
|
||||
"committed/".to_owned(),
|
||||
"mixed/".to_owned(),
|
||||
"named/".to_owned(),
|
||||
"residue/".to_owned(),
|
||||
"uploads/".to_owned(),
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_scan_dir_nonrecursive_skips_dirs_with_only_hidden_delete_markers() {
|
||||
use rustfs_filemeta::MetacacheReader;
|
||||
@@ -22166,6 +22426,110 @@ mod test {
|
||||
));
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn conditional_file_update_dir_fsync_failure_restores_previous_bytes() {
|
||||
use tempfile::tempdir;
|
||||
|
||||
let dir = tempdir().expect("temp dir should be created");
|
||||
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
|
||||
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
|
||||
let previous = Bytes::from_static(b"previous-owner");
|
||||
let successor = Bytes::from_static(b"successor-owner");
|
||||
|
||||
assert_eq!(
|
||||
disk.compare_and_update_file(RUSTFS_META_BUCKET, HEALING_MARKER_PATH, None, Some(previous.clone()))
|
||||
.await
|
||||
.expect("previous owner should commit"),
|
||||
ConditionalFileUpdate::Updated
|
||||
);
|
||||
|
||||
let marker_path = disk
|
||||
.get_object_path(RUSTFS_META_BUCKET, HEALING_MARKER_PATH)
|
||||
.expect("marker path should resolve");
|
||||
let parent = marker_path.parent().expect("marker path should have a parent");
|
||||
os::fsync_dir_recorder::set_failure(parent, ErrorKind::Other);
|
||||
|
||||
let err = disk
|
||||
.compare_and_update_file(RUSTFS_META_BUCKET, HEALING_MARKER_PATH, Some(previous.clone()), Some(successor))
|
||||
.await
|
||||
.expect_err("directory fsync failure must fail the CAS update");
|
||||
assert!(matches!(err, DiskError::Io(ref err) if err.kind() == ErrorKind::Other));
|
||||
assert_eq!(
|
||||
disk.read_all(RUSTFS_META_BUCKET, HEALING_MARKER_PATH)
|
||||
.await
|
||||
.expect("previous bytes should remain readable after rollback"),
|
||||
previous
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn conditional_file_update_dir_fsync_failure_removes_new_file_without_anchor() {
|
||||
use tempfile::tempdir;
|
||||
|
||||
let dir = tempdir().expect("temp dir should be created");
|
||||
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
|
||||
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
|
||||
ensure_test_volume(&disk, RUSTFS_META_BUCKET).await;
|
||||
let marker_path = disk
|
||||
.get_object_path(RUSTFS_META_BUCKET, HEALING_MARKER_PATH)
|
||||
.expect("marker path should resolve");
|
||||
let parent = marker_path.parent().expect("marker path should have a parent");
|
||||
os::fsync_dir_recorder::set_failure(parent, ErrorKind::Other);
|
||||
|
||||
let err = disk
|
||||
.compare_and_update_file(
|
||||
RUSTFS_META_BUCKET,
|
||||
HEALING_MARKER_PATH,
|
||||
None,
|
||||
Some(Bytes::from_static(b"successor-owner")),
|
||||
)
|
||||
.await
|
||||
.expect_err("directory fsync failure must fail the CAS create");
|
||||
assert!(matches!(err, DiskError::Io(ref err) if err.kind() == ErrorKind::Other));
|
||||
assert!(
|
||||
matches!(disk.read_all(RUSTFS_META_BUCKET, HEALING_MARKER_PATH).await, Err(DiskError::FileNotFound)),
|
||||
"uncommitted successor bytes must be removed when no previous anchor exists"
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn conditional_file_delete_dir_fsync_failure_restores_previous_bytes() {
|
||||
use tempfile::tempdir;
|
||||
|
||||
let dir = tempdir().expect("temp dir should be created");
|
||||
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
|
||||
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
|
||||
let previous = Bytes::from_static(b"previous-owner");
|
||||
|
||||
assert_eq!(
|
||||
disk.compare_and_update_file(RUSTFS_META_BUCKET, HEALING_MARKER_PATH, None, Some(previous.clone()))
|
||||
.await
|
||||
.expect("previous owner should commit"),
|
||||
ConditionalFileUpdate::Updated
|
||||
);
|
||||
|
||||
let marker_path = disk
|
||||
.get_object_path(RUSTFS_META_BUCKET, HEALING_MARKER_PATH)
|
||||
.expect("marker path should resolve");
|
||||
let parent = marker_path.parent().expect("marker path should have a parent");
|
||||
os::fsync_dir_recorder::set_failure(parent, ErrorKind::Other);
|
||||
|
||||
let err = disk
|
||||
.compare_and_update_file(RUSTFS_META_BUCKET, HEALING_MARKER_PATH, Some(previous.clone()), None)
|
||||
.await
|
||||
.expect_err("directory fsync failure must fail the CAS delete");
|
||||
assert!(matches!(err, DiskError::Io(ref err) if err.kind() == ErrorKind::Other));
|
||||
assert_eq!(
|
||||
disk.read_all(RUSTFS_META_BUCKET, HEALING_MARKER_PATH)
|
||||
.await
|
||||
.expect("previous bytes should be restored after failed delete"),
|
||||
previous
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn conditional_file_update_returns_would_block_when_marker_lock_is_contended() {
|
||||
|
||||
@@ -92,6 +92,9 @@ pub(crate) mod fsync_dir_recorder {
|
||||
static LIMITED: Mutex<Vec<PathBuf>> = Mutex::new(Vec::new());
|
||||
static GROUPED: Mutex<Vec<(PathBuf, usize)>> = Mutex::new(Vec::new());
|
||||
#[cfg(unix)]
|
||||
static FAILURES: std::sync::LazyLock<Mutex<HashMap<PathBuf, io::ErrorKind>>> =
|
||||
std::sync::LazyLock::new(|| Mutex::new(HashMap::new()));
|
||||
#[cfg(unix)]
|
||||
static BEFORE_LIMITED: std::sync::LazyLock<Mutex<HashMap<PathBuf, Hook>>> =
|
||||
std::sync::LazyLock::new(|| Mutex::new(HashMap::new()));
|
||||
static BEFORE_GROUP_BATCH: std::sync::LazyLock<Mutex<HashMap<PathBuf, Hook>>> =
|
||||
@@ -151,6 +154,19 @@ pub(crate) mod fsync_dir_recorder {
|
||||
contains_path(&RECORDED.lock().expect("fsync dir recorder poisoned"), dir)
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
pub(crate) fn set_failure(dir: &Path, kind: io::ErrorKind) {
|
||||
FAILURES
|
||||
.lock()
|
||||
.expect("fsync dir failure hook poisoned")
|
||||
.insert(dir.to_path_buf(), kind);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
pub(crate) fn take_failure(dir: &Path) -> Option<io::ErrorKind> {
|
||||
remove_path_keyed(&FAILURES, dir, "fsync dir failure hook poisoned")
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
pub(crate) fn record_limited(dir: &Path) {
|
||||
record_path(&LIMITED, dir, "limited fsync dir recorder");
|
||||
@@ -247,7 +263,7 @@ pub(crate) mod fsync_dir_recorder {
|
||||
}
|
||||
|
||||
/// Pause a real namespace mutation inside its physical executor.
|
||||
#[cfg(all(any(test, feature = "test-util"), not(windows)))]
|
||||
#[cfg(all(test, not(windows)))]
|
||||
pub(crate) mod prepared_publication_test_hooks {
|
||||
use super::*;
|
||||
|
||||
@@ -256,9 +272,7 @@ pub(crate) mod prepared_publication_test_hooks {
|
||||
PreparedRename,
|
||||
Rename,
|
||||
Remove,
|
||||
#[cfg(test)]
|
||||
Rollback,
|
||||
#[cfg(test)]
|
||||
DirFsync,
|
||||
}
|
||||
|
||||
@@ -274,7 +288,6 @@ pub(crate) mod prepared_publication_test_hooks {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn install(path: &Path, hook: impl FnOnce() + Send + 'static) -> Guard {
|
||||
install_at(Stage::PreparedRename, path, hook)
|
||||
}
|
||||
@@ -333,51 +346,6 @@ pub(crate) mod prepared_publication_test_hooks {
|
||||
}
|
||||
}
|
||||
|
||||
/// Controlled application-test pause at an existing physical executor boundary.
|
||||
#[cfg(all(feature = "test-util", not(windows)))]
|
||||
pub struct LocalPublicationPause {
|
||||
_hook: prepared_publication_test_hooks::Guard,
|
||||
entered: oneshot::Receiver<()>,
|
||||
_release: std::sync::mpsc::Sender<()>,
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "test-util", not(windows)))]
|
||||
#[derive(Clone, Copy)]
|
||||
pub enum LocalPublicationStage {
|
||||
PreparedRename,
|
||||
Rename,
|
||||
Remove,
|
||||
}
|
||||
|
||||
#[cfg(all(feature = "test-util", not(windows)))]
|
||||
impl LocalPublicationPause {
|
||||
pub fn install(disk: &crate::disk::Disk, volume: &str, path: &str, stage: LocalPublicationStage) -> Result<Self> {
|
||||
let path = disk
|
||||
.get_object_path_for_io_if_local(volume, path)
|
||||
.ok_or(DiskError::DiskNotFound)??;
|
||||
let stage = match stage {
|
||||
LocalPublicationStage::PreparedRename => prepared_publication_test_hooks::Stage::PreparedRename,
|
||||
LocalPublicationStage::Rename => prepared_publication_test_hooks::Stage::Rename,
|
||||
LocalPublicationStage::Remove => prepared_publication_test_hooks::Stage::Remove,
|
||||
};
|
||||
let (entered_tx, entered) = oneshot::channel();
|
||||
let (release, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let hook = prepared_publication_test_hooks::install_at(stage, &path, move || {
|
||||
let _ = entered_tx.send(());
|
||||
let _ = release_rx.recv();
|
||||
});
|
||||
Ok(Self {
|
||||
_hook: hook,
|
||||
entered,
|
||||
_release: release,
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn entered(&mut self) -> std::result::Result<(), oneshot::error::RecvError> {
|
||||
(&mut self.entered).await
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, windows))]
|
||||
pub(crate) mod windows_rename_test_hooks {
|
||||
use super::*;
|
||||
@@ -474,6 +442,10 @@ pub fn fsync_dir_std(dir: impl AsRef<Path>) -> io::Result<()> {
|
||||
fsync_dir_recorder::record(dir.as_ref());
|
||||
#[cfg(unix)]
|
||||
{
|
||||
#[cfg(test)]
|
||||
if let Some(kind) = fsync_dir_recorder::take_failure(dir.as_ref()) {
|
||||
return Err(io::Error::from(kind));
|
||||
}
|
||||
std::fs::File::open(dir.as_ref())?.sync_all()?;
|
||||
}
|
||||
#[cfg(not(unix))]
|
||||
@@ -2004,7 +1976,7 @@ pub(crate) async fn remove_file_with_owner(
|
||||
let path = path.as_ref().to_path_buf();
|
||||
let lease = acquire_namespace_mutation_lease_with_owner(&path, namespace_owner).await;
|
||||
run_blocking_namespace_operation(lease, move || {
|
||||
#[cfg(all(any(test, feature = "test-util"), not(windows)))]
|
||||
#[cfg(all(test, not(windows)))]
|
||||
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Remove, &path);
|
||||
std::fs::remove_file(path)
|
||||
})
|
||||
@@ -2264,7 +2236,7 @@ pub(crate) async fn rename_all_with_prepared_source(
|
||||
move || {
|
||||
validate_prepared_rename_source(&prepared_source, &src_file_path)?;
|
||||
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
#[cfg(test)]
|
||||
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::PreparedRename, &dst_file_path);
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation)
|
||||
}
|
||||
@@ -2397,7 +2369,7 @@ async fn reliable_rename_inner_with_lease(
|
||||
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
#[cfg(all(test, not(windows)))]
|
||||
prepared_publication_test_hooks::run_rename_destination(&src_file_path, &dst_file_path);
|
||||
#[cfg(all(any(test, feature = "test-util"), not(windows)))]
|
||||
#[cfg(all(test, not(windows)))]
|
||||
{
|
||||
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Rename, &src_file_path);
|
||||
prepared_publication_test_hooks::run(prepared_publication_test_hooks::Stage::Rename, &dst_file_path);
|
||||
|
||||
@@ -23,17 +23,59 @@ use s3s::S3ErrorCode;
|
||||
pub type Error = StorageError;
|
||||
pub type Result<T> = core::result::Result<T, Error>;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum PoolMetadataFailure {
|
||||
ReadUnavailable,
|
||||
RecoveryRequired,
|
||||
TransactionUnknown,
|
||||
FenceLost,
|
||||
}
|
||||
|
||||
impl PoolMetadataFailure {
|
||||
fn recovery_hint(self) -> &'static str {
|
||||
match self {
|
||||
Self::ReadUnavailable => "read unavailable; retry after the replicas are readable",
|
||||
Self::TransactionUnknown => "writes remain blocked pending fenced transaction recovery",
|
||||
Self::RecoveryRequired | Self::FenceLost => {
|
||||
"writes remain blocked after a recovery-required replica state; restart after all replicas are readable and consistent, with compatible formats"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub fn as_str(self) -> &'static str {
|
||||
match self {
|
||||
Self::ReadUnavailable => "read_unavailable",
|
||||
Self::RecoveryRequired => "recovery_required",
|
||||
Self::TransactionUnknown => "transaction_unknown",
|
||||
Self::FenceLost => "fence_lost",
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Local control-plane context. Keep the existing storage error wire codes;
|
||||
/// the HTTP boundary recognizes this typed source, not an error-message prefix.
|
||||
#[derive(Debug, Clone, thiserror::Error)]
|
||||
#[error("{operation}: pool metadata {hint} ({reason}, {phase}): {detail}", hint = kind.recovery_hint(), reason = kind.as_str(), detail = source.as_ref().map(ToString::to_string).unwrap_or_default())]
|
||||
pub struct PoolMetadataError {
|
||||
pub kind: PoolMetadataFailure,
|
||||
pub operation: String,
|
||||
pub phase: &'static str,
|
||||
pub since: time::OffsetDateTime,
|
||||
#[source]
|
||||
pub source: Option<std::sync::Arc<StorageError>>,
|
||||
}
|
||||
|
||||
/// Keeps high-cardinality diagnostic detail in the error source while making
|
||||
/// the rendered `io::Error` stable for quorum aggregation.
|
||||
#[derive(Debug)]
|
||||
struct StableIoContextError {
|
||||
message: &'static str,
|
||||
message: std::borrow::Cow<'static, str>,
|
||||
source: Box<dyn std::error::Error + Send + Sync>,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for StableIoContextError {
|
||||
fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
formatter.write_str(self.message)
|
||||
formatter.write_str(&self.message)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +90,7 @@ where
|
||||
E: Into<Box<dyn std::error::Error + Send + Sync>>,
|
||||
{
|
||||
std::io::Error::other(StableIoContextError {
|
||||
message,
|
||||
message: message.into(),
|
||||
source: source.into(),
|
||||
})
|
||||
}
|
||||
@@ -203,6 +245,19 @@ pub enum StorageError {
|
||||
NotFirstDisk,
|
||||
#[error("first disk wait")]
|
||||
FirstDiskWait,
|
||||
#[error(
|
||||
"unsupported pool expansion: an existing single-node single-drive (SNSD) deployment cannot be expanded in place (configured {configured_drives} drive endpoints); restart with the original single local path, or create a new multi-drive deployment and migrate data through S3"
|
||||
)]
|
||||
UnsupportedSnsdExpansion { configured_drives: usize },
|
||||
#[error(
|
||||
"pool topology mismatch: stored {stored_drives} drives with {stored_set_drive_count} drives per erasure set, configured {configured_drives} drives with {configured_set_drive_count} drives per erasure set; an existing pool's drive count and erasure set width cannot be changed in place; restore its original endpoints and RUSTFS_ERASURE_SET_DRIVE_COUNT setting; to expand a multi-drive deployment, append a new pool with at least 2 drive endpoints"
|
||||
)]
|
||||
PoolTopologyMismatch {
|
||||
stored_drives: usize,
|
||||
stored_set_drive_count: usize,
|
||||
configured_drives: usize,
|
||||
configured_set_drive_count: usize,
|
||||
},
|
||||
|
||||
// ── Operational ──────────────────────────────────────────────────
|
||||
#[error("Storage reached its minimum free drive threshold.")]
|
||||
@@ -287,6 +342,22 @@ impl From<crate::erasure::coding::ErasureConstructionError> for StorageError {
|
||||
}
|
||||
|
||||
impl StorageError {
|
||||
pub fn pool_metadata_failure(&self) -> Option<&PoolMetadataError> {
|
||||
let mut current: Option<&(dyn std::error::Error + 'static)> = Some(self);
|
||||
while let Some(error) = current {
|
||||
if let Some(context) = error.downcast_ref::<PoolMetadataError>() {
|
||||
return Some(context);
|
||||
}
|
||||
// io::Error::source skips its boxed context itself.
|
||||
current = if let Some(io) = error.downcast_ref::<std::io::Error>() {
|
||||
io.get_ref().map(|inner| inner as &(dyn std::error::Error + 'static))
|
||||
} else {
|
||||
error.source()
|
||||
};
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
pub fn other<E>(error: E) -> Self
|
||||
where
|
||||
E: Into<Box<dyn std::error::Error + Send + Sync>>,
|
||||
@@ -517,6 +588,7 @@ impl From<rustfs_filemeta::Error> for StorageError {
|
||||
rustfs_filemeta::Error::FileVersionNotFound => StorageError::FileVersionNotFound,
|
||||
rustfs_filemeta::Error::FileCorrupt => StorageError::FileCorrupt,
|
||||
rustfs_filemeta::Error::Unexpected => StorageError::Unexpected,
|
||||
rustfs_filemeta::Error::MaxVersionsExceeded => StorageError::MaxVersionsExceeded,
|
||||
rustfs_filemeta::Error::Io(io_error) => io_error.into(),
|
||||
_ => StorageError::Io(std::io::Error::other(e)),
|
||||
}
|
||||
@@ -535,7 +607,19 @@ impl PartialEq for StorageError {
|
||||
impl Clone for StorageError {
|
||||
fn clone(&self) -> Self {
|
||||
match self {
|
||||
StorageError::Io(e) => StorageError::Io(std::io::Error::new(e.kind(), e.to_string())),
|
||||
StorageError::Io(e) => {
|
||||
if let Some(context) = self.pool_metadata_failure() {
|
||||
Self::Io(std::io::Error::new(
|
||||
e.kind(),
|
||||
StableIoContextError {
|
||||
message: e.to_string().into(),
|
||||
source: Box::new(context.clone()),
|
||||
},
|
||||
))
|
||||
} else {
|
||||
StorageError::Io(std::io::Error::new(e.kind(), e.to_string()))
|
||||
}
|
||||
}
|
||||
StorageError::FaultyDisk => StorageError::FaultyDisk,
|
||||
StorageError::DiskFull => StorageError::DiskFull,
|
||||
StorageError::VolumeNotFound => StorageError::VolumeNotFound,
|
||||
@@ -629,6 +713,20 @@ impl Clone for StorageError {
|
||||
StorageError::ErasureWriteQuorum => StorageError::ErasureWriteQuorum,
|
||||
StorageError::NotFirstDisk => StorageError::NotFirstDisk,
|
||||
StorageError::FirstDiskWait => StorageError::FirstDiskWait,
|
||||
StorageError::UnsupportedSnsdExpansion { configured_drives } => StorageError::UnsupportedSnsdExpansion {
|
||||
configured_drives: *configured_drives,
|
||||
},
|
||||
StorageError::PoolTopologyMismatch {
|
||||
stored_drives,
|
||||
stored_set_drive_count,
|
||||
configured_drives,
|
||||
configured_set_drive_count,
|
||||
} => StorageError::PoolTopologyMismatch {
|
||||
stored_drives: *stored_drives,
|
||||
stored_set_drive_count: *stored_set_drive_count,
|
||||
configured_drives: *configured_drives,
|
||||
configured_set_drive_count: *configured_set_drive_count,
|
||||
},
|
||||
StorageError::TooManyOpenFiles => StorageError::TooManyOpenFiles,
|
||||
StorageError::NoHealRequired => StorageError::NoHealRequired,
|
||||
StorageError::Lock(e) => StorageError::Lock(e.clone()),
|
||||
@@ -662,7 +760,8 @@ impl Clone for StorageError {
|
||||
}
|
||||
|
||||
impl StorageError {
|
||||
fn code(&self) -> StorageErrorCode {
|
||||
/// Stable classification without error payloads or storage paths.
|
||||
pub fn code(&self) -> StorageErrorCode {
|
||||
match self {
|
||||
StorageError::Io(_) => StorageErrorCode::Io,
|
||||
StorageError::FaultyDisk => StorageErrorCode::FaultyDisk,
|
||||
@@ -735,6 +834,11 @@ impl StorageError {
|
||||
StorageError::ErasureWriteQuorum => StorageErrorCode::ErasureWriteQuorum,
|
||||
StorageError::NotFirstDisk => StorageErrorCode::NotFirstDisk,
|
||||
StorageError::FirstDiskWait => StorageErrorCode::FirstDiskWait,
|
||||
// Topology diagnostics reuse the existing wire code; they are
|
||||
// not disk errors and must retain their local identity for retry classification.
|
||||
StorageError::UnsupportedSnsdExpansion { .. } | StorageError::PoolTopologyMismatch { .. } => {
|
||||
StorageErrorCode::InvalidArgument
|
||||
}
|
||||
StorageError::ConfigNotFound => StorageErrorCode::ConfigNotFound,
|
||||
StorageError::TooManyOpenFiles => StorageErrorCode::TooManyOpenFiles,
|
||||
StorageError::NoHealRequired => StorageErrorCode::NoHealRequired,
|
||||
@@ -1215,6 +1319,29 @@ mod tests {
|
||||
use super::*;
|
||||
use std::io::{Error as IoError, ErrorKind};
|
||||
|
||||
#[test]
|
||||
fn startup_topology_errors_preserve_identity_and_guidance() {
|
||||
for error in [
|
||||
StorageError::UnsupportedSnsdExpansion { configured_drives: 4 },
|
||||
StorageError::PoolTopologyMismatch {
|
||||
stored_drives: 4,
|
||||
stored_set_drive_count: 4,
|
||||
configured_drives: 8,
|
||||
configured_set_drive_count: 8,
|
||||
},
|
||||
] {
|
||||
let io_error: IoError = error.clone().into();
|
||||
let restored = StorageError::from(io_error);
|
||||
assert_eq!(std::mem::discriminant(&restored), std::mem::discriminant(&error));
|
||||
assert_eq!(restored.to_string(), error.to_string());
|
||||
assert_eq!(restored.code(), StorageErrorCode::InvalidArgument);
|
||||
assert!(
|
||||
restored.narrow_to_disk().is_err(),
|
||||
"startup diagnostics must not become disk/quorum errors"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn other_preserves_erasure_construction_source_chain() {
|
||||
use crate::erasure::coding::ErasureConstructionError;
|
||||
|
||||
@@ -25,6 +25,20 @@ pub(crate) const MAX_ERASURE_SET_DRIVE_COUNT: usize = 16;
|
||||
const SET_SIZES: [usize; 15] = [2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, MAX_ERASURE_SET_DRIVE_COUNT];
|
||||
const ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT: &str = "RUSTFS_ERASURE_SET_DRIVE_COUNT";
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
enum PoolDriveCountError {
|
||||
#[error(
|
||||
"Incorrect number of endpoints provided, size {size}; an erasure pool requires at least {} drive endpoints on one or more nodes; for a standalone single-drive deployment, use a single local path without ellipses",
|
||||
SET_SIZES[0]
|
||||
)]
|
||||
BelowMinimum { size: usize },
|
||||
#[error(
|
||||
"Incorrect number of endpoints provided, size {size}; {}={set_drive_count} requires at least {set_drive_count} drive endpoints per pool",
|
||||
ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT
|
||||
)]
|
||||
BelowSetWidth { size: usize, set_drive_count: usize },
|
||||
}
|
||||
|
||||
#[derive(Deserialize, Debug, Default)]
|
||||
pub struct PoolDisksLayout {
|
||||
cmd_line: String,
|
||||
@@ -132,7 +146,7 @@ impl DisksLayout {
|
||||
for arg in args.iter() {
|
||||
if !has_ellipses(&[arg]) && args.len() > 1 {
|
||||
return Err(Error::other(
|
||||
"all args must have ellipses for pool expansion (Invalid arguments specified)",
|
||||
"all args must have ellipses for pool expansion (Invalid arguments specified); each pool must expand to at least 2 drive endpoints on one or more nodes; a single-drive pool cannot be added to a multi-pool deployment",
|
||||
));
|
||||
}
|
||||
|
||||
@@ -396,9 +410,11 @@ fn get_set_indexes<T: AsRef<str>>(
|
||||
}
|
||||
|
||||
for &size in total_sizes {
|
||||
// Check if total_sizes has minimum range upto set_size
|
||||
if size < SET_SIZES[0] || size < set_drive_count {
|
||||
return Err(Error::other(format!("Incorrect number of endpoints provided, size {size}")));
|
||||
if size < SET_SIZES[0] {
|
||||
return Err(Error::other(PoolDriveCountError::BelowMinimum { size }));
|
||||
}
|
||||
if size < set_drive_count {
|
||||
return Err(Error::other(PoolDriveCountError::BelowSetWidth { size, set_drive_count }));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -707,7 +723,7 @@ mod test {
|
||||
arg: "http://rustfs{2...3}/export/set{1...0}",
|
||||
..Default::default()
|
||||
},
|
||||
// Range cannot be smaller than 4 minimum.
|
||||
// Ranges must use three dots.
|
||||
TestCase {
|
||||
num: 4,
|
||||
arg: "/export{1..2}",
|
||||
@@ -926,11 +942,146 @@ mod test {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pool_expansion_accepts_single_node_multi_drive_pools() {
|
||||
temp_env::with_var(ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT, Some("0"), || {
|
||||
for (volumes, drives) in [
|
||||
(["http://node1:9000/data{1...2}", "http://node2:9000/data{1...2}"], 2),
|
||||
(["http://node1:9000/data{1...4}", "http://node2:9000/data{1...4}"], 4),
|
||||
(["http://node{1...4}:9000/data", "http://node5:9000/data{1...4}"], 4),
|
||||
(["http://node5:9000/data{1...4}", "http://node{1...4}:9000/data"], 4),
|
||||
] {
|
||||
let layout = DisksLayout::from_volumes(&volumes).expect("single-node multi-drive pools are valid");
|
||||
|
||||
assert!(!layout.legacy);
|
||||
assert_eq!(layout.pools.len(), 2);
|
||||
for (index, volume) in volumes.iter().enumerate() {
|
||||
assert_eq!(layout.get_set_count(index), 1);
|
||||
assert_eq!(layout.get_drives_per_set(index), drives);
|
||||
assert_eq!(layout.get_cmd_line(index), *volume);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pool_expansion_accepts_multi_node_single_drive_pools() {
|
||||
temp_env::with_var(ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT, Some("0"), || {
|
||||
for nodes in [2, 3, 4] {
|
||||
let volumes = [
|
||||
format!("http://pool1-node{{1...{nodes}}}:9000/data"),
|
||||
format!("http://pool2-node{{1...{nodes}}}:9000/data"),
|
||||
];
|
||||
let layout = DisksLayout::from_volumes(&volumes).expect("each node may contribute one drive to a pool");
|
||||
|
||||
assert_eq!(layout.pools.len(), 2);
|
||||
for pool in 0..2 {
|
||||
assert_eq!(layout.get_set_count(pool), 1);
|
||||
assert_eq!(layout.get_drives_per_set(pool), nodes);
|
||||
let expected = (1..=nodes)
|
||||
.map(|node| format!("http://pool{}-node{node}:9000/data", pool + 1))
|
||||
.collect::<Vec<_>>();
|
||||
assert_eq!(layout.pools[pool].layout, vec![expected]);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn explicit_endpoints_without_ellipses_form_one_pool() {
|
||||
temp_env::with_var(ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT, Some("0"), || {
|
||||
let volumes = ["http://node1:9000/data", "http://node2:9000/data"];
|
||||
let layout = DisksLayout::from_volumes(&volumes).expect("explicit endpoints form one legacy pool");
|
||||
|
||||
assert!(layout.legacy);
|
||||
assert_eq!(layout.pools.len(), 1);
|
||||
assert_eq!(layout.pools[0].layout, vec![volumes.to_vec()]);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn standalone_single_drive_path_remains_supported() {
|
||||
temp_env::with_var(ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT, Some("0"), || {
|
||||
let layout = DisksLayout::from_volumes(&["/data"]).expect("standalone single-drive deployment is valid");
|
||||
|
||||
assert!(layout.is_single_drive_layout());
|
||||
assert_eq!(layout.get_single_drive_layout(), "/data");
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pool_expansion_rejects_plain_single_drive_pool_with_notice() {
|
||||
temp_env::with_var(ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT, Some("0"), || {
|
||||
for volumes in [
|
||||
["http://node{1...2}:9000/data", "http://node3:9000/data"],
|
||||
["http://node3:9000/data", "http://node{1...2}:9000/data"],
|
||||
] {
|
||||
let err = DisksLayout::from_volumes(&volumes).expect_err("a plain endpoint cannot be an expansion pool");
|
||||
let message = err.to_string();
|
||||
|
||||
assert!(message.contains("all args must have ellipses for pool expansion"), "{message}");
|
||||
assert!(message.contains("at least 2 drive endpoints"), "{message}");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pool_expansion_rejects_singleton_ellipsis_pool_with_notice() {
|
||||
temp_env::with_var(ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT, Some("0"), || {
|
||||
for singleton in ["http://node{3...3}:9000/data", "http://node3:9000/data{1...1}"] {
|
||||
for volumes in [
|
||||
vec!["http://node{1...2}:9000/data", singleton],
|
||||
vec![singleton, "http://node{1...2}:9000/data"],
|
||||
vec![singleton],
|
||||
] {
|
||||
let err = DisksLayout::from_volumes(&volumes).expect_err("a singleton range still contains one drive");
|
||||
let message = err.to_string();
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::Other);
|
||||
assert!(matches!(
|
||||
err.get_ref().and_then(|source| source.downcast_ref::<PoolDriveCountError>()),
|
||||
Some(PoolDriveCountError::BelowMinimum { size: 1 })
|
||||
));
|
||||
assert!(message.contains("at least 2 drive endpoints"), "{message}");
|
||||
assert!(message.contains("single local path without ellipses"), "{message}");
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn explicit_set_size_counts_drives_not_nodes() {
|
||||
for volume in ["http://node1:9000/data{1...4}", "http://node{1...4}:9000/data"] {
|
||||
let sets = get_all_sets(2, true, &[volume]).expect("four endpoints can form two two-drive sets");
|
||||
assert_eq!(sets.iter().map(Vec::len).collect::<Vec<_>>(), vec![2, 2]);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn undersized_pool_error_identifies_requested_set_size() {
|
||||
let err =
|
||||
get_all_sets(4, true, &["http://node{1...2}:9000/data"]).expect_err("two endpoints cannot fill a four-drive set");
|
||||
let message = err.to_string();
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::Other);
|
||||
assert!(matches!(
|
||||
err.get_ref().and_then(|source| source.downcast_ref::<PoolDriveCountError>()),
|
||||
Some(PoolDriveCountError::BelowSetWidth {
|
||||
size: 2,
|
||||
set_drive_count: 4
|
||||
})
|
||||
));
|
||||
assert!(message.contains("size 2"), "{message}");
|
||||
assert!(message.contains("RUSTFS_ERASURE_SET_DRIVE_COUNT=4"), "{message}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn layout_errors_do_not_echo_url_credentials() {
|
||||
for volumes in [
|
||||
vec!["http://:duplicate-secret@server/path", "http://:duplicate-secret@server/path"],
|
||||
vec!["http://:ellipsis...secret@server/path"],
|
||||
vec!["http://server{1...2}/data", "http://:plain-secret@server3/data"],
|
||||
vec!["http://server{1...2}/data", "http://:singleton-secret@server{3...3}/data"],
|
||||
] {
|
||||
let err = DisksLayout::from_volumes(&volumes).unwrap_err();
|
||||
assert!(!err.to_string().contains("secret"), "layout error leaked endpoint credentials: {err}");
|
||||
|
||||
@@ -2432,6 +2432,41 @@ mod test {
|
||||
assert_eq!(local_endpoints[0].pool_idx, 1);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn pool_expansion_resolves_single_node_multi_drive_and_multi_node_single_drive_pools() {
|
||||
for (additional_pool, expected_nodes) in [
|
||||
("http://rustfs-5.example.invalid:9000/data{1...4}", 5),
|
||||
("http://rustfs-{5...8}.example.invalid:9000/data", 8),
|
||||
] {
|
||||
let layout = temp_env::with_var("RUSTFS_ERASURE_SET_DRIVE_COUNT", Some("0"), || {
|
||||
DisksLayout::from_volumes(&["http://rustfs-{1...4}.example.invalid:9000/data", additional_pool])
|
||||
})
|
||||
.expect("both single-node multi-drive and multi-node single-drive pools should parse");
|
||||
|
||||
let (pools, setup_type) = EndpointServerPools::create_server_endpoints_with(
|
||||
"0.0.0.0:9000",
|
||||
&layout,
|
||||
Some(orchestrated_test_policy()),
|
||||
Some("rustfs-1.example.invalid"),
|
||||
)
|
||||
.await
|
||||
.expect("pool admission must not impose a minimum node count or drives per node");
|
||||
|
||||
assert_eq!(setup_type, SetupType::DistErasure);
|
||||
assert_eq!(pools.0.len(), 2);
|
||||
assert_eq!(pools.get_nodes().len(), expected_nodes);
|
||||
for (pool_index, pool) in (0_i32..).zip(&pools.0) {
|
||||
assert_eq!((pool.set_count, pool.drives_per_set), (1, 4));
|
||||
assert_eq!(pool.endpoints.as_ref().len(), 4);
|
||||
for (disk_index, endpoint) in (0_i32..).zip(pool.endpoints.as_ref()) {
|
||||
assert_eq!(endpoint.pool_idx, pool_index);
|
||||
assert_eq!(endpoint.set_idx, 0);
|
||||
assert_eq!(endpoint.disk_idx, disk_index);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn explicit_local_endpoint_host_fails_closed_for_invalid_context_or_zero_match() {
|
||||
let args = vec![
|
||||
|
||||
@@ -55,6 +55,8 @@ mod set_disk;
|
||||
mod storage_api_contracts;
|
||||
mod store;
|
||||
|
||||
pub use store::PoolMetaWriteGateStatus;
|
||||
|
||||
// pub mod checksum;
|
||||
mod event;
|
||||
|
||||
|
||||
@@ -2278,6 +2278,121 @@ mod tests {
|
||||
assert_eq!(read, fixture.plaintext, "SSE-C + compression full GET must reassemble all parts");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn multipart_empty_tail_full_reads_preserve_plaintext() {
|
||||
let key = [0x6Eu8; 32];
|
||||
let part_sizes = [5 * 1024 * 1024, 0];
|
||||
let encrypted = build_legacy_ssec_multipart_fixture(key, &part_sizes).await;
|
||||
for (kind, mut fixture, headers) in [
|
||||
(
|
||||
"encrypted",
|
||||
CompressedMultipartFixture {
|
||||
object_info: encrypted.object_info,
|
||||
stored: encrypted.ciphertext,
|
||||
plaintext: encrypted.plaintext,
|
||||
},
|
||||
ssec_headers_from_key(key),
|
||||
),
|
||||
("compressed", compressed_multipart_fixture(&part_sizes).await, HeaderMap::new()),
|
||||
(
|
||||
"compressed and encrypted",
|
||||
compressed_encrypted_multipart_fixture(key, &part_sizes).await,
|
||||
ssec_headers_from_key(key),
|
||||
),
|
||||
] {
|
||||
fixture.object_info.etag = Some(faster_hex::hex_string(Md5::digest(&fixture.plaintext).as_ref()));
|
||||
assert_eq!(fixture.object_info.etag.as_ref().expect("source ETag").len(), 32);
|
||||
assert_eq!(fixture.object_info.parts.len(), 2);
|
||||
let tail = &fixture.object_info.parts[1];
|
||||
assert_eq!(tail.actual_size, 0, "{kind}: final part has no plaintext");
|
||||
if kind == "compressed" {
|
||||
assert_eq!(tail.size, 0, "unpadded compression emits no bytes for an empty part");
|
||||
} else {
|
||||
assert!(tail.size > 0, "{kind}: the empty part still has a stored frame");
|
||||
}
|
||||
let stored_size = i64::try_from(fixture.stored.len()).expect("fixture size fits i64");
|
||||
let (mut reader, offset, length) = GetObjectReader::new(
|
||||
Box::new(Cursor::new(fixture.stored)),
|
||||
None,
|
||||
&fixture.object_info,
|
||||
&ObjectOptions::default(),
|
||||
&headers,
|
||||
)
|
||||
.await
|
||||
.expect("full transformed read must include the empty tail");
|
||||
assert_eq!((offset, length), (0, stored_size), "{kind}: full read includes all stored parts");
|
||||
let mut body = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut body)
|
||||
.await
|
||||
.expect("read through the complete decoder EOF");
|
||||
assert_eq!(body, fixture.plaintext, "{kind}: no plaintext is added or lost by the empty tail");
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn multipart_empty_tail_full_read_authenticates_v2_final_frame() {
|
||||
let key = [0x6Eu8; 32];
|
||||
let plaintext = legacy_fixture_part_plaintext(1, 5 * 1024 * 1024);
|
||||
let mut ciphertext = Vec::new();
|
||||
let mut parts = Vec::new();
|
||||
for (number, body) in [(1, plaintext.as_slice()), (2, b"".as_slice())] {
|
||||
let start = ciphertext.len();
|
||||
rustfs_rio::EncryptReader::new_multipart_v2(Cursor::new(body), key, LEGACY_FIXTURE_BASE_NONCE, number)
|
||||
.read_to_end(&mut ciphertext)
|
||||
.await
|
||||
.expect("encrypt a v2 fixture part with an authenticated final frame");
|
||||
parts.push(ObjectPartInfo {
|
||||
number,
|
||||
size: ciphertext.len() - start,
|
||||
actual_size: i64::try_from(body.len()).expect("fixture plaintext size fits"),
|
||||
..Default::default()
|
||||
});
|
||||
}
|
||||
let tail_start = parts[0].size;
|
||||
assert_eq!(parts[1].actual_size, 0);
|
||||
assert!(parts[1].size > 8, "the empty final frame carries more than an END marker");
|
||||
let object_info = ObjectInfo {
|
||||
bucket: "bucket".to_string(),
|
||||
name: "v2-empty-tail".to_string(),
|
||||
size: i64::try_from(ciphertext.len()).expect("fixture ciphertext size fits"),
|
||||
etag: Some(faster_hex::hex_string(Md5::digest(&plaintext).as_ref())),
|
||||
parts: Arc::new(parts),
|
||||
user_defined: Arc::new(legacy_ssec_multipart_metadata(key, plaintext.len())),
|
||||
..Default::default()
|
||||
};
|
||||
for corrupt_tail in [false, true] {
|
||||
let mut stored = ciphertext.clone();
|
||||
if corrupt_tail {
|
||||
// The v2 header is authenticated associated data, including
|
||||
// the header of a final frame containing zero plaintext.
|
||||
stored[tail_start + 5] ^= 1;
|
||||
}
|
||||
let (mut reader, offset, length) = GetObjectReader::new(
|
||||
Box::new(Cursor::new(stored)),
|
||||
None,
|
||||
&object_info,
|
||||
&ObjectOptions::default(),
|
||||
&ssec_headers_from_key(key),
|
||||
)
|
||||
.await
|
||||
.expect("construct the full reader before consuming the final frame");
|
||||
assert_eq!((offset, length), (0, object_info.size));
|
||||
let result = tokio::io::copy(&mut reader.stream, &mut tokio::io::sink()).await;
|
||||
if corrupt_tail {
|
||||
let err = result.expect_err("EOF must authenticate the empty final frame after all plaintext is returned");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||
assert_eq!(err.to_string(), "v2 encrypted frame failed authentication");
|
||||
} else {
|
||||
assert_eq!(
|
||||
result.expect("valid empty final frame must reach EOF"),
|
||||
u64::try_from(plaintext.len()).expect("plaintext length fits")
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn compressed_encrypted_multipart_range_crosses_part_boundary() {
|
||||
let key_bytes = [0x6Eu8; 32];
|
||||
@@ -3656,6 +3771,61 @@ mod tests {
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn multipart_full_read_preserves_legacy_zero_and_negative_part_sizes() {
|
||||
let key = [0x77; 32];
|
||||
let part_sizes = [5 * 1024 * 1024, 1024 * 1024];
|
||||
let encrypted = build_legacy_ssec_multipart_fixture(key, &part_sizes).await;
|
||||
// The encrypted case supplies the fixture key explicitly. This covers
|
||||
// full decrypted reads, not managed-key acquisition.
|
||||
for (kind, fixture, headers) in [
|
||||
("compressed", compressed_multipart_fixture(&part_sizes).await, HeaderMap::new()),
|
||||
(
|
||||
"encrypted with supplied key",
|
||||
CompressedMultipartFixture {
|
||||
object_info: encrypted.object_info,
|
||||
stored: encrypted.ciphertext,
|
||||
plaintext: encrypted.plaintext,
|
||||
},
|
||||
ssec_headers_from_key(key),
|
||||
),
|
||||
] {
|
||||
let source_etag = faster_hex::hex_string(Md5::digest(&fixture.plaintext).as_ref());
|
||||
assert_eq!(source_etag.len(), 32);
|
||||
assert_eq!(fixture.plaintext.len(), 6 * 1024 * 1024);
|
||||
for part_index in 0..part_sizes.len() {
|
||||
assert!(fixture.object_info.parts[part_index].actual_size > 0, "the selected part is nonempty");
|
||||
for actual_size in [0, -1] {
|
||||
let mut object_info = fixture.object_info.clone();
|
||||
object_info.etag = Some(source_etag.clone());
|
||||
Arc::make_mut(&mut object_info.parts)[part_index].actual_size = actual_size;
|
||||
let (mut reader, offset, length) = GetObjectReader::new(
|
||||
Box::new(Cursor::new(fixture.stored.clone())),
|
||||
None,
|
||||
&object_info,
|
||||
&ObjectOptions::default(),
|
||||
&headers,
|
||||
)
|
||||
.await
|
||||
.expect("the authoritative total size must keep full legacy reads available");
|
||||
assert_eq!(offset, 0);
|
||||
assert_eq!(length, i64::try_from(fixture.stored.len()).expect("stored size fits"));
|
||||
let mut body = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut body)
|
||||
.await
|
||||
.expect("full read must reach EOF despite an unspecified per-part logical size");
|
||||
assert_eq!(
|
||||
body, fixture.plaintext,
|
||||
"{kind}: part {part_index} with actual_size={actual_size} must not lose readable data"
|
||||
);
|
||||
assert_eq!(reader.object_info.etag.as_deref(), Some(source_etag.as_str()));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The physical part sizes must add up to `oi.size` for a seek to be safe;
|
||||
/// inconsistent metadata must fall back to the previous full-object read
|
||||
/// instead of scheduling an erasure read past the object end.
|
||||
|
||||
@@ -1597,7 +1597,7 @@ impl ObjectInfo {
|
||||
}
|
||||
|
||||
pub fn is_multipart(&self) -> bool {
|
||||
self.etag.as_ref().is_some_and(|v| v.len() != 32)
|
||||
self.parts.len() > 1 || self.etag.as_ref().is_some_and(|v| v.len() != 32)
|
||||
}
|
||||
|
||||
pub fn is_encrypted(&self) -> bool {
|
||||
@@ -2235,6 +2235,35 @@ mod tests {
|
||||
}
|
||||
use rustfs_filemeta::{FileInfo, FileMeta, MetaCacheEntry, TRANSITION_COMPLETE};
|
||||
|
||||
#[test]
|
||||
fn multipart_identity_uses_stored_parts_and_preserves_the_etag_fallback() {
|
||||
let plain_etag = "0123456789abcdef0123456789abcdef";
|
||||
let multipart_etag = "0123456789abcdef0123456789abcdef-1";
|
||||
for (case, part_count, etag, expected) in [
|
||||
("preserved source ETag", 2, Some(plain_etag), true),
|
||||
("missing ETag", 2, None, true),
|
||||
("ordinary PUT", 1, Some(plain_etag), false),
|
||||
("ordinary PUT without ETag", 1, None, false),
|
||||
("single-part MPU", 1, Some(multipart_etag), true),
|
||||
("legacy MPU without parts", 0, Some(multipart_etag), true),
|
||||
] {
|
||||
let object = ObjectInfo {
|
||||
etag: etag.map(str::to_string),
|
||||
parts: Arc::new(
|
||||
(1..=part_count)
|
||||
.map(|number| ObjectPartInfo {
|
||||
number,
|
||||
..Default::default()
|
||||
})
|
||||
.collect(),
|
||||
),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
assert_eq!(object.is_multipart(), expected, "{case}");
|
||||
}
|
||||
}
|
||||
|
||||
fn inline_fast_path_object(size: i64, versioned: bool) -> ObjectInfo {
|
||||
ObjectInfo {
|
||||
size,
|
||||
|
||||
@@ -20,9 +20,10 @@ use chrono::Utc;
|
||||
use jiff::Timestamp;
|
||||
use rustfs_heal_contracts::heal_channel::DriveState;
|
||||
use rustfs_io_metrics::internode_metrics::global_internode_metrics;
|
||||
use rustfs_io_metrics::s3_http_metrics::s3_http_metrics_snapshot;
|
||||
use rustfs_madmin::metrics::{
|
||||
DiskIOStats, DiskMetric, LastMinute as MadminLastMinute, NetDevLine, NetMetrics, RPCMetrics, RealtimeMetrics,
|
||||
ScannerCheckpointReport as MadminScannerCheckpointReport,
|
||||
DiskIOStats, DiskMetric, HttpMetrics, HttpRequestMetric, LastMinute as MadminLastMinute, NetDevLine, NetMetrics, RPCMetrics,
|
||||
RealtimeMetrics, ScannerCheckpointReport as MadminScannerCheckpointReport,
|
||||
ScannerLifecycleExpirySnapshot as MadminScannerLifecycleExpirySnapshot,
|
||||
ScannerLifecycleTransitionSnapshot as MadminScannerLifecycleTransitionSnapshot,
|
||||
ScannerMaintenanceControlSnapshot as MadminScannerMaintenanceControlSnapshot,
|
||||
@@ -61,9 +62,10 @@ impl MetricType {
|
||||
pub const MEM: MetricType = MetricType(1 << 6);
|
||||
pub const CPU: MetricType = MetricType(1 << 7);
|
||||
pub const RPC: MetricType = MetricType(1 << 8);
|
||||
pub const HTTP: MetricType = MetricType(1 << 9);
|
||||
|
||||
// MetricsAll must be last.
|
||||
pub const ALL: MetricType = MetricType((1 << 9) - 1);
|
||||
pub const ALL: MetricType = MetricType((1 << 10) - 1);
|
||||
|
||||
pub fn new(t: u32) -> Self {
|
||||
Self(t)
|
||||
@@ -410,6 +412,21 @@ pub async fn collect_local_metrics(types: MetricType, opts: &CollectMetricsOpts)
|
||||
by_host_name = local_node_name;
|
||||
}
|
||||
|
||||
if types.contains(&MetricType::HTTP) {
|
||||
real_time_metrics.aggregated.http = Some(HttpMetrics {
|
||||
collected_at: Timestamp::now(),
|
||||
requests: s3_http_metrics_snapshot()
|
||||
.into_iter()
|
||||
.map(|series| HttpRequestMetric {
|
||||
method: series.method.to_string(),
|
||||
operation: series.operation.to_string(),
|
||||
outcome: series.outcome.to_string(),
|
||||
total: series.total,
|
||||
})
|
||||
.collect(),
|
||||
});
|
||||
}
|
||||
|
||||
if types.contains(&MetricType::DISK) {
|
||||
debug!("start get disk metrics");
|
||||
let mut aggr = DiskMetric {
|
||||
@@ -585,11 +602,47 @@ mod test {
|
||||
assert!(t.contains(&MetricType::MEM));
|
||||
assert!(t.contains(&MetricType::CPU));
|
||||
assert!(t.contains(&MetricType::RPC));
|
||||
assert!(t.contains(&MetricType::HTTP));
|
||||
|
||||
let disk = MetricType::new(1 << 1);
|
||||
assert!(disk.contains(&MetricType::DISK));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn collect_local_metrics_reports_the_same_http_outcome_counters() {
|
||||
let mut request = rustfs_io_metrics::s3_http_metrics::S3HttpRequestGuard::new("PUT");
|
||||
request.response(503);
|
||||
drop(request);
|
||||
let snapshot = s3_http_metrics_snapshot();
|
||||
let realtime = collect_local_metrics(MetricType::HTTP, &CollectMetricsOpts::default()).await;
|
||||
let http = realtime.aggregated.http.as_ref().expect("HTTP selection must report support");
|
||||
assert_eq!(http.requests.len(), snapshot.len());
|
||||
for (actual, expected) in http.requests.iter().zip(&snapshot) {
|
||||
assert_eq!(actual.method, expected.method);
|
||||
assert_eq!(actual.operation, expected.operation);
|
||||
assert_eq!(actual.outcome, expected.outcome);
|
||||
assert_eq!(actual.total, expected.total);
|
||||
}
|
||||
assert_eq!(realtime.by_host.len(), 1);
|
||||
assert_eq!(
|
||||
realtime
|
||||
.by_host
|
||||
.values()
|
||||
.next()
|
||||
.expect("local host")
|
||||
.http
|
||||
.as_ref()
|
||||
.expect("host HTTP")
|
||||
.requests,
|
||||
http.requests
|
||||
);
|
||||
let encoded = rmp_serde::to_vec_named(&realtime).expect("RPC metric map");
|
||||
let decoded: RealtimeMetrics = rmp_serde::from_slice(&encoded).expect("RPC metric roundtrip");
|
||||
assert_eq!(decoded.aggregated.http.expect("HTTP field survives RPC").requests, http.requests);
|
||||
let excluded = collect_local_metrics(MetricType::NET, &CollectMetricsOpts::default()).await;
|
||||
assert!(excluded.aggregated.http.is_none());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn collect_local_metrics_reports_internode_net_and_rpc() {
|
||||
let metrics = global_internode_metrics();
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -300,11 +300,13 @@ pub(super) fn ensure_rebalance_worker_active(meta: Option<&RebalanceMeta>, expec
|
||||
let Some(meta) = meta else {
|
||||
return Err(rebalance_metadata_not_initialized_error(stage));
|
||||
};
|
||||
if meta.stopped_at.is_some()
|
||||
|| meta
|
||||
.cancel
|
||||
.as_ref()
|
||||
.is_some_and(tokio_util::sync::CancellationToken::is_cancelled)
|
||||
if meta.stopped_at.is_some() || meta.stop_requested {
|
||||
return Err(Error::OperationCanceled);
|
||||
}
|
||||
if meta
|
||||
.cancel
|
||||
.as_ref()
|
||||
.is_some_and(tokio_util::sync::CancellationToken::is_cancelled)
|
||||
|| !is_rebalance_conflicting_with_decommission(meta)
|
||||
{
|
||||
return Err(Error::other(format!("inactive rebalance worker rejected during {stage}: {expected_id}")));
|
||||
@@ -629,6 +631,7 @@ impl ECStore {
|
||||
if let Some(meta) = rebalance_meta.as_mut()
|
||||
&& is_rebalance_conflicting_with_decommission(meta)
|
||||
{
|
||||
meta.stop_requested = true;
|
||||
meta.cancel
|
||||
.get_or_insert_with(tokio_util::sync::CancellationToken::new)
|
||||
.cancel();
|
||||
@@ -643,12 +646,13 @@ impl ECStore {
|
||||
let Some(meta) = rebalance_meta.as_mut() else {
|
||||
return Ok(None);
|
||||
};
|
||||
if !is_rebalance_conflicting_with_decommission(meta) {
|
||||
if meta.stopped_at.is_some() || (!is_rebalance_conflicting_with_decommission(meta) && !meta.stop_requested) {
|
||||
return Ok(None);
|
||||
}
|
||||
if meta.id.is_empty() {
|
||||
return Err(Error::other("active rebalance metadata has no activation id"));
|
||||
}
|
||||
meta.stop_requested = true;
|
||||
meta.cancel
|
||||
.get_or_insert_with(tokio_util::sync::CancellationToken::new)
|
||||
.cancel();
|
||||
@@ -673,7 +677,13 @@ impl ECStore {
|
||||
let movement_changed = rebalance_movement_snapshot_changed(self.rebalance_meta.read().await.as_ref(), &meta);
|
||||
{
|
||||
let mut rebalance_meta = self.rebalance_meta.write().await;
|
||||
|
||||
if let Some(current) = rebalance_meta.as_ref()
|
||||
&& current.id == meta.id
|
||||
{
|
||||
meta.cancel = current.cancel.clone();
|
||||
meta.activation_gate = Arc::clone(¤t.activation_gate);
|
||||
meta.stop_requested = current.stop_requested;
|
||||
}
|
||||
*rebalance_meta = Some(meta);
|
||||
|
||||
drop(rebalance_meta);
|
||||
@@ -1188,11 +1198,12 @@ impl ECStore {
|
||||
let meta = rebalance_meta
|
||||
.as_mut()
|
||||
.ok_or_else(|| rebalance_metadata_not_initialized_error("cancel rebalance admission"))?;
|
||||
if meta.stopped_at.is_some() || !is_rebalance_conflicting_with_decommission(meta) {
|
||||
if meta.stopped_at.is_some() || (!is_rebalance_conflicting_with_decommission(meta) && !meta.stop_requested) {
|
||||
return Err(Error::other(format!(
|
||||
"inactive rebalance rejected while cancelling admission: {expected_id}"
|
||||
)));
|
||||
}
|
||||
meta.stop_requested = true;
|
||||
meta.cancel
|
||||
.get_or_insert_with(tokio_util::sync::CancellationToken::new)
|
||||
.cancel();
|
||||
@@ -1213,6 +1224,7 @@ impl ECStore {
|
||||
ensure_rebalance_run_id(rebalance_meta.as_ref(), expected_id, "stop rebalance")?;
|
||||
}
|
||||
rebalance_meta.as_mut().map(|meta| {
|
||||
meta.stop_requested |= is_rebalance_conflicting_with_decommission(meta);
|
||||
let cancel = meta.cancel.get_or_insert_with(tokio_util::sync::CancellationToken::new);
|
||||
cancel.cancel();
|
||||
Arc::clone(&meta.activation_gate)
|
||||
@@ -1377,6 +1389,79 @@ mod tests {
|
||||
probe.wait_until_attempted().await;
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rebalance_stop_classification_checks_identity_and_explicit_intent() {
|
||||
let mut meta = RebalanceMeta {
|
||||
id: "current".to_string(),
|
||||
cancel: Some(tokio_util::sync::CancellationToken::new()),
|
||||
pool_stats: vec![RebalanceStats {
|
||||
participating: true,
|
||||
info: RebalanceInfo {
|
||||
status: RebalStatus::Started,
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
}],
|
||||
..Default::default()
|
||||
};
|
||||
ensure_rebalance_worker_active(Some(&meta), "current", "test").expect("active worker");
|
||||
meta.cancel.as_ref().unwrap().cancel();
|
||||
assert!(
|
||||
!matches!(
|
||||
ensure_rebalance_worker_active(Some(&meta), "current", "test"),
|
||||
Err(Error::OperationCanceled)
|
||||
),
|
||||
"a sibling failure is not an operator stop"
|
||||
);
|
||||
meta.stop_requested = true;
|
||||
assert!(matches!(
|
||||
ensure_rebalance_worker_active(Some(&meta), "current", "test"),
|
||||
Err(Error::OperationCanceled)
|
||||
));
|
||||
assert!(
|
||||
!matches!(ensure_rebalance_worker_active(Some(&meta), "old", "test"), Err(Error::OperationCanceled)),
|
||||
"stale identity remains a failure even during stop"
|
||||
);
|
||||
assert!(!matches!(
|
||||
ensure_rebalance_worker_active(None, "current", "test"),
|
||||
Err(Error::OperationCanceled)
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn rebalance_stop_intent_does_not_survive_replacement_run_reload() {
|
||||
let (_temp_dirs, store) = crate::services::rebalance::test_store_with_persisted_rebalance_meta(RebalanceMeta {
|
||||
id: "replacement".to_string(),
|
||||
pool_stats: vec![RebalanceStats {
|
||||
participating: true,
|
||||
info: RebalanceInfo {
|
||||
status: RebalStatus::Completed,
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
}],
|
||||
..Default::default()
|
||||
})
|
||||
.await;
|
||||
let previous_gate = {
|
||||
let mut meta = store.rebalance_meta.write().await;
|
||||
let meta = meta.as_mut().unwrap();
|
||||
meta.id = "previous".to_string();
|
||||
meta.stop_requested = true;
|
||||
let cancel = tokio_util::sync::CancellationToken::new();
|
||||
cancel.cancel();
|
||||
meta.cancel = Some(cancel);
|
||||
Arc::clone(&meta.activation_gate)
|
||||
};
|
||||
store.load_rebalance_meta().await.expect("reload replacement run");
|
||||
let meta = store.rebalance_meta.read().await;
|
||||
let meta = meta.as_ref().unwrap();
|
||||
assert_eq!(meta.id, "replacement");
|
||||
assert!(!meta.stop_requested);
|
||||
assert!(meta.cancel.is_none());
|
||||
assert!(!Arc::ptr_eq(&previous_gate, &meta.activation_gate));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn cancel_rebalance_admission_is_id_checked_and_idempotent() {
|
||||
let rebalance_id = "rebalance-admission-current";
|
||||
@@ -1415,6 +1500,59 @@ mod tests {
|
||||
.await
|
||||
.expect("retrying admission cancellation should be idempotent");
|
||||
assert!(cancel.is_cancelled());
|
||||
let err = store
|
||||
.update_pool_stats_batch_for_rebalance(0, "bucket".to_string(), &[&FileInfo::default()], rebalance_id)
|
||||
.await
|
||||
.expect_err("stop racing with a final stats update must cancel that update");
|
||||
assert!(matches!(err, Error::OperationCanceled), "operator stop lost its cancellation type: {err}");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn prepare_rebalance_stop_preserves_intent_when_worker_stops_before_reload() {
|
||||
let id = "stop-worker-before-reload";
|
||||
let cancel = tokio_util::sync::CancellationToken::new();
|
||||
let (_temp_dirs, store) = crate::services::rebalance::test_store_with_persisted_rebalance_meta(RebalanceMeta {
|
||||
id: id.to_string(),
|
||||
cancel: Some(cancel.clone()),
|
||||
pool_stats: vec![RebalanceStats {
|
||||
participating: true,
|
||||
info: RebalanceInfo {
|
||||
status: RebalStatus::Stopped,
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
}],
|
||||
..Default::default()
|
||||
})
|
||||
.await;
|
||||
let gate = {
|
||||
let mut meta = store.rebalance_meta.write().await;
|
||||
let meta = meta.as_mut().expect("local rebalance metadata");
|
||||
meta.pool_stats[0].info.status = RebalStatus::Started;
|
||||
Arc::clone(&meta.activation_gate)
|
||||
};
|
||||
assert_eq!(
|
||||
store.prepare_rebalance_stop().await.expect("prepare the same run stop"),
|
||||
Some(id.to_string())
|
||||
);
|
||||
{
|
||||
let meta = store.rebalance_meta.read().await;
|
||||
let meta = meta.as_ref().expect("reloaded stop target");
|
||||
assert!(Arc::ptr_eq(&gate, &meta.activation_gate), "reload must retain the drained run's gate");
|
||||
assert!(meta.cancel.as_ref().is_some_and(|token| token.is_cancelled()));
|
||||
}
|
||||
store
|
||||
.stop_rebalance_for_id(Some(id))
|
||||
.await
|
||||
.expect("finish the stop after the worker's terminal event");
|
||||
store
|
||||
.load_rebalance_meta()
|
||||
.await
|
||||
.expect("reload the acknowledged durable stop");
|
||||
let meta = store.rebalance_meta.read().await;
|
||||
let meta = meta.as_ref().expect("durable stopped metadata");
|
||||
assert!(meta.stopped_at.is_some(), "a successful stop must retain its durable timestamp");
|
||||
assert_eq!(meta.pool_stats[0].info.status, RebalStatus::Stopped);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -2031,7 +2169,10 @@ mod tests {
|
||||
let err = acquire_persisted_rebalance_run_guard(set_disks, active.id.as_str(), "cross-node stale snapshot")
|
||||
.await
|
||||
.expect_err("persisted stop must fence a node that missed stop propagation");
|
||||
assert!(err.to_string().contains("inactive rebalance worker rejected"));
|
||||
assert!(
|
||||
matches!(err, Error::OperationCanceled),
|
||||
"a durable remote stop cancels the same run: {err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -19,11 +19,11 @@ use super::meta::{
|
||||
use super::migration::{RebalanceMigrationBackend, migrate_entry_version};
|
||||
use super::worker::{
|
||||
RebalanceEntryCleanupResult, RebalanceEntryTask, load_rebalance_bucket_configs, rebalance_max_attempts,
|
||||
resolve_rebalance_bucket_error, resolve_rebalance_entry_cleanup_delete_result, resolve_rebalance_file_info_versions_result,
|
||||
resolve_rebalance_migrate_result_error, resolve_rebalance_stats_update_result, resolve_rebalance_worker_result,
|
||||
run_rebalance_listing_with_retry, should_cleanup_rebalance_source_entry, should_count_rebalance_version_complete,
|
||||
should_defer_rebalance_entry_failure, should_skip_rebalance_delete_marker, wait_rebalance_entry_tasks,
|
||||
with_rebalance_entry_context,
|
||||
record_rebalance_error, resolve_rebalance_bucket_error, resolve_rebalance_entry_cleanup_delete_result,
|
||||
resolve_rebalance_file_info_versions_result, resolve_rebalance_migrate_result_error, resolve_rebalance_stats_update_result,
|
||||
resolve_rebalance_worker_result, run_rebalance_listing_with_retry, should_cleanup_rebalance_source_entry,
|
||||
should_count_rebalance_version_complete, should_defer_rebalance_entry_failure, should_skip_rebalance_delete_marker,
|
||||
wait_rebalance_entry_tasks, with_rebalance_entry_context,
|
||||
};
|
||||
use super::{
|
||||
EVENT_REBALANCE_BUCKET, EVENT_REBALANCE_ENTRY, EVENT_REBALANCE_STATE, LOG_COMPONENT_ECSTORE, LOG_SUBSYSTEM_REBALANCE,
|
||||
@@ -676,10 +676,8 @@ impl ECStore {
|
||||
}
|
||||
error!("rebalance_entry: data movement admission failed: {err}");
|
||||
let mut first_err = entry_error.lock().await;
|
||||
if first_err.is_none() {
|
||||
*first_err = Some(err);
|
||||
callback_rx.cancel();
|
||||
}
|
||||
record_rebalance_error(&mut first_err, err);
|
||||
callback_rx.cancel();
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -721,10 +719,8 @@ impl ECStore {
|
||||
if let Err(err) = &result {
|
||||
error!("rebalance_entry: rebalance entry failed: {err}");
|
||||
let mut first_err = entry_error.lock().await;
|
||||
if first_err.is_none() {
|
||||
*first_err = Some(err.clone());
|
||||
callback_rx.cancel();
|
||||
}
|
||||
record_rebalance_error(&mut first_err, err.clone());
|
||||
callback_rx.cancel();
|
||||
}
|
||||
debug!(
|
||||
event = EVENT_REBALANCE_ENTRY,
|
||||
@@ -793,10 +789,7 @@ impl ECStore {
|
||||
deferred_error = Some(last_error);
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(err) if worker_error.is_none() => {
|
||||
worker_error = Some(err);
|
||||
}
|
||||
Err(_) => {}
|
||||
Err(err) => record_rebalance_error(&mut worker_error, err),
|
||||
}
|
||||
}
|
||||
let entry_error = entry_error.lock().await.clone();
|
||||
|
||||
@@ -643,16 +643,12 @@ pub(super) fn should_skip_start_rebalance(cancel_attached: bool, in_progress: bo
|
||||
cancel_attached && in_progress
|
||||
}
|
||||
|
||||
pub(super) fn is_rebalance_stopped_terminal_event(terminal_event: &RebalanceTerminalEvent) -> bool {
|
||||
matches!(terminal_event, RebalanceTerminalEvent::Stopped { .. })
|
||||
}
|
||||
|
||||
pub(super) fn should_preserve_rebalance_stopped_state(
|
||||
meta_stopped: bool,
|
||||
status: RebalStatus,
|
||||
terminal_event: &RebalanceTerminalEvent,
|
||||
) -> bool {
|
||||
(meta_stopped || status == RebalStatus::Stopped) && !is_rebalance_stopped_terminal_event(terminal_event)
|
||||
(meta_stopped || status == RebalStatus::Stopped) && matches!(terminal_event, RebalanceTerminalEvent::Completed { .. })
|
||||
}
|
||||
|
||||
pub(super) fn resolve_rebalance_participants(pool_stats: &[RebalanceStats], pool_count: usize) -> Vec<bool> {
|
||||
@@ -920,7 +916,7 @@ pub(super) fn clear_rebalance_cancel_token(meta: Option<&mut RebalanceMeta>) ->
|
||||
|
||||
pub(super) fn stop_rebalance_state(meta: &mut RebalanceMeta, now: OffsetDateTime) {
|
||||
clear_rebalance_cancel_token(Some(meta));
|
||||
if meta.stopped_at.is_none() && is_rebalance_in_progress(meta) {
|
||||
if meta.stopped_at.is_none() && (meta.stop_requested || is_rebalance_in_progress(meta)) {
|
||||
apply_stopped_at(meta, now);
|
||||
} else if meta.stopped_at.is_some() {
|
||||
mark_started_rebalance_pools_stopping(meta);
|
||||
|
||||
@@ -19,14 +19,14 @@ use super::meta::{
|
||||
complete_rebalance_pools_at_goal, complete_rebalance_pools_with_empty_queue, defer_bucket_in_rebalance_queue,
|
||||
ensure_rebalance_not_decommissioning, ensure_valid_rebalance_pool_index, first_rebalance_bucket,
|
||||
has_deferred_rebalance_error, is_rebalance_actively_running, is_rebalance_conflicting_with_decommission,
|
||||
is_rebalance_in_progress, is_rebalance_meta_replaceable_for_new_id, is_rebalance_stopped_terminal_event,
|
||||
mark_rebalance_bucket_done, merge_rebalance_bucket_lists, merge_rebalance_meta, next_rebal_bucket_from_stat,
|
||||
percent_free_ratio, rebalance_goal_reached, rebalance_meta_load_no_data_error, rebalance_meta_load_unknown_format_error,
|
||||
rebalance_meta_load_unknown_version_error, rebalance_requires_worker_activation, record_rebalance_cleanup_warning_in_meta,
|
||||
remove_rebalanced_buckets_from_queue, resolve_next_rebalance_bucket, resolve_rebalance_participants,
|
||||
should_accept_rebalance_stats_update, should_ignore_rebalance_data_usage_cache, should_pool_participate,
|
||||
should_preserve_rebalance_stopped_state, should_skip_start_rebalance, stop_rebalance_meta_snapshot, stop_rebalance_state,
|
||||
take_bucket_from_rebalance_queue, validate_init_rebalance_state, validate_start_rebalance_state,
|
||||
is_rebalance_in_progress, is_rebalance_meta_replaceable_for_new_id, mark_rebalance_bucket_done, merge_rebalance_bucket_lists,
|
||||
merge_rebalance_meta, next_rebal_bucket_from_stat, percent_free_ratio, rebalance_goal_reached,
|
||||
rebalance_meta_load_no_data_error, rebalance_meta_load_unknown_format_error, rebalance_meta_load_unknown_version_error,
|
||||
rebalance_requires_worker_activation, record_rebalance_cleanup_warning_in_meta, remove_rebalanced_buckets_from_queue,
|
||||
resolve_next_rebalance_bucket, resolve_rebalance_participants, should_accept_rebalance_stats_update,
|
||||
should_ignore_rebalance_data_usage_cache, should_pool_participate, should_preserve_rebalance_stopped_state,
|
||||
should_skip_start_rebalance, stop_rebalance_meta_snapshot, stop_rebalance_state, take_bucket_from_rebalance_queue,
|
||||
validate_init_rebalance_state, validate_start_rebalance_state,
|
||||
};
|
||||
use super::migration::{
|
||||
MigrationBackend, MigrationVersionResult, migrate_entry_version, migrate_entry_version_with_retry_wait,
|
||||
@@ -1676,6 +1676,30 @@ fn test_resolve_rebalance_stats_update_result_passthrough() {
|
||||
assert!(resolve_rebalance_stats_update_result(Ok(()), 0, "bucket", "object").is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_rebalance_stop_preserves_cancellation_through_entry_context() {
|
||||
let err = resolve_rebalance_stats_update_result(Err(Error::OperationCanceled), 0, "bucket", "object")
|
||||
.expect_err("canceled stats update");
|
||||
let err = with_rebalance_entry_context("stats", "bucket", "object", err);
|
||||
assert!(matches!(err, Error::OperationCanceled));
|
||||
assert!(matches!(
|
||||
classify_rebalance_terminal_event(Some(Err(err)), OffsetDateTime::now_utc()),
|
||||
RebalanceTerminalEvent::Stopped { .. }
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_rebalance_stop_does_not_hide_later_entry_failure() {
|
||||
let tasks = Arc::new(tokio::sync::Mutex::new(vec![
|
||||
tokio::spawn(async { Err(Error::OperationCanceled) }),
|
||||
tokio::spawn(async { Err(Error::ErasureWriteQuorum) }),
|
||||
]));
|
||||
let err = wait_rebalance_entry_tasks(0, tasks)
|
||||
.await
|
||||
.expect_err("entry I/O failure must survive sibling cancellation");
|
||||
assert!(matches!(err, Error::ErasureWriteQuorum));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_rebalance_stats_update_result_wraps_error_context() {
|
||||
let err = resolve_rebalance_stats_update_result(Err(Error::SlowDown), 2, "bucket-a", "obj.txt")
|
||||
@@ -2365,9 +2389,9 @@ fn test_resolve_rebalance_terminal_error_wraps_signal_failure_context() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_rebalance_bucket_error_prefers_entry_error() {
|
||||
fn test_resolve_rebalance_bucket_error_prefers_real_failure_over_entry_cancellation() {
|
||||
let err = resolve_rebalance_bucket_error(Some(Error::OperationCanceled), Some(Error::SlowDown)).unwrap_err();
|
||||
assert!(matches!(err, Error::OperationCanceled));
|
||||
assert!(matches!(err, Error::SlowDown));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2512,19 +2536,6 @@ fn test_apply_rebalance_terminal_event_stopped_clears_error() {
|
||||
assert_eq!(last_error, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_is_rebalance_stopped_terminal_event_only_matches_stopped_variant() {
|
||||
let stopped = RebalanceTerminalEvent::Stopped {
|
||||
msg: "stopped".to_string(),
|
||||
};
|
||||
let completed = RebalanceTerminalEvent::Completed {
|
||||
msg: "completed".to_string(),
|
||||
};
|
||||
|
||||
assert!(is_rebalance_stopped_terminal_event(&stopped));
|
||||
assert!(!is_rebalance_stopped_terminal_event(&completed));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_should_preserve_rebalance_stopped_state_when_meta_marked_stopped() {
|
||||
let event = RebalanceTerminalEvent::Completed {
|
||||
@@ -2535,13 +2546,14 @@ fn test_should_preserve_rebalance_stopped_state_when_meta_marked_stopped() {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_should_preserve_rebalance_stopped_state_when_pool_already_stopped() {
|
||||
fn test_rebalance_stop_does_not_hide_real_terminal_failure() {
|
||||
let event = RebalanceTerminalEvent::Failed {
|
||||
msg: "failed".to_string(),
|
||||
last_error: "boom".to_string(),
|
||||
};
|
||||
|
||||
assert!(should_preserve_rebalance_stopped_state(false, RebalStatus::Stopped, &event));
|
||||
assert!(!should_preserve_rebalance_stopped_state(false, RebalStatus::Stopped, &event));
|
||||
assert!(!should_preserve_rebalance_stopped_state(true, RebalStatus::Started, &event));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2716,6 +2728,32 @@ async fn test_start_rebalance_for_id_rejects_stopped_metadata() {
|
||||
assert!(err.to_string().contains("was stopped before start"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_rebalance_stop_intent_blocks_activation_before_durable_timestamp() {
|
||||
let mut meta = RebalanceMeta {
|
||||
id: "stopping".to_string(),
|
||||
stop_requested: true,
|
||||
pool_stats: vec![RebalanceStats {
|
||||
participating: true,
|
||||
buckets: vec!["pending".to_string()],
|
||||
info: RebalanceInfo {
|
||||
status: RebalStatus::Started,
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
}],
|
||||
..Default::default()
|
||||
};
|
||||
let outcome = commit_local_rebalance_worker_activation(&mut meta, "stopping", CancellationToken::new())
|
||||
.expect("stop must prevent activation without a new error");
|
||||
assert_eq!(outcome, RebalanceLocalActivationOutcome::NotStartedTerminal);
|
||||
assert!(meta.cancel.is_none());
|
||||
assert!(meta.stopped_at.is_none());
|
||||
let bytes = rmp_serde::to_vec_named(&meta).expect("encode legacy-compatible metadata");
|
||||
let reloaded: RebalanceMeta = rmp_serde::from_slice(&bytes).expect("decode metadata");
|
||||
assert!(!reloaded.stop_requested, "operator intent is local, not a new persisted field");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_stopped_activation_state_prevents_worker_token_commit() {
|
||||
let mut meta = RebalanceMeta {
|
||||
|
||||
@@ -57,7 +57,7 @@ pub(super) fn commit_local_rebalance_worker_activation(
|
||||
meta.id
|
||||
)));
|
||||
}
|
||||
if meta.stopped_at.is_some() || !is_rebalance_in_progress(meta) {
|
||||
if meta.stopped_at.is_some() || meta.stop_requested || !is_rebalance_in_progress(meta) {
|
||||
return Ok(RebalanceLocalActivationOutcome::NotStartedTerminal);
|
||||
}
|
||||
meta.cancel = Some(cancel);
|
||||
|
||||
@@ -143,6 +143,10 @@ pub struct DiskStat {
|
||||
pub struct RebalanceMeta {
|
||||
#[serde(skip)]
|
||||
pub cancel: Option<CancellationToken>, // To be invoked on rebalance-stop
|
||||
/// Local operator intent, scoped to this run ID; a worker failure also cancels
|
||||
/// `cancel`, so the token alone cannot identify an administrative stop.
|
||||
#[serde(skip)]
|
||||
pub stop_requested: bool,
|
||||
#[serde(skip)]
|
||||
pub activation_gate: std::sync::Arc<tokio::sync::RwLock<()>>,
|
||||
#[serde(skip)]
|
||||
|
||||
@@ -38,6 +38,17 @@ pub(super) fn resolve_rebalance_worker_result<T>(
|
||||
|
||||
pub(super) type RebalanceEntryTask = tokio::task::JoinHandle<Result<RebalanceEntryOutcome>>;
|
||||
|
||||
/// Preserve the first real failure even when another task observes cancellation
|
||||
/// first. Cancellation is an outcome only when no entry or worker failed.
|
||||
pub(super) fn record_rebalance_error(first_error: &mut Option<Error>, err: Error) {
|
||||
if first_error
|
||||
.as_ref()
|
||||
.is_none_or(|first| is_err_operation_canceled(first) && !is_err_operation_canceled(&err))
|
||||
{
|
||||
*first_error = Some(err);
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub(super) enum RebalanceEntryCleanupResult {
|
||||
Completed { warning: Option<String> },
|
||||
@@ -65,16 +76,12 @@ pub(super) async fn wait_rebalance_entry_tasks(
|
||||
}
|
||||
Ok(Err(err)) => {
|
||||
error!("rebalance entry task failed for set {}: {}", set_idx, err);
|
||||
if first_error.is_none() {
|
||||
first_error = Some(err);
|
||||
}
|
||||
record_rebalance_error(&mut first_error, err);
|
||||
}
|
||||
Err(err) => {
|
||||
let err = Error::other(format!("rebalance entry task join error for set {set_idx}: {err}"));
|
||||
error!("{}", err);
|
||||
if first_error.is_none() {
|
||||
first_error = Some(err);
|
||||
}
|
||||
record_rebalance_error(&mut first_error, err);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -135,6 +142,9 @@ pub(super) fn resolve_rebalance_stats_update_result(
|
||||
object_name: &str,
|
||||
) -> Result<()> {
|
||||
result.map_err(|err| {
|
||||
if is_err_operation_canceled(&err) {
|
||||
return err;
|
||||
}
|
||||
Error::other(format!(
|
||||
"rebalance stats update failed for pool {pool_idx} bucket {bucket} object {object_name}: {err}"
|
||||
))
|
||||
@@ -214,16 +224,11 @@ pub(super) fn resolve_rebalance_terminal_error(primary_err: Error, signal_result
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn resolve_rebalance_bucket_error(entry_error: Option<Error>, worker_error: Option<Error>) -> Result<()> {
|
||||
if let Some(err) = entry_error {
|
||||
return Err(err);
|
||||
}
|
||||
|
||||
pub(super) fn resolve_rebalance_bucket_error(mut entry_error: Option<Error>, worker_error: Option<Error>) -> Result<()> {
|
||||
if let Some(err) = worker_error {
|
||||
return Err(err);
|
||||
record_rebalance_error(&mut entry_error, err);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
entry_error.map_or(Ok(()), Err)
|
||||
}
|
||||
|
||||
pub(super) fn resolve_rebalance_bucket_result(
|
||||
@@ -362,6 +367,9 @@ pub(super) fn ensure_rebalance_listing_disks_available(has_disks: bool, bucket:
|
||||
}
|
||||
|
||||
pub(super) fn with_rebalance_entry_context(stage: &str, bucket: &str, object_name: &str, err: Error) -> Error {
|
||||
if is_err_operation_canceled(&err) {
|
||||
return err;
|
||||
}
|
||||
Error::other(format!("rebalance entry {stage} failed for {bucket}/{object_name}: {err}"))
|
||||
}
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pub mod test_util;
|
||||
#[allow(clippy::module_inception, reason = "preserve the public services::tier::tier path")]
|
||||
pub mod tier;
|
||||
pub mod tier_admin;
|
||||
pub mod tier_config;
|
||||
|
||||
@@ -15,8 +15,6 @@
|
||||
#![allow(unused_variables)]
|
||||
#![allow(unused_mut)]
|
||||
#![allow(unused_assignments)]
|
||||
#![allow(unused_must_use)]
|
||||
#![allow(clippy::all)]
|
||||
|
||||
use byteorder::{ByteOrder, LittleEndian};
|
||||
use bytes::Bytes;
|
||||
@@ -802,7 +800,7 @@ pub enum TierConfigUpdateError {
|
||||
}
|
||||
|
||||
enum TierCandidateMutation {
|
||||
Add(TierConfig, bool),
|
||||
Add(Box<TierConfig>, bool),
|
||||
Edit(String, TierCreds),
|
||||
Remove(String, bool),
|
||||
Clear(bool),
|
||||
@@ -823,7 +821,7 @@ struct PrevalidatedTierCandidateMutation {
|
||||
impl TierCandidateMutation {
|
||||
fn add(mut config: TierConfig, force: bool) -> std::result::Result<Self, AdminError> {
|
||||
normalize_s3_gcs_add_tier_name(&mut config)?;
|
||||
Ok(Self::Add(config, force))
|
||||
Ok(Self::Add(Box::new(config), force))
|
||||
}
|
||||
|
||||
fn normalize_add_tier_name(&mut self) -> std::result::Result<(), AdminError> {
|
||||
@@ -910,7 +908,7 @@ impl TierCandidateMutation {
|
||||
match self {
|
||||
Self::Add(config, force) => {
|
||||
let tier_name = config.name.clone();
|
||||
candidate.add_with_deadline(config, force, deadline).await?;
|
||||
candidate.add_with_deadline(*config, force, deadline).await?;
|
||||
Ok(Some(tier_name))
|
||||
}
|
||||
Self::Edit(tier_name, credentials) => {
|
||||
@@ -2990,7 +2988,7 @@ fn from_external_tier_config(name: String, ext: ExternalTierConfig) -> io::Resul
|
||||
let tier_type = if wasabi_version {
|
||||
TierType::Wasabi
|
||||
} else {
|
||||
tier_type_from_hint(ext.tier_type_hint.as_deref()).unwrap_or_else(|| match ext.tier_type {
|
||||
tier_type_from_hint(ext.tier_type_hint.as_deref()).unwrap_or(match ext.tier_type {
|
||||
EXTERNAL_TIER_TYPE_S3 => TierType::S3,
|
||||
EXTERNAL_TIER_TYPE_AZURE => TierType::Azure,
|
||||
EXTERNAL_TIER_TYPE_GCS => TierType::GCS,
|
||||
@@ -3372,28 +3370,23 @@ impl TierConfigMgr {
|
||||
|
||||
pub async fn remove(&mut self, tier_name: &str, force: bool) -> std::result::Result<(), AdminError> {
|
||||
self.ensure_generation_is_idle(tier_name)?;
|
||||
let d = self.get_driver(tier_name).await;
|
||||
if let Err(err) = d {
|
||||
if err.code == ERR_TIER_NOT_FOUND.code {
|
||||
return Ok(());
|
||||
} else {
|
||||
return Err(err);
|
||||
}
|
||||
}
|
||||
let driver = match self.get_driver(tier_name).await {
|
||||
Ok(driver) => driver,
|
||||
Err(err) if err.code == ERR_TIER_NOT_FOUND.code => return Ok(()),
|
||||
Err(err) => return Err(err),
|
||||
};
|
||||
if !force {
|
||||
if let Ok(driver) = d {
|
||||
match driver.in_use().await {
|
||||
Err(err) => {
|
||||
let mut e = ERR_TIER_PERM_ERR.clone();
|
||||
e.message.push('.');
|
||||
e.message.push_str(&err.to_string());
|
||||
return Err(e);
|
||||
}
|
||||
Ok(in_use) if in_use => {
|
||||
return Err(ERR_TIER_BACKEND_NOT_EMPTY.clone());
|
||||
}
|
||||
_ => {}
|
||||
match driver.in_use().await {
|
||||
Err(err) => {
|
||||
let mut e = ERR_TIER_PERM_ERR.clone();
|
||||
e.message.push('.');
|
||||
e.message.push_str(&err.to_string());
|
||||
return Err(e);
|
||||
}
|
||||
Ok(in_use) if in_use => {
|
||||
return Err(ERR_TIER_BACKEND_NOT_EMPTY.clone());
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
self.tiers.remove(tier_name);
|
||||
@@ -3402,21 +3395,12 @@ impl TierConfigMgr {
|
||||
}
|
||||
|
||||
pub async fn verify(&mut self, tier_name: &str) -> std::result::Result<(), std::io::Error> {
|
||||
let d = match self.get_driver(tier_name).await {
|
||||
Ok(d) => d,
|
||||
Err(err) => {
|
||||
return Err(std::io::Error::other(err));
|
||||
}
|
||||
};
|
||||
if let Err(err) = check_warm_backend(Some(d)).await {
|
||||
return Err(std::io::Error::other(err));
|
||||
} else {
|
||||
return Ok(());
|
||||
}
|
||||
let driver = self.get_driver(tier_name).await.map_err(std::io::Error::other)?;
|
||||
check_warm_backend(Some(driver)).await.map_err(std::io::Error::other)
|
||||
}
|
||||
|
||||
pub fn empty(&self) -> bool {
|
||||
self.list_tiers().len() == 0
|
||||
self.tiers.is_empty()
|
||||
}
|
||||
|
||||
pub fn tier_type(&self, tier_name: &str) -> String {
|
||||
@@ -3429,7 +3413,7 @@ impl TierConfigMgr {
|
||||
|
||||
pub fn list_tiers(&self) -> Vec<TierConfig> {
|
||||
let mut tier_cfgs = Vec::<TierConfig>::new();
|
||||
for (_, tier) in self.tiers.iter() {
|
||||
for tier in self.tiers.values() {
|
||||
let tier = tier.redacted();
|
||||
tier_cfgs.push(tier);
|
||||
}
|
||||
@@ -7135,7 +7119,8 @@ mod tests {
|
||||
let err = expect_decode_err(&encode_fixture(&wrong_hint));
|
||||
assert!(err.to_string().contains("inconsistent Wasabi type discriminators"), "{err}");
|
||||
|
||||
let poison_fields: [(&str, fn(&mut ExternalTierS3)); 6] = [
|
||||
type WasabiPoisonField = (&'static str, fn(&mut ExternalTierS3));
|
||||
let poison_fields: [WasabiPoisonField; 6] = [
|
||||
("storage_class", |s3| s3.storage_class = "GLACIER".to_string()),
|
||||
("aws_role", |s3| s3.aws_role = true),
|
||||
("web_identity_token", |s3| s3.aws_role_web_identity_token_file = "/tmp/token".to_string()),
|
||||
@@ -8256,7 +8241,11 @@ mod tests {
|
||||
peer_calls.clone(),
|
||||
Ok(PeerTierMutationState::Committed),
|
||||
)],
|
||||
TierConfigMgr::update_candidate_with_config_lock(&manager, store, TierCandidateMutation::Add(tier, true)),
|
||||
TierConfigMgr::update_candidate_with_config_lock(
|
||||
&manager,
|
||||
store,
|
||||
TierCandidateMutation::Add(Box::new(tier), true),
|
||||
),
|
||||
),
|
||||
)
|
||||
.await
|
||||
@@ -8295,7 +8284,7 @@ mod tests {
|
||||
let add = TIER_DRIVER_TEST_FACTORY.scope(
|
||||
factory,
|
||||
apply_tier_candidate_mutation(
|
||||
TierCandidateMutation::Add(build_rustfs_tier("COLD-DEADLINE"), false),
|
||||
TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-DEADLINE")), false),
|
||||
&mut candidate,
|
||||
deadline,
|
||||
),
|
||||
@@ -9114,7 +9103,9 @@ mod tests {
|
||||
fn decode_hex_fixture(hex: &str) -> Vec<u8> {
|
||||
assert_eq!(hex.len() % 2, 0, "hex fixture must contain complete bytes");
|
||||
hex.as_bytes()
|
||||
.chunks_exact(2)
|
||||
.as_chunks::<2>()
|
||||
.0
|
||||
.iter()
|
||||
.map(|pair| {
|
||||
let pair = std::str::from_utf8(pair).expect("hex fixture should be ASCII");
|
||||
u8::from_str_radix(pair, 16).expect("hex fixture should contain only hexadecimal digits")
|
||||
@@ -11128,7 +11119,7 @@ mod tests {
|
||||
store.clone(),
|
||||
candidate,
|
||||
version,
|
||||
TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true),
|
||||
TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-A")), true),
|
||||
update,
|
||||
None,
|
||||
)
|
||||
@@ -11725,8 +11716,9 @@ mod tests {
|
||||
assert!(merged[0].has_peer_record && merged[0].has_coordinator_record);
|
||||
}
|
||||
|
||||
let err = TierConfigMgr::merge_mutation_recovery_intents(&[committed.clone()], &[prepared.clone()])
|
||||
.expect_err("a peer committed record cannot outrun the coordinator commit order");
|
||||
let err =
|
||||
TierConfigMgr::merge_mutation_recovery_intents(std::slice::from_ref(&committed), std::slice::from_ref(&prepared))
|
||||
.expect_err("a peer committed record cannot outrun the coordinator commit order");
|
||||
assert!(err.to_string().contains("conflicting states"), "{err}");
|
||||
|
||||
let mut conflicting_identity = prepared.clone();
|
||||
@@ -13571,9 +13563,11 @@ mod tests {
|
||||
let build = tokio::spawn(async move { TierConfigMgr::acquire_operation_lease(&build_manager, cold_tier).await });
|
||||
barrier.arrived.notified().await;
|
||||
|
||||
tokio::time::timeout(Duration::from_millis(100), manager.read())
|
||||
.await
|
||||
.expect("cold driver construction must not block manager readers");
|
||||
drop(
|
||||
tokio::time::timeout(Duration::from_millis(100), manager.read())
|
||||
.await
|
||||
.expect("cold driver construction must not block manager readers"),
|
||||
);
|
||||
let tier_b = tokio::time::timeout(Duration::from_millis(100), TierConfigMgr::acquire_operation_lease(&manager, "COLD-B"))
|
||||
.await
|
||||
.expect("cold tier A construction must not block tier B")
|
||||
@@ -13776,9 +13770,11 @@ mod tests {
|
||||
let verify_manager = manager.clone();
|
||||
let verify = tokio::spawn(async move { TierConfigMgr::verify_without_manager_lock(&verify_manager, "COLD-A").await });
|
||||
started.notified().await;
|
||||
tokio::time::timeout(Duration::from_millis(100), manager.read())
|
||||
.await
|
||||
.expect("slow verify must not hold the manager lock");
|
||||
drop(
|
||||
tokio::time::timeout(Duration::from_millis(100), manager.read())
|
||||
.await
|
||||
.expect("slow verify must not hold the manager lock"),
|
||||
);
|
||||
release.add_permits(1);
|
||||
verify.await.expect("verify task should join").expect("verify should finish");
|
||||
}
|
||||
@@ -14186,9 +14182,11 @@ mod tests {
|
||||
vec!["COLD-A".to_string()]
|
||||
);
|
||||
}
|
||||
tokio::time::timeout(Duration::from_secs(1), manager.read())
|
||||
.await
|
||||
.expect("manager reads must not wait for tier A leases");
|
||||
drop(
|
||||
tokio::time::timeout(Duration::from_secs(1), manager.read())
|
||||
.await
|
||||
.expect("manager reads must not wait for tier A leases"),
|
||||
);
|
||||
let next_b = tokio::time::timeout(Duration::from_secs(1), TierConfigMgr::acquire_operation_lease(&manager, "COLD-B"))
|
||||
.await
|
||||
.expect("tier B lease acquisition must not wait for tier A")
|
||||
@@ -14621,7 +14619,7 @@ mod tests {
|
||||
"https://example-compat.invalid"
|
||||
);
|
||||
let runtime = registered_tier_driver_runtime(&manager_guard).expect("runtime sidecar should remain registered");
|
||||
assert!(lock_unpoisoned(&runtime).generations.get("COLD-A").is_none());
|
||||
assert!(!lock_unpoisoned(&runtime).generations.contains_key("COLD-A"));
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
@@ -15261,15 +15259,12 @@ mod tests {
|
||||
.filter(|object| object.bucket == bucket && object.name.starts_with(prefix))
|
||||
.cloned()
|
||||
.collect();
|
||||
objects.sort_by(|left, right| tier_test_object_marker(left).cmp(&tier_test_object_marker(right)));
|
||||
objects.sort_by_key(tier_test_object_marker);
|
||||
if marker.is_some() || version_marker.is_some() {
|
||||
let marker = (marker.unwrap_or_default(), version_marker.unwrap_or_default());
|
||||
objects.retain(|object| tier_test_object_marker(object) > marker);
|
||||
}
|
||||
let limit = match usize::try_from(max_keys) {
|
||||
Ok(limit) => limit,
|
||||
Err(_) => 0,
|
||||
};
|
||||
let limit: usize = usize::try_from(max_keys).unwrap_or_default();
|
||||
let is_truncated = objects.len() > limit;
|
||||
if is_truncated {
|
||||
objects.truncate(limit);
|
||||
@@ -15299,17 +15294,16 @@ mod tests {
|
||||
result: Self::WalkResultSender,
|
||||
opts: Self::WalkOptions,
|
||||
) -> Result<()> {
|
||||
if self.fail_reference_walk.load(Ordering::SeqCst) {
|
||||
if result
|
||||
if self.fail_reference_walk.load(Ordering::SeqCst)
|
||||
&& result
|
||||
.send(StorageObjectInfoOrErr {
|
||||
item: None,
|
||||
err: Some(Error::other("injected tier reference walk failure")),
|
||||
})
|
||||
.await
|
||||
.is_err()
|
||||
{
|
||||
return Ok(());
|
||||
}
|
||||
{
|
||||
return Ok(());
|
||||
}
|
||||
let mut objects = self
|
||||
.listed_versions
|
||||
@@ -15320,7 +15314,7 @@ mod tests {
|
||||
.filter(|object| opts.include_free_versions || !object.transitioned_object.free_version)
|
||||
.cloned()
|
||||
.collect::<Vec<_>>();
|
||||
objects.sort_by(|left, right| tier_test_object_marker(left).cmp(&tier_test_object_marker(right)));
|
||||
objects.sort_by_key(tier_test_object_marker);
|
||||
if let Some(marker) = opts.marker.as_deref() {
|
||||
objects.retain(|object| object.name.as_str() > marker);
|
||||
}
|
||||
@@ -15498,17 +15492,18 @@ mod tests {
|
||||
api_view.rustfs.expect("admin RustFS payload should exist").secret_key,
|
||||
TIER_CREDENTIAL_REDACTED
|
||||
);
|
||||
let observed = lock_unpoisoned(&observed);
|
||||
assert_eq!(observed.len(), 1);
|
||||
assert_eq!(
|
||||
observed[0]
|
||||
.rustfs
|
||||
.as_ref()
|
||||
.expect("backend factory should observe the RustFS payload")
|
||||
.secret_key,
|
||||
SECRET_KEY
|
||||
);
|
||||
drop(observed);
|
||||
{
|
||||
let observed = lock_unpoisoned(&observed);
|
||||
assert_eq!(observed.len(), 1);
|
||||
assert_eq!(
|
||||
observed[0]
|
||||
.rustfs
|
||||
.as_ref()
|
||||
.expect("backend factory should observe the RustFS payload")
|
||||
.secret_key,
|
||||
SECRET_KEY
|
||||
);
|
||||
}
|
||||
|
||||
let operations = backend.op_log().await;
|
||||
assert_eq!(operations.len(), 5);
|
||||
@@ -16709,7 +16704,7 @@ mod tests {
|
||||
candidate.tiers.insert("COLD-A".to_string(), build_rustfs_tier("COLD-A"));
|
||||
candidate.tiers.insert("COLD-B".to_string(), build_rustfs_tier("COLD-B"));
|
||||
|
||||
let targets = TierCandidateMutation::Add(build_rustfs_tier("COLD-B"), true)
|
||||
let targets = TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-B")), true)
|
||||
.affected_targets(¤t, &candidate)
|
||||
.expect("add proof should ignore unchanged durable tiers");
|
||||
assert_eq!(targets.len(), 1);
|
||||
@@ -16735,7 +16730,7 @@ mod tests {
|
||||
TierConfigMgr::update_candidate_with_config_lock(
|
||||
&manager,
|
||||
store.clone(),
|
||||
TierCandidateMutation::Add(build_rustfs_tier("COLD-B"), true),
|
||||
TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-B")), true),
|
||||
),
|
||||
)
|
||||
.await
|
||||
@@ -16794,14 +16789,15 @@ mod tests {
|
||||
.await
|
||||
.expect("legacy nested-name Add must run the full coordinator fanout");
|
||||
|
||||
let prepared_intents = lock_unpoisoned(&prepared_intents);
|
||||
assert_eq!(prepared_intents.len(), 1);
|
||||
assert_eq!(prepared_intents[0].kind, TierMutationIntentKind::Add);
|
||||
assert_eq!(prepared_intents[0].affected_targets.len(), 1);
|
||||
assert_eq!(prepared_intents[0].affected_targets[0].tier_name, "COLD-LEGACY");
|
||||
assert!(prepared_intents[0].affected_targets[0].old_backend_identity.is_none());
|
||||
assert!(prepared_intents[0].affected_targets[0].new_backend_identity.is_some());
|
||||
drop(prepared_intents);
|
||||
{
|
||||
let prepared_intents = lock_unpoisoned(&prepared_intents);
|
||||
assert_eq!(prepared_intents.len(), 1);
|
||||
assert_eq!(prepared_intents[0].kind, TierMutationIntentKind::Add);
|
||||
assert_eq!(prepared_intents[0].affected_targets.len(), 1);
|
||||
assert_eq!(prepared_intents[0].affected_targets[0].tier_name, "COLD-LEGACY");
|
||||
assert!(prepared_intents[0].affected_targets[0].old_backend_identity.is_none());
|
||||
assert!(prepared_intents[0].affected_targets[0].new_backend_identity.is_some());
|
||||
}
|
||||
|
||||
let peer_calls = lock_unpoisoned(&peer_calls).clone();
|
||||
let prepare_index = peer_calls
|
||||
@@ -16883,7 +16879,7 @@ mod tests {
|
||||
let err = TierConfigMgr::update_candidate_with_config_lock(
|
||||
&manager,
|
||||
store.clone(),
|
||||
TierCandidateMutation::Add(build_rustfs_tier("COLD-B"), true),
|
||||
TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-B")), true),
|
||||
)
|
||||
.await
|
||||
.expect_err("a new tier config update must wait for pending mutation recovery");
|
||||
@@ -17159,7 +17155,7 @@ mod tests {
|
||||
TierConfigMgr::update_candidate_with_config_lock(
|
||||
&update_manager,
|
||||
update_store,
|
||||
TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true),
|
||||
TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-A")), true),
|
||||
),
|
||||
)
|
||||
.await
|
||||
@@ -17210,7 +17206,7 @@ mod tests {
|
||||
TierConfigMgr::update_candidate_with_config_lock(
|
||||
&update_manager,
|
||||
update_store,
|
||||
TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true),
|
||||
TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-A")), true),
|
||||
),
|
||||
),
|
||||
)
|
||||
@@ -17273,7 +17269,7 @@ mod tests {
|
||||
TierConfigMgr::prevalidate_candidate_owned(
|
||||
empty_mgr(),
|
||||
None,
|
||||
TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true),
|
||||
TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-A")), true),
|
||||
),
|
||||
)
|
||||
.await;
|
||||
@@ -17937,7 +17933,7 @@ mod tests {
|
||||
|
||||
#[tokio::test]
|
||||
async fn tier_add_succeeds_with_refresh_during_coordinator_commit() {
|
||||
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true)).await;
|
||||
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Add(Box::new(build_rustfs_tier("COLD-A")), true)).await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
|
||||
@@ -15,8 +15,6 @@
|
||||
#![allow(unused_variables)]
|
||||
#![allow(unused_mut)]
|
||||
#![allow(unused_assignments)]
|
||||
#![allow(unused_must_use)]
|
||||
#![allow(clippy::all)]
|
||||
|
||||
use crate::error::is_err_bucket_not_found;
|
||||
#[cfg(feature = "gcs")]
|
||||
@@ -719,17 +717,7 @@ async fn check_warm_backend_with_deadlines(
|
||||
if !matches!(cleanup_result, Ok(Ok(()))) {
|
||||
return Err(probe_cleanup_incomplete_error());
|
||||
}
|
||||
if let Err(err) = read_result {
|
||||
//if is_err_bucket_not_found(&err) {
|
||||
// return Err(ERR_TIER_BUCKET_NOT_FOUND);
|
||||
//}
|
||||
/*else if is_err_signature_does_not_match(err) {
|
||||
return Err(ERR_TIER_MISSING_CREDENTIALS);
|
||||
}*/
|
||||
//else {
|
||||
return Err(err);
|
||||
//}
|
||||
}
|
||||
read_result?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -759,7 +747,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -800,7 +788,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -820,7 +808,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -840,7 +828,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -860,7 +848,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -880,7 +868,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -900,7 +888,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -929,7 +917,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
@@ -949,7 +937,7 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
|
||||
warn!("{}", err);
|
||||
return Err(AdminError {
|
||||
code: "XRustFSAdminTierInvalidConfig".to_string(),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {}", err.to_string()),
|
||||
message: format!("Unable to setup remote tier, check tier configuration: {err}"),
|
||||
status_code: StatusCode::BAD_REQUEST,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -15,8 +15,6 @@
|
||||
#![allow(unused_variables)]
|
||||
#![allow(unused_mut)]
|
||||
#![allow(unused_assignments)]
|
||||
#![allow(unused_must_use)]
|
||||
#![allow(clippy::all)]
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::sync::Arc;
|
||||
@@ -106,39 +104,35 @@ impl WarmBackendS3 {
|
||||
};
|
||||
validate_outbound_url(&u).map_err(|err| std::io::Error::other(format!("tier endpoint is not allowed: {err}")))?;
|
||||
|
||||
if conf.aws_role_web_identity_token_file == "" && conf.aws_role_arn != ""
|
||||
|| conf.aws_role_web_identity_token_file != "" && conf.aws_role_arn == ""
|
||||
{
|
||||
let has_web_identity_token_file = !conf.aws_role_web_identity_token_file.is_empty();
|
||||
let has_role_arn = !conf.aws_role_arn.is_empty();
|
||||
let has_access_key = !conf.access_key.is_empty();
|
||||
let has_secret_key = !conf.secret_key.is_empty();
|
||||
|
||||
if has_web_identity_token_file != has_role_arn {
|
||||
return Err(std::io::Error::other("both the token file and the role ARN are required"));
|
||||
} else if conf.access_key == "" && conf.secret_key != "" || conf.access_key != "" && conf.secret_key == "" {
|
||||
} else if has_access_key != has_secret_key {
|
||||
return Err(std::io::Error::other("both the access and secret keys are required"));
|
||||
} else if conf.aws_role
|
||||
&& (conf.aws_role_web_identity_token_file != ""
|
||||
|| conf.aws_role_arn != ""
|
||||
|| conf.access_key != ""
|
||||
|| conf.secret_key != "")
|
||||
{
|
||||
} else if conf.aws_role && (has_web_identity_token_file || has_role_arn || has_access_key || has_secret_key) {
|
||||
return Err(std::io::Error::other(
|
||||
"AWS Role cannot be activated with static credentials or the web identity token file",
|
||||
));
|
||||
} else if conf.bucket == "" {
|
||||
} else if conf.bucket.is_empty() {
|
||||
return Err(std::io::Error::other("no bucket name was provided"));
|
||||
}
|
||||
|
||||
let creds: Credentials<Static>;
|
||||
|
||||
if conf.access_key != "" && conf.secret_key != "" {
|
||||
let creds = if has_access_key && has_secret_key {
|
||||
//creds = Credentials::new_static_v4(conf.access_key, conf.secret_key, "");
|
||||
creds = Credentials::new(Static(Value {
|
||||
Credentials::new(Static(Value {
|
||||
access_key_id: conf.access_key.clone(),
|
||||
secret_access_key: conf.secret_key.clone(),
|
||||
session_token: "".to_string(),
|
||||
signer_type: SignatureType::SignatureV4,
|
||||
..Default::default()
|
||||
}));
|
||||
}))
|
||||
} else {
|
||||
return Err(std::io::Error::other("insufficient parameters for S3 backend authentication"));
|
||||
}
|
||||
};
|
||||
let timeouts = transition_client_timeouts_from_env();
|
||||
let opts = Options {
|
||||
creds,
|
||||
@@ -162,11 +156,11 @@ impl WarmBackendS3 {
|
||||
}
|
||||
|
||||
pub fn get_dest(&self, object: &str) -> String {
|
||||
let mut dest_obj = object.to_string();
|
||||
if self.prefix != "" {
|
||||
dest_obj = format!("{}/{}", &self.prefix, object);
|
||||
if self.prefix.is_empty() {
|
||||
object.to_string()
|
||||
} else {
|
||||
format!("{}/{}", self.prefix, object)
|
||||
}
|
||||
return dest_obj;
|
||||
}
|
||||
|
||||
pub(crate) async fn remove_with_result(&self, object: &str, rv: &str) -> Result<RemoveObjectResult, std::io::Error> {
|
||||
@@ -413,6 +407,10 @@ impl TransitionCandidateVersions {
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
#[allow(
|
||||
clippy::items_after_test_module,
|
||||
reason = "keep parsing tests adjacent to the helpers they cover"
|
||||
)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use rustfs_s3_client::api_s3_datatypes::{ListVersionsResult, Version};
|
||||
@@ -917,7 +915,7 @@ impl WarmBackend for WarmBackendS3 {
|
||||
.list_objects_v2(&self.bucket, &self.prefix, "", "", SLASH_SEPARATOR, 1)
|
||||
.await?;
|
||||
|
||||
Ok(result.common_prefixes.len() > 0 || result.contents.len() > 0)
|
||||
Ok(!result.common_prefixes.is_empty() || !result.contents.is_empty())
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+150
-102
@@ -883,6 +883,17 @@ pub(crate) use ops::object::body_cache_plaintext_len;
|
||||
pub(crate) use ops::object::cleanup_rejected_transition_upload_durably;
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pub use ops::object::{PutObjectCommitBarrier, PutObjectCommitPause};
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) use ops::object::{
|
||||
TransitionTransactionKillPoint as SetDiskTransitionTransactionKillPoint,
|
||||
TransitionTransactionKillPointBarrier as SetDiskTransitionTransactionKillPointBarrier,
|
||||
};
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) use ops::object::{
|
||||
TransitionTransactionMutationKind as SetDiskTransitionTransactionMutationKind,
|
||||
TransitionTransactionMutationObservation as SetDiskTransitionTransactionMutationObservation,
|
||||
TransitionTransactionMutationProbe as SetDiskTransitionTransactionMutationProbe,
|
||||
};
|
||||
mod read;
|
||||
mod replication;
|
||||
pub(crate) mod shard_source;
|
||||
@@ -6452,113 +6463,114 @@ pub fn should_heal_object_on_disk(
|
||||
(false, false, None)
|
||||
}
|
||||
|
||||
/// Probe every drive of the set at once. Each live probe is bounded by the
|
||||
/// drive `disk_info` timeout, and the admin peer probe budget only covers one
|
||||
/// such timeout; a sequential walk over several stalled drives after a power
|
||||
/// cut would exceed it and make healthy peers render as unknown (#6488).
|
||||
async fn get_disks_info(disks: &[Option<DiskStore>], eps: &[Endpoint]) -> Vec<rustfs_madmin::Disk> {
|
||||
let mut ret = Vec::new();
|
||||
join_all(disks.iter().zip(eps).map(|(disk, ep)| disk_admin_info(disk.as_ref(), ep))).await
|
||||
}
|
||||
|
||||
for (i, pool) in disks.iter().enumerate() {
|
||||
if let Some(disk) = pool {
|
||||
let runtime_state = disk.runtime_state();
|
||||
let offline_duration_seconds = disk.offline_duration_secs();
|
||||
let capacity_snapshot = disk.last_capacity_snapshot();
|
||||
let cached_disk_id = disk.cached_disk_id().await;
|
||||
if runtime_state.should_probe_for_admin() || runtime_state == disk::health_state::RuntimeDriveHealthState::Suspect {
|
||||
match disk
|
||||
.disk_info(&DiskInfoOptions {
|
||||
metrics: true,
|
||||
..Default::default()
|
||||
})
|
||||
.await
|
||||
{
|
||||
Ok(res) => {
|
||||
disk.record_capacity_probe(res.total, res.used, res.free);
|
||||
ret.push(rustfs_madmin::Disk {
|
||||
endpoint: eps[i].to_string(),
|
||||
local: eps[i].is_local,
|
||||
pool_index: eps[i].pool_idx,
|
||||
set_index: eps[i].set_idx,
|
||||
disk_index: eps[i].disk_idx,
|
||||
state: "ok".to_owned(),
|
||||
async fn disk_admin_info(disk: Option<&DiskStore>, ep: &Endpoint) -> rustfs_madmin::Disk {
|
||||
let Some(disk) = disk else {
|
||||
return rustfs_madmin::Disk {
|
||||
endpoint: ep.to_string(),
|
||||
drive_path: ep.get_file_path(),
|
||||
local: ep.is_local,
|
||||
pool_index: ep.pool_idx,
|
||||
set_index: ep.set_idx,
|
||||
disk_index: ep.disk_idx,
|
||||
runtime_state: None,
|
||||
offline_duration_seconds: None,
|
||||
state: DiskError::DiskNotFound.to_string(),
|
||||
capacity_observation_source: Some("missing".to_owned()),
|
||||
capacity_observation_age_seconds: Some(0),
|
||||
..Default::default()
|
||||
};
|
||||
};
|
||||
|
||||
root_disk: res.root_disk,
|
||||
drive_path: res.mount_path.clone(),
|
||||
healing: res.healing,
|
||||
scanning: res.scanning,
|
||||
runtime_state: Some(runtime_state.as_str().to_string()),
|
||||
offline_duration_seconds,
|
||||
capacity_observation_source: Some("live_probe".to_owned()),
|
||||
capacity_observation_age_seconds: Some(0),
|
||||
|
||||
uuid: res.id.map_or_else(|| "".to_string(), |id| id.to_string()),
|
||||
major: res.major as u32,
|
||||
minor: res.minor as u32,
|
||||
model: None,
|
||||
total_space: res.total,
|
||||
used_space: res.used,
|
||||
available_space: res.free,
|
||||
physical_device_ids: (!res.physical_device_ids.is_empty()).then_some(res.physical_device_ids.clone()),
|
||||
utilization: utilization_percent(res.total, res.used),
|
||||
used_inodes: res.used_inodes,
|
||||
free_inodes: res.free_inodes,
|
||||
metrics: Some(res.metrics),
|
||||
..Default::default()
|
||||
});
|
||||
}
|
||||
Err(err) => {
|
||||
let mut disk_info = rustfs_madmin::Disk {
|
||||
state: err.to_string(),
|
||||
endpoint: eps[i].to_string(),
|
||||
drive_path: eps[i].get_file_path(),
|
||||
local: eps[i].is_local,
|
||||
pool_index: eps[i].pool_idx,
|
||||
set_index: eps[i].set_idx,
|
||||
disk_index: eps[i].disk_idx,
|
||||
runtime_state: Some(runtime_state.as_str().to_string()),
|
||||
offline_duration_seconds,
|
||||
metrics: disk.metrics_snapshot(),
|
||||
uuid: cached_disk_id.map_or_else(String::new, |id| id.to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
if let Some((total, used, free, _)) = capacity_snapshot {
|
||||
disk_info.total_space = total;
|
||||
disk_info.used_space = used;
|
||||
disk_info.available_space = free;
|
||||
disk_info.utilization = utilization_percent(total, used);
|
||||
disk_info.capacity_observation_source = Some("snapshot".to_owned());
|
||||
disk_info.capacity_observation_age_seconds = capacity_snapshot
|
||||
.map(|(_, _, _, probe_unix_secs)| capacity_snapshot_age_seconds(probe_unix_secs));
|
||||
} else {
|
||||
disk_info.capacity_observation_source = Some("missing".to_owned());
|
||||
disk_info.capacity_observation_age_seconds = Some(0);
|
||||
}
|
||||
ret.push(disk_info);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
let mut disk_info =
|
||||
build_runtime_snapshot_disk(&eps[i], runtime_state, offline_duration_seconds, capacity_snapshot);
|
||||
disk_info.metrics = disk.metrics_snapshot();
|
||||
disk_info.uuid = cached_disk_id.map_or_else(String::new, |id| id.to_string());
|
||||
ret.push(disk_info);
|
||||
}
|
||||
} else {
|
||||
ret.push(rustfs_madmin::Disk {
|
||||
endpoint: eps[i].to_string(),
|
||||
drive_path: eps[i].get_file_path(),
|
||||
local: eps[i].is_local,
|
||||
pool_index: eps[i].pool_idx,
|
||||
set_index: eps[i].set_idx,
|
||||
disk_index: eps[i].disk_idx,
|
||||
runtime_state: None,
|
||||
offline_duration_seconds: None,
|
||||
state: DiskError::DiskNotFound.to_string(),
|
||||
capacity_observation_source: Some("missing".to_owned()),
|
||||
capacity_observation_age_seconds: Some(0),
|
||||
..Default::default()
|
||||
})
|
||||
}
|
||||
let runtime_state = disk.runtime_state();
|
||||
let offline_duration_seconds = disk.offline_duration_secs();
|
||||
let capacity_snapshot = disk.last_capacity_snapshot();
|
||||
let cached_disk_id = disk.cached_disk_id().await;
|
||||
if !(runtime_state.should_probe_for_admin() || runtime_state == disk::health_state::RuntimeDriveHealthState::Suspect) {
|
||||
let mut disk_info = build_runtime_snapshot_disk(ep, runtime_state, offline_duration_seconds, capacity_snapshot);
|
||||
disk_info.metrics = disk.metrics_snapshot();
|
||||
disk_info.uuid = cached_disk_id.map_or_else(String::new, |id| id.to_string());
|
||||
return disk_info;
|
||||
}
|
||||
|
||||
ret
|
||||
match disk
|
||||
.disk_info(&DiskInfoOptions {
|
||||
metrics: true,
|
||||
..Default::default()
|
||||
})
|
||||
.await
|
||||
{
|
||||
Ok(res) => {
|
||||
disk.record_capacity_probe(res.total, res.used, res.free);
|
||||
rustfs_madmin::Disk {
|
||||
endpoint: ep.to_string(),
|
||||
local: ep.is_local,
|
||||
pool_index: ep.pool_idx,
|
||||
set_index: ep.set_idx,
|
||||
disk_index: ep.disk_idx,
|
||||
state: "ok".to_owned(),
|
||||
|
||||
root_disk: res.root_disk,
|
||||
drive_path: res.mount_path.clone(),
|
||||
healing: res.healing,
|
||||
scanning: res.scanning,
|
||||
runtime_state: Some(runtime_state.as_str().to_string()),
|
||||
offline_duration_seconds,
|
||||
capacity_observation_source: Some("live_probe".to_owned()),
|
||||
capacity_observation_age_seconds: Some(0),
|
||||
|
||||
uuid: res.id.map_or_else(|| "".to_string(), |id| id.to_string()),
|
||||
major: res.major as u32,
|
||||
minor: res.minor as u32,
|
||||
model: None,
|
||||
total_space: res.total,
|
||||
used_space: res.used,
|
||||
available_space: res.free,
|
||||
physical_device_ids: (!res.physical_device_ids.is_empty()).then_some(res.physical_device_ids.clone()),
|
||||
utilization: utilization_percent(res.total, res.used),
|
||||
used_inodes: res.used_inodes,
|
||||
free_inodes: res.free_inodes,
|
||||
metrics: Some(res.metrics),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
Err(err) => {
|
||||
let mut disk_info = rustfs_madmin::Disk {
|
||||
state: err.to_string(),
|
||||
endpoint: ep.to_string(),
|
||||
drive_path: ep.get_file_path(),
|
||||
local: ep.is_local,
|
||||
pool_index: ep.pool_idx,
|
||||
set_index: ep.set_idx,
|
||||
disk_index: ep.disk_idx,
|
||||
runtime_state: Some(runtime_state.as_str().to_string()),
|
||||
offline_duration_seconds,
|
||||
metrics: disk.metrics_snapshot(),
|
||||
uuid: cached_disk_id.map_or_else(String::new, |id| id.to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
if let Some((total, used, free, _)) = capacity_snapshot {
|
||||
disk_info.total_space = total;
|
||||
disk_info.used_space = used;
|
||||
disk_info.available_space = free;
|
||||
disk_info.utilization = utilization_percent(total, used);
|
||||
disk_info.capacity_observation_source = Some("snapshot".to_owned());
|
||||
disk_info.capacity_observation_age_seconds =
|
||||
capacity_snapshot.map(|(_, _, _, probe_unix_secs)| capacity_snapshot_age_seconds(probe_unix_secs));
|
||||
} else {
|
||||
disk_info.capacity_observation_source = Some("missing".to_owned());
|
||||
disk_info.capacity_observation_age_seconds = Some(0);
|
||||
}
|
||||
disk_info
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn build_runtime_snapshot_disk(
|
||||
@@ -6638,6 +6650,7 @@ async fn get_storage_info(disks: &[Option<DiskStore>], eps: &[Endpoint]) -> rust
|
||||
total_sets,
|
||||
..Default::default()
|
||||
},
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
pub async fn stat_all_dirs(disks: &[Option<DiskStore>], bucket: &str, prefix: &str) -> Vec<Option<DiskError>> {
|
||||
@@ -10680,6 +10693,41 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn test_get_disks_info_probes_drives_concurrently() {
|
||||
use crate::disk::disk_store::DISK_INFO_PROBE_DELAY_FOR_TEST;
|
||||
|
||||
let format = FormatV3::new(1, 4);
|
||||
let mut temp_dirs = Vec::new();
|
||||
let mut endpoints = Vec::new();
|
||||
let mut disks = Vec::new();
|
||||
for disk_idx in 0..4 {
|
||||
let (dir, endpoint, disk) = make_formatted_local_disk_for_info_test(disk_idx, &format).await;
|
||||
temp_dirs.push(dir);
|
||||
endpoints.push(endpoint);
|
||||
disks.push(Some(disk));
|
||||
}
|
||||
|
||||
let probe_delay = std::time::Duration::from_secs(2);
|
||||
let started = tokio::time::Instant::now();
|
||||
let info = DISK_INFO_PROBE_DELAY_FOR_TEST
|
||||
.scope(probe_delay, get_disks_info(&disks, &endpoints))
|
||||
.await;
|
||||
let elapsed = started.elapsed();
|
||||
|
||||
assert_eq!(info.len(), 4);
|
||||
assert!(info.iter().all(|disk| disk.state == "ok"), "every drive should still report a live probe");
|
||||
assert_eq!(
|
||||
info.iter().map(|disk| disk.disk_index).collect::<Vec<_>>(),
|
||||
endpoints.iter().map(|ep| ep.disk_idx).collect::<Vec<_>>(),
|
||||
"concurrent probes must keep endpoint order"
|
||||
);
|
||||
assert!(
|
||||
elapsed < probe_delay * 2,
|
||||
"four stalled drives must cost one probe delay, not four; took {elapsed:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_get_disks_info_preserves_remote_cached_disk_id_when_offline() {
|
||||
let (endpoint, disk) = make_remote_disk_for_info_test(0).await;
|
||||
|
||||
@@ -24,7 +24,7 @@ use super::super::{
|
||||
};
|
||||
use crate::disk::DataDirDeleteStatus;
|
||||
use crate::disk::DiskAPI;
|
||||
use crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX;
|
||||
use crate::disk::local::{DELETE_DATA_DIR_MARKER_PREFIX, metadata_less_part_file};
|
||||
use crate::io_support::bitrot::object_mmap_read_enabled;
|
||||
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
|
||||
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, trace_emit};
|
||||
@@ -301,12 +301,6 @@ struct MetadataLessDataDirCleanup {
|
||||
touched_disks: Vec<bool>,
|
||||
}
|
||||
|
||||
fn metadata_less_part_file(entry: &str) -> bool {
|
||||
entry
|
||||
.strip_prefix("part.")
|
||||
.is_some_and(|part_number| part_number.parse::<usize>().is_ok_and(|part_number| part_number > 0))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
struct DanglingCheckPartsFailure {
|
||||
key: DanglingCheckPartsFailureKey,
|
||||
|
||||
@@ -3301,13 +3301,18 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
// (rustfs/backlog#1009): CompleteMultipartUpload keeps its pre-commit
|
||||
// `get_object_info` lookup, so the backfill has no consumer here yet.
|
||||
Self::assign_rename_data_indexes(&mut parts_metadatas);
|
||||
let mut rename_result = SetDisks::rename_data_owned(
|
||||
// Disk deadlines can expire before physical publication or failure undo drains.
|
||||
let mut rename_result = SetDisks::rename_data_owned_with_fence(
|
||||
&commit_disks,
|
||||
(RUSTFS_META_MULTIPART_BUCKET, &commit_upload_id_path),
|
||||
parts_metadatas,
|
||||
(&commit_bucket, &commit_object),
|
||||
write_quorum,
|
||||
commit_allows_early_ack,
|
||||
crate::set_disk::core::io_primitives::RenameDataFenceOptions::new(write_quorum, None)
|
||||
.with_namespace_commit_guard(
|
||||
(!crate::bucket::utils::is_meta_bucketname(&commit_bucket))
|
||||
.then(|| commit_set.ctx.begin_namespace_commit()),
|
||||
),
|
||||
)
|
||||
.await;
|
||||
if let Ok(rename_commit) = rename_result.as_mut() {
|
||||
@@ -6758,6 +6763,301 @@ mod tests {
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial(capacity_dirty_scope)]
|
||||
async fn complete_multipart_advances_namespace_generation_after_commit() {
|
||||
let (dirs, disks, set_disks) = hermetic_set_disks(4).await;
|
||||
let bucket = "multipart-namespace-commit";
|
||||
let object = "completed-object";
|
||||
let body = vec![0x65; 4096];
|
||||
make_bucket_on_all(&disks, bucket).await;
|
||||
let before = set_disks.ctx.namespace_commit_generation();
|
||||
let (upload_id, parts) =
|
||||
stage_upload_with_create_opts(&set_disks, bucket, object, &body, &ObjectOptions::default()).await;
|
||||
assert_eq!(set_disks.ctx.namespace_commit_generation(), before, "staging is not publication");
|
||||
assert!(!set_disks.ctx.namespace_commits_pending());
|
||||
tokio::time::timeout(
|
||||
Duration::from_secs(10),
|
||||
set_disks
|
||||
.clone()
|
||||
.complete_multipart_upload(bucket, object, &upload_id, parts, &ObjectOptions::default()),
|
||||
)
|
||||
.await
|
||||
.expect("completion must finish")
|
||||
.expect("the four real shards must commit");
|
||||
let mut reader = tokio::time::timeout(
|
||||
Duration::from_secs(5),
|
||||
set_disks.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default()),
|
||||
)
|
||||
.await
|
||||
.expect("GET after completion must finish")
|
||||
.expect("a successful completion must be immediately readable");
|
||||
let mut actual = Vec::new();
|
||||
tokio::time::timeout(Duration::from_secs(5), reader.stream.read_to_end(&mut actual))
|
||||
.await
|
||||
.expect("the completed body stream must finish")
|
||||
.expect("read completed body");
|
||||
assert_eq!(actual, body);
|
||||
let upload_path = SetDisks::get_upload_id_dir(bucket, object, &upload_id);
|
||||
for dir in &dirs {
|
||||
assert!(!dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_path).exists());
|
||||
}
|
||||
assert!(matches!(
|
||||
set_disks.check_upload_id_exists(bucket, object, &upload_id, false).await,
|
||||
Err(StorageError::InvalidUploadID(..))
|
||||
));
|
||||
assert!(!set_disks.ctx.namespace_commits_pending());
|
||||
assert_eq!(
|
||||
set_disks.ctx.namespace_commit_generation(),
|
||||
before + 2,
|
||||
"one completed MPU must invalidate snapshots at namespace admission and physical retirement"
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
async fn assert_complete_multipart_physical_namespace_owner(undo: bool) {
|
||||
use crate::disk::os::{self, prepared_publication_test_hooks as hooks};
|
||||
use crate::set_disk::core::io_primitives::rename_fault_injection;
|
||||
use futures::FutureExt;
|
||||
|
||||
temp_env::async_with_vars([(rustfs_config::ENV_DRIVE_MAX_TIMEOUT_DURATION, Some("60"))], async {
|
||||
let (dirs, disks, set_disks) = hermetic_set_disks(4).await;
|
||||
let bucket = "multipart-physical-namespace";
|
||||
let object = if undo { "undo-tail" } else { "publication-tail" };
|
||||
let old_body = vec![0x41; 1024];
|
||||
let new_body = vec![0x62; 4096];
|
||||
make_bucket_on_all(&disks, bucket).await;
|
||||
let old = set_disks
|
||||
.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut PutObjReader::from_vec(old_body.clone()),
|
||||
&ObjectOptions {
|
||||
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("seed a real readable old version");
|
||||
let old_etag = old.etag.expect("the committed old object must have an ETag");
|
||||
let before = set_disks.ctx.namespace_commit_generation();
|
||||
assert!(!set_disks.ctx.namespace_commits_pending());
|
||||
let (upload_id, parts) =
|
||||
stage_upload_with_create_opts(&set_disks, bucket, object, &new_body, &ObjectOptions::default()).await;
|
||||
let new_etag = get_complete_multipart_md5(&parts);
|
||||
let upload_path = SetDisks::get_upload_id_dir(bucket, object, &upload_id);
|
||||
for dir in &dirs {
|
||||
assert!(dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_path).exists());
|
||||
}
|
||||
assert_eq!(set_disks.ctx.namespace_commit_generation(), before);
|
||||
let _fault = undo.then(|| rename_fault_injection::fail_rename_on(object, &[2, 3]));
|
||||
let stage = if undo {
|
||||
hooks::Stage::Rename
|
||||
} else {
|
||||
hooks::Stage::PreparedRename
|
||||
};
|
||||
let (entered_tx, mut entered_rx) = tokio::sync::mpsc::unbounded_channel();
|
||||
let mut hooks = Vec::new();
|
||||
let mut releases = Vec::new();
|
||||
let mut mutation_paths = Vec::new();
|
||||
for (index, disk) in disks.iter().enumerate() {
|
||||
let disk::Disk::Local(local) = disk.as_ref() else {
|
||||
panic!("physical MPU fixture requires local disks");
|
||||
};
|
||||
let destination = local
|
||||
.get_disk()
|
||||
.get_object_path_for_io(bucket, object)
|
||||
.expect("leased IO path");
|
||||
let entered_tx = entered_tx.clone();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
hooks.push(hooks::install_at(stage, &destination.join(STORAGE_FORMAT_FILE), move || {
|
||||
let _ = entered_tx.send(index);
|
||||
// Dropping senders releases every syscall on assertion failure, too.
|
||||
let _ = release_rx.recv();
|
||||
}));
|
||||
releases.push(release_tx);
|
||||
// Canonical rename serializes the object directory; backup restore
|
||||
// serializes its xl.meta destination. Drain the actual executor key.
|
||||
mutation_paths.push(if undo {
|
||||
destination.join(STORAGE_FORMAT_FILE)
|
||||
} else {
|
||||
destination
|
||||
});
|
||||
}
|
||||
drop(entered_tx);
|
||||
let complete_set = set_disks.clone();
|
||||
let complete_upload = upload_id.clone();
|
||||
let mut complete = tokio::spawn(async move {
|
||||
complete_set
|
||||
.complete_multipart_upload(bucket, object, &complete_upload, parts, &ObjectOptions::default())
|
||||
.await
|
||||
});
|
||||
let mut complete_joined = false;
|
||||
let mut observed_counts = None;
|
||||
let observations = std::panic::AssertUnwindSafe(async {
|
||||
let expected_publishers = if undo { 2 } else { 4 };
|
||||
let entered = tokio::time::timeout(Duration::from_secs(10), async {
|
||||
let mut entered = HashSet::new();
|
||||
while entered.len() < expected_publishers {
|
||||
tokio::select! {
|
||||
index = entered_rx.recv() => {
|
||||
assert!(entered.insert(index.expect("physical publisher must signal entry")));
|
||||
}
|
||||
result = &mut complete => {
|
||||
complete_joined = true;
|
||||
panic!("completion returned before physical entry: {result:?}");
|
||||
}
|
||||
}
|
||||
}
|
||||
entered
|
||||
})
|
||||
.await
|
||||
.expect("all expected physical metadata operations must enter");
|
||||
let pending_at_entry = set_disks.ctx.namespace_commits_pending();
|
||||
let generation_at_entry = set_disks.ctx.namespace_commit_generation();
|
||||
for &index in &entered {
|
||||
let metadata = tokio::time::timeout(
|
||||
Duration::from_secs(5),
|
||||
disks[index].read_version("", bucket, object, "", &ReadOptions::default()),
|
||||
)
|
||||
.await
|
||||
.expect("metadata observation must finish while publication is paused")
|
||||
.expect("metadata before the paused physical action must be readable");
|
||||
assert_eq!(
|
||||
metadata.metadata.get("etag"),
|
||||
Some(if undo { &new_etag } else { &old_etag }),
|
||||
"undo must follow actual publication; prepared rename must precede publication"
|
||||
);
|
||||
}
|
||||
|
||||
// The entry signals run inside the real blocking closures, after each
|
||||
// wrapper installed its normal deadline. No quota/external guard disables it.
|
||||
tokio::time::pause();
|
||||
tokio::time::advance(Duration::from_secs(61)).await;
|
||||
tokio::time::resume();
|
||||
let joined = tokio::time::timeout(Duration::from_secs(5), &mut complete).await;
|
||||
complete_joined = joined.is_ok();
|
||||
let result = joined
|
||||
.expect("ordinary MPU disk/undo deadlines must still return before physical drain")
|
||||
.expect("completion task must not panic");
|
||||
assert!(result.is_err(), "a timed-out or two-shard commit cannot acknowledge success");
|
||||
let pending_after_timeout = set_disks.ctx.namespace_commits_pending();
|
||||
let generation_after_timeout = set_disks.ctx.namespace_commit_generation();
|
||||
observed_counts = Some((pending_at_entry, generation_at_entry, pending_after_timeout, generation_after_timeout));
|
||||
for &index in &entered {
|
||||
assert!(
|
||||
os::acquire_rename_data_mutation_lease(&disks[index].path(), bucket, &mutation_paths[index])
|
||||
.now_or_never()
|
||||
.is_none(),
|
||||
"timed-out physical work must still own its object serialization"
|
||||
);
|
||||
}
|
||||
for dir in &dirs {
|
||||
assert!(
|
||||
dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_path).exists(),
|
||||
"failed completion must not clean the upload staging"
|
||||
);
|
||||
}
|
||||
})
|
||||
.catch_unwind()
|
||||
.await;
|
||||
|
||||
// Release even after a failed observation, then finish dispatch before
|
||||
// draining every physical key. No per-disk assertion may skip a later drain.
|
||||
drop(releases);
|
||||
drop(hooks);
|
||||
let coordinator_drained = complete_joined
|
||||
|| tokio::time::timeout(Duration::from_secs(10), &mut complete).await.is_ok();
|
||||
if !coordinator_drained {
|
||||
complete.abort();
|
||||
let _ = tokio::time::timeout(Duration::from_secs(5), &mut complete).await;
|
||||
}
|
||||
let drains = futures::future::join_all(disks.iter().zip(&mutation_paths).map(|(disk, destination)| async move {
|
||||
tokio::time::timeout(
|
||||
Duration::from_secs(5),
|
||||
os::acquire_rename_data_mutation_lease(&disk.path(), bucket, destination),
|
||||
)
|
||||
.await
|
||||
.map(drop)
|
||||
}))
|
||||
.await;
|
||||
let owner_drained = tokio::time::timeout(Duration::from_secs(5), async {
|
||||
while set_disks.ctx.namespace_commits_pending() {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await;
|
||||
let physical_drained = drains.iter().all(|drain| drain.is_ok());
|
||||
if !coordinator_drained || !physical_drained || owner_drained.is_err() {
|
||||
// A bounded cleanup failure cannot justify deleting roots that a
|
||||
// detached executor might still use. Keep them for diagnosis.
|
||||
let retained = dirs.into_iter().map(TempDir::keep).collect::<Vec<_>>();
|
||||
eprintln!("MPU cleanup incomplete: coordinator={coordinator_drained}, physical={physical_drained}, retained={retained:?}");
|
||||
if let Err(panic) = observations {
|
||||
std::panic::resume_unwind(panic);
|
||||
}
|
||||
panic!("MPU cleanup did not drain: coordinator={coordinator_drained}, physical={physical_drained}, retained={retained:?}");
|
||||
}
|
||||
if let Err(panic) = observations {
|
||||
std::panic::resume_unwind(panic);
|
||||
}
|
||||
// Preserve the original drain checks after collecting every result.
|
||||
for drained in drains {
|
||||
drained.expect("released physical MPU work must drain");
|
||||
}
|
||||
owner_drained.expect("physical retirement must finish its namespace counter decrement");
|
||||
for disk in &disks {
|
||||
let metadata = disk
|
||||
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
.expect("all disks must expose the expected final metadata");
|
||||
assert_eq!(metadata.metadata.get("etag"), Some(if undo { &old_etag } else { &new_etag }));
|
||||
}
|
||||
let (pending_at_entry, generation_at_entry, pending_after_timeout, generation_after_timeout) =
|
||||
observed_counts.expect("successful observations must record namespace counters");
|
||||
let mut reader = tokio::time::timeout(
|
||||
Duration::from_secs(5),
|
||||
set_disks.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default()),
|
||||
)
|
||||
.await
|
||||
.expect("GET after physical drain must finish")
|
||||
.expect("the real final object must be readable");
|
||||
let mut actual = Vec::new();
|
||||
tokio::time::timeout(Duration::from_secs(5), reader.stream.read_to_end(&mut actual))
|
||||
.await
|
||||
.expect("the final object stream must finish")
|
||||
.expect("read final object bytes");
|
||||
assert_eq!(actual, if undo { old_body } else { new_body });
|
||||
assert!(
|
||||
pending_at_entry && pending_after_timeout,
|
||||
"physical MPU work outlived namespace accounting: undo={undo}"
|
||||
);
|
||||
assert_eq!(generation_at_entry, before + 1);
|
||||
assert_eq!(
|
||||
generation_after_timeout, generation_at_entry,
|
||||
"the blocked physical owner cannot retire early"
|
||||
);
|
||||
assert_eq!(set_disks.ctx.namespace_commit_generation(), before + 2);
|
||||
assert!(!set_disks.ctx.namespace_commits_pending());
|
||||
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
#[tokio::test]
|
||||
#[serial(capacity_dirty_scope)]
|
||||
async fn complete_multipart_timeout_keeps_namespace_owner_until_physical_publication() {
|
||||
assert_complete_multipart_physical_namespace_owner(false).await;
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
#[tokio::test]
|
||||
#[serial(capacity_dirty_scope)]
|
||||
async fn complete_multipart_failed_quorum_keeps_namespace_owner_until_physical_undo() {
|
||||
assert_complete_multipart_physical_namespace_owner(true).await;
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
#[serial]
|
||||
async fn complete_multipart_releases_disk_snapshot_before_cleanup() {
|
||||
|
||||
@@ -5666,6 +5666,10 @@ impl Drop for TransitionUploadCleanup {
|
||||
if !self.armed {
|
||||
return;
|
||||
}
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
if transition_transaction_kill_point_is_active(self.cleanup_transaction.as_ref()) {
|
||||
return;
|
||||
}
|
||||
let Some(candidate) = self.candidate.as_ref() else {
|
||||
return;
|
||||
};
|
||||
@@ -5870,17 +5874,29 @@ fn transition_source_identity(
|
||||
}
|
||||
|
||||
async fn save_transition_transaction_if_available(api: Option<&Arc<ECStore>>, transaction: &TransitionTransaction) -> Result<()> {
|
||||
if let Some(api) = api {
|
||||
return save_transition_transaction_record(api.clone(), transaction).await;
|
||||
}
|
||||
#[cfg(test)]
|
||||
{
|
||||
Ok(())
|
||||
}
|
||||
#[cfg(not(test))]
|
||||
{
|
||||
Err(Error::other("transition transaction store is unavailable"))
|
||||
}
|
||||
let started = std::time::Instant::now();
|
||||
let result = if let Some(api) = api {
|
||||
save_transition_transaction_record(api.clone(), transaction).await
|
||||
} else {
|
||||
#[cfg(test)]
|
||||
{
|
||||
Ok(())
|
||||
}
|
||||
#[cfg(not(test))]
|
||||
{
|
||||
Err(Error::other("transition transaction store is unavailable"))
|
||||
}
|
||||
};
|
||||
#[cfg(test)]
|
||||
record_transition_transaction_mutation(
|
||||
transaction,
|
||||
TransitionTransactionMutationKind::Create,
|
||||
None,
|
||||
started.elapsed(),
|
||||
result.is_ok(),
|
||||
);
|
||||
result
|
||||
}
|
||||
|
||||
async fn compare_and_save_transition_transaction_if_available(
|
||||
@@ -5888,19 +5904,31 @@ async fn compare_and_save_transition_transaction_if_available(
|
||||
expected: &TransitionTransaction,
|
||||
next: &TransitionTransaction,
|
||||
) -> Result<()> {
|
||||
if let Some(api) = api {
|
||||
#[cfg(test)]
|
||||
let started = std::time::Instant::now();
|
||||
let result = if let Some(api) = api {
|
||||
// The transition worker already has a deep poll chain. Keep the CAS
|
||||
// read/write/receipt future off Tokio's default worker stack.
|
||||
return Box::pin(save_transition_transaction_record_if_current(api.clone(), expected, next)).await;
|
||||
}
|
||||
Box::pin(save_transition_transaction_record_if_current(api.clone(), expected, next)).await
|
||||
} else {
|
||||
#[cfg(test)]
|
||||
{
|
||||
Ok(())
|
||||
}
|
||||
#[cfg(not(test))]
|
||||
{
|
||||
Err(Error::other("transition transaction store is unavailable"))
|
||||
}
|
||||
};
|
||||
#[cfg(test)]
|
||||
{
|
||||
Ok(())
|
||||
}
|
||||
#[cfg(not(test))]
|
||||
{
|
||||
Err(Error::other("transition transaction store is unavailable"))
|
||||
}
|
||||
record_transition_transaction_mutation(
|
||||
next,
|
||||
TransitionTransactionMutationKind::CompareAndSave,
|
||||
Some(expected.state),
|
||||
started.elapsed(),
|
||||
result.is_ok(),
|
||||
);
|
||||
result
|
||||
}
|
||||
|
||||
async fn advance_and_save_transition_transaction(
|
||||
@@ -5909,8 +5937,6 @@ async fn advance_and_save_transition_transaction(
|
||||
next: TransitionTransactionState,
|
||||
remote_version: Option<TransitionRemoteVersion>,
|
||||
) -> Result<()> {
|
||||
#[cfg(test)]
|
||||
record_transition_uploaded_save_attempt(transaction, next);
|
||||
let expected = transaction.clone();
|
||||
let mut advanced = expected.clone();
|
||||
advanced
|
||||
@@ -5922,10 +5948,33 @@ async fn advance_and_save_transition_transaction(
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
struct TransitionUploadedSaveProbeState {
|
||||
struct TransitionTransactionMutationProbeState {
|
||||
bucket: String,
|
||||
object: String,
|
||||
attempts: std::sync::atomic::AtomicUsize,
|
||||
observations: std::sync::Mutex<Vec<TransitionTransactionMutationObservation>>,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||
pub(crate) enum TransitionTransactionMutationKind {
|
||||
Create,
|
||||
CompareAndSave,
|
||||
Delete,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
#[derive(Clone, Debug)]
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "full mutation measurements are consumed by tests behind `--features test-util`"
|
||||
)]
|
||||
pub(crate) struct TransitionTransactionMutationObservation {
|
||||
pub(crate) kind: TransitionTransactionMutationKind,
|
||||
pub(crate) previous_state: Option<TransitionTransactionState>,
|
||||
pub(crate) state: TransitionTransactionState,
|
||||
pub(crate) encoded_bytes: usize,
|
||||
pub(crate) elapsed: std::time::Duration,
|
||||
pub(crate) succeeded: bool,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -5933,31 +5982,35 @@ struct TransitionUploadedSaveProbeState {
|
||||
dead_code,
|
||||
reason = "installed by set_disk tests behind `--features test-util` (backlog#1823)"
|
||||
)]
|
||||
struct TransitionUploadedSaveProbe {
|
||||
state: Arc<TransitionUploadedSaveProbeState>,
|
||||
pub(crate) struct TransitionTransactionMutationProbe {
|
||||
state: Arc<TransitionTransactionMutationProbeState>,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
static TRANSITION_UPLOADED_SAVE_PROBE: std::sync::OnceLock<std::sync::Mutex<Option<Arc<TransitionUploadedSaveProbeState>>>> =
|
||||
std::sync::OnceLock::new();
|
||||
static TRANSITION_TRANSACTION_MUTATION_PROBE: std::sync::OnceLock<
|
||||
std::sync::Mutex<Option<Arc<TransitionTransactionMutationProbeState>>>,
|
||||
> = std::sync::OnceLock::new();
|
||||
|
||||
#[cfg(test)]
|
||||
impl TransitionUploadedSaveProbe {
|
||||
impl TransitionTransactionMutationProbe {
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "installed by set_disk tests behind `--features test-util` (backlog#1823)"
|
||||
)]
|
||||
fn install(bucket: &str, object: &str) -> Self {
|
||||
let state = Arc::new(TransitionUploadedSaveProbeState {
|
||||
pub(crate) fn install(bucket: &str, object: &str) -> Self {
|
||||
let state = Arc::new(TransitionTransactionMutationProbeState {
|
||||
bucket: bucket.to_string(),
|
||||
object: object.to_string(),
|
||||
attempts: std::sync::atomic::AtomicUsize::new(0),
|
||||
observations: std::sync::Mutex::new(Vec::new()),
|
||||
});
|
||||
let mut slot = TRANSITION_UPLOADED_SAVE_PROBE
|
||||
let mut slot = TRANSITION_TRANSACTION_MUTATION_PROBE
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("transition uploaded-save probe mutex should not poison");
|
||||
assert!(slot.is_none(), "transition uploaded-save probe must be installed by one test at a time");
|
||||
.expect("transition transaction mutation probe mutex should not poison");
|
||||
assert!(
|
||||
slot.is_none(),
|
||||
"transition transaction mutation probe must be installed by one test at a time"
|
||||
);
|
||||
*slot = Some(Arc::clone(&state));
|
||||
drop(slot);
|
||||
Self { state }
|
||||
@@ -5968,17 +6021,31 @@ impl TransitionUploadedSaveProbe {
|
||||
reason = "installed by set_disk tests behind `--features test-util` (backlog#1823)"
|
||||
)]
|
||||
fn attempts(&self) -> usize {
|
||||
self.state.attempts.load(std::sync::atomic::Ordering::Acquire)
|
||||
self.observations()
|
||||
.into_iter()
|
||||
.filter(|observation| {
|
||||
observation.kind == TransitionTransactionMutationKind::CompareAndSave
|
||||
&& observation.state == TransitionTransactionState::Uploaded
|
||||
})
|
||||
.count()
|
||||
}
|
||||
|
||||
pub(crate) fn observations(&self) -> Vec<TransitionTransactionMutationObservation> {
|
||||
self.state
|
||||
.observations
|
||||
.lock()
|
||||
.expect("transition transaction mutation observations mutex should not poison")
|
||||
.clone()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
impl Drop for TransitionUploadedSaveProbe {
|
||||
impl Drop for TransitionTransactionMutationProbe {
|
||||
fn drop(&mut self) {
|
||||
let mut slot = TRANSITION_UPLOADED_SAVE_PROBE
|
||||
let mut slot = TRANSITION_TRANSACTION_MUTATION_PROBE
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("transition uploaded-save probe mutex should not poison");
|
||||
.expect("transition transaction mutation probe mutex should not poison");
|
||||
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
|
||||
*slot = None;
|
||||
}
|
||||
@@ -5986,19 +6053,34 @@ impl Drop for TransitionUploadedSaveProbe {
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
fn record_transition_uploaded_save_attempt(transaction: &TransitionTransaction, next: TransitionTransactionState) {
|
||||
if next != TransitionTransactionState::Uploaded {
|
||||
return;
|
||||
}
|
||||
let state = TRANSITION_UPLOADED_SAVE_PROBE
|
||||
fn record_transition_transaction_mutation(
|
||||
transaction: &TransitionTransaction,
|
||||
kind: TransitionTransactionMutationKind,
|
||||
previous_state: Option<TransitionTransactionState>,
|
||||
elapsed: std::time::Duration,
|
||||
succeeded: bool,
|
||||
) {
|
||||
let state = TRANSITION_TRANSACTION_MUTATION_PROBE
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("transition uploaded-save probe mutex should not poison")
|
||||
.expect("transition transaction mutation probe mutex should not poison")
|
||||
.as_ref()
|
||||
.filter(|state| state.bucket == transaction.source.bucket && state.object == transaction.source.object)
|
||||
.cloned();
|
||||
if let Some(state) = state {
|
||||
state.attempts.fetch_add(1, std::sync::atomic::Ordering::AcqRel);
|
||||
let encoded_bytes = transaction.encode().map_or(0, |encoded| encoded.len());
|
||||
state
|
||||
.observations
|
||||
.lock()
|
||||
.expect("transition transaction mutation observations mutex should not poison")
|
||||
.push(TransitionTransactionMutationObservation {
|
||||
kind,
|
||||
previous_state,
|
||||
state: transaction.state,
|
||||
encoded_bytes,
|
||||
elapsed,
|
||||
succeeded,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6006,12 +6088,24 @@ async fn delete_transition_transaction_if_available(
|
||||
api: Option<&Arc<ECStore>>,
|
||||
transaction: &TransitionTransaction,
|
||||
) -> Result<()> {
|
||||
if let Some(api) = api {
|
||||
#[cfg(test)]
|
||||
let started = std::time::Instant::now();
|
||||
let result = if let Some(api) = api {
|
||||
// Conditional delete now includes a read and terminal receipt; box it
|
||||
// for the same transition-worker stack bound as the CAS path above.
|
||||
return Box::pin(delete_transition_transaction_record(api.clone(), transaction)).await;
|
||||
}
|
||||
Ok(())
|
||||
Box::pin(delete_transition_transaction_record(api.clone(), transaction)).await
|
||||
} else {
|
||||
Ok(())
|
||||
};
|
||||
#[cfg(test)]
|
||||
record_transition_transaction_mutation(
|
||||
transaction,
|
||||
TransitionTransactionMutationKind::Delete,
|
||||
Some(transaction.state),
|
||||
started.elapsed(),
|
||||
result.is_ok(),
|
||||
);
|
||||
result
|
||||
}
|
||||
|
||||
async fn delete_transition_transaction_after_remote_cleanup(
|
||||
@@ -6239,6 +6333,103 @@ async fn pause_after_transition_uploaded_persisted(bucket: &str, object: &str) {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||
pub(crate) enum TransitionTransactionKillPoint {
|
||||
PrePutFence,
|
||||
UploadBeforeCommitFence,
|
||||
CommitFenceBeforeLocalCommit,
|
||||
LocalCommitBeforeDelete,
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
struct TransitionTransactionKillPointBarrierState {
|
||||
bucket: String,
|
||||
object: String,
|
||||
point: TransitionTransactionKillPoint,
|
||||
arrived: tokio::sync::Notify,
|
||||
release: tokio::sync::Notify,
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) struct TransitionTransactionKillPointBarrier {
|
||||
state: Arc<TransitionTransactionKillPointBarrierState>,
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
static TRANSITION_TRANSACTION_KILL_POINT_BARRIER: std::sync::OnceLock<
|
||||
std::sync::Mutex<Option<Arc<TransitionTransactionKillPointBarrierState>>>,
|
||||
> = std::sync::OnceLock::new();
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
impl TransitionTransactionKillPointBarrier {
|
||||
pub(crate) fn install(bucket: &str, object: &str, point: TransitionTransactionKillPoint) -> Self {
|
||||
let state = Arc::new(TransitionTransactionKillPointBarrierState {
|
||||
bucket: bucket.to_string(),
|
||||
object: object.to_string(),
|
||||
point,
|
||||
arrived: tokio::sync::Notify::new(),
|
||||
release: tokio::sync::Notify::new(),
|
||||
});
|
||||
let mut slot = TRANSITION_TRANSACTION_KILL_POINT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("transition transaction kill-point barrier mutex should not poison");
|
||||
assert!(slot.is_none(), "one transition transaction kill-point may be installed at a time");
|
||||
*slot = Some(Arc::clone(&state));
|
||||
drop(slot);
|
||||
Self { state }
|
||||
}
|
||||
|
||||
pub(crate) async fn wait_until_paused(&self) {
|
||||
tokio::time::timeout(Duration::from_secs(30), self.state.arrived.notified())
|
||||
.await
|
||||
.expect("transition should reach the requested transaction kill-point");
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
impl Drop for TransitionTransactionKillPointBarrier {
|
||||
fn drop(&mut self) {
|
||||
self.state.release.notify_one();
|
||||
let mut slot = TRANSITION_TRANSACTION_KILL_POINT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("transition transaction kill-point barrier mutex should not poison");
|
||||
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
|
||||
*slot = None;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
async fn pause_transition_transaction_at(bucket: &str, object: &str, point: TransitionTransactionKillPoint) {
|
||||
let barrier = TRANSITION_TRANSACTION_KILL_POINT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("transition transaction kill-point barrier mutex should not poison")
|
||||
.as_ref()
|
||||
.filter(|barrier| barrier.bucket == bucket && barrier.object == object && barrier.point == point)
|
||||
.cloned();
|
||||
if let Some(barrier) = barrier {
|
||||
barrier.arrived.notify_one();
|
||||
barrier.release.notified().await;
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
fn transition_transaction_kill_point_is_active(transaction: Option<&TransitionTransaction>) -> bool {
|
||||
let Some(transaction) = transaction else {
|
||||
return false;
|
||||
};
|
||||
TRANSITION_TRANSACTION_KILL_POINT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("transition transaction kill-point barrier mutex should not poison")
|
||||
.as_ref()
|
||||
.is_some_and(|barrier| barrier.bucket == transaction.source.bucket && barrier.object == transaction.source.object)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
enum TransitionCommitPause {
|
||||
@@ -8987,7 +9178,12 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
||||
|
||||
let oi = ObjectInfo::from_file_info(&fi, bucket, object, opts.versioned || opts.version_suspended);
|
||||
let transaction_api = transition_object_store(&self.ctx).await;
|
||||
let mut transaction = TransitionTransaction::new(TransitionTransactionInit {
|
||||
let transition_compaction_fleet_proof =
|
||||
crate::services::notification_sys::acquire_transition_transaction_compaction_fleet_proof();
|
||||
let compact_transition_transaction = transition_compaction_fleet_proof
|
||||
.as_ref()
|
||||
.is_some_and(crate::services::notification_sys::transition_transaction_compaction_fleet_proof_matches);
|
||||
let transaction_init = TransitionTransactionInit {
|
||||
deployment_id: transition_deployment_id(&self.ctx)?,
|
||||
transaction_id: Uuid::new_v4(),
|
||||
owner_epoch: Uuid::new_v4(),
|
||||
@@ -8996,9 +9192,16 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
||||
tier_name: opts.transition.tier.clone(),
|
||||
backend_fingerprint: tgt_client.backend_identity(),
|
||||
not_after_unix_nanos: transition_transaction_not_after_unix_nanos()?,
|
||||
})
|
||||
};
|
||||
let mut transaction = if compact_transition_transaction {
|
||||
TransitionTransaction::new_compact(transaction_init)
|
||||
} else {
|
||||
TransitionTransaction::new(transaction_init)
|
||||
}
|
||||
.map_err(Error::other)?;
|
||||
save_transition_transaction_if_available(transaction_api.as_ref(), &transaction).await?;
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pause_transition_transaction_at(bucket, object, TransitionTransactionKillPoint::PrePutFence).await;
|
||||
let transaction_id = transaction.transaction_id;
|
||||
let dest_obj = transaction.remote_object.clone();
|
||||
let mut transition_meta = (*oi.user_defined).clone();
|
||||
@@ -9075,14 +9278,16 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
||||
|
||||
let mut upload_cleanup = TransitionUploadCleanup::new(tgt_client, &dest_obj);
|
||||
upload_cleanup.set_cleanup_owner(transaction_api.clone(), &transaction);
|
||||
advance_and_save_transition_transaction(
|
||||
transaction_api.as_ref(),
|
||||
&mut transaction,
|
||||
TransitionTransactionState::UploadOutcomeUnknown,
|
||||
None,
|
||||
)
|
||||
.await?;
|
||||
upload_cleanup.update_cleanup_transaction(&transaction);
|
||||
if !compact_transition_transaction {
|
||||
advance_and_save_transition_transaction(
|
||||
transaction_api.as_ref(),
|
||||
&mut transaction,
|
||||
TransitionTransactionState::UploadOutcomeUnknown,
|
||||
None,
|
||||
)
|
||||
.await?;
|
||||
upload_cleanup.update_cleanup_transaction(&transaction);
|
||||
}
|
||||
let remote_upload = {
|
||||
let lease = &upload_cleanup.lease;
|
||||
let recorded_candidate = &mut upload_cleanup.candidate;
|
||||
@@ -9142,27 +9347,31 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
||||
return Err(err.into());
|
||||
}
|
||||
};
|
||||
if let Err(err) = advance_and_save_transition_transaction(
|
||||
transaction_api.as_ref(),
|
||||
&mut transaction,
|
||||
TransitionTransactionState::Uploaded,
|
||||
Some(TransitionRemoteVersion::known_from_put_response(candidate.remote_version().to_string())),
|
||||
)
|
||||
.await
|
||||
{
|
||||
let cleanup_api = transition_cleanup_store(&self.ctx).await;
|
||||
if let Err(cleanup_err) = upload_cleanup.cleanup_rejected_upload(cleanup_api, &mut transaction).await {
|
||||
return Err(StorageError::Io(std::io::Error::other(format!(
|
||||
"{err}; uploaded transition transaction persist failed and cleanup failed: {cleanup_err}"
|
||||
))));
|
||||
if !compact_transition_transaction {
|
||||
if let Err(err) = advance_and_save_transition_transaction(
|
||||
transaction_api.as_ref(),
|
||||
&mut transaction,
|
||||
TransitionTransactionState::Uploaded,
|
||||
Some(TransitionRemoteVersion::known_from_put_response(candidate.remote_version().to_string())),
|
||||
)
|
||||
.await
|
||||
{
|
||||
let cleanup_api = transition_cleanup_store(&self.ctx).await;
|
||||
if let Err(cleanup_err) = upload_cleanup.cleanup_rejected_upload(cleanup_api, &mut transaction).await {
|
||||
return Err(StorageError::Io(std::io::Error::other(format!(
|
||||
"{err}; uploaded transition transaction persist failed and cleanup failed: {cleanup_err}"
|
||||
))));
|
||||
}
|
||||
delete_transition_transaction_after_remote_cleanup(transaction_api.as_ref(), &transaction, bucket, object).await;
|
||||
return Err(err);
|
||||
}
|
||||
delete_transition_transaction_after_remote_cleanup(transaction_api.as_ref(), &transaction, bucket, object).await;
|
||||
return Err(err);
|
||||
upload_cleanup.update_cleanup_transaction(&transaction);
|
||||
}
|
||||
upload_cleanup.update_cleanup_transaction(&transaction);
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pause_after_transition_uploaded_persisted(bucket, object).await;
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pause_transition_transaction_at(bucket, object, TransitionTransactionKillPoint::UploadBeforeCommitFence).await;
|
||||
|
||||
let commit_opts = opts.as_commit_opts();
|
||||
// Note: Using clone() here is necessary because ObjectOptions has 124 fields.
|
||||
@@ -9273,11 +9482,25 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
||||
}
|
||||
return Err(Error::other("remote version state fleet capability changed during transition"));
|
||||
}
|
||||
if compact_transition_transaction
|
||||
&& !transition_compaction_fleet_proof
|
||||
.as_ref()
|
||||
.is_some_and(crate::services::notification_sys::transition_transaction_compaction_fleet_proof_matches)
|
||||
{
|
||||
drop(transition_lock_guard);
|
||||
if upload_cleanup.cleanup().await.is_ok() {
|
||||
delete_transition_transaction_after_remote_cleanup(transaction_api.as_ref(), &transaction, bucket, object).await;
|
||||
}
|
||||
return Err(Error::other(
|
||||
"transition transaction compaction fleet capability changed during transition",
|
||||
));
|
||||
}
|
||||
if let Err(err) = advance_and_save_transition_transaction(
|
||||
transaction_api.as_ref(),
|
||||
&mut transaction,
|
||||
TransitionTransactionState::LocalCommitStarted,
|
||||
None,
|
||||
compact_transition_transaction
|
||||
.then(|| TransitionRemoteVersion::known_from_put_response(candidate.remote_version().to_string())),
|
||||
)
|
||||
.await
|
||||
{
|
||||
@@ -9287,6 +9510,9 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
upload_cleanup.update_cleanup_transaction(&transaction);
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pause_transition_transaction_at(bucket, object, TransitionTransactionKillPoint::CommitFenceBeforeLocalCommit).await;
|
||||
upload_cleanup.disarm();
|
||||
if let Err(err) = self.delete_object_version(bucket, object, &fi, false).await {
|
||||
warn!(
|
||||
@@ -9299,34 +9525,48 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
||||
drop(transition_lock_guard);
|
||||
return Err(err);
|
||||
}
|
||||
match advance_and_save_transition_transaction(
|
||||
transaction_api.as_ref(),
|
||||
&mut transaction,
|
||||
TransitionTransactionState::Committed,
|
||||
None,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
if let Err(err) = delete_transition_transaction_if_available(transaction_api.as_ref(), &transaction).await {
|
||||
warn!(
|
||||
bucket = bucket,
|
||||
object = object,
|
||||
transaction_id = %transaction_id,
|
||||
error = ?err,
|
||||
"transition committed locally but transaction cleanup failed"
|
||||
);
|
||||
}
|
||||
}
|
||||
Err(err) => {
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pause_transition_transaction_at(bucket, object, TransitionTransactionKillPoint::LocalCommitBeforeDelete).await;
|
||||
if compact_transition_transaction {
|
||||
if let Err(err) = delete_transition_transaction_if_available(transaction_api.as_ref(), &transaction).await {
|
||||
warn!(
|
||||
bucket = bucket,
|
||||
object = object,
|
||||
transaction_id = %transaction_id,
|
||||
error = ?err,
|
||||
"transition committed locally but transaction committed-state advance failed"
|
||||
"transition committed locally but compact transaction cleanup failed"
|
||||
);
|
||||
}
|
||||
} else {
|
||||
match advance_and_save_transition_transaction(
|
||||
transaction_api.as_ref(),
|
||||
&mut transaction,
|
||||
TransitionTransactionState::Committed,
|
||||
None,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
if let Err(err) = delete_transition_transaction_if_available(transaction_api.as_ref(), &transaction).await {
|
||||
warn!(
|
||||
bucket = bucket,
|
||||
object = object,
|
||||
transaction_id = %transaction_id,
|
||||
error = ?err,
|
||||
"transition committed locally but transaction cleanup failed"
|
||||
);
|
||||
}
|
||||
}
|
||||
Err(err) => {
|
||||
warn!(
|
||||
bucket = bucket,
|
||||
object = object,
|
||||
transaction_id = %transaction_id,
|
||||
error = ?err,
|
||||
"transition committed locally but transaction committed-state advance failed"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// delete_object_version persisted transition_status=complete and freed the
|
||||
@@ -14929,6 +15169,7 @@ mod transition_upload_integrity_tests {
|
||||
use super::*;
|
||||
use crate::bucket::lifecycle::lifecycle::{TRANSITION_PENDING, TransitionOptions};
|
||||
use crate::layout::endpoints::SetupType;
|
||||
use crate::services::notification_sys::install_transition_transaction_compaction_fleet_proof_for_test;
|
||||
use crate::services::tier::test_util::register_mock_tier;
|
||||
use crate::set_disk::replication::RestoreFinalizeBarrier;
|
||||
use http::HeaderMap;
|
||||
@@ -16195,6 +16436,7 @@ mod transition_upload_integrity_tests {
|
||||
let original = write_source(&set_disks, &disk_stores, bucket, object, &payload).await;
|
||||
let tier_name = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
||||
let backend = register_mock_tier(&runtime_sources::global_tier_config_mgr(), &tier_name).await;
|
||||
let _compaction_proof = install_transition_transaction_compaction_fleet_proof_for_test("object-transaction-fencing-test");
|
||||
let barrier = TransitionCommitBarrier::install(bucket, object);
|
||||
|
||||
let transition_set = Arc::clone(&set_disks);
|
||||
@@ -16397,7 +16639,7 @@ mod transition_upload_integrity_tests {
|
||||
let tier_name = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
||||
let backend = register_mock_tier(&runtime_sources::global_tier_config_mgr(), &tier_name).await;
|
||||
backend.set_put_remote_version(Some(String::new())).await;
|
||||
let save_probe = TransitionUploadedSaveProbe::install(bucket, object);
|
||||
let save_probe = TransitionTransactionMutationProbe::install(bucket, object);
|
||||
|
||||
set_disks
|
||||
.transition_object(bucket, object, &transition_options(&original, tier_name))
|
||||
@@ -16461,7 +16703,7 @@ mod transition_upload_integrity_tests {
|
||||
let remote_version = Uuid::nil().to_string();
|
||||
let backend = register_mock_tier(&runtime_sources::global_tier_config_mgr(), &tier_name).await;
|
||||
backend.set_put_remote_version(Some(remote_version.clone())).await;
|
||||
let save_probe = TransitionUploadedSaveProbe::install(bucket, object);
|
||||
let save_probe = TransitionTransactionMutationProbe::install(bucket, object);
|
||||
|
||||
set_disks
|
||||
.transition_object(bucket, object, &transition_options(&original, tier_name))
|
||||
|
||||
@@ -2353,11 +2353,22 @@ mod tests {
|
||||
.await
|
||||
.expect("quorum boundary heal should return a mapped result");
|
||||
*store.pools[0].disk_set[0].disks.write().await = original_quorum_disks;
|
||||
let quorum_err = quorum_err
|
||||
.as_ref()
|
||||
.expect("heal must fail closed when capacity admission cannot verify pool metadata");
|
||||
let quorum_failure = quorum_err
|
||||
.pool_metadata_failure()
|
||||
.expect("capacity admission failure should preserve typed pool metadata context");
|
||||
assert_eq!(
|
||||
quorum_failure.kind,
|
||||
crate::error::PoolMetadataFailure::ReadUnavailable,
|
||||
"read-only capacity admission failure must remain retryable"
|
||||
);
|
||||
assert_eq!(quorum_failure.operation, "target capacity admission failed");
|
||||
assert_eq!(quorum_failure.phase, "pool_read");
|
||||
assert!(
|
||||
quorum_err.as_ref().is_some_and(|err| err
|
||||
.to_string()
|
||||
.contains("pool metadata writes remain blocked after a recovery-required replica state")),
|
||||
"heal must fail closed when capacity admission cannot verify pool metadata, got {quorum_err:?}"
|
||||
store.pool_meta_writes_ready().await,
|
||||
"read-only capacity admission failure must not latch the pool metadata writer"
|
||||
);
|
||||
shutdown.cancel();
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -109,6 +109,21 @@ pub(crate) async fn connect_load_init_formats_with_instance_ctx(
|
||||
let fresh_bootstrap_proven = should_init_erasure_disks(&errs);
|
||||
let formats_present = formats.iter().flatten().count();
|
||||
let mut format_quorum = (formats_present > 0).then(|| select_format_erasure_in_quorum(&formats, 0));
|
||||
// A resized pool may never reach quorum under its new endpoint count.
|
||||
// Diagnose a valid, unambiguous stored layout before migration or waiting.
|
||||
// A healthy quorum still takes precedence over foreign minority formats;
|
||||
// conflicting or malformed observations retain their existing error path.
|
||||
if format_quorum.as_ref().is_some_and(Result::is_err)
|
||||
&& let Some(reference) = formats.iter().flatten().next()
|
||||
&& formats.iter().flatten().all(|format| {
|
||||
format.shared_identity() == reference.shared_identity()
|
||||
&& reference.erasure.sets.iter().flatten().any(|id| *id == format.erasure.this)
|
||||
})
|
||||
&& let Err(err @ (Error::UnsupportedSnsdExpansion { .. } | Error::PoolTopologyMismatch { .. })) =
|
||||
check_format_erasure_value_for_topology(reference, formats.len(), set_drive_count)
|
||||
{
|
||||
return Err(err);
|
||||
}
|
||||
if format_quorum.as_ref().is_none_or(Result::is_err)
|
||||
&& errs.iter().any(|error| {
|
||||
matches!(
|
||||
@@ -661,15 +676,18 @@ fn check_format_erasure_value_for_topology(format: &FormatV3, format_count: usiz
|
||||
.len()
|
||||
.checked_mul(set_drive_count_in_format)
|
||||
.ok_or_else(|| Error::other("erasure set drive count overflow"))?;
|
||||
if format_count != format_drive_count {
|
||||
return Err(Error::other(format!(
|
||||
"formats length for erasure.sets does not match: got {format_count}, expected {format_drive_count}"
|
||||
)));
|
||||
if format_drive_count == 1 && format_count > 1 {
|
||||
return Err(Error::UnsupportedSnsdExpansion {
|
||||
configured_drives: format_count,
|
||||
});
|
||||
}
|
||||
if set_drive_count_in_format != set_drive_count {
|
||||
return Err(Error::other(format!(
|
||||
"erasure set length for set_drive_count does not match: got {set_drive_count_in_format}, expected {set_drive_count}"
|
||||
)));
|
||||
if format_count != format_drive_count || set_drive_count_in_format != set_drive_count {
|
||||
return Err(Error::PoolTopologyMismatch {
|
||||
stored_drives: format_drive_count,
|
||||
stored_set_drive_count: set_drive_count_in_format,
|
||||
configured_drives: format_count,
|
||||
configured_set_drive_count: set_drive_count,
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -877,6 +895,10 @@ mod tests {
|
||||
use serial_test::serial;
|
||||
|
||||
async fn local_disks(count: usize) -> (tempfile::TempDir, Vec<Option<DiskStore>>) {
|
||||
local_disks_with_set_width(count, count).await
|
||||
}
|
||||
|
||||
async fn local_disks_with_set_width(count: usize, set_width: usize) -> (tempfile::TempDir, Vec<Option<DiskStore>>) {
|
||||
let temp_dir = tempfile::tempdir().expect("temporary disk root should be created");
|
||||
let mut endpoints = Vec::with_capacity(count);
|
||||
for disk_index in 0..count {
|
||||
@@ -887,8 +909,8 @@ mod tests {
|
||||
let mut endpoint =
|
||||
Endpoint::try_from(path.to_str().expect("temporary disk path should be UTF-8")).expect("endpoint should parse");
|
||||
endpoint.set_pool_index(0);
|
||||
endpoint.set_set_index(0);
|
||||
endpoint.set_disk_index(disk_index);
|
||||
endpoint.set_set_index(disk_index / set_width);
|
||||
endpoint.set_disk_index(disk_index % set_width);
|
||||
endpoints.push(endpoint);
|
||||
}
|
||||
|
||||
@@ -912,6 +934,21 @@ mod tests {
|
||||
(temp_dir, disks)
|
||||
}
|
||||
|
||||
async fn format_bytes(disks: &[Option<DiskStore>]) -> Vec<Option<Vec<u8>>> {
|
||||
let mut snapshots = Vec::with_capacity(disks.len());
|
||||
for disk in disks {
|
||||
let disk = disk.as_ref().expect("snapshot disk should exist");
|
||||
// Inspect bytes even when the disk wrapper rejects a format whose
|
||||
// stored slot differs from the attempted new endpoint geometry.
|
||||
match tokio::fs::read(disk.path().join(RUSTFS_META_BUCKET).join(FORMAT_CONFIG_FILE)).await {
|
||||
Ok(data) => snapshots.push(Some(data)),
|
||||
Err(err) if err.kind() == std::io::ErrorKind::NotFound => snapshots.push(None),
|
||||
Err(err) => panic!("format snapshot failed: {err}"),
|
||||
}
|
||||
}
|
||||
snapshots
|
||||
}
|
||||
|
||||
async fn write_legacy_format(disk: &Option<DiskStore>, format: &FormatV3) {
|
||||
write_legacy_bytes(disk, bytes::Bytes::from(format.to_json().expect("legacy format should serialize"))).await;
|
||||
}
|
||||
@@ -1116,6 +1153,212 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn single_drive_format_rejects_in_place_expansion_without_writes() {
|
||||
for configured_drives in [2, 4] {
|
||||
for first_disk in [false, true] {
|
||||
let (_temp_dir, mut disks) = local_disks(configured_drives).await;
|
||||
let mut original = FormatV3::new(1, 1);
|
||||
original.erasure.this = original.erasure.sets[0][0];
|
||||
save_format_file(&disks[0], &Some(original))
|
||||
.await
|
||||
.expect("SNSD format should be written");
|
||||
let before = format_bytes(&disks).await;
|
||||
|
||||
let err = connect_load_init_formats(first_disk, &mut disks, 1, configured_drives, None)
|
||||
.await
|
||||
.expect_err("an existing SNSD deployment cannot expand in place");
|
||||
let message = err.to_string();
|
||||
assert!(message.contains("SNSD"), "expected a single-drive expansion error: {message}");
|
||||
assert!(message.contains("migrate data through S3"), "expected actionable guidance: {message}");
|
||||
assert_eq!(format_bytes(&disks).await, before, "neither old nor new formats may be written");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn existing_pool_rejects_drive_count_or_set_width_changes_without_writes() {
|
||||
for (stored_sets, stored_width, configured_sets, configured_width) in
|
||||
[(1, 4, 1, 6), (1, 4, 1, 8), (1, 4, 1, 2), (1, 4, 2, 2), (2, 2, 1, 4)]
|
||||
{
|
||||
for first_disk in [false, true] {
|
||||
let (_temp_dir, mut disks) =
|
||||
local_disks_with_set_width(configured_sets * configured_width, configured_width).await;
|
||||
let original = FormatV3::new(stored_sets, stored_width);
|
||||
for (disk, disk_id) in disks.iter().zip(original.erasure.sets.iter().flatten()) {
|
||||
let mut format = original.clone();
|
||||
format.erasure.this = *disk_id;
|
||||
save_format_file(disk, &Some(format))
|
||||
.await
|
||||
.expect("existing format should be written");
|
||||
}
|
||||
let before = format_bytes(&disks).await;
|
||||
|
||||
let err = connect_load_init_formats(first_disk, &mut disks, configured_sets, configured_width, None)
|
||||
.await
|
||||
.expect_err("an existing pool's geometry is immutable");
|
||||
let message = err.to_string();
|
||||
assert!(message.contains("pool topology mismatch"), "expected a topology error: {message}");
|
||||
assert!(
|
||||
message.contains(&format!("stored 4 drives with {stored_width} drives per erasure set")),
|
||||
"expected stored geometry: {message}"
|
||||
);
|
||||
assert!(message.contains("append a new pool"), "expected expansion guidance: {message}");
|
||||
assert_eq!(format_bytes(&disks).await, before, "rejection must not rewrite any format");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn subquorum_existing_layout_with_missing_drives_is_not_expansion() {
|
||||
let (_temp_dir, mut disks) = local_disks(1).await;
|
||||
let mut original = FormatV3::new(1, 4);
|
||||
original.erasure.this = original.erasure.sets[0][0];
|
||||
save_format_file(&disks[0], &Some(original))
|
||||
.await
|
||||
.expect("existing format should be written");
|
||||
disks.extend([None, None, None]);
|
||||
|
||||
for first_disk in [false, true] {
|
||||
assert!(matches!(
|
||||
connect_load_init_formats(first_disk, &mut disks, 1, 4, None).await,
|
||||
Err(Error::ErasureReadQuorum)
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn conflicting_layouts_without_quorum_are_not_expansion_proof() {
|
||||
let (_temp_dir, mut disks) = local_disks(2).await;
|
||||
for (index, (disk, width)) in disks.iter().zip([4, 2]).enumerate() {
|
||||
let mut format = FormatV3::new(1, width);
|
||||
format.erasure.this = format.erasure.sets[0][index];
|
||||
save_format_file(disk, &Some(format))
|
||||
.await
|
||||
.expect("existing format should be written");
|
||||
}
|
||||
disks.extend([None, None]);
|
||||
|
||||
let result = connect_load_init_formats(true, &mut disks, 1, 4, None).await;
|
||||
assert!(matches!(result, Err(Error::ErasureReadQuorum)), "conflicting layout result: {result:?}");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn existing_format_quorum_ignores_single_drive_outlier() {
|
||||
let (_temp_dir, mut disks) = local_disks(3).await;
|
||||
let majority = FormatV3::new(1, 3);
|
||||
for (index, disk) in disks.iter().enumerate() {
|
||||
// Slot zero lets the SNSD outlier pass the disk wrapper's own
|
||||
// slot check, so quorum selection must exclude the parsed format.
|
||||
let mut format = if index == 0 { FormatV3::new(1, 1) } else { majority.clone() };
|
||||
format.erasure.this = format.erasure.sets[0][index];
|
||||
save_format_file(disk, &Some(format))
|
||||
.await
|
||||
.expect("existing format should be written");
|
||||
}
|
||||
|
||||
let loaded = connect_load_init_formats(true, &mut disks, 1, 3, None)
|
||||
.await
|
||||
.expect("a foreign SNSD outlier must not block a healthy majority");
|
||||
assert_eq!(loaded.shared_identity(), majority.shared_identity());
|
||||
assert!(disks[0].is_none(), "the foreign single-drive format must be quarantined");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn multi_drive_pool_expansion_preserves_existing_format() {
|
||||
let (_original_dir, mut disks) = local_disks(4).await;
|
||||
let (_new_dir, mut new_disks) = local_disks(4).await;
|
||||
let original = connect_load_init_formats(true, &mut disks, 1, 4, None)
|
||||
.await
|
||||
.expect("original multi-drive pool should initialize");
|
||||
let before = format_bytes(&disks).await;
|
||||
|
||||
let added = connect_load_init_formats(true, &mut new_disks, 1, 4, Some(original.id))
|
||||
.await
|
||||
.expect("a new multi-drive pool should initialize with the existing deployment ID");
|
||||
assert_eq!(added.id, original.id);
|
||||
assert_ne!(added.erasure.sets, original.erasure.sets);
|
||||
assert_eq!(format_bytes(&disks).await, before);
|
||||
assert_eq!(
|
||||
connect_load_init_formats(true, &mut disks, 1, 4, Some(original.id))
|
||||
.await
|
||||
.expect("the original pool should restart with unchanged geometry"),
|
||||
original
|
||||
);
|
||||
assert_eq!(
|
||||
connect_load_init_formats(true, &mut new_disks, 1, 4, Some(original.id))
|
||||
.await
|
||||
.expect("the new pool should restart with its own format"),
|
||||
added
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn store_startup_rejects_pool_resize_before_retry_loop() {
|
||||
use crate::layout::endpoints::{EndpointServerPools, PoolEndpoints};
|
||||
use tokio_util::sync::CancellationToken;
|
||||
|
||||
for (stored_width, configured_width) in [(1, 4), (4, 8)] {
|
||||
let (_temp_dir, disks) = local_disks(configured_width).await;
|
||||
let original = FormatV3::new(1, stored_width);
|
||||
for (disk, disk_id) in disks.iter().zip(&original.erasure.sets[0]) {
|
||||
let mut format = original.clone();
|
||||
format.erasure.this = *disk_id;
|
||||
save_format_file(disk, &Some(format))
|
||||
.await
|
||||
.expect("old format should be written");
|
||||
}
|
||||
let before = format_bytes(&disks).await;
|
||||
let endpoints = disks.iter().flatten().map(|disk| disk.endpoint()).collect::<Vec<_>>();
|
||||
let pools = EndpointServerPools::from(vec![PoolEndpoints {
|
||||
legacy: true,
|
||||
set_count: 1,
|
||||
drives_per_set: configured_width,
|
||||
endpoints: Endpoints::from(endpoints),
|
||||
cmd_line: "test-pool".to_string(),
|
||||
platform: String::new(),
|
||||
}]);
|
||||
let shutdown = CancellationToken::new();
|
||||
let result = temp_env::async_with_vars(
|
||||
[
|
||||
(storageclass::STANDARD_ENV, None::<&str>),
|
||||
(storageclass::RRS_ENV, None::<&str>),
|
||||
(storageclass::OPTIMIZE_ENV, None::<&str>),
|
||||
(storageclass::INLINE_BLOCK_ENV, None::<&str>),
|
||||
],
|
||||
tokio::time::timeout(
|
||||
std::time::Duration::from_secs(5),
|
||||
crate::store::ECStore::new_with_instance_ctx(
|
||||
"127.0.0.1:0".parse().expect("test address"),
|
||||
pools,
|
||||
shutdown.clone(),
|
||||
Arc::new(InstanceContext::new()),
|
||||
),
|
||||
),
|
||||
)
|
||||
.await;
|
||||
shutdown.cancel();
|
||||
let err = result
|
||||
.expect("invalid topology must abort without the format retry backoff")
|
||||
.expect_err("resize must fail");
|
||||
match stored_width {
|
||||
1 => assert!(matches!(err, Error::UnsupportedSnsdExpansion { configured_drives: 4 }), "{err}"),
|
||||
_ => assert!(
|
||||
matches!(
|
||||
err,
|
||||
Error::PoolTopologyMismatch {
|
||||
stored_drives: 4,
|
||||
configured_drives: 8,
|
||||
..
|
||||
}
|
||||
),
|
||||
"{err}"
|
||||
),
|
||||
}
|
||||
assert_eq!(format_bytes(&disks).await, before, "failed store startup must not write formats");
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn existing_format_load_rejects_conflicting_formats_without_a_majority() {
|
||||
let (_temp_dir, mut disks) = two_local_disks_with_missing_third().await;
|
||||
|
||||
@@ -316,10 +316,15 @@ async fn can_skip_hidden_prefix_check(options: &ListPathOptions) -> bool {
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
/// Whether an empty listing of `prefix` is the proof that lets the caller
|
||||
/// reclaim delete residue under it. Any first page that scanned the whole
|
||||
/// prefix without finding an object or a sub-prefix qualifies, with or without
|
||||
/// a delimiter: that is the shape clients issue when they stat, browse, or
|
||||
/// recursively remove a phantom folder. The purge itself re-verifies every
|
||||
/// directory on every disk before deleting anything.
|
||||
fn should_purge_empty_directory_listing(
|
||||
prefix: &str,
|
||||
marker: Option<&str>,
|
||||
delimiter: Option<&str>,
|
||||
max_keys: i32,
|
||||
incl_deleted: bool,
|
||||
result: &ListObjectsInfo,
|
||||
@@ -327,8 +332,7 @@ fn should_purge_empty_directory_listing(
|
||||
!prefix.is_empty()
|
||||
&& prefix.ends_with(SLASH_SEPARATOR)
|
||||
&& marker.is_none()
|
||||
&& delimiter.is_none_or(str::is_empty)
|
||||
&& max_keys == 1
|
||||
&& max_keys > 0
|
||||
&& !incl_deleted
|
||||
&& !result.is_truncated
|
||||
&& result.objects.is_empty()
|
||||
@@ -3847,16 +3851,10 @@ impl ECStore {
|
||||
.list_objects_from_opt_in_key_only_provider(&opts, mode, max_keys, incl_deleted)
|
||||
.await?
|
||||
{
|
||||
if should_purge_empty_directory_listing(
|
||||
prefix,
|
||||
opts.marker.as_deref(),
|
||||
delimiter.as_deref(),
|
||||
max_keys,
|
||||
incl_deleted,
|
||||
&result,
|
||||
) && has_authoritative_never_versioned_state_in(&self.ctx, bucket)
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
if should_purge_empty_directory_listing(prefix, opts.marker.as_deref(), max_keys, incl_deleted, &result)
|
||||
&& has_authoritative_never_versioned_state_in(&self.ctx, bucket)
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
{
|
||||
self.purge_orphan_dir_object(bucket, prefix).await;
|
||||
}
|
||||
@@ -3943,16 +3941,10 @@ impl ECStore {
|
||||
objects,
|
||||
prefixes,
|
||||
};
|
||||
if should_purge_empty_directory_listing(
|
||||
prefix,
|
||||
opts.marker.as_deref(),
|
||||
delimiter.as_deref(),
|
||||
max_keys,
|
||||
incl_deleted,
|
||||
&result,
|
||||
) && has_authoritative_never_versioned_state_in(&self.ctx, bucket)
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
if should_purge_empty_directory_listing(prefix, opts.marker.as_deref(), max_keys, incl_deleted, &result)
|
||||
&& has_authoritative_never_versioned_state_in(&self.ctx, bucket)
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
{
|
||||
self.purge_orphan_dir_object(bucket, prefix).await;
|
||||
}
|
||||
@@ -8843,26 +8835,28 @@ mod test {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_directory_listing_purge_requires_complete_exact_recursive_request() {
|
||||
fn empty_directory_listing_purge_requires_complete_first_page_of_prefix() {
|
||||
let empty = ListObjectsInfo::default();
|
||||
assert!(should_purge_empty_directory_listing("ghost/", None, None, 1, false, &empty));
|
||||
assert!(should_purge_empty_directory_listing("ghost/", None, Some(""), 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost", None, None, 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", Some("marker"), None, 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, Some("/"), 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 0, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 2, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 1, true, &empty));
|
||||
assert!(should_purge_empty_directory_listing("ghost/", None, 1, false, &empty));
|
||||
assert!(should_purge_empty_directory_listing("ghost/", None, 1000, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost", None, 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", Some("marker"), 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, 0, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, 1, true, &empty));
|
||||
|
||||
let mut live = ListObjectsInfo::default();
|
||||
live.objects.push(ObjectInfo::default());
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 1, false, &live));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, 1, false, &live));
|
||||
|
||||
let mut prefixed = ListObjectsInfo::default();
|
||||
prefixed.prefixes.push("ghost/child/".to_owned());
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, 1, false, &prefixed));
|
||||
|
||||
let truncated = ListObjectsInfo {
|
||||
is_truncated: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 1, false, &truncated));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, 1, false, &truncated));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -8916,6 +8910,59 @@ mod test {
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn empty_delimiter_listing_hides_and_purges_committed_delete_residue() {
|
||||
use crate::bucket::metadata_sys::{init_bucket_metadata_sys, test_support::isolated_store_over_temp_disks};
|
||||
use crate::storage_api_contracts::bucket::{BucketOperations as _, MakeBucketOptions};
|
||||
|
||||
let (dirs, store) = isolated_store_over_temp_disks().await;
|
||||
let bucket = "listing-purge-delimiter-bucket";
|
||||
init_bucket_metadata_sys(store.clone(), Vec::new()).await;
|
||||
store
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created with authoritative metadata");
|
||||
let data_dir = uuid::Uuid::new_v4();
|
||||
let transaction = uuid::Uuid::new_v4();
|
||||
for dir in &dirs {
|
||||
let residue = dir
|
||||
.path()
|
||||
.join(bucket)
|
||||
.join("metrics")
|
||||
.join("2026")
|
||||
.join("object.parquet")
|
||||
.join(data_dir.to_string());
|
||||
tokio::fs::create_dir_all(&residue)
|
||||
.await
|
||||
.expect("committed delete residue should be created");
|
||||
tokio::fs::write(residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
tokio::fs::write(
|
||||
residue.join(format!("{}{}", crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX, transaction)),
|
||||
[],
|
||||
)
|
||||
.await
|
||||
.expect("committed delete marker should be written");
|
||||
}
|
||||
|
||||
// The object directory holds only a deleted version's data dir, so a
|
||||
// console-style browse of its parent must not show it as a folder.
|
||||
let result = store
|
||||
.clone()
|
||||
.list_objects_generic(bucket, "metrics/2026/", None, Some("/".to_owned()), 1000, false)
|
||||
.await
|
||||
.expect("delimiter listing should succeed");
|
||||
assert!(result.objects.is_empty());
|
||||
assert!(result.prefixes.is_empty(), "delete residue must not surface as a prefix");
|
||||
for dir in &dirs {
|
||||
assert!(
|
||||
!dir.path().join(bucket).join("metrics").join("2026").exists(),
|
||||
"the empty delimiter listing should reclaim the committed delete residue under it"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_objects_index_provider_state_uses_lifecycle_active_generation() {
|
||||
let provider = ListObjectsIndexProviderState::walker_key_only();
|
||||
|
||||
@@ -442,7 +442,7 @@ pub(crate) mod utils;
|
||||
|
||||
use peer::init_local_peer;
|
||||
pub use peer::{
|
||||
BootstrapLocalTarget, all_local_disk, all_local_disk_path, find_local_disk_by_ref, get_disk_infos, init_local_disks,
|
||||
all_local_disk, all_local_disk_path, find_local_disk_by_ref, get_disk_infos, init_local_disks,
|
||||
init_local_disks_with_instance_ctx, init_lock_clients, prewarm_local_disk_id_map,
|
||||
prewarm_local_disk_id_map_with_instance_ctx,
|
||||
};
|
||||
@@ -527,6 +527,31 @@ impl Default for ScannerDataMovementPauseStatus {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize)]
|
||||
pub struct PoolMetaWriteGateStatus {
|
||||
pub writes_ready: bool,
|
||||
pub write_blocked: bool,
|
||||
pub transaction_aborted: bool,
|
||||
pub pool_meta_absent: bool,
|
||||
pub identity_initialized: Option<bool>,
|
||||
pub identity_needs_repair: bool,
|
||||
pub cluster_epoch: Option<u64>,
|
||||
}
|
||||
|
||||
impl Default for PoolMetaWriteGateStatus {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
writes_ready: true,
|
||||
write_blocked: false,
|
||||
transaction_aborted: false,
|
||||
pool_meta_absent: false,
|
||||
identity_initialized: None,
|
||||
identity_needs_repair: false,
|
||||
cluster_epoch: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn offset_unix_seconds(value: OffsetDateTime) -> u64 {
|
||||
u64::try_from(value.unix_timestamp()).unwrap_or(0)
|
||||
}
|
||||
@@ -1787,7 +1812,7 @@ mod tests {
|
||||
|
||||
// Build a minimal ECStore carrying an explicit instance context. Empty
|
||||
// pools/disks are sufficient: the Phase 5 accessors read only `self.ctx`.
|
||||
pub(super) fn build_store_with_ctx(ctx: Arc<InstanceContext>) -> Arc<ECStore> {
|
||||
fn build_store_with_ctx(ctx: Arc<InstanceContext>) -> Arc<ECStore> {
|
||||
let endpoint_pools = EndpointServerPools::default();
|
||||
Arc::new(ECStore {
|
||||
id: uuid::Uuid::new_v4(),
|
||||
|
||||
@@ -526,12 +526,8 @@ impl ECStore {
|
||||
if self.single_pool() {
|
||||
self.apply_decommission_target_mutation_fence(0, object, &mut opts, mutation_fence)
|
||||
.await;
|
||||
return self
|
||||
.run_decommission_capacity_admitted_mutation(0, None, None, || async {
|
||||
self.pools[0].new_multipart_upload(bucket, object, &opts).await
|
||||
})
|
||||
.await
|
||||
.map(|res| (res, 0, opts.expected_bucket_incarnation_id));
|
||||
let result = self.pools[0].new_multipart_upload(bucket, object, &opts).await?;
|
||||
return Ok((result, 0, opts.expected_bucket_incarnation_id));
|
||||
}
|
||||
|
||||
if opts.data_movement && opts.version_id.is_some() {
|
||||
@@ -658,7 +654,9 @@ impl ECStore {
|
||||
) -> Result<PartInfo> {
|
||||
check_put_object_part_args(bucket, object, upload_id)?;
|
||||
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
if !self.single_pool() {
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
}
|
||||
let opts = &opts;
|
||||
|
||||
if self.single_pool() {
|
||||
@@ -982,7 +980,9 @@ impl ECStore {
|
||||
) -> Result<ObjectInfo> {
|
||||
check_complete_multipart_args(bucket, object, upload_id)?;
|
||||
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
if !self.single_pool() {
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
}
|
||||
let opts = &opts;
|
||||
|
||||
if self.single_pool() {
|
||||
|
||||
@@ -3123,6 +3123,9 @@ impl ECStore {
|
||||
Fut: std::future::Future<Output = Result<T>>,
|
||||
{
|
||||
let (lock_object, target_object) = objects;
|
||||
if self.single_pool() {
|
||||
return operation(opts).await;
|
||||
}
|
||||
let (capacity_guard, has_active_decommission) = if capacity_releasing {
|
||||
self.acquire_decommission_capacity_release_fence_with_active_source().await?
|
||||
} else {
|
||||
@@ -3220,6 +3223,9 @@ impl ECStore {
|
||||
F: FnOnce(HealOpts) -> Fut,
|
||||
Fut: std::future::Future<Output = Result<T>>,
|
||||
{
|
||||
if self.single_pool() {
|
||||
return operation(opts).await;
|
||||
}
|
||||
let (capacity_guard, has_active_decommission) = self
|
||||
.acquire_external_decommission_capacity_fence_with_active_source(&[target_pool_idx], "heal")
|
||||
.await?;
|
||||
@@ -4162,7 +4168,9 @@ impl ECStore {
|
||||
.select_put_object_pool_idx(bucket, object.as_str(), data.size(), &opts)
|
||||
.await?;
|
||||
let mut opts = opts;
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
if !self.single_pool() {
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
}
|
||||
self.pools[idx]
|
||||
.put_object_with_old_current_size(bucket, object.as_str(), data, &opts)
|
||||
.await
|
||||
@@ -4170,6 +4178,24 @@ impl ECStore {
|
||||
|
||||
#[instrument(level = "trace", skip(self))]
|
||||
pub(super) async fn handle_get_object_info(&self, bucket: &str, object: &str, opts: &ObjectOptions) -> Result<ObjectInfo> {
|
||||
self.get_object_info_snapshot(bucket, object, opts, false).await
|
||||
}
|
||||
|
||||
/// Return metadata for DELETE preflight, including an explicitly addressed
|
||||
/// delete marker. Read APIs must keep using `get_object_info`; authorization
|
||||
/// and Object Lock enforcement still belong to the caller and locked delete.
|
||||
#[instrument(level = "trace", skip_all)]
|
||||
pub async fn get_object_info_for_delete(&self, bucket: &str, object: &str, opts: &ObjectOptions) -> Result<ObjectInfo> {
|
||||
self.get_object_info_snapshot(bucket, object, opts, true).await
|
||||
}
|
||||
|
||||
async fn get_object_info_snapshot(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
opts: &ObjectOptions,
|
||||
allow_delete_marker: bool,
|
||||
) -> Result<ObjectInfo> {
|
||||
check_object_args(bucket, object)?;
|
||||
|
||||
let object = encode_dir_object(object);
|
||||
@@ -4180,6 +4206,8 @@ impl ECStore {
|
||||
|
||||
let info = if self.single_pool() {
|
||||
self.pools[0].get_object_info(bucket, object.as_str(), &opts).await?
|
||||
} else if allow_delete_marker {
|
||||
self.get_latest_object_info_with_idx(bucket, object.as_str(), &opts).await?.0
|
||||
} else {
|
||||
self.get_latest_accessible_object_info_with_idx(bucket, object.as_str(), &opts)
|
||||
.await?
|
||||
@@ -4340,8 +4368,10 @@ impl ECStore {
|
||||
object_lock_config_snapshot: dst_opts.object_lock_config_snapshot.clone(),
|
||||
..Default::default()
|
||||
};
|
||||
put_opts.decommission_capacity_admission =
|
||||
crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
if !self.single_pool() {
|
||||
put_opts.decommission_capacity_admission =
|
||||
crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
}
|
||||
return if let Some(reader) = src_info.put_object_reader.as_mut() {
|
||||
self.pools[pool_idx]
|
||||
.put_object(dst_bucket, &dst_object, reader, &put_opts)
|
||||
@@ -4376,8 +4406,10 @@ impl ECStore {
|
||||
object_lock_config_snapshot: dst_opts.object_lock_config_snapshot.clone(),
|
||||
..Default::default()
|
||||
};
|
||||
put_opts.decommission_capacity_admission =
|
||||
crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
if !self.single_pool() {
|
||||
put_opts.decommission_capacity_admission =
|
||||
crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
}
|
||||
return self.pools[pool_idx]
|
||||
.put_object(dst_bucket, &dst_object, reader, &put_opts)
|
||||
.await;
|
||||
@@ -4422,7 +4454,10 @@ impl ECStore {
|
||||
object_lock_config_snapshot: dst_opts.object_lock_config_snapshot.clone(),
|
||||
..Default::default()
|
||||
};
|
||||
put_opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
if !self.single_pool() {
|
||||
put_opts.decommission_capacity_admission =
|
||||
crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
}
|
||||
|
||||
if let Some(put_object_reader) = src_info.put_object_reader.as_mut() {
|
||||
return self.pools[pool_idx]
|
||||
@@ -5057,7 +5092,7 @@ impl ECStore {
|
||||
}
|
||||
}
|
||||
|
||||
let _capacity_fence = if latest_marker_objects.iter().any(|creates_marker| *creates_marker) {
|
||||
let _capacity_fence = if !self.single_pool() && latest_marker_objects.iter().any(|creates_marker| *creates_marker) {
|
||||
let target_pool_indices = (0..self.pools.len()).collect::<Vec<_>>();
|
||||
match self
|
||||
.acquire_external_decommission_capacity_fence(&target_pool_indices, "batch_delete")
|
||||
@@ -5375,7 +5410,6 @@ impl ECStore {
|
||||
// self-deadlocked on the inner commits.
|
||||
let object_name = object.as_str();
|
||||
if self.single_pool() {
|
||||
opts.decommission_capacity_admission = Some(Arc::clone(&self));
|
||||
return self.pools[0]
|
||||
.clone()
|
||||
.restore_transitioned_object(bucket, object_name, &opts)
|
||||
|
||||
@@ -13,10 +13,7 @@
|
||||
// limitations under the License.
|
||||
|
||||
use super::*;
|
||||
use crate::bucket::utils::has_bad_path_component;
|
||||
use crate::disk::error::{DiskError, Result as DiskResult};
|
||||
use crate::disk::{DeleteOptions, Disk, RenameDataGuards, RenameDataResp};
|
||||
use crate::runtime::instance::{InstanceContext, NamespaceCommitGuard};
|
||||
use crate::runtime::instance::InstanceContext;
|
||||
use crate::runtime::sources as runtime_sources;
|
||||
use tracing::{debug, error};
|
||||
|
||||
@@ -25,203 +22,6 @@ const LOG_SUBSYSTEM_DISK_STARTUP: &str = "disk_startup";
|
||||
const EVENT_LOCAL_DISK_ID_PREWARM_SKIPPED: &str = "local_disk_id_prewarm_skipped";
|
||||
const EVENT_LOCK_CLIENT_INITIALIZATION_FAILED: &str = "lock_client_initialization_failed";
|
||||
|
||||
/// An instance-bound capability for internal writes before ECStore/IAM startup.
|
||||
/// Its private context and volume checks cannot be replaced by a caller guard.
|
||||
#[derive(Clone)]
|
||||
pub struct BootstrapLocalTarget {
|
||||
ctx: Arc<InstanceContext>,
|
||||
}
|
||||
|
||||
impl BootstrapLocalTarget {
|
||||
pub fn new(ctx: Arc<InstanceContext>) -> Self {
|
||||
Self { ctx }
|
||||
}
|
||||
|
||||
pub fn is_for_store(&self, store: &ECStore) -> bool {
|
||||
Arc::ptr_eq(&self.ctx, &store.ctx)
|
||||
}
|
||||
|
||||
pub async fn rename_local_data(
|
||||
&self,
|
||||
disk_ref: &str,
|
||||
source: (&str, &str),
|
||||
fi: &FileInfo,
|
||||
destination: (&str, &str),
|
||||
scanner_token: Option<Uuid>,
|
||||
) -> DiskResult<RenameDataResp> {
|
||||
if scanner_token.is_some() {
|
||||
return Err(DiskError::other("bootstrap rename cannot use a scanner publication lease"));
|
||||
}
|
||||
validate_bootstrap_volume(source.0)?;
|
||||
validate_bootstrap_volume(destination.0)?;
|
||||
rename_local_data_with_ctx(&self.ctx, disk_ref, source, fi, destination, RenameDataGuards::default()).await
|
||||
}
|
||||
|
||||
pub async fn undo_local_write(
|
||||
&self,
|
||||
disk_ref: &str,
|
||||
volume: &str,
|
||||
path: &str,
|
||||
fi: FileInfo,
|
||||
opts: DeleteOptions,
|
||||
) -> DiskResult<()> {
|
||||
validate_bootstrap_volume(volume)?;
|
||||
undo_local_write_with_ctx(&self.ctx, disk_ref, volume, path, fi, opts).await
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_bootstrap_volume(volume: &str) -> DiskResult<()> {
|
||||
// Prefix membership alone permits aliases such as .rustfs.sys/../bucket.
|
||||
// Validate both raw rename volumes before any disk lookup or admission.
|
||||
if has_bad_path_component(volume) || !is_meta_bucketname(volume) {
|
||||
return Err(DiskError::FileAccessDenied);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
impl ECStore {
|
||||
/// Execute on this instance's active local disk through the physical owner.
|
||||
pub async fn rename_local_data(
|
||||
&self,
|
||||
disk_ref: &str,
|
||||
source: (&str, &str),
|
||||
fi: &FileInfo,
|
||||
destination: (&str, &str),
|
||||
scanner_token: Option<Uuid>,
|
||||
) -> DiskResult<RenameDataResp> {
|
||||
let external_guard: Option<Arc<dyn Send + Sync>> = if let Some(token) = scanner_token {
|
||||
Some(Arc::new(
|
||||
self.acquire_scanner_publication_lease_guard(token)
|
||||
.await
|
||||
.map_err(|err| DiskError::other(err.to_string()))?,
|
||||
))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
rename_local_data_with_ctx(
|
||||
&self.ctx,
|
||||
disk_ref,
|
||||
source,
|
||||
fi,
|
||||
destination,
|
||||
RenameDataGuards {
|
||||
scanner_publication_lease_token: scanner_token,
|
||||
external_guard,
|
||||
namespace_owner: None,
|
||||
},
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn undo_local_write(
|
||||
&self,
|
||||
disk_ref: &str,
|
||||
volume: &str,
|
||||
path: &str,
|
||||
fi: FileInfo,
|
||||
opts: DeleteOptions,
|
||||
) -> DiskResult<()> {
|
||||
undo_local_write_with_ctx(&self.ctx, disk_ref, volume, path, fi, opts).await
|
||||
}
|
||||
}
|
||||
|
||||
// The optional ID is a cold lookup to cache only after final admission.
|
||||
async fn local_disk_candidate(ctx: &Arc<InstanceContext>, disk_ref: &str) -> DiskResult<(DiskStore, Option<Uuid>)> {
|
||||
let map = ctx.local_disk_map();
|
||||
if let Some(disk) = map.read().await.get(disk_ref).and_then(Option::as_ref).cloned() {
|
||||
return Ok((disk, None));
|
||||
}
|
||||
let disk_id = Uuid::parse_str(disk_ref).map_err(|_| DiskError::DiskNotFound)?;
|
||||
let cached_path = ctx.local_disk_id_map().read().await.get(&disk_id).cloned();
|
||||
if let Some(path) = cached_path {
|
||||
let cached_disk = map.read().await.get(&path).and_then(Option::as_ref).cloned();
|
||||
if let Some(disk) = cached_disk
|
||||
&& matches!(disk.as_ref(), Disk::Local(_))
|
||||
&& disk.get_disk_id().await? == Some(disk_id)
|
||||
{
|
||||
return Ok((disk, None));
|
||||
}
|
||||
}
|
||||
let disks: Vec<_> = map.read().await.values().filter_map(Clone::clone).collect();
|
||||
// Disk identity may perform format I/O. No registry guard spans this await.
|
||||
for disk in disks {
|
||||
if matches!(disk.as_ref(), Disk::Local(_)) && disk.get_disk_id().await.ok().flatten() == Some(disk_id) {
|
||||
return Ok((disk, Some(disk_id)));
|
||||
}
|
||||
}
|
||||
Err(DiskError::DiskNotFound)
|
||||
}
|
||||
|
||||
async fn admit_local_disk(
|
||||
ctx: &Arc<InstanceContext>,
|
||||
disk: &DiskStore,
|
||||
disk_id: Option<Uuid>,
|
||||
volume: &str,
|
||||
) -> DiskResult<Option<Arc<NamespaceCommitGuard>>> {
|
||||
if !matches!(disk.as_ref(), Disk::Local(_)) {
|
||||
return Err(DiskError::DiskNotFound);
|
||||
}
|
||||
let map = ctx.local_disk_map();
|
||||
let active = map.read().await;
|
||||
if !active
|
||||
.get(&disk.endpoint().to_string())
|
||||
.and_then(Option::as_ref)
|
||||
.is_some_and(|current| Arc::ptr_eq(current, disk))
|
||||
{
|
||||
return Err(DiskError::DiskNotFound);
|
||||
}
|
||||
// Preserve registry -> ID-cache lock order; no filesystem I/O under either.
|
||||
if let Some(disk_id) = disk_id {
|
||||
ctx.local_disk_id_map()
|
||||
.write()
|
||||
.await
|
||||
.insert(disk_id, disk.endpoint().to_string());
|
||||
}
|
||||
// Admission linearizes under the registry read: replacement/quarantine
|
||||
// before this point rejects; later changes do not revoke physical I/O.
|
||||
Ok((!is_meta_bucketname(volume)).then(|| ctx.begin_namespace_commit()))
|
||||
}
|
||||
|
||||
async fn rename_local_data_with_ctx(
|
||||
ctx: &Arc<InstanceContext>,
|
||||
disk_ref: &str,
|
||||
source: (&str, &str),
|
||||
fi: &FileInfo,
|
||||
destination: (&str, &str),
|
||||
mut guards: RenameDataGuards,
|
||||
) -> DiskResult<RenameDataResp> {
|
||||
let (disk, disk_id) = local_disk_candidate(ctx, disk_ref).await?;
|
||||
let owner = admit_local_disk(ctx, &disk, disk_id, destination.0).await?;
|
||||
guards.namespace_owner = owner.as_ref().map(|owner| owner.clone() as Arc<dyn Send + Sync>);
|
||||
let result = disk
|
||||
.rename_data_borrowed_with_fence_observed(source.0, source.1, fi, destination.0, destination.1, guards)
|
||||
.await
|
||||
.result;
|
||||
drop(owner);
|
||||
result
|
||||
}
|
||||
|
||||
async fn undo_local_write_with_ctx(
|
||||
ctx: &Arc<InstanceContext>,
|
||||
disk_ref: &str,
|
||||
volume: &str,
|
||||
path: &str,
|
||||
fi: FileInfo,
|
||||
opts: DeleteOptions,
|
||||
) -> DiskResult<()> {
|
||||
if !opts.undo_write {
|
||||
return Err(DiskError::other("target undo requires undo_write"));
|
||||
}
|
||||
let (disk, disk_id) = local_disk_candidate(ctx, disk_ref).await?;
|
||||
let owner = admit_local_disk(ctx, &disk, disk_id, volume).await?;
|
||||
let physical_owner = owner.as_ref().map(|owner| owner.clone() as Arc<dyn Send + Sync>);
|
||||
let result = disk
|
||||
.undo_write_with_namespace_owner(volume, path, fi, opts, physical_owner)
|
||||
.await;
|
||||
drop(owner);
|
||||
result
|
||||
}
|
||||
|
||||
async fn remember_local_disk_id(disk: &DiskStore) -> Option<Uuid> {
|
||||
remember_local_disk_id_with_instance_ctx(&crate::runtime::global::current_ctx(), disk).await
|
||||
}
|
||||
@@ -465,522 +265,6 @@ mod tests {
|
||||
}])
|
||||
}
|
||||
|
||||
async fn target_disk(ctx: &Arc<InstanceContext>, root: &std::path::Path, id: Uuid) -> DiskStore {
|
||||
let mut format = crate::layout::format::FormatV3::new(1, 1);
|
||||
format.erasure.this = id;
|
||||
format.erasure.sets[0][0] = id;
|
||||
let meta = root.join(crate::disk::RUSTFS_META_BUCKET);
|
||||
tokio::fs::create_dir_all(&meta).await.expect("create format volume");
|
||||
tokio::fs::write(
|
||||
meta.join(crate::disk::FORMAT_CONFIG_FILE),
|
||||
serde_json::to_vec(&format).expect("encode format"),
|
||||
)
|
||||
.await
|
||||
.expect("write real disk identity");
|
||||
let mut endpoint = Endpoint::try_from(root.to_str().expect("UTF-8 root")).expect("endpoint");
|
||||
endpoint.set_pool_index(0);
|
||||
endpoint.set_set_index(0);
|
||||
endpoint.set_disk_index(0);
|
||||
let disk = new_disk(
|
||||
&endpoint,
|
||||
&DiskOption {
|
||||
cleanup: false,
|
||||
health_check: false,
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("open real local disk");
|
||||
assert_eq!(disk.get_disk_id().await.expect("read disk format identity"), Some(id));
|
||||
ctx.local_disk_map()
|
||||
.write()
|
||||
.await
|
||||
.insert(disk.endpoint().to_string(), Some(disk.clone()));
|
||||
disk
|
||||
}
|
||||
|
||||
fn target_file_info(object: &str, version: Uuid, body: &'static [u8]) -> FileInfo {
|
||||
let mut fi = FileInfo::new(object, 1, 0);
|
||||
fi.erasure.index = 1;
|
||||
fi.version_id = Some(version);
|
||||
fi.mod_time = Some(OffsetDateTime::now_utc());
|
||||
fi.size = i64::try_from(body.len()).expect("fixture length");
|
||||
fi.parts = vec![rustfs_filemeta::ObjectPartInfo {
|
||||
number: 1,
|
||||
size: body.len(),
|
||||
actual_size: fi.size,
|
||||
..Default::default()
|
||||
}];
|
||||
fi.data = Some(bytes::Bytes::from_static(body));
|
||||
fi.set_inline_data();
|
||||
fi
|
||||
}
|
||||
|
||||
async fn seed_target(disk: &DiskStore, volume: &str, object: &str, fi: FileInfo) -> Vec<u8> {
|
||||
let dir = disk.path().join(volume);
|
||||
tokio::fs::create_dir_all(&dir).await.expect("real fixture volume");
|
||||
disk.write_metadata(volume, volume, object, fi.clone())
|
||||
.await
|
||||
.expect("seed real metadata");
|
||||
let read = disk
|
||||
.read_version(
|
||||
volume,
|
||||
volume,
|
||||
object,
|
||||
&fi.version_id.expect("fixture version").to_string(),
|
||||
&crate::disk::ReadOptions {
|
||||
read_data: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("read fixture before mutation");
|
||||
assert_eq!(read.data, fi.data, "fixture must contain readable inline bytes");
|
||||
tokio::fs::read(dir.join(object).join(crate::disk::STORAGE_FORMAT_FILE))
|
||||
.await
|
||||
.expect("seeded metadata bytes")
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn target_uuid_lookup_binds_real_disk_and_owner_to_one_instance() {
|
||||
for warm in [false, true] {
|
||||
let ctx_a = Arc::new(InstanceContext::new());
|
||||
let ctx_b = Arc::new(InstanceContext::new());
|
||||
let a = tempfile::tempdir().expect("A root");
|
||||
let b = tempfile::tempdir().expect("B root");
|
||||
let id = Uuid::new_v4();
|
||||
let disk_a = target_disk(&ctx_a, a.path(), id).await;
|
||||
let disk_b = target_disk(&ctx_b, b.path(), id).await;
|
||||
if warm {
|
||||
assert!(record_local_disk_id_if_active(&ctx_a, &disk_a, id).await);
|
||||
assert!(record_local_disk_id_if_active(&ctx_b, &disk_b, id).await);
|
||||
}
|
||||
let version = Uuid::new_v4();
|
||||
let fi = target_file_info("destination", version, b"new-A");
|
||||
for disk in [&disk_a, &disk_b] {
|
||||
seed_target(disk, "target-bucket", "staged", fi.clone()).await;
|
||||
}
|
||||
let b_before = seed_target(
|
||||
&disk_b,
|
||||
"target-bucket",
|
||||
"destination",
|
||||
target_file_info("destination", version, b"old-B"),
|
||||
)
|
||||
.await;
|
||||
let store = super::super::tests::build_store_with_ctx(ctx_a.clone());
|
||||
store
|
||||
.rename_local_data(&id.to_string(), ("target-bucket", "staged"), &fi, ("target-bucket", "destination"), None)
|
||||
.await
|
||||
.expect("rename on A");
|
||||
let read = disk_a
|
||||
.read_version(
|
||||
"target-bucket",
|
||||
"target-bucket",
|
||||
"destination",
|
||||
&version.to_string(),
|
||||
&crate::disk::ReadOptions {
|
||||
read_data: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("read committed A");
|
||||
assert_eq!(read.data, fi.data, "warm={warm}");
|
||||
assert_eq!(
|
||||
tokio::fs::read(b.path().join("target-bucket/destination/xl.meta"))
|
||||
.await
|
||||
.expect("B metadata"),
|
||||
b_before
|
||||
);
|
||||
assert!(b.path().join("target-bucket/staged/xl.meta").exists());
|
||||
assert!(ctx_a.namespace_commit_generation() > 0);
|
||||
assert_eq!(ctx_b.namespace_commit_generation(), 0);
|
||||
assert!(!ctx_a.namespace_commits_pending());
|
||||
assert!(!ctx_b.namespace_commits_pending());
|
||||
assert_eq!(ctx_a.local_disk_id_map().read().await.get(&id), Some(&disk_a.endpoint().to_string()));
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn target_admission_rejects_removed_quarantined_and_replaced_arcs() {
|
||||
let ctx = Arc::new(InstanceContext::new());
|
||||
let root = tempfile::tempdir().expect("root");
|
||||
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
|
||||
let endpoint = disk.endpoint().to_string();
|
||||
for state in ["removed", "quarantined", "replaced"] {
|
||||
let replacement = new_disk(
|
||||
&disk.endpoint(),
|
||||
&DiskOption {
|
||||
cleanup: false,
|
||||
health_check: false,
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("separate active Arc");
|
||||
let map = ctx.local_disk_map();
|
||||
let mut entries = map.write().await;
|
||||
match state {
|
||||
"removed" => {
|
||||
entries.remove(&endpoint);
|
||||
}
|
||||
"quarantined" => {
|
||||
entries.insert(endpoint.clone(), None);
|
||||
}
|
||||
_ => {
|
||||
entries.insert(endpoint.clone(), Some(replacement));
|
||||
}
|
||||
}
|
||||
drop(entries);
|
||||
assert!(
|
||||
matches!(admit_local_disk(&ctx, &disk, None, "target-bucket").await, Err(DiskError::DiskNotFound)),
|
||||
"{state}"
|
||||
);
|
||||
assert!(!ctx.namespace_commits_pending());
|
||||
assert_eq!(ctx.namespace_commit_generation(), 0);
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn target_uuid_cache_cannot_admit_a_different_format_at_the_same_path() {
|
||||
let ctx = Arc::new(InstanceContext::new());
|
||||
let root = tempfile::tempdir().expect("root");
|
||||
let old_id = Uuid::new_v4();
|
||||
let old = target_disk(&ctx, root.path(), old_id).await;
|
||||
assert!(record_local_disk_id_if_active(&ctx, &old, old_id).await);
|
||||
let replacement_id = Uuid::new_v4();
|
||||
let replacement = target_disk(&ctx, root.path(), replacement_id).await;
|
||||
assert!(!Arc::ptr_eq(&old, &replacement));
|
||||
assert!(matches!(
|
||||
local_disk_candidate(&ctx, &old_id.to_string()).await,
|
||||
Err(DiskError::DiskNotFound)
|
||||
));
|
||||
let (candidate, verified) = local_disk_candidate(&ctx, &replacement_id.to_string())
|
||||
.await
|
||||
.expect("replacement UUID");
|
||||
assert!(Arc::ptr_eq(&candidate, &replacement));
|
||||
assert_eq!(verified, Some(replacement_id));
|
||||
assert!(!ctx.namespace_commits_pending());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn bootstrap_rejects_user_volumes_aliases_and_scanner_tokens_without_mutation() {
|
||||
let ctx = Arc::new(InstanceContext::new());
|
||||
let root = tempfile::tempdir().expect("root");
|
||||
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
|
||||
let target = BootstrapLocalTarget::new(ctx.clone());
|
||||
let fi = target_file_info("destination", Uuid::new_v4(), b"body");
|
||||
let user_before = seed_target(&disk, "victim", "staged", fi.clone()).await;
|
||||
let meta_before = seed_target(&disk, ".rustfs.sys/tmp", "staged", fi.clone()).await;
|
||||
for invalid in [
|
||||
"victim",
|
||||
".rustfs.sys/../victim",
|
||||
".rustfs.sys/./tmp",
|
||||
".rustfs.sys/ .. /victim",
|
||||
".rustfs.sys\\..\\victim",
|
||||
".minio.sys/../victim",
|
||||
] {
|
||||
for (src, dst) in [(invalid, ".rustfs.sys/tmp"), (".rustfs.sys/tmp", invalid)] {
|
||||
assert!(
|
||||
target
|
||||
.rename_local_data(&disk.endpoint().to_string(), (src, "staged"), &fi, (dst, "destination"), None)
|
||||
.await
|
||||
.is_err(),
|
||||
"src={src}, dst={dst}"
|
||||
);
|
||||
}
|
||||
assert!(
|
||||
target
|
||||
.undo_local_write(
|
||||
&disk.endpoint().to_string(),
|
||||
invalid,
|
||||
"staged",
|
||||
fi.clone(),
|
||||
DeleteOptions {
|
||||
undo_write: true,
|
||||
..Default::default()
|
||||
}
|
||||
)
|
||||
.await
|
||||
.is_err(),
|
||||
"{invalid}"
|
||||
);
|
||||
}
|
||||
assert!(
|
||||
target
|
||||
.rename_local_data(
|
||||
&disk.endpoint().to_string(),
|
||||
(".rustfs.sys/tmp", "staged"),
|
||||
&fi,
|
||||
(".rustfs.sys/tmp", "destination"),
|
||||
Some(Uuid::new_v4())
|
||||
)
|
||||
.await
|
||||
.is_err()
|
||||
);
|
||||
assert_eq!(
|
||||
tokio::fs::read(root.path().join("victim/staged/xl.meta"))
|
||||
.await
|
||||
.expect("user source"),
|
||||
user_before
|
||||
);
|
||||
assert_eq!(
|
||||
tokio::fs::read(root.path().join(".rustfs.sys/tmp/staged/xl.meta"))
|
||||
.await
|
||||
.expect("metadata source"),
|
||||
meta_before
|
||||
);
|
||||
assert!(!root.path().join("victim/destination").exists());
|
||||
assert!(!root.path().join(".rustfs.sys/tmp/destination").exists());
|
||||
assert_eq!(ctx.namespace_commit_generation(), 0);
|
||||
assert!(!ctx.namespace_commits_pending());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn bootstrap_allows_internal_multisegment_rename_without_namespace_owner() {
|
||||
for volume in [".rustfs.sys/tmp", ".rustfs.sys/multipart", ".minio.sys/config"] {
|
||||
let ctx = Arc::new(InstanceContext::new());
|
||||
let root = tempfile::tempdir().expect("root");
|
||||
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
|
||||
let fi = target_file_info("destination", Uuid::new_v4(), b"internal-CAS-body");
|
||||
seed_target(&disk, volume, "staged", fi.clone()).await;
|
||||
BootstrapLocalTarget::new(ctx.clone())
|
||||
.rename_local_data(&disk.endpoint().to_string(), (volume, "staged"), &fi, (volume, "destination"), None)
|
||||
.await
|
||||
.expect("legitimate bootstrap metadata write");
|
||||
let read = disk
|
||||
.read_version(
|
||||
volume,
|
||||
volume,
|
||||
"destination",
|
||||
&fi.version_id.expect("version").to_string(),
|
||||
&crate::disk::ReadOptions {
|
||||
read_data: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("read bootstrap result");
|
||||
assert_eq!(read.data, fi.data);
|
||||
assert_eq!(ctx.namespace_commit_generation(), 0);
|
||||
assert!(!ctx.namespace_commits_pending());
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
#[tokio::test]
|
||||
async fn target_rename_cancellation_retains_real_namespace_and_scanner_owners() {
|
||||
use crate::disk::os::prepared_publication_test_hooks as hooks;
|
||||
let ctx = Arc::new(InstanceContext::new());
|
||||
let sibling = Arc::new(InstanceContext::new());
|
||||
let root = tempfile::tempdir().expect("root");
|
||||
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
|
||||
let store = super::super::tests::build_store_with_ctx(ctx.clone());
|
||||
let fi = target_file_info("destination", Uuid::new_v4(), b"physically-owned");
|
||||
seed_target(&disk, "target-bucket", "staged", fi.clone()).await;
|
||||
let (token, _) = store
|
||||
.acquire_scanner_publication_lease(0, crate::runtime::instance::SCANNER_PUBLICATION_LEASE_TTL)
|
||||
.await
|
||||
.expect("real scanner token in A");
|
||||
let destination = disk
|
||||
.get_object_path_for_io_if_local("target-bucket", "destination/xl.meta")
|
||||
.expect("local disk")
|
||||
.expect("destination IO path");
|
||||
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let _hook = hooks::install(&destination, move || {
|
||||
let _ = entered_tx.send(());
|
||||
let _ = release_rx.recv();
|
||||
});
|
||||
let disk_ref = disk.endpoint().to_string();
|
||||
let mut rename = Box::pin(store.rename_local_data(
|
||||
&disk_ref,
|
||||
("target-bucket", "staged"),
|
||||
&fi,
|
||||
("target-bucket", "destination"),
|
||||
Some(token),
|
||||
));
|
||||
tokio::time::timeout(std::time::Duration::from_secs(10), async {
|
||||
tokio::select! {
|
||||
result = &mut rename => panic!("rename completed before physical pause: {result:?}"),
|
||||
entered = entered_rx => entered.expect("physical rename entered"),
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("bounded physical entry");
|
||||
drop(rename);
|
||||
assert!(store.scanner_data_usage_publication_blocked().await);
|
||||
assert!(ctx.namespace_commits_pending());
|
||||
assert!(!sibling.namespace_commits_pending());
|
||||
assert!(
|
||||
store
|
||||
.rename_local_data(&disk_ref, ("target-bucket", "staged"), &fi, ("target-bucket", "another"), Some(token))
|
||||
.await
|
||||
.is_err(),
|
||||
"real pending rename blocks another scanner publication"
|
||||
);
|
||||
assert!(store.release_scanner_publication_lease(token).await, "remove registered token");
|
||||
let gate = ctx.data_movement_operation_gate();
|
||||
assert!(
|
||||
gate.clone().try_write_owned().is_err(),
|
||||
"physical operation still owns the scanner read guard"
|
||||
);
|
||||
drop(release_tx);
|
||||
let _drained = tokio::time::timeout(std::time::Duration::from_secs(10), gate.write_owned())
|
||||
.await
|
||||
.expect("physical tail must release scanner guard");
|
||||
tokio::time::timeout(std::time::Duration::from_secs(10), async {
|
||||
while ctx.namespace_commits_pending() {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("namespace owner drains");
|
||||
let read = disk
|
||||
.read_version(
|
||||
"target-bucket",
|
||||
"target-bucket",
|
||||
"destination",
|
||||
&fi.version_id.expect("version").to_string(),
|
||||
&crate::disk::ReadOptions {
|
||||
read_data: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("read actual late commit");
|
||||
assert_eq!(read.data, fi.data);
|
||||
assert!(ctx.namespace_commit_generation() >= 2);
|
||||
assert_eq!(sibling.namespace_commit_generation(), 0);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn target_ready_rejects_unknown_foreign_released_and_expired_scanner_tokens() {
|
||||
let ctx = Arc::new(InstanceContext::new());
|
||||
let other = Arc::new(InstanceContext::new());
|
||||
let store = super::super::tests::build_store_with_ctx(ctx.clone());
|
||||
let other_store = super::super::tests::build_store_with_ctx(other);
|
||||
let root = tempfile::tempdir().expect("root");
|
||||
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
|
||||
let fi = target_file_info("destination", Uuid::new_v4(), b"unchanged");
|
||||
let before = seed_target(&disk, "target-bucket", "staged", fi.clone()).await;
|
||||
let ttl = crate::runtime::instance::SCANNER_PUBLICATION_LEASE_TTL;
|
||||
let (foreign, _) = other_store.acquire_scanner_publication_lease(0, ttl).await.expect("B token");
|
||||
let (released, _) = store.acquire_scanner_publication_lease(0, ttl).await.expect("A token");
|
||||
assert!(store.release_scanner_publication_lease(released).await);
|
||||
let (valid, _) = store.acquire_scanner_publication_lease(0, ttl).await.expect("new A token");
|
||||
for token in [Uuid::new_v4(), foreign, released] {
|
||||
assert!(
|
||||
store
|
||||
.rename_local_data(
|
||||
&disk.endpoint().to_string(),
|
||||
("target-bucket", "staged"),
|
||||
&fi,
|
||||
("target-bucket", "destination"),
|
||||
Some(token)
|
||||
)
|
||||
.await
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
tokio::time::pause();
|
||||
tokio::time::advance(ttl + std::time::Duration::from_secs(1)).await;
|
||||
tokio::time::resume();
|
||||
assert!(
|
||||
store
|
||||
.rename_local_data(
|
||||
&disk.endpoint().to_string(),
|
||||
("target-bucket", "staged"),
|
||||
&fi,
|
||||
("target-bucket", "destination"),
|
||||
Some(valid)
|
||||
)
|
||||
.await
|
||||
.is_err(),
|
||||
"expired real token"
|
||||
);
|
||||
let _ = other_store.release_scanner_publication_lease(foreign).await;
|
||||
assert_eq!(
|
||||
tokio::fs::read(root.path().join("target-bucket/staged/xl.meta"))
|
||||
.await
|
||||
.expect("source bytes"),
|
||||
before
|
||||
);
|
||||
assert!(!root.path().join("target-bucket/destination").exists());
|
||||
assert!(!ctx.namespace_commits_pending());
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn target_ordinary_timeout_keeps_its_physical_namespace_owner() {
|
||||
use crate::disk::os::prepared_publication_test_hooks as hooks;
|
||||
temp_env::async_with_vars([(rustfs_config::ENV_DRIVE_MAX_TIMEOUT_DURATION, Some("1"))], async {
|
||||
let ctx = Arc::new(InstanceContext::new());
|
||||
let store = super::super::tests::build_store_with_ctx(ctx.clone());
|
||||
let root = tempfile::tempdir().expect("root");
|
||||
let disk = target_disk(&ctx, root.path(), Uuid::new_v4()).await;
|
||||
let fi = target_file_info("destination", Uuid::new_v4(), b"timed-out-physical-commit");
|
||||
seed_target(&disk, "target-bucket", "staged", fi.clone()).await;
|
||||
let path = disk
|
||||
.get_object_path_for_io_if_local("target-bucket", "destination/xl.meta")
|
||||
.expect("local")
|
||||
.expect("destination IO path");
|
||||
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
let _hook = hooks::install(&path, move || {
|
||||
let _ = entered_tx.send(());
|
||||
let _ = release_rx.recv();
|
||||
});
|
||||
let disk_ref = disk.endpoint().to_string();
|
||||
let mut rename = Box::pin(store.rename_local_data(
|
||||
&disk_ref,
|
||||
("target-bucket", "staged"),
|
||||
&fi,
|
||||
("target-bucket", "destination"),
|
||||
None,
|
||||
));
|
||||
tokio::time::timeout(std::time::Duration::from_secs(10), async {
|
||||
tokio::select! {
|
||||
result = &mut rename => panic!("completed before physical pause: {result:?}"),
|
||||
entered = entered_rx => entered.expect("physical entry"),
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("bounded entry");
|
||||
tokio::time::pause();
|
||||
tokio::time::advance(std::time::Duration::from_secs(2)).await;
|
||||
tokio::time::resume();
|
||||
let result = tokio::time::timeout(std::time::Duration::from_secs(5), &mut rename)
|
||||
.await
|
||||
.expect("ordinary deadline remains enabled");
|
||||
assert!(matches!(result, Err(DiskError::Timeout)), "{result:?}");
|
||||
drop(rename);
|
||||
assert!(ctx.namespace_commits_pending(), "timeout is not a physical drain");
|
||||
drop(release_tx);
|
||||
tokio::time::timeout(std::time::Duration::from_secs(10), async {
|
||||
while ctx.namespace_commits_pending() {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("late physical owner drains");
|
||||
let read = disk
|
||||
.read_version(
|
||||
"target-bucket",
|
||||
"target-bucket",
|
||||
"destination",
|
||||
&fi.version_id.expect("version").to_string(),
|
||||
&crate::disk::ReadOptions {
|
||||
read_data: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("read actual timeout tail");
|
||||
assert_eq!(read.data, fi.data);
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn endpoint_rpc_authority_preserves_port_and_ipv6_brackets() {
|
||||
let endpoint = Endpoint::try_from("https://127.0.0.1:9001/d1").expect("URL endpoint");
|
||||
|
||||
@@ -1146,7 +1146,11 @@ impl ECStore {
|
||||
}
|
||||
|
||||
let backend = StorageAdminApi::backend_info(self).await;
|
||||
rustfs_madmin::StorageInfo { backend, disks }
|
||||
rustfs_madmin::StorageInfo {
|
||||
backend,
|
||||
disks,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[instrument(skip(self))]
|
||||
|
||||
@@ -37,6 +37,9 @@ pub enum Error {
|
||||
#[error("Method not allowed")]
|
||||
MethodNotAllowed,
|
||||
|
||||
#[error("You've exceeded the limit on the number of versions you can create on this object")]
|
||||
MaxVersionsExceeded,
|
||||
|
||||
#[error("Unexpected error")]
|
||||
Unexpected,
|
||||
|
||||
@@ -86,6 +89,7 @@ impl PartialEq for Error {
|
||||
(Error::FileCorrupt, Error::FileCorrupt) => true,
|
||||
(Error::DoneForNow, Error::DoneForNow) => true,
|
||||
(Error::MethodNotAllowed, Error::MethodNotAllowed) => true,
|
||||
(Error::MaxVersionsExceeded, Error::MaxVersionsExceeded) => true,
|
||||
(Error::FileNotFound, Error::FileNotFound) => true,
|
||||
(Error::FileVersionNotFound, Error::FileVersionNotFound) => true,
|
||||
(Error::VolumeNotFound, Error::VolumeNotFound) => true,
|
||||
@@ -111,6 +115,7 @@ impl Clone for Error {
|
||||
Error::FileCorrupt => Error::FileCorrupt,
|
||||
Error::DoneForNow => Error::DoneForNow,
|
||||
Error::MethodNotAllowed => Error::MethodNotAllowed,
|
||||
Error::MaxVersionsExceeded => Error::MaxVersionsExceeded,
|
||||
Error::VolumeNotFound => Error::VolumeNotFound,
|
||||
Error::Io(e) => Error::Io(std::io::Error::new(e.kind(), e.to_string())),
|
||||
Error::RmpSerdeDecode(s) => Error::RmpSerdeDecode(s.clone()),
|
||||
|
||||
+193
-16
@@ -34,11 +34,14 @@ use rustfs_utils::http::{
|
||||
};
|
||||
use s3s::header::X_AMZ_RESTORE;
|
||||
use serde::{Deserialize, Serialize};
|
||||
#[cfg(test)]
|
||||
use std::cell::Cell;
|
||||
use std::cmp::Ordering;
|
||||
use std::collections::BTreeMap;
|
||||
use std::convert::TryFrom;
|
||||
use std::hash::Hasher;
|
||||
use std::io::{Read, Write};
|
||||
use std::sync::atomic::{AtomicUsize, Ordering as AtomicOrdering};
|
||||
use std::{collections::HashMap, io::Cursor};
|
||||
use time::OffsetDateTime;
|
||||
use time::format_description::well_known::Rfc3339;
|
||||
@@ -67,8 +70,46 @@ const _XL_FLAG_INLINE_DATA: u8 = 1 << 2;
|
||||
const META_DATA_READ_DEFAULT: usize = 4 << 10;
|
||||
const MSGP_UINT32_SIZE: usize = 5;
|
||||
|
||||
/// Max object versions per object, default is 10000
|
||||
const DEFAULT_OBJECT_MAX_VERSIONS: usize = 10000;
|
||||
/// Default max object versions per object, aligned with MinIO's default.
|
||||
pub const DEFAULT_OBJECT_MAX_VERSIONS: usize = if usize::BITS >= 64 {
|
||||
9_223_372_036_854_775_807
|
||||
} else {
|
||||
usize::MAX
|
||||
};
|
||||
|
||||
static OBJECT_MAX_VERSIONS: AtomicUsize = AtomicUsize::new(DEFAULT_OBJECT_MAX_VERSIONS);
|
||||
|
||||
#[cfg(test)]
|
||||
thread_local! {
|
||||
static OBJECT_MAX_VERSIONS_OVERRIDE: Cell<Option<usize>> = const { Cell::new(None) };
|
||||
}
|
||||
|
||||
#[inline]
|
||||
pub fn object_max_versions() -> usize {
|
||||
#[cfg(test)]
|
||||
if let Some(limit) = OBJECT_MAX_VERSIONS_OVERRIDE.with(Cell::get) {
|
||||
return limit;
|
||||
}
|
||||
|
||||
OBJECT_MAX_VERSIONS.load(AtomicOrdering::Relaxed)
|
||||
}
|
||||
|
||||
pub fn set_object_max_versions(limit: usize) -> Result<()> {
|
||||
if limit == 0 {
|
||||
return Err(Error::other("object max versions must be greater than 0"));
|
||||
}
|
||||
OBJECT_MAX_VERSIONS.store(limit, AtomicOrdering::Relaxed);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
fn set_object_max_versions_override_for_test(limit: Option<usize>) -> Option<usize> {
|
||||
OBJECT_MAX_VERSIONS_OVERRIDE.with(|override_limit| {
|
||||
let previous = override_limit.get();
|
||||
override_limit.set(limit);
|
||||
previous
|
||||
})
|
||||
}
|
||||
|
||||
/// Returns the inline data map key for a version_id. "null" for null version.
|
||||
pub(crate) fn data_key_for_version(version_id: Option<Uuid>) -> String {
|
||||
@@ -460,18 +501,6 @@ impl FileMeta {
|
||||
return Err(Error::other("file meta version invalid"));
|
||||
}
|
||||
|
||||
// check max versions limit
|
||||
if self.versions.len() + 1 > DEFAULT_OBJECT_MAX_VERSIONS {
|
||||
return Err(Error::other(
|
||||
"You've exceeded the limit on the number of versions you can create on this object",
|
||||
));
|
||||
}
|
||||
|
||||
if self.versions.is_empty() {
|
||||
self.versions.push(FileMetaShallowVersion::try_from(version)?);
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let vid = version.get_version_id();
|
||||
let vid_is_null = vid.is_none() || vid == Some(Uuid::nil());
|
||||
let existing_idx = if vid_is_null {
|
||||
@@ -490,6 +519,15 @@ impl FileMeta {
|
||||
return self.set_idx(fidx, version);
|
||||
}
|
||||
|
||||
if self.versions.len() >= object_max_versions() {
|
||||
return Err(Error::MaxVersionsExceeded);
|
||||
}
|
||||
|
||||
if self.versions.is_empty() {
|
||||
self.versions.push(FileMetaShallowVersion::try_from(version)?);
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let new_shallow = FileMetaShallowVersion::try_from(version)?;
|
||||
let insert_pos = self
|
||||
.versions
|
||||
@@ -692,10 +730,15 @@ impl FileMeta {
|
||||
}
|
||||
}
|
||||
|
||||
let old_dir = v.object.as_ref().map(|v| v.data_dir).unwrap_or_default();
|
||||
// The version stays on disk while the purge replicates
|
||||
// (status PENDING/FAILED); its data dir must stay with
|
||||
// it. Returning the dir here made the disk layer delete
|
||||
// it, which turned every non-inline retained version
|
||||
// into an unreadable zombie: the purge state could never
|
||||
// be applied and the bucket could never be deleted.
|
||||
self.set_idx(i, v)?;
|
||||
|
||||
return Ok(old_dir);
|
||||
return Ok(None);
|
||||
}
|
||||
found_index = Some(i);
|
||||
}
|
||||
@@ -1325,6 +1368,88 @@ mod test {
|
||||
}
|
||||
}
|
||||
|
||||
struct ObjectMaxVersionsRestore {
|
||||
previous: Option<usize>,
|
||||
}
|
||||
|
||||
impl Drop for ObjectMaxVersionsRestore {
|
||||
fn drop(&mut self) {
|
||||
set_object_max_versions_override_for_test(self.previous);
|
||||
}
|
||||
}
|
||||
|
||||
fn with_object_max_versions_for_test<R>(limit: usize, test: impl FnOnce() -> R) -> R {
|
||||
let previous = set_object_max_versions_override_for_test(Some(limit));
|
||||
let _restore = ObjectMaxVersionsRestore { previous };
|
||||
test()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_version_filemata_rejects_new_version_above_configured_limit() {
|
||||
with_object_max_versions_for_test(2, || {
|
||||
let mut fm = FileMeta::new();
|
||||
fm.add_version_filemata(valid_object_version(Uuid::from_u128(1), vec![10, 20]))
|
||||
.expect("add first version within limit");
|
||||
fm.add_version_filemata(valid_object_version(Uuid::from_u128(2), vec![10, 20]))
|
||||
.expect("add second version at limit");
|
||||
|
||||
let err = fm
|
||||
.add_version_filemata(valid_object_version(Uuid::from_u128(3), vec![10, 20]))
|
||||
.expect_err("new version above limit must fail");
|
||||
|
||||
assert_eq!(err, Error::MaxVersionsExceeded);
|
||||
assert_eq!(fm.versions.len(), 2, "failed insert must not mutate version list");
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_version_filemata_allows_same_version_replacement_at_limit() {
|
||||
with_object_max_versions_for_test(2, || {
|
||||
let mut fm = FileMeta::new();
|
||||
let target = Uuid::from_u128(10);
|
||||
fm.add_version_filemata(valid_object_version(target, vec![10, 20]))
|
||||
.expect("add target version");
|
||||
fm.add_version_filemata(valid_object_version(Uuid::from_u128(20), vec![10, 20]))
|
||||
.expect("add peer version at limit");
|
||||
|
||||
fm.add_version_filemata(valid_object_version(target, vec![30, 40]))
|
||||
.expect("same version replacement at limit must succeed");
|
||||
|
||||
assert_eq!(fm.versions.len(), 2);
|
||||
let replaced = fm
|
||||
.versions
|
||||
.iter()
|
||||
.find(|version| version.header.version_id == Some(target))
|
||||
.expect("target version must remain present")
|
||||
.parse_version_meta()
|
||||
.expect("parse replaced version");
|
||||
assert_eq!(replaced.object.expect("object version").part_sizes, vec![30, 40]);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_version_allows_null_version_replacement_at_limit() {
|
||||
with_object_max_versions_for_test(1, || {
|
||||
let mut fm = FileMeta::new();
|
||||
let mut first = FileInfo::new("object", 2, 2);
|
||||
first.mod_time = Some(OffsetDateTime::now_utc());
|
||||
first.version_id = None;
|
||||
fm.add_version(first).expect("add initial null version");
|
||||
|
||||
let mut replacement = FileInfo::new("object", 2, 2);
|
||||
replacement.mod_time = Some(OffsetDateTime::now_utc());
|
||||
replacement.version_id = None;
|
||||
replacement.size = 42;
|
||||
fm.add_version(replacement)
|
||||
.expect("null version replacement at limit must succeed");
|
||||
|
||||
assert_eq!(fm.versions.len(), 1);
|
||||
assert_eq!(fm.versions[0].header.version_id, Some(Uuid::nil()));
|
||||
let replaced = fm.versions[0].parse_version_meta().expect("parse null replacement");
|
||||
assert_eq!(replaced.object.expect("object version").size, 42);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn add_version_filemata_uses_canonical_equal_time_order() {
|
||||
let mod_time = OffsetDateTime::from_unix_timestamp(1_700_000_000).expect("valid test timestamp");
|
||||
@@ -2702,6 +2827,58 @@ mod test {
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression for rustfs/backlog#2340: a version purge that still awaits
|
||||
/// the replication target keeps the object version on disk with a pending
|
||||
/// purge status. Its data dir must be retained with it; handing the dir
|
||||
/// back here made the disk layer delete it, leaving every non-inline
|
||||
/// retained version unreadable. The dir is released only once the purge
|
||||
/// completes and the version itself goes away.
|
||||
#[test]
|
||||
fn delete_version_pending_version_purge_retains_object_data_dir() {
|
||||
let version_id = Uuid::new_v4();
|
||||
let data_dir = Uuid::new_v4();
|
||||
let mut fm = FileMeta::new();
|
||||
let mut fi = FileInfo::new("object", 2, 2);
|
||||
fi.version_id = Some(version_id);
|
||||
fi.data_dir = Some(data_dir);
|
||||
fi.mod_time = Some(OffsetDateTime::now_utc());
|
||||
fm.add_version(fi).unwrap();
|
||||
|
||||
let pending_purge = FileInfo {
|
||||
name: "object".to_string(),
|
||||
version_id: Some(version_id),
|
||||
mark_deleted: true,
|
||||
replication_state_internal: Some(ReplicationState {
|
||||
version_purge_status_internal: Some("target=PENDING;".to_string()),
|
||||
purge_targets: version_purge_statuses_map("target=PENDING;"),
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
let freed = fm.delete_version(&pending_purge).unwrap();
|
||||
assert_eq!(freed, None, "a pending purge must not release the retained version's data dir");
|
||||
assert_eq!(fm.versions.len(), 1, "the version must stay until the purge replicates");
|
||||
let retained = fm
|
||||
.into_fileinfo("vol", "object", &version_id.to_string(), false, false, true)
|
||||
.unwrap();
|
||||
assert_eq!(retained.data_dir, Some(data_dir));
|
||||
assert_eq!(retained.version_purge_status(), VersionPurgeStatusType::Pending);
|
||||
|
||||
let completed_purge = FileInfo {
|
||||
name: "object".to_string(),
|
||||
version_id: Some(version_id),
|
||||
replication_state_internal: Some(ReplicationState {
|
||||
version_purge_status_internal: Some("target=COMPLETE;".to_string()),
|
||||
purge_targets: version_purge_statuses_map("target=COMPLETE;"),
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
let freed = fm.delete_version(&completed_purge).unwrap();
|
||||
assert_eq!(freed, Some(data_dir), "a completed purge removes the version and releases its data dir");
|
||||
assert!(fm.versions.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn delete_version_accepts_delete_only_marker_and_free_version_paths() {
|
||||
let marker_version_id = Uuid::new_v4();
|
||||
|
||||
@@ -442,6 +442,10 @@ pub struct ReplicatedTargetInfo {
|
||||
pub error: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub target_delete_marker_version_id: Option<String>,
|
||||
/// Kept in step with the replication crate's copy: the id a target that
|
||||
/// mints its own version ids assigned to this object version.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub target_version_id: Option<String>,
|
||||
}
|
||||
|
||||
impl ReplicatedTargetInfo {
|
||||
|
||||
@@ -20,7 +20,7 @@ use crate::heal::{
|
||||
task::{HealOptions, HealPriority, HealRequest, HealTask, HealTaskStatus, HealType, demote_to_debug_when},
|
||||
};
|
||||
use crate::{Error, Result};
|
||||
use metrics::{counter, gauge};
|
||||
use metrics::{counter, gauge, histogram};
|
||||
use rustfs_concurrency::WorkloadAdmissionSnapshotProvider;
|
||||
use rustfs_concurrency::workload::{ForegroundPressure, foreground_pressure};
|
||||
#[cfg(test)]
|
||||
@@ -34,7 +34,7 @@ use std::sync::LazyLock;
|
||||
use std::{
|
||||
collections::{BinaryHeap, HashMap, HashSet},
|
||||
sync::{Arc, Mutex as StdMutex, MutexGuard as StdMutexGuard},
|
||||
time::{Duration, SystemTime},
|
||||
time::{Duration, Instant, SystemTime},
|
||||
};
|
||||
use tokio::{
|
||||
sync::{Mutex, Notify, RwLock},
|
||||
@@ -181,6 +181,13 @@ fn lock_displaced_terminals(
|
||||
}
|
||||
}
|
||||
|
||||
fn lock_admission_telemetry(registry: &StdMutex<HealAdmissionTelemetry>) -> StdMutexGuard<'_, HealAdmissionTelemetry> {
|
||||
match registry.lock() {
|
||||
Ok(guard) => guard,
|
||||
Err(poisoned) => poisoned.into_inner(),
|
||||
}
|
||||
}
|
||||
|
||||
fn record_displaced_terminal(
|
||||
registry: &StdMutex<HashMap<String, Arc<CompletedHealStatus>>>,
|
||||
request: &HealRequest,
|
||||
@@ -384,6 +391,61 @@ impl HealSourceCounts {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, serde::Deserialize, serde::Serialize)]
|
||||
#[serde(rename_all = "camelCase", deny_unknown_fields)]
|
||||
pub struct HealAdmissionTelemetry {
|
||||
pub accepted: u64,
|
||||
pub merged: u64,
|
||||
pub full: u64,
|
||||
pub dropped: u64,
|
||||
pub duplicate: u64,
|
||||
pub overlap_rejected: u64,
|
||||
pub displaced: u64,
|
||||
pub force_start: u64,
|
||||
pub max_start_duration_micros: u64,
|
||||
pub max_lock_phase_micros: u64,
|
||||
}
|
||||
|
||||
impl HealAdmissionTelemetry {
|
||||
fn record(&mut self, observation: HealAdmissionObservation) {
|
||||
match observation.result {
|
||||
HealAdmissionResult::Accepted => self.accepted = self.accepted.saturating_add(1),
|
||||
HealAdmissionResult::Merged => self.merged = self.merged.saturating_add(1),
|
||||
HealAdmissionResult::Full => self.full = self.full.saturating_add(1),
|
||||
HealAdmissionResult::Dropped(_) => self.dropped = self.dropped.saturating_add(1),
|
||||
}
|
||||
if observation.context == "duplicate" {
|
||||
self.duplicate = self.duplicate.saturating_add(1);
|
||||
}
|
||||
if observation.context == "overlap_rejected" {
|
||||
self.overlap_rejected = self.overlap_rejected.saturating_add(1);
|
||||
}
|
||||
if observation.displaced {
|
||||
self.displaced = self.displaced.saturating_add(1);
|
||||
}
|
||||
if observation.force_start {
|
||||
self.force_start = self.force_start.saturating_add(1);
|
||||
}
|
||||
self.max_start_duration_micros = self
|
||||
.max_start_duration_micros
|
||||
.max(duration_micros_saturated(observation.start_duration));
|
||||
self.max_lock_phase_micros = self
|
||||
.max_lock_phase_micros
|
||||
.max(duration_micros_saturated(observation.lock_phase));
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
struct HealAdmissionObservation {
|
||||
source: HealRequestSource,
|
||||
result: HealAdmissionResult,
|
||||
context: &'static str,
|
||||
force_start: bool,
|
||||
displaced: bool,
|
||||
start_duration: Duration,
|
||||
lock_phase: Duration,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, serde::Deserialize, serde::Serialize)]
|
||||
#[serde(rename_all = "camelCase", deny_unknown_fields)]
|
||||
pub struct HealOperationsSnapshot {
|
||||
@@ -396,12 +458,18 @@ pub struct HealOperationsSnapshot {
|
||||
pub queued_by_source: HealSourceCounts,
|
||||
pub active_by_source: HealSourceCounts,
|
||||
pub retrying_by_source: HealSourceCounts,
|
||||
#[serde(default)]
|
||||
pub admission: HealAdmissionTelemetry,
|
||||
}
|
||||
|
||||
fn usize_to_u64_saturated(value: usize) -> u64 {
|
||||
u64::try_from(value).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
fn duration_micros_saturated(duration: Duration) -> u64 {
|
||||
u64::try_from(duration.as_micros()).unwrap_or(u64::MAX)
|
||||
}
|
||||
|
||||
fn heal_type_matches_path(heal_type: &HealType, heal_path: &str) -> bool {
|
||||
let heal_path = heal_path.trim_matches('/');
|
||||
if heal_path.is_empty() || heal_path == LEGACY_ROOT_HEAL_PATH {
|
||||
@@ -764,6 +832,9 @@ pub struct HealManager {
|
||||
notify: Arc<Notify>,
|
||||
/// Optional runtime workload snapshot provider used to protect foreground data-plane work.
|
||||
workload_provider: Option<WorkloadSnapshotProviderRef>,
|
||||
/// Bounded, low-cardinality admission telemetry exposed through the
|
||||
/// existing operations snapshot for cluster E2E assertions.
|
||||
admission_telemetry: Arc<StdMutex<HealAdmissionTelemetry>>,
|
||||
}
|
||||
|
||||
/// Where a task-id lookup resolved. The variants carry the resolved state
|
||||
@@ -919,6 +990,33 @@ impl HealManager {
|
||||
.increment(1);
|
||||
}
|
||||
|
||||
fn record_admission_observation(&self, observation: HealAdmissionObservation) {
|
||||
let result = observation.result.result_label().to_string();
|
||||
let reason = observation.result.reason_label().to_string();
|
||||
let source = observation.source.as_str().to_string();
|
||||
let context = observation.context.to_string();
|
||||
let force_start = observation.force_start.to_string();
|
||||
histogram!(
|
||||
"rustfs_heal_admission_start_duration_seconds",
|
||||
"source" => source.clone(),
|
||||
"result" => result.clone(),
|
||||
"reason" => reason.clone(),
|
||||
"context" => context.clone(),
|
||||
"force_start" => force_start.clone()
|
||||
)
|
||||
.record(observation.start_duration.as_secs_f64());
|
||||
histogram!(
|
||||
"rustfs_heal_admission_lock_phase_seconds",
|
||||
"source" => source,
|
||||
"result" => result,
|
||||
"reason" => reason,
|
||||
"context" => context,
|
||||
"force_start" => force_start
|
||||
)
|
||||
.record(observation.lock_phase.as_secs_f64());
|
||||
lock_admission_telemetry(&self.admission_telemetry).record(observation);
|
||||
}
|
||||
|
||||
fn remove_mrf_repair_notice_targets_for_task(&self, task_id: &str) {
|
||||
let targets = lock_mrf_repair_notice_targets(&self.mrf_repair_notice_targets).remove(task_id);
|
||||
if let Some(targets) = targets {
|
||||
@@ -1265,6 +1363,7 @@ impl HealManager {
|
||||
statistics: Arc::new(RwLock::new(HealStatistics::new())),
|
||||
notify: Arc::new(Notify::new()),
|
||||
workload_provider,
|
||||
admission_telemetry: Arc::new(StdMutex::new(HealAdmissionTelemetry::default())),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1455,6 +1554,9 @@ impl HealManager {
|
||||
preserve_alias: bool,
|
||||
mrf_notice_target: Option<MrfRepairNoticeTarget>,
|
||||
) -> Result<HealAdmissionReceipt> {
|
||||
let admission_start = Instant::now();
|
||||
let source = request.source;
|
||||
let force_start = request.force_start;
|
||||
// HS-06 forceStart semantics (admin only): MinIO stops the old task
|
||||
// first and then starts the new one. Cancel any active admin task
|
||||
// overlapping this request's path before entering admission, so the
|
||||
@@ -1505,6 +1607,7 @@ impl HealManager {
|
||||
// Match the scheduler's active -> queue order and keep retry ownership
|
||||
// in the same atomic view. Otherwise queue -> active and
|
||||
// active -> retrying transitions can slip between duplicate checks.
|
||||
let lock_phase_start = Instant::now();
|
||||
let active_heals = self.active_heals.lock().await;
|
||||
#[cfg(test)]
|
||||
pause_duplicate_admission_after_active_lock(&request.id).await;
|
||||
@@ -1539,7 +1642,17 @@ impl HealManager {
|
||||
drop(retrying_heals);
|
||||
drop(queue);
|
||||
drop(active_heals);
|
||||
let lock_phase = lock_phase_start.elapsed();
|
||||
Self::record_admission_metric(request.source, admission, "duplicate");
|
||||
self.record_admission_observation(HealAdmissionObservation {
|
||||
source,
|
||||
result: admission,
|
||||
context: "duplicate",
|
||||
force_start,
|
||||
displaced: false,
|
||||
start_duration: admission_start.elapsed(),
|
||||
lock_phase,
|
||||
});
|
||||
|
||||
match admission {
|
||||
HealAdmissionResult::Merged => {
|
||||
@@ -1618,7 +1731,17 @@ impl HealManager {
|
||||
drop(retrying_heals);
|
||||
drop(queue);
|
||||
drop(active_heals);
|
||||
let lock_phase = lock_phase_start.elapsed();
|
||||
Self::record_admission_metric(request.source, HealAdmissionResult::Dropped(reason), "overlap_rejected");
|
||||
self.record_admission_observation(HealAdmissionObservation {
|
||||
source,
|
||||
result: HealAdmissionResult::Dropped(reason),
|
||||
context: "overlap_rejected",
|
||||
force_start,
|
||||
displaced: false,
|
||||
start_duration: admission_start.elapsed(),
|
||||
lock_phase,
|
||||
});
|
||||
warn!(
|
||||
target: "rustfs::heal::manager",
|
||||
event = EVENT_HEAL_QUEUE_ADMISSION,
|
||||
@@ -1663,6 +1786,8 @@ impl HealManager {
|
||||
drop(retrying_heals);
|
||||
drop(queue);
|
||||
drop(active_heals);
|
||||
let lock_phase = lock_phase_start.elapsed();
|
||||
let displaced = displaced_terminal.is_some();
|
||||
|
||||
if let (Some(displaced_task_id), Some(displaced_terminal)) = (displaced_task_id, displaced_terminal) {
|
||||
// The queue has already removed the displaced request, so the
|
||||
@@ -1676,6 +1801,16 @@ impl HealManager {
|
||||
self.notify.notify_one();
|
||||
}
|
||||
|
||||
self.record_admission_observation(HealAdmissionObservation {
|
||||
source,
|
||||
result: admission,
|
||||
context: "submit",
|
||||
force_start,
|
||||
displaced,
|
||||
start_duration: admission_start.elapsed(),
|
||||
lock_phase,
|
||||
});
|
||||
|
||||
Ok(HealAdmissionReceipt {
|
||||
result: admission,
|
||||
task_id,
|
||||
@@ -2111,17 +2246,25 @@ impl HealManager {
|
||||
}
|
||||
publish_active_heal_count(&active_heals);
|
||||
publish_heal_queue_length(&queue);
|
||||
let queue_length = usize_to_u64_saturated(queue.len());
|
||||
let active_tasks = usize_to_u64_saturated(active_heals.len());
|
||||
let retrying_tasks = usize_to_u64_saturated(retrying_heals.len());
|
||||
drop(retrying_heals);
|
||||
drop(queue);
|
||||
drop(active_heals);
|
||||
let admission = *lock_admission_telemetry(&self.admission_telemetry);
|
||||
|
||||
HealOperationsSnapshot {
|
||||
queue_length: usize_to_u64_saturated(queue.len()),
|
||||
active_tasks: usize_to_u64_saturated(active_heals.len()),
|
||||
retrying_tasks: usize_to_u64_saturated(retrying_heals.len()),
|
||||
queue_length,
|
||||
active_tasks,
|
||||
retrying_tasks,
|
||||
queued_by_priority,
|
||||
active_by_priority,
|
||||
retrying_by_priority,
|
||||
queued_by_source,
|
||||
active_by_source,
|
||||
retrying_by_source,
|
||||
admission,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -150,6 +150,7 @@ impl HealManager {
|
||||
Err(DiskError::UnformattedDisk) => {
|
||||
if !super::super::replacement_readiness::auto_replacement_target_ready(disk, &local_disks)
|
||||
.await
|
||||
&& !super::super::replacement_readiness::directory_backed_replacement_fallback_enabled()
|
||||
{
|
||||
deferred_replacement_endpoints.insert(endpoint.to_string());
|
||||
skipped_invalid_count += 1;
|
||||
|
||||
@@ -731,11 +731,8 @@ pub(super) fn prune_completed_heal_statuses(completed_heals: &mut HashMap<String
|
||||
}
|
||||
|
||||
pub(super) fn prune_completed_heal_statuses_at(completed_heals: &mut HashMap<String, Arc<CompletedHealStatus>>, now: SystemTime) {
|
||||
completed_heals.retain(|_, completed| {
|
||||
now.duration_since(completed.completed_at)
|
||||
.map(|age| age <= KEEP_HEAL_TASK_STATUS_DURATION)
|
||||
.unwrap_or(false)
|
||||
});
|
||||
completed_heals
|
||||
.retain(|_, completed| now.duration_since(completed.completed_at).unwrap_or_default() <= KEEP_HEAL_TASK_STATUS_DURATION);
|
||||
let entry_bytes = |key: &String, value: &Arc<CompletedHealStatus>| {
|
||||
key.capacity()
|
||||
.saturating_add(size_of::<(String, Arc<CompletedHealStatus>)>())
|
||||
|
||||
@@ -189,37 +189,104 @@ fn completed_retention_count_ttl_and_alias_eviction_are_bounded() {
|
||||
);
|
||||
entries.insert("future".to_string(), Arc::new(completed_retention_fixture(now + Duration::from_nanos(1))));
|
||||
prune_completed_heal_statuses_at(&mut entries, now);
|
||||
assert_eq!(entries.len(), 1);
|
||||
assert_eq!(entries.len(), 2);
|
||||
assert!(entries.contains_key("ttl-boundary"));
|
||||
assert!(entries.contains_key("future"), "clock rollback must not expire a new completion");
|
||||
prune_completed_heal_statuses_at(&mut entries, now + Duration::from_nanos(1));
|
||||
assert_eq!(entries.len(), 1);
|
||||
assert!(entries.contains_key("future"));
|
||||
prune_completed_heal_statuses_at(&mut entries, now + Duration::from_nanos(1) + KEEP_HEAL_TASK_STATUS_DURATION);
|
||||
assert!(entries.contains_key("future"), "the exact TTL boundary remains retained");
|
||||
prune_completed_heal_statuses_at(&mut entries, now + Duration::from_nanos(2) + KEEP_HEAL_TASK_STATUS_DURATION);
|
||||
assert!(entries.is_empty());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn completed_retention_clock_rollback_preserves_terminal_alias_queries() {
|
||||
let completed_at = SystemTime::now() + Duration::from_secs(3600);
|
||||
for status in [
|
||||
HealTaskStatus::Completed,
|
||||
HealTaskStatus::Failed {
|
||||
error: "fixture failure".to_string(),
|
||||
},
|
||||
HealTaskStatus::Cancelled,
|
||||
] {
|
||||
let manager = HealManager::new(Arc::new(MockStorage), None);
|
||||
let mut snapshot = completed_retention_fixture(completed_at);
|
||||
snapshot.status = status.clone();
|
||||
let expected_progress = snapshot.progress.clone();
|
||||
let snapshot = Arc::new(snapshot);
|
||||
{
|
||||
let mut completed = manager.completed_heals.lock().await;
|
||||
completed.insert("canonical".to_string(), Arc::clone(&snapshot));
|
||||
completed.insert("alias".to_string(), Arc::clone(&snapshot));
|
||||
}
|
||||
for token in ["canonical", "alias"] {
|
||||
let report = manager
|
||||
.get_task_report_since(token, Some(3))
|
||||
.await
|
||||
.expect("a clock rollback must retain terminal queries");
|
||||
assert_eq!(report.status, status);
|
||||
assert_eq!(report.progress, expected_progress);
|
||||
assert_eq!(report.result_items.len(), 1);
|
||||
assert_eq!((report.min_seq, report.next_seq), (3, 5));
|
||||
assert!(!report.result_items_truncated);
|
||||
}
|
||||
let mut completed = manager.completed_heals.lock().await;
|
||||
prune_completed_heal_statuses_at(&mut completed, completed_at + KEEP_HEAL_TASK_STATUS_DURATION);
|
||||
assert_eq!(completed.len(), 2, "both tokens remain at the exact TTL boundary");
|
||||
prune_completed_heal_statuses_at(&mut completed, completed_at + KEEP_HEAL_TASK_STATUS_DURATION + Duration::from_nanos(1));
|
||||
assert!(completed.is_empty(), "both tokens expire after the TTL");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn completed_retention_clock_rollback_keeps_count_and_alias_eviction_bounded() {
|
||||
let now = SystemTime::UNIX_EPOCH + Duration::from_secs(3600);
|
||||
let oldest = Arc::new(completed_retention_fixture(now + Duration::from_secs(1)));
|
||||
let mut entries = HashMap::from([
|
||||
("oldest".to_string(), Arc::clone(&oldest)),
|
||||
("oldest-alias".to_string(), oldest),
|
||||
]);
|
||||
for index in 2..=MAX_COMPLETED_HEAL_TOKENS {
|
||||
entries.insert(
|
||||
format!("task-{index}"),
|
||||
Arc::new(completed_retention_fixture(now + Duration::from_secs(2))),
|
||||
);
|
||||
}
|
||||
prune_completed_heal_statuses_at(&mut entries, now);
|
||||
assert_eq!(entries.len(), MAX_COMPLETED_HEAL_TOKENS - 1);
|
||||
assert!(!entries.contains_key("oldest"));
|
||||
assert!(!entries.contains_key("oldest-alias"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn completed_retention_total_byte_cap_and_cap_plus_one() {
|
||||
let now = SystemTime::now();
|
||||
let key = "large".to_string();
|
||||
let mut entry = completed_retention_fixture(now);
|
||||
let base_bytes = entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>();
|
||||
entry.retained_bytes.take();
|
||||
entry.status = HealTaskStatus::Failed {
|
||||
error: "x".repeat(MAX_COMPLETED_HEAL_BYTES - base_bytes),
|
||||
};
|
||||
assert_eq!(
|
||||
entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>(),
|
||||
MAX_COMPLETED_HEAL_BYTES
|
||||
);
|
||||
let mut entries = HashMap::from([(key, Arc::new(entry))]);
|
||||
prune_completed_heal_statuses_at(&mut entries, now);
|
||||
assert_eq!(entries.len(), 1, "exact byte cap remains retained");
|
||||
let mut over = Arc::try_unwrap(entries.remove("large").expect("entry retained")).expect("entry not shared");
|
||||
over.retained_bytes.take();
|
||||
if let HealTaskStatus::Failed { error } = &mut over.status {
|
||||
*error = "x".repeat(error.len() + 1);
|
||||
for completed_at in [now, now + Duration::from_secs(1)] {
|
||||
let key = "large".to_string();
|
||||
let mut entry = completed_retention_fixture(completed_at);
|
||||
let base_bytes = entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>();
|
||||
entry.retained_bytes.take();
|
||||
entry.status = HealTaskStatus::Failed {
|
||||
error: "x".repeat(MAX_COMPLETED_HEAL_BYTES - base_bytes),
|
||||
};
|
||||
assert_eq!(
|
||||
entry.retained_bytes() + key.capacity() + size_of::<(String, Arc<CompletedHealStatus>)>(),
|
||||
MAX_COMPLETED_HEAL_BYTES
|
||||
);
|
||||
let mut entries = HashMap::from([(key, Arc::new(entry))]);
|
||||
prune_completed_heal_statuses_at(&mut entries, now);
|
||||
assert_eq!(entries.len(), 1, "exact byte cap remains retained");
|
||||
let mut over = Arc::try_unwrap(entries.remove("large").expect("entry retained")).expect("entry not shared");
|
||||
over.retained_bytes.take();
|
||||
if let HealTaskStatus::Failed { error } = &mut over.status {
|
||||
*error = "x".repeat(error.len() + 1);
|
||||
}
|
||||
entries.insert("large".to_string(), Arc::new(over));
|
||||
prune_completed_heal_statuses_at(&mut entries, now);
|
||||
assert!(entries.is_empty(), "oversized metadata cannot escape total byte bound");
|
||||
}
|
||||
entries.insert("large".to_string(), Arc::new(over));
|
||||
prune_completed_heal_statuses_at(&mut entries, now);
|
||||
assert!(entries.is_empty(), "oversized metadata cannot escape total byte bound");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -877,6 +944,84 @@ fn bucket_request(bucket: &str, priority: HealPriority, source: HealRequestSourc
|
||||
request
|
||||
}
|
||||
|
||||
fn scoped_object_request(bucket: &str, object: &str, pool_index: usize, set_index: usize) -> HealRequest {
|
||||
HealRequest::new(
|
||||
HealType::Object {
|
||||
bucket: bucket.to_string(),
|
||||
object: object.to_string(),
|
||||
version_id: None,
|
||||
},
|
||||
HealOptions {
|
||||
pool_index: Some(pool_index),
|
||||
set_index: Some(set_index),
|
||||
..Default::default()
|
||||
},
|
||||
HealPriority::Normal,
|
||||
)
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn scheduler_bulkhead_starts_other_sets_and_retains_same_set_tail() {
|
||||
let manager = HealManager::new(Arc::new(MockStorage), None);
|
||||
{
|
||||
let mut config = manager.config.write().await;
|
||||
config.max_concurrent_heals = 2;
|
||||
config.max_concurrent_per_set = 1;
|
||||
config.set_bulkhead_enable = true;
|
||||
config.event_driven_scheduler_enable = false;
|
||||
config.mainline_throttle_enable = false;
|
||||
}
|
||||
|
||||
let first_set = scoped_object_request("scheduler-bulkhead-set-a-first", "object-a", 0, 1);
|
||||
let first_set_id = first_set.id.clone();
|
||||
let same_set_tail = scoped_object_request("scheduler-bulkhead-set-a-tail", "object-b", 0, 1);
|
||||
let same_set_tail_id = same_set_tail.id.clone();
|
||||
let other_set = scoped_object_request("scheduler-bulkhead-set-b", "object-c", 0, 2);
|
||||
let other_set_id = other_set.id.clone();
|
||||
|
||||
let first_hook = Arc::new(CompletedRetentionHook::default());
|
||||
let other_hook = Arc::new(CompletedRetentionHook::default());
|
||||
{
|
||||
let mut hooks = COMPLETED_RETENTION_HOOKS.lock().await;
|
||||
hooks.insert("scheduler-bulkhead-set-a-first".to_string(), first_hook.clone());
|
||||
hooks.insert("scheduler-bulkhead-set-b".to_string(), other_hook.clone());
|
||||
}
|
||||
{
|
||||
let mut queue = manager.heal_queue.lock().await;
|
||||
assert_eq!(queue.push(first_set), QueuePushOutcome::Accepted);
|
||||
assert_eq!(queue.push(same_set_tail), QueuePushOutcome::Accepted);
|
||||
assert_eq!(queue.push(other_set), QueuePushOutcome::Accepted);
|
||||
}
|
||||
|
||||
process_manager_queue_once(&manager).await;
|
||||
tokio::time::timeout(Duration::from_secs(5), first_hook.started.notified())
|
||||
.await
|
||||
.expect("first set task should start");
|
||||
tokio::time::timeout(Duration::from_secs(5), other_hook.started.notified())
|
||||
.await
|
||||
.expect("other set task should start despite same-set tail");
|
||||
|
||||
assert_eq!(manager.get_active_task_count().await, 2);
|
||||
assert_eq!(manager.get_queue_length().await, 1);
|
||||
assert!(matches!(manager.get_task_status(&same_set_tail_id).await, Ok(HealTaskStatus::Pending)));
|
||||
{
|
||||
let active = manager.active_heals.lock().await;
|
||||
assert!(active.contains_key(&first_set_id));
|
||||
assert!(active.contains_key(&other_set_id));
|
||||
assert!(!active.contains_key(&same_set_tail_id));
|
||||
let counts = running_heal_set_counts(&active);
|
||||
assert_eq!(counts.get("pool_0_set_1"), Some(&1));
|
||||
assert_eq!(counts.get("pool_0_set_2"), Some(&1));
|
||||
}
|
||||
|
||||
manager.cancel_task(&first_set_id).await.expect("cancel first active task");
|
||||
manager.cancel_task(&other_set_id).await.expect("cancel other active task");
|
||||
COMPLETED_RETENTION_HOOKS
|
||||
.lock()
|
||||
.await
|
||||
.retain(|bucket, _| !bucket.starts_with("scheduler-bulkhead-"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_push_displacing_lower_priority_actually_enqueues_new_request() {
|
||||
// Regression for the release-build defect where the enqueue side effect lived inside
|
||||
@@ -2647,6 +2792,88 @@ async fn admin_force_start_cancels_overlapping_active_task_first() {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn admission_snapshot_tracks_start_duplicate_force_start_and_displacement() {
|
||||
let storage: Arc<dyn HealStorageAPI> = Arc::new(MockStorage);
|
||||
let manager = Arc::new(HealManager::new(
|
||||
storage,
|
||||
Some(HealConfig {
|
||||
queue_size: 1,
|
||||
..Default::default()
|
||||
}),
|
||||
));
|
||||
|
||||
let mut paused = admin_prefix_request("bucket-a", "logs/");
|
||||
paused.priority = HealPriority::Low;
|
||||
let hook = Arc::new(DuplicateAdmissionTestHook {
|
||||
request_id: paused.id.clone(),
|
||||
active_lock_reached: Notify::new(),
|
||||
active_lock_release: Notify::new(),
|
||||
});
|
||||
*DUPLICATE_ADMISSION_TEST_HOOK.lock().await = Some(hook.clone());
|
||||
|
||||
let submit_manager = Arc::clone(&manager);
|
||||
let mut paused_submission = tokio::spawn(async move { submit_manager.submit_heal_request(paused).await });
|
||||
tokio::time::timeout(Duration::from_secs(1), hook.active_lock_reached.notified())
|
||||
.await
|
||||
.expect("admission should reach the test-only lock phase hook");
|
||||
assert!(
|
||||
tokio::time::timeout(Duration::from_millis(10), &mut paused_submission)
|
||||
.await
|
||||
.is_err(),
|
||||
"admission must wait while the lock-phase hook is held"
|
||||
);
|
||||
hook.active_lock_release.notify_one();
|
||||
assert_eq!(
|
||||
paused_submission
|
||||
.await
|
||||
.expect("paused admission task should join")
|
||||
.expect("paused admission should succeed"),
|
||||
HealAdmissionResult::Accepted
|
||||
);
|
||||
*DUPLICATE_ADMISSION_TEST_HOOK.lock().await = None;
|
||||
|
||||
let duplicate = admin_prefix_request("bucket-a", "logs/");
|
||||
let duplicate_receipt = manager
|
||||
.submit_heal_request_with_receipt(duplicate)
|
||||
.await
|
||||
.expect("duplicate admission should return a canonical receipt");
|
||||
assert_eq!(duplicate_receipt.result, HealAdmissionResult::Merged);
|
||||
|
||||
let mut high = admin_prefix_request("bucket-b", "logs/");
|
||||
high.priority = HealPriority::High;
|
||||
assert_eq!(
|
||||
manager
|
||||
.submit_heal_request(high)
|
||||
.await
|
||||
.expect("higher priority admin request should displace queued low-priority work"),
|
||||
HealAdmissionResult::Accepted
|
||||
);
|
||||
|
||||
let mut forced = admin_prefix_request("bucket-c", "logs/");
|
||||
forced.force_start = true;
|
||||
assert_eq!(
|
||||
manager
|
||||
.submit_heal_request(forced)
|
||||
.await
|
||||
.expect("forceStart should keep explicit admission semantics"),
|
||||
HealAdmissionResult::Accepted
|
||||
);
|
||||
|
||||
let admission = manager.operations_snapshot().await.admission;
|
||||
assert_eq!(admission.accepted, 3);
|
||||
assert_eq!(admission.merged, 1);
|
||||
assert_eq!(admission.full, 0);
|
||||
assert_eq!(admission.dropped, 0);
|
||||
assert_eq!(admission.duplicate, 1);
|
||||
assert_eq!(admission.displaced, 1);
|
||||
assert_eq!(admission.force_start, 1);
|
||||
assert!(
|
||||
admission.max_lock_phase_micros > 0,
|
||||
"snapshot should expose a measurable queue/admission lock phase for p95-style external aggregation"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_operations_snapshot_counts_active_by_source_and_priority() {
|
||||
let storage: Arc<dyn HealStorageAPI> = Arc::new(MockStorage);
|
||||
|
||||
@@ -34,7 +34,7 @@ use storage_api::owner::{
|
||||
};
|
||||
|
||||
pub use erasure_healer::ErasureSetHealer;
|
||||
pub use manager::{HealManager, HealOperationsSnapshot, HealPriorityCounts, HealSourceCounts};
|
||||
pub use manager::{HealAdmissionTelemetry, HealManager, HealOperationsSnapshot, HealPriorityCounts, HealSourceCounts};
|
||||
pub use resume::{CheckpointManager, ResumeCheckpoint, ResumeManager, ResumeState, ResumeUtils};
|
||||
pub use task::{HealOptions, HealPriority, HealRequest, HealTask, HealType};
|
||||
|
||||
|
||||
@@ -186,6 +186,11 @@ impl MrfQueue {
|
||||
MrfQueuePushResult::Enqueued
|
||||
}
|
||||
|
||||
fn raise_limits_for_replay(&mut self, intents: usize, bytes: usize) {
|
||||
self.capacity = self.capacity.max(self.pending.len().saturating_add(intents));
|
||||
self.byte_budget = self.byte_budget.max(self.bytes.saturating_add(bytes));
|
||||
}
|
||||
|
||||
/// Bool compatibility adapter: only a newly executable queue item is
|
||||
/// reported as accepted; a coalesced duplicate is not durable admission.
|
||||
#[cfg(test)]
|
||||
@@ -511,6 +516,7 @@ async fn submit_mrf_heal_request(manager: &HealManager, intent: &MrfIntent) -> c
|
||||
|
||||
struct MrfRuntime {
|
||||
queue: MrfQueue,
|
||||
retained_replay_intents: Vec<MrfIntent>,
|
||||
config: MrfConsumerConfig,
|
||||
new_since_flush: usize,
|
||||
/// True while the in-memory pending set has changed since the last
|
||||
@@ -519,9 +525,8 @@ struct MrfRuntime {
|
||||
/// waiting out an admission backoff must not re-fsync every local disk
|
||||
/// twice a second.
|
||||
dirty: bool,
|
||||
/// True while a journal snapshot exists on disk that no longer reflects
|
||||
/// an all-consumed pending set; the next idle tick removes it (MinIO
|
||||
/// deletes its `list.bin` after replay for the same reason).
|
||||
/// True while a journal snapshot exists on disk that may still be needed
|
||||
/// for replay or cleanup.
|
||||
journal_on_disk: bool,
|
||||
/// Earliest instant a full-admission retry may proceed.
|
||||
backoff_until: Option<tokio::time::Instant>,
|
||||
@@ -531,7 +536,7 @@ impl MrfRuntime {
|
||||
fn snapshot(&self) -> (Vec<u8>, Vec<u8>) {
|
||||
let mut authoritative = Vec::new();
|
||||
let mut legacy = Vec::new();
|
||||
for intent in self.queue.intents() {
|
||||
for intent in self.retained_replay_intents.iter().chain(self.queue.intents()) {
|
||||
let scoped_identity =
|
||||
!matches!(intent.kind, rustfs_common::mrf_channel::MrfKind::MetadataCorruption) && intent.scope.is_some();
|
||||
if !encode_intent(intent, &mut authoritative) {
|
||||
@@ -655,9 +660,10 @@ pub fn spawn_mrf_consumer(manager: Arc<HealManager>) {
|
||||
|
||||
/// Replay the durable journal into a fresh pending queue and submit whatever
|
||||
/// it armed. Returns the number of intact intents replayed. Duplicates are
|
||||
/// merged by the manager's dedup key; the journal file is removed once read
|
||||
/// (torn tails truncate via the per-record CRC). Public for integration tests;
|
||||
/// the live consumer invokes this through [`replay_into`] at startup.
|
||||
/// merged by the manager's dedup key; the journal is retained whenever replay
|
||||
/// cannot fully hand off a successor in-memory snapshot (torn tails truncate
|
||||
/// via the per-record CRC). Public for integration tests; the live consumer
|
||||
/// invokes this through [`replay_into`] at startup.
|
||||
pub async fn replay_journal_once(manager: &Arc<HealManager>) -> usize {
|
||||
let config = MrfConsumerConfig::default();
|
||||
let mut queue = MrfQueue::new(config.queue_capacity, config.journal_max_bytes);
|
||||
@@ -668,9 +674,16 @@ pub async fn replay_journal_once(manager: &Arc<HealManager>) -> usize {
|
||||
struct ReplayOutcome {
|
||||
replayed: usize,
|
||||
journal_on_disk: bool,
|
||||
retained_replay_intents: Vec<MrfIntent>,
|
||||
}
|
||||
|
||||
/// Shared replay core: read + decode + re-arm + delete, then drain what fits.
|
||||
fn replay_must_retain_journal(rearm_incomplete: bool, pending_depth: usize, retained_replay_depth: usize) -> bool {
|
||||
rearm_incomplete || pending_depth > 0 || retained_replay_depth > 0
|
||||
}
|
||||
|
||||
/// Shared replay core: read + decode + re-arm, then drain what fits. The
|
||||
/// startup journal is removed only after every replayed record has either
|
||||
/// reached the manager or been proven redundant inside the in-memory queue.
|
||||
async fn replay_into(
|
||||
manager: &Arc<HealManager>,
|
||||
queue: &mut MrfQueue,
|
||||
@@ -687,6 +700,7 @@ async fn replay_into(
|
||||
return ReplayOutcome {
|
||||
replayed: 0,
|
||||
journal_on_disk: false,
|
||||
retained_replay_intents: Vec::new(),
|
||||
};
|
||||
}
|
||||
},
|
||||
@@ -703,35 +717,73 @@ async fn replay_into(
|
||||
}
|
||||
counter!("rustfs_heal_mrf_replayed_total").increment(u64::try_from(intents.len()).unwrap_or(u64::MAX));
|
||||
let replayed = intents.len();
|
||||
let replay_bytes = intents
|
||||
.iter()
|
||||
.fold(0usize, |total, intent| total.saturating_add(intent.estimated_bytes()));
|
||||
// The decoded journal is already resident in memory. Allow the startup
|
||||
// queue to arm that full bounded snapshot so a later flush can become the
|
||||
// successor anchor instead of overwriting the old journal with only a
|
||||
// prefix.
|
||||
queue.raise_limits_for_replay(intents.len(), replay_bytes);
|
||||
let mut rearm_incomplete = false;
|
||||
for intent in intents {
|
||||
let result = queue.try_push_typed(intent.clone());
|
||||
if !matches!(result, MrfQueuePushResult::Enqueued) {
|
||||
rustfs_common::mrf_channel::release_mrf_intent(&intent);
|
||||
match result {
|
||||
MrfQueuePushResult::Enqueued => {}
|
||||
MrfQueuePushResult::Coalesced => rustfs_common::mrf_channel::release_mrf_intent(&intent),
|
||||
MrfQueuePushResult::Rejected => {
|
||||
rearm_incomplete = true;
|
||||
rustfs_common::mrf_channel::release_mrf_intent(&intent);
|
||||
}
|
||||
}
|
||||
}
|
||||
let journal_on_disk = !delete_journals().await;
|
||||
|
||||
// Drain the replayed intents immediately; whatever the manager refuses
|
||||
// stays armed in `queue` for the consumer's retry loop.
|
||||
let mut retained_replay_intents = Vec::new();
|
||||
if backoff_until.is_none() {
|
||||
while let Some(mut intent) = queue.pop_front() {
|
||||
match submit_mrf_heal_request(manager, &intent).await {
|
||||
Ok(HealAdmissionResult::Accepted) | Ok(HealAdmissionResult::Merged) => {}
|
||||
Ok(HealAdmissionResult::Accepted) | Ok(HealAdmissionResult::Merged) => {
|
||||
retained_replay_intents.push(intent);
|
||||
}
|
||||
Ok(HealAdmissionResult::Full) | Ok(HealAdmissionResult::Dropped(HealAdmissionDropReason::QueueFull)) => {
|
||||
intent.attempts = intent.attempts.saturating_add(1);
|
||||
if intent.attempts < MRF_MAX_ATTEMPTS {
|
||||
queue.push_back(intent);
|
||||
*backoff_until = Some(tokio::time::Instant::now());
|
||||
} else {
|
||||
rearm_incomplete = true;
|
||||
counter!("rustfs_heal_mrf_dropped_total", "reason" => "attempts_exhausted").increment(1);
|
||||
rustfs_common::mrf_channel::release_mrf_intent(&intent);
|
||||
}
|
||||
break;
|
||||
}
|
||||
Ok(HealAdmissionResult::Dropped(_)) => {}
|
||||
Err(_) => {
|
||||
intent.attempts = intent.attempts.saturating_add(1);
|
||||
if intent.attempts < MRF_MAX_ATTEMPTS {
|
||||
queue.push_back(intent);
|
||||
*backoff_until = Some(tokio::time::Instant::now());
|
||||
} else {
|
||||
rearm_incomplete = true;
|
||||
counter!("rustfs_heal_mrf_dropped_total", "reason" => "attempts_exhausted").increment(1);
|
||||
rustfs_common::mrf_channel::release_mrf_intent(&intent);
|
||||
}
|
||||
break;
|
||||
}
|
||||
Ok(HealAdmissionResult::Dropped(_)) | Err(_) => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
let journal_on_disk = if replay_must_retain_journal(rearm_incomplete, queue.depth(), retained_replay_intents.len()) {
|
||||
true
|
||||
} else {
|
||||
!delete_journals().await
|
||||
};
|
||||
ReplayOutcome {
|
||||
replayed,
|
||||
journal_on_disk,
|
||||
retained_replay_intents,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -741,6 +793,7 @@ async fn run_mrf_consumer(manager: Arc<HealManager>, mut receiver: mpsc::Receive
|
||||
let config = MrfConsumerConfig::default();
|
||||
let mut runtime = MrfRuntime {
|
||||
queue: MrfQueue::new(config.queue_capacity, config.journal_max_bytes),
|
||||
retained_replay_intents: Vec::new(),
|
||||
config: config.clone(),
|
||||
new_since_flush: 0,
|
||||
dirty: false,
|
||||
@@ -748,13 +801,14 @@ async fn run_mrf_consumer(manager: Arc<HealManager>, mut receiver: mpsc::Receive
|
||||
backoff_until: None,
|
||||
};
|
||||
|
||||
// Replay: read the journal, re-arm intents (duplicates are merged by the
|
||||
// manager's dedup key), then drop the file so the next flush starts clean.
|
||||
// Replay reads the journal and re-arms intents. The startup journal stays
|
||||
// on disk whenever any replayed intent still needs a successor snapshot.
|
||||
let replay = replay_into(&manager, &mut runtime.queue, &mut runtime.backoff_until).await;
|
||||
runtime.journal_on_disk = replay.journal_on_disk;
|
||||
// The replay deleted the journal file; anything still pending (e.g. the
|
||||
// manager was full and backoff armed) must be re-persisted by the next
|
||||
// flush or a crash before it would lose those intents.
|
||||
runtime.retained_replay_intents = replay.retained_replay_intents;
|
||||
// Anything still pending (e.g. the manager was full and backoff armed)
|
||||
// must be re-persisted by the next flush before replay can delete the
|
||||
// startup anchor.
|
||||
runtime.dirty = runtime.queue.depth() > 0;
|
||||
|
||||
let mut flush_tick = tokio::time::interval(runtime.config.flush_interval);
|
||||
@@ -769,7 +823,7 @@ async fn run_mrf_consumer(manager: Arc<HealManager>, mut receiver: mpsc::Receive
|
||||
// provably current AND idle (a dirty or pending state
|
||||
// gets one last persist attempt, matching the shutdown
|
||||
// retry the unconditional flush used to provide).
|
||||
if runtime.dirty || runtime.queue.depth() > 0 {
|
||||
if runtime.dirty || runtime.queue.depth() > 0 || !runtime.retained_replay_intents.is_empty() {
|
||||
runtime.flush().await;
|
||||
}
|
||||
tracing::info!(
|
||||
@@ -795,7 +849,12 @@ async fn run_mrf_consumer(manager: Arc<HealManager>, mut receiver: mpsc::Receive
|
||||
}
|
||||
}
|
||||
_ = flush_tick.tick() => {
|
||||
match tick_action(runtime.dirty, runtime.queue.depth(), runtime.journal_on_disk) {
|
||||
match tick_action(
|
||||
runtime.dirty,
|
||||
runtime.queue.depth(),
|
||||
runtime.retained_replay_intents.len(),
|
||||
runtime.journal_on_disk,
|
||||
) {
|
||||
TickAction::Flush => {
|
||||
runtime.flush().await;
|
||||
runtime.dispatch(manager.as_ref()).await;
|
||||
@@ -808,8 +867,8 @@ async fn run_mrf_consumer(manager: Arc<HealManager>, mut receiver: mpsc::Receive
|
||||
runtime.dispatch(manager.as_ref()).await;
|
||||
}
|
||||
TickAction::DeleteJournal => {
|
||||
// All intents consumed: remove the journal so a restart
|
||||
// replays nothing (mirrors MinIO's post-replay unlink).
|
||||
// Only remove a stale journal after every replayed
|
||||
// intent has a durable successor proof.
|
||||
if delete_journals().await {
|
||||
runtime.journal_on_disk = false;
|
||||
gauge!("rustfs_heal_mrf_journal_bytes").set(0.0);
|
||||
@@ -838,11 +897,13 @@ enum TickAction {
|
||||
Idle,
|
||||
}
|
||||
|
||||
fn tick_action(dirty: bool, depth: usize, journal_on_disk: bool) -> TickAction {
|
||||
fn tick_action(dirty: bool, depth: usize, retained_replay_depth: usize, journal_on_disk: bool) -> TickAction {
|
||||
if dirty {
|
||||
TickAction::Flush
|
||||
} else if depth > 0 {
|
||||
TickAction::Retry
|
||||
} else if retained_replay_depth > 0 {
|
||||
TickAction::Idle
|
||||
} else if journal_on_disk {
|
||||
TickAction::DeleteJournal
|
||||
} else {
|
||||
@@ -875,19 +936,90 @@ mod tests {
|
||||
|
||||
// Dirty dominates: a changed pending set flushes even when idle
|
||||
// otherwise.
|
||||
assert!(matches!(tick_action(true, 0, false), Flush));
|
||||
assert!(matches!(tick_action(true, 3, true), Flush));
|
||||
assert!(matches!(tick_action(true, 0, 0, false), Flush));
|
||||
assert!(matches!(tick_action(true, 3, 0, true), Flush));
|
||||
|
||||
// Clean backlog: no rewrite, but keep draining so an expired
|
||||
// admission backoff retries on time.
|
||||
assert!(matches!(tick_action(false, 1, false), Retry));
|
||||
assert!(matches!(tick_action(false, 2, true), Retry));
|
||||
assert!(matches!(tick_action(false, 1, 0, false), Retry));
|
||||
assert!(matches!(tick_action(false, 2, 0, true), Retry));
|
||||
|
||||
// Replayed records accepted by the manager are still restart anchors
|
||||
// until a durable successor proof can tombstone them.
|
||||
assert!(matches!(tick_action(false, 0, 1, true), Idle));
|
||||
|
||||
// Quiescent with a stale journal file on disk: remove it.
|
||||
assert!(matches!(tick_action(false, 0, true), DeleteJournal));
|
||||
assert!(matches!(tick_action(false, 0, 0, true), DeleteJournal));
|
||||
|
||||
// Fully quiescent: nothing to do.
|
||||
assert!(matches!(tick_action(false, 0, false), Idle));
|
||||
assert!(matches!(tick_action(false, 0, 0, false), Idle));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn replay_cleanup_retains_journal_for_unarmed_or_refused_records() {
|
||||
assert!(
|
||||
replay_must_retain_journal(true, 0, 0),
|
||||
"a rejected replay record still needs its disk anchor"
|
||||
);
|
||||
assert!(
|
||||
replay_must_retain_journal(false, 1, 0),
|
||||
"a Full admission retry must keep the startup journal until the next snapshot"
|
||||
);
|
||||
assert!(
|
||||
replay_must_retain_journal(false, 0, 1),
|
||||
"an accepted replay record still needs a durable successor before cleanup"
|
||||
);
|
||||
assert!(
|
||||
!replay_must_retain_journal(false, 0, 0),
|
||||
"only a fully consumed replay snapshot with no retained anchors may be deleted"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn retained_replay_anchor_remains_in_successor_snapshot() {
|
||||
let retained = intent("accepted-replay", "object", 0);
|
||||
let mut runtime = MrfRuntime {
|
||||
queue: MrfQueue::new(8, 8192),
|
||||
retained_replay_intents: vec![retained.clone()],
|
||||
config: MrfConsumerConfig::default(),
|
||||
new_since_flush: 0,
|
||||
dirty: false,
|
||||
journal_on_disk: true,
|
||||
backoff_until: None,
|
||||
};
|
||||
assert_eq!(
|
||||
runtime.queue.try_push_typed(intent("new-pending", "object", 0)),
|
||||
MrfQueuePushResult::Enqueued
|
||||
);
|
||||
|
||||
let (authoritative, legacy) = runtime.snapshot();
|
||||
let (decoded, truncated) = decode_journal(&authoritative);
|
||||
let (legacy_decoded, legacy_truncated) = decode_journal(&legacy);
|
||||
|
||||
assert_eq!(truncated, 0);
|
||||
assert_eq!(legacy_truncated, 0);
|
||||
assert_eq!(decoded.len(), 2);
|
||||
assert_eq!(legacy_decoded.len(), 2);
|
||||
assert!(
|
||||
decoded.iter().any(|intent| intent.bucket == retained.bucket),
|
||||
"accepted replay anchor must remain crash-replayable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn replay_can_arm_more_records_than_live_queue_budget() {
|
||||
let mut queue = MrfQueue::new(1, intent("bucket", "object-0", 0).estimated_bytes());
|
||||
let intents = vec![intent("bucket", "object-0", 0), intent("bucket", "object-1", 0)];
|
||||
let bytes = intents
|
||||
.iter()
|
||||
.fold(0usize, |total, intent| total.saturating_add(intent.estimated_bytes()));
|
||||
|
||||
queue.raise_limits_for_replay(intents.len(), bytes);
|
||||
|
||||
for intent in intents {
|
||||
assert_eq!(queue.try_push_typed(intent), MrfQueuePushResult::Enqueued);
|
||||
}
|
||||
assert_eq!(queue.depth(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -123,6 +123,13 @@ pub struct CommittedSnapshot {
|
||||
payload: Vec<u8>,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
struct SnapshotReadStats {
|
||||
file_reads: usize,
|
||||
bytes_read: usize,
|
||||
peak_file_bytes: usize,
|
||||
}
|
||||
|
||||
impl CommittedSnapshot {
|
||||
/// Persistent single-writer sequence, not a process UUID ordering.
|
||||
pub fn sequence(&self) -> u64 {
|
||||
@@ -162,6 +169,15 @@ pub enum RecoverySnapshot {
|
||||
}
|
||||
|
||||
async fn read_bounded(disk: &EcstoreDiskStore, path: &str, limit: usize) -> Result<Option<Vec<u8>>, SnapshotError> {
|
||||
read_bounded_with_stats(disk, path, limit, None).await
|
||||
}
|
||||
|
||||
async fn read_bounded_with_stats(
|
||||
disk: &EcstoreDiskStore,
|
||||
path: &str,
|
||||
limit: usize,
|
||||
mut stats: Option<&mut SnapshotReadStats>,
|
||||
) -> Result<Option<Vec<u8>>, SnapshotError> {
|
||||
let reader = match EcstoreDiskAPI::read_file(disk.as_ref(), RUSTFS_META_BUCKET, path).await {
|
||||
Ok(reader) => reader,
|
||||
Err(EcstoreDiskError::FileNotFound | EcstoreDiskError::VolumeNotFound) => return Ok(None),
|
||||
@@ -178,6 +194,11 @@ async fn read_bounded(disk: &EcstoreDiskStore, path: &str, limit: usize) -> Resu
|
||||
if bytes.len() > limit {
|
||||
return Err(SnapshotError::TooLarge);
|
||||
}
|
||||
if let Some(stats) = stats.as_mut() {
|
||||
stats.file_reads += 1;
|
||||
stats.bytes_read = stats.bytes_read.checked_add(bytes.len()).ok_or(SnapshotError::TooLarge)?;
|
||||
stats.peak_file_bytes = stats.peak_file_bytes.max(bytes.len());
|
||||
}
|
||||
Ok(Some(bytes))
|
||||
}
|
||||
|
||||
@@ -197,17 +218,26 @@ fn select_snapshot(selected: &mut Option<CommittedSnapshot>, candidate: Committe
|
||||
}
|
||||
|
||||
async fn read_committed(disks: &[EcstoreDiskStore], limit: usize) -> Result<Option<CommittedSnapshot>, SnapshotError> {
|
||||
read_committed_with_stats(disks, limit, None).await
|
||||
}
|
||||
|
||||
async fn read_committed_with_stats(
|
||||
disks: &[EcstoreDiskStore],
|
||||
limit: usize,
|
||||
mut stats: Option<&mut SnapshotReadStats>,
|
||||
) -> Result<Option<CommittedSnapshot>, SnapshotError> {
|
||||
let mut selected = None;
|
||||
let mut damaged = None;
|
||||
let mut identities = HashMap::new();
|
||||
for disk in disks {
|
||||
for (manifest_path, payload_path) in MANIFEST_PATHS.into_iter().zip(PAYLOAD_PATHS) {
|
||||
let candidate = async {
|
||||
let Some(manifest) = read_bounded(disk, manifest_path, MANIFEST_LEN).await? else {
|
||||
let Some(manifest) = read_bounded_with_stats(disk, manifest_path, MANIFEST_LEN, stats.as_deref_mut()).await?
|
||||
else {
|
||||
return Ok(None);
|
||||
};
|
||||
let header = Manifest::decode(&manifest, limit)?;
|
||||
let payload = read_bounded(disk, payload_path, header.payload_len)
|
||||
let payload = read_bounded_with_stats(disk, payload_path, header.payload_len, stats.as_deref_mut())
|
||||
.await?
|
||||
.ok_or(SnapshotError::Corrupt)?;
|
||||
CommittedSnapshot::decode(&manifest, payload, limit).map(Some)
|
||||
@@ -492,6 +522,69 @@ mod tests {
|
||||
assert_eq!(recovered.manifest.sequence, 1);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn committed_reader_reopens_previous_anchor_across_publication_boundaries() {
|
||||
let owner = Uuid::new_v4();
|
||||
let old = payload("old");
|
||||
let next = payload("next");
|
||||
let boundaries = [
|
||||
("payload-only", next.clone(), None),
|
||||
("torn-manifest", next.clone(), Some(manifest(owner, 2, &next)[..20].to_vec())),
|
||||
("stale-payload", old.clone(), Some(manifest(owner, 2, &next))),
|
||||
];
|
||||
for (case, successor_payload, successor_manifest) in boundaries {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let store = disk(&root, "disk").await;
|
||||
commit(&store, 0, owner, 1, &old).await;
|
||||
install(&store, PAYLOAD_PATHS[1], &successor_payload).await;
|
||||
if let Some(manifest) = &successor_manifest {
|
||||
install(&store, MANIFEST_PATHS[1], manifest).await;
|
||||
}
|
||||
|
||||
let reopened = disk(&root, "disk").await;
|
||||
let recovered = read_committed(std::slice::from_ref(&reopened), 4096)
|
||||
.await
|
||||
.unwrap_or_else(|error| panic!("{case}: old anchor must remain readable after reopen: {error:?}"))
|
||||
.unwrap_or_else(|| panic!("{case}: previous committed anchor missing after reopen"));
|
||||
assert_eq!(recovered.manifest.sequence, 1, "{case}: successor must not become authoritative");
|
||||
assert_eq!(recovered.payload, old, "{case}: previous payload must survive");
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(reopened.as_ref(), RUSTFS_META_BUCKET, PAYLOAD_PATHS[0])
|
||||
.await
|
||||
.expect("old payload retained")
|
||||
.as_ref(),
|
||||
old.as_slice(),
|
||||
"{case}: previous payload bytes changed"
|
||||
);
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(reopened.as_ref(), RUSTFS_META_BUCKET, MANIFEST_PATHS[0])
|
||||
.await
|
||||
.expect("old manifest retained")
|
||||
.as_ref(),
|
||||
manifest(owner, 1, &old).as_slice(),
|
||||
"{case}: previous manifest bytes changed"
|
||||
);
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(reopened.as_ref(), RUSTFS_META_BUCKET, PAYLOAD_PATHS[1])
|
||||
.await
|
||||
.expect("successor payload retained")
|
||||
.as_ref(),
|
||||
successor_payload.as_slice(),
|
||||
"{case}: successor evidence changed"
|
||||
);
|
||||
if let Some(manifest) = &successor_manifest {
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(reopened.as_ref(), RUSTFS_META_BUCKET, MANIFEST_PATHS[1])
|
||||
.await
|
||||
.expect("successor manifest retained")
|
||||
.as_ref(),
|
||||
manifest.as_slice(),
|
||||
"{case}: successor manifest evidence changed"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn stale_manifest_cas_cannot_replace_committed_anchor() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
@@ -571,6 +664,35 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn committed_reader_resource_bounds_are_measured() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let first = disk(&root, "first").await;
|
||||
let second = disk(&root, "second").await;
|
||||
let owner = Uuid::new_v4();
|
||||
let old = [payload("old-0"), payload("old-1")].concat();
|
||||
let new = [payload("new-0"), payload("new-1"), payload("new-2")].concat();
|
||||
commit(&first, 0, owner, 1, &old).await;
|
||||
commit(&second, 1, owner, 2, &new).await;
|
||||
|
||||
let mut stats = SnapshotReadStats::default();
|
||||
let recovered = read_committed_with_stats(&[first, second], 4096, Some(&mut stats))
|
||||
.await
|
||||
.expect("read committed replicas")
|
||||
.expect("committed snapshot");
|
||||
|
||||
assert_eq!(recovered.sequence(), 2);
|
||||
assert_eq!(recovered.payload(), new.as_slice());
|
||||
assert_eq!(recovered.manifest.payload_len, new.len());
|
||||
assert_eq!(stats.file_reads, 4, "only committed manifests and their payloads are materialized");
|
||||
assert_eq!(stats.bytes_read, (MANIFEST_LEN * 2) + old.len() + new.len());
|
||||
assert_eq!(
|
||||
stats.peak_file_bytes,
|
||||
new.len().max(MANIFEST_LEN),
|
||||
"reader peak allocation remains bounded by one manifest or payload file"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn legacy_inspection_rejects_complete_subsets_and_scope_ambiguity() {
|
||||
let scoped = |set_index| {
|
||||
|
||||
@@ -91,6 +91,29 @@ pub struct HealObjectOutcome {
|
||||
pub detail: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct HealObjectReceipt {
|
||||
pub identity: HealObjectIdentity,
|
||||
pub disposition: HealObjectDisposition,
|
||||
}
|
||||
|
||||
impl HealObjectReceipt {
|
||||
pub(crate) fn verified_for(&self, expected: &HealObjectIdentity) -> bool {
|
||||
matches!(
|
||||
self.disposition,
|
||||
HealObjectDisposition::Repaired
|
||||
| HealObjectDisposition::VerifiedHealthy
|
||||
| HealObjectDisposition::AuthoritativelyAbsent
|
||||
) && self.identity.kind == expected.kind
|
||||
&& self.identity.bucket == expected.bucket
|
||||
&& self.identity.object == expected.object
|
||||
&& self.identity.version_id == expected.version_id
|
||||
&& self.identity.pool_index == expected.pool_index
|
||||
&& self.identity.set_index == expected.set_index
|
||||
&& self.identity.bucket_incarnation_id.is_some()
|
||||
}
|
||||
}
|
||||
|
||||
impl HealObjectOutcome {
|
||||
fn retained_bytes(&self) -> usize {
|
||||
size_of::<Self>()
|
||||
@@ -470,4 +493,48 @@ mod canonical_outcome_tests {
|
||||
assert_eq!(outcome.counters.processed, u64::MAX);
|
||||
assert_eq!(outcome.coverage, HealTraversalCoverage::Partial);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn positive_receipt_requires_exact_identity_and_bucket_incarnation() {
|
||||
let expected = item(HealObjectDisposition::Unknown).identity;
|
||||
let mut receipt = HealObjectReceipt {
|
||||
identity: expected.clone(),
|
||||
disposition: HealObjectDisposition::Repaired,
|
||||
};
|
||||
|
||||
assert!(
|
||||
!receipt.verified_for(&expected),
|
||||
"a positive storage receipt without bucket incarnation must remain untrusted"
|
||||
);
|
||||
|
||||
let incarnation = Uuid::new_v4();
|
||||
receipt.identity.bucket_incarnation_id = Some(incarnation);
|
||||
assert!(receipt.verified_for(&expected));
|
||||
|
||||
receipt.identity.version_id = Some("older-version".to_string());
|
||||
assert!(
|
||||
!receipt.verified_for(&expected),
|
||||
"a storage receipt for a different object/version tuple must not clear the requested responsibility"
|
||||
);
|
||||
|
||||
receipt.identity = HealObjectIdentity {
|
||||
bucket_incarnation_id: Some(incarnation),
|
||||
pool_index: Some(1),
|
||||
..expected.clone()
|
||||
};
|
||||
assert!(
|
||||
!receipt.verified_for(&expected),
|
||||
"a storage receipt for a different erasure location must not clear the requested responsibility"
|
||||
);
|
||||
|
||||
receipt.identity = HealObjectIdentity {
|
||||
bucket_incarnation_id: Some(incarnation),
|
||||
..expected
|
||||
};
|
||||
receipt.disposition = HealObjectDisposition::Unknown;
|
||||
assert!(
|
||||
!receipt.verified_for(&receipt.identity),
|
||||
"legacy success without a positive disposition remains unknown"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -18,6 +18,25 @@ use super::{
|
||||
DiskOption, DiskStore, Endpoint, HealDiskExt as _, local_disk_map_read, new_disk, resume::ReplacementTargetIdentity,
|
||||
};
|
||||
|
||||
/// Whether automatic replacement may fall back to the set-wide format heal when
|
||||
/// a target cannot pass the independent-mount admission.
|
||||
///
|
||||
/// Directory-backed deployments already declare, through
|
||||
/// `RUSTFS_UNSAFE_BYPASS_DISK_CHECK`, that their endpoints are plain
|
||||
/// directories sharing a device with the host root. Those endpoints can never
|
||||
/// satisfy [`auto_replacement_target_identity`], so without this fallback a
|
||||
/// runtime-wiped or replaced directory disk would stay deferred forever. The
|
||||
/// admission check itself is never bypassed; the fallback only routes the heal
|
||||
/// through the ordinary format path that formats every unformatted disk in the
|
||||
/// set, which is exactly what the pre-admission `heal_disk` path did.
|
||||
pub(crate) fn directory_backed_replacement_fallback_enabled() -> bool {
|
||||
rustfs_utils::get_env_bool_with_aliases(
|
||||
rustfs_config::ENV_UNSAFE_BYPASS_DISK_CHECK,
|
||||
&[rustfs_config::ENV_MINIO_CI],
|
||||
rustfs_config::DEFAULT_UNSAFE_BYPASS_DISK_CHECK,
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) async fn auto_replacement_target_ready(disk: &DiskStore, local_disks: &[DiskStore]) -> bool {
|
||||
auto_replacement_target_identity(disk, local_disks).await.is_some()
|
||||
}
|
||||
@@ -184,12 +203,47 @@ mod tests {
|
||||
assert!(endpoint.is_local);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn directory_backed_fallback_is_off_by_default() {
|
||||
temp_env::with_vars(
|
||||
[
|
||||
(rustfs_config::ENV_UNSAFE_BYPASS_DISK_CHECK, None::<&str>),
|
||||
(rustfs_config::ENV_MINIO_CI, None::<&str>),
|
||||
],
|
||||
|| assert!(!directory_backed_replacement_fallback_enabled()),
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn directory_backed_fallback_follows_the_disk_check_bypass() {
|
||||
temp_env::with_vars(
|
||||
[
|
||||
(rustfs_config::ENV_UNSAFE_BYPASS_DISK_CHECK, Some("true")),
|
||||
(rustfs_config::ENV_MINIO_CI, None::<&str>),
|
||||
],
|
||||
|| assert!(directory_backed_replacement_fallback_enabled()),
|
||||
);
|
||||
temp_env::with_vars(
|
||||
[
|
||||
(rustfs_config::ENV_UNSAFE_BYPASS_DISK_CHECK, Some("false")),
|
||||
(rustfs_config::ENV_MINIO_CI, Some("true")),
|
||||
],
|
||||
|| {
|
||||
assert!(
|
||||
!directory_backed_replacement_fallback_enabled(),
|
||||
"the canonical key must win over the alias"
|
||||
)
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn runtime_environment_cannot_bypass_mount_admission() {
|
||||
temp_env::async_with_vars(
|
||||
[
|
||||
("RUSTFS_TEST_AUTO_REPLACEMENT_READINESS_BYPASS", Some("1")),
|
||||
("RUSTFS_E2E_AUTO_REPLACEMENT_READINESS_BYPASS", Some("1")),
|
||||
(rustfs_config::ENV_UNSAFE_BYPASS_DISK_CHECK, Some("true")),
|
||||
],
|
||||
async {
|
||||
let temp = TempDir::new().expect("temporary replacement root should be created");
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user