Compare commits

..

1 Commits

Author SHA1 Message Date
overtrue 195f19217f fix(ecstore): collapse ILM expiry worker env knobs to canonical name 2026-08-13 03:04:09 +08:00
256 changed files with 7100 additions and 24341 deletions
+1 -1
View File
@@ -85,7 +85,7 @@ runs:
repo-token: ${{ github.token }} repo-token: ${{ github.token }}
- name: Install flatc - name: Install flatc
uses: Nugine/setup-flatc@698800de72a96bfb22cf60431dc21a2ff9a7e07b # v1 uses: Nugine/setup-flatc@e7855e994773ce90094a3f1626d4afc9080c23ae # v1
with: with:
version: "25.12.19" version: "25.12.19"
+1 -6
View File
@@ -182,12 +182,7 @@ jobs:
echo '```' echo '```'
} >> "$GITHUB_STEP_SUMMARY" } >> "$GITHUB_STEP_SUMMARY"
# Readers: test-and-lint-rio-v2 (per-PR), build-rustfs-debug-binary-rio-v2 # Readers: test-and-lint-rio-v2, build-rustfs-debug-binary-rio-v2.
# (weekly schedule / manual dispatch only — dormant rio-v2 variant, see
# rustfs/backlog#1835 and docs/architecture/minio-file-format-compat.md).
# The second build below stays despite the reduced cadence: it warms the
# rio-v2,e2e-test-hooks feature resolution the scheduled build restores,
# which keeps that lane inside its 30-minute timeout.
warm-ci-feat-rio: warm-ci-feat-rio:
name: Warm ci-feat-rio name: Warm ci-feat-rio
runs-on: sm-standard-4 runs-on: sm-standard-4
+1 -9
View File
@@ -533,12 +533,7 @@ jobs:
build-rustfs-debug-binary-rio-v2: build-rustfs-debug-binary-rio-v2:
name: Build RustFS Debug Binary (rio-v2) name: Build RustFS Debug Binary (rio-v2)
# Dormant rio-v2 variant (rustfs/backlog#1835): the feature ships in no if: github.event_name != 'pull_request' || github.event.action != 'closed'
# default build, so this full-suite lane runs only on the weekly schedule
# and manual dispatch. Per-PR cfg-seam coverage stays with
# test-and-lint-rio-v2. Lifecycle and the promote-or-delete condition:
# docs/architecture/minio-file-format-compat.md ("rio-v2 variant lifecycle").
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
needs: [ quick-checks ] needs: [ quick-checks ]
runs-on: sm-standard-4 runs-on: sm-standard-4
timeout-minutes: 30 timeout-minutes: 30
@@ -829,9 +824,6 @@ jobs:
e2e-tests-rio-v2: e2e-tests-rio-v2:
name: End-to-End Tests (rio-v2) name: End-to-End Tests (rio-v2)
# Inherits the schedule/dispatch-only gate through needs: on every other
# event build-rustfs-debug-binary-rio-v2 is skipped, so this job skips
# with it (see the dormant-variant comment on that job).
needs: [ build-rustfs-debug-binary-rio-v2 ] needs: [ build-rustfs-debug-binary-rio-v2 ]
runs-on: sm-standard-2 runs-on: sm-standard-2
timeout-minutes: 30 timeout-minutes: 30
+4 -7
View File
@@ -101,10 +101,7 @@ refactors.
The `rustfs` binary crate composes these libraries into the running server. The `rustfs` binary crate composes these libraries into the running server.
`ecstore` remains the storage engine at the architectural center; its internal `ecstore` remains the storage engine at the architectural center; its internal
module split is tracked under `docs/architecture/`. `rio-v2` is the module split is tracked under `docs/architecture/`.
feature-gated MinIO on-disk format compatibility I/O layer; it ships in no
default build (lifecycle:
[docs/architecture/minio-file-format-compat.md](docs/architecture/minio-file-format-compat.md)).
## Architecture Invariants ## Architecture Invariants
@@ -134,9 +131,9 @@ default build (lifecycle:
why it stays local). why it stays local).
- ✅ RESOLVED: `BackpressureConfig` and `DataUsageInfo` each have exactly one - ✅ RESOLVED: `BackpressureConfig` and `DataUsageInfo` each have exactly one
definition (`crates/io-core/src/backpressure.rs`, definition (`crates/io-core/src/backpressure.rs`,
`crates/data-usage/src/data_usage.rs`). The zero-consumer `crates/data-usage/src/data_usage.rs`). A zero-consumer
`BackpressureSettings` copy that lingered in io-metrics was removed `BackpressureSettings` copy lingers in `crates/io-metrics/src/config.rs`;
(rustfs/backlog#1833). its removal is tracked in rustfs/backlog#1833.
4. **ecstore does not know about HTTP or S3 protocol details.** It operates on 4. **ecstore does not know about HTTP or S3 protocol details.** It operates on
storage-level abstractions (objects, buckets, disks, pools). storage-level abstractions (objects, buckets, disks, pools).
Generated
+31 -39
View File
@@ -1162,9 +1162,9 @@ dependencies = [
[[package]] [[package]]
name = "aws-smithy-eventstream" name = "aws-smithy-eventstream"
version = "0.61.2" version = "0.61.1"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6de526c7b567420a31bc283657a7921b45c4cafe0827fdf2490713dcc770c28f" checksum = "5a9381123ab62d20c13082b151f30f962a3b112b727345394536dfa39a482944"
dependencies = [ dependencies = [
"aws-smithy-types", "aws-smithy-types",
"bytes", "bytes",
@@ -1195,9 +1195,9 @@ dependencies = [
[[package]] [[package]]
name = "aws-smithy-http-client" name = "aws-smithy-http-client"
version = "1.3.0" version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3c1c8a04cb31ba74d0115af5a890bb8c0d48fba64b52812fa13929a6ef0cc83c" checksum = "635d23afda0a6ab48d666c4d447c4873e8d1e83518a2be2093122397e50b838e"
dependencies = [ dependencies = [
"aws-smithy-async", "aws-smithy-async",
"aws-smithy-protocol-test", "aws-smithy-protocol-test",
@@ -1277,9 +1277,9 @@ dependencies = [
[[package]] [[package]]
name = "aws-smithy-runtime" name = "aws-smithy-runtime"
version = "1.13.1" version = "1.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "483b858ff67522011c4786310c5cd8fd88d0be7ea3d5f1a48328446300c4269e" checksum = "07505b34e8f4b3591a4fa69e9792b52289b95488dbbc68c3c0075b7bedb245e1"
dependencies = [ dependencies = [
"aws-smithy-async", "aws-smithy-async",
"aws-smithy-http", "aws-smithy-http",
@@ -1343,9 +1343,9 @@ dependencies = [
[[package]] [[package]]
name = "aws-smithy-types" name = "aws-smithy-types"
version = "1.6.2" version = "1.6.1"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "fce83ce9abbb198d25bc7131e468d0f9fe1257125e58c39f3f9fc9f5098c9647" checksum = "d6dc683efb34b9e755675b37fedbe0103141e5b6df7bdc9eb6967756a8c167d8"
dependencies = [ dependencies = [
"base64-simd", "base64-simd",
"bytes", "bytes",
@@ -5133,9 +5133,9 @@ dependencies = [
[[package]] [[package]]
name = "http-body-util" name = "http-body-util"
version = "0.1.5" version = "0.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "23169fe34a5fbcdd3f3862e78fb9b6fccd5f02a6dc6f732547005d45631ce71c" checksum = "e9f41fd6a08e4d4ec69df65976da761afd5ad5e58a9d4acb46bd1c953a9e3ff2"
dependencies = [ dependencies = [
"bytes", "bytes",
"futures-core", "futures-core",
@@ -5957,7 +5957,7 @@ checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981"
[[package]] [[package]]
name = "libmimalloc-sys" name = "libmimalloc-sys"
version = "0.1.49" version = "0.1.49"
source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11#6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=ce6338661179c8be22e516b00af7483f151485a7#ce6338661179c8be22e516b00af7483f151485a7"
dependencies = [ dependencies = [
"cc", "cc",
"cty", "cty",
@@ -6259,9 +6259,9 @@ dependencies = [
[[package]] [[package]]
name = "metrique" name = "metrique"
version = "0.1.30" version = "0.1.29"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dedbf06ffeef4c37990c73636fbd993aa34fb1948afd736e6114f239220993db" checksum = "d2e394c63e2d1a30aeb3b9392ecf3439d8475d2df810a8f4f6e66d6866754017"
dependencies = [ dependencies = [
"itoa", "itoa",
"jiff", "jiff",
@@ -6289,9 +6289,9 @@ dependencies = [
[[package]] [[package]]
name = "metrique-macro" name = "metrique-macro"
version = "0.1.21" version = "0.1.20"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f4fb1f30185f53f7f6e4c9e46745c1a1350af8e77fda5a88aded44b0637a82e0" checksum = "786df1fd0abebd0db685f7e9a353c78756d4b370fb98a52376c2015fa55f141f"
dependencies = [ dependencies = [
"Inflector", "Inflector",
"darling 0.23.0", "darling 0.23.0",
@@ -6318,9 +6318,9 @@ checksum = "2faca4e4480069ff02b1763b3b79f5cec7e8628e24d9dc5b6073f53d2577a4d9"
[[package]] [[package]]
name = "metrique-writer" name = "metrique-writer"
version = "0.1.26" version = "0.1.25"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "20bd17c1a3ca2719e31f19ce77a853948dc2102f35976b92276c42a64fdc5f3f" checksum = "82cdde44d241dab7fc8b7a32e0eb5dae6cd28f8de80b59f9a1e9f2f0b05e485e"
dependencies = [ dependencies = [
"ahash", "ahash",
"crossbeam-queue", "crossbeam-queue",
@@ -6339,9 +6339,9 @@ dependencies = [
[[package]] [[package]]
name = "metrique-writer-core" name = "metrique-writer-core"
version = "0.1.20" version = "0.1.19"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f1a55b6aae1d85c557c729564c4e2b32a26dc65ba2d90d9647ca01f2bd4854c4" checksum = "e57379b7ee2272efaeaaa6de062503563e57333b24aadc7f2255b3d602899e8b"
dependencies = [ dependencies = [
"derive-where", "derive-where",
"itertools 0.14.0", "itertools 0.14.0",
@@ -6366,7 +6366,7 @@ dependencies = [
[[package]] [[package]]
name = "mimalloc" name = "mimalloc"
version = "0.1.52" version = "0.1.52"
source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11#6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=ce6338661179c8be22e516b00af7483f151485a7#ce6338661179c8be22e516b00af7483f151485a7"
dependencies = [ dependencies = [
"libmimalloc-sys", "libmimalloc-sys",
] ]
@@ -6882,7 +6882,7 @@ version = "5.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d" checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d"
dependencies = [ dependencies = [
"base64 0.21.7", "base64 0.22.1",
"chrono", "chrono",
"getrandom 0.2.17", "getrandom 0.2.17",
"http 1.5.0", "http 1.5.0",
@@ -8043,7 +8043,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf" checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
dependencies = [ dependencies = [
"heck 0.5.0", "heck 0.5.0",
"itertools 0.10.5", "itertools 0.14.0",
"log", "log",
"multimap", "multimap",
"once_cell", "once_cell",
@@ -8063,7 +8063,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042" checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042"
dependencies = [ dependencies = [
"heck 0.5.0", "heck 0.5.0",
"itertools 0.10.5", "itertools 0.14.0",
"log", "log",
"multimap", "multimap",
"petgraph 0.8.3", "petgraph 0.8.3",
@@ -8084,7 +8084,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d" checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d"
dependencies = [ dependencies = [
"anyhow", "anyhow",
"itertools 0.10.5", "itertools 0.14.0",
"proc-macro2", "proc-macro2",
"quote", "quote",
"syn 2.0.119", "syn 2.0.119",
@@ -8097,7 +8097,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf" checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf"
dependencies = [ dependencies = [
"anyhow", "anyhow",
"itertools 0.10.5", "itertools 0.14.0",
"proc-macro2", "proc-macro2",
"quote", "quote",
"syn 2.0.119", "syn 2.0.119",
@@ -9201,6 +9201,7 @@ dependencies = [
"sha2 0.11.0", "sha2 0.11.0",
"shadow-rs", "shadow-rs",
"socket2", "socket2",
"starshard",
"subtle", "subtle",
"sysinfo", "sysinfo",
"temp-env", "temp-env",
@@ -9620,7 +9621,9 @@ dependencies = [
"metrics", "metrics",
"metrics-util", "metrics-util",
"num_cpus", "num_cpus",
"rustfs-common",
"rustfs-s3-ops", "rustfs-s3-ops",
"rustfs-utils",
"sysinfo", "sysinfo",
"thiserror 2.0.20", "thiserror 2.0.20",
"tokio", "tokio",
@@ -9734,7 +9737,6 @@ dependencies = [
"rustfs-utils", "rustfs-utils",
"rustify", "rustify",
"serde", "serde",
"serde_ignored",
"serde_json", "serde_json",
"sha2 0.11.0", "sha2 0.11.0",
"subtle", "subtle",
@@ -10559,9 +10561,9 @@ dependencies = [
[[package]] [[package]]
name = "rustls-connector" name = "rustls-connector"
version = "0.23.8" version = "0.23.7"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1babecfcc65b139b812e74bcc7f9ec7b4e00db659fd42d99567b7e77f0c714c6" checksum = "09a5abe04eec18f8b9fbe87885bcaee6426de80bbc579958c0bc064b728ee617"
dependencies = [ dependencies = [
"futures-io", "futures-io",
"futures-rustls", "futures-rustls",
@@ -10937,16 +10939,6 @@ dependencies = [
"syn 3.0.3", "syn 3.0.3",
] ]
[[package]]
name = "serde_ignored"
version = "0.1.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "115dffd5f3853e06e746965a20dcbae6ee747ae30b543d91b0e089668bb07798"
dependencies = [
"serde",
"serde_core",
]
[[package]] [[package]]
name = "serde_json" name = "serde_json"
version = "1.0.151" version = "1.0.151"
@@ -11800,7 +11792,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
dependencies = [ dependencies = [
"fastrand", "fastrand",
"getrandom 0.3.4", "getrandom 0.4.3",
"once_cell", "once_cell",
"rustix", "rustix",
"windows-sys 0.61.2", "windows-sys 0.61.2",
+6 -7
View File
@@ -41,7 +41,7 @@ members = [
"crates/protocols", # Protocol implementations (FTPS, SFTP, etc.) "crates/protocols", # Protocol implementations (FTPS, SFTP, etc.)
"crates/protos", # Protocol buffer definitions "crates/protos", # Protocol buffer definitions
"crates/rio", # Rust I/O utilities and abstractions "crates/rio", # Rust I/O utilities and abstractions
"crates/rio-v2", # MinIO on-disk format compatibility I/O layer (feature-gated, ships in no default build) "crates/rio-v2", # Next-generation Rust I/O compatibility layer
"crates/replication", # Replication contracts and wire formats "crates/replication", # Replication contracts and wire formats
"crates/concurrency", # Concurrency management for RustFS - timeout, locking, backpressure, and I/O scheduling "crates/concurrency", # Concurrency management for RustFS - timeout, locking, backpressure, and I/O scheduling
"crates/s3-types", # S3 event type definitions "crates/s3-types", # S3 event type definitions
@@ -154,7 +154,7 @@ hyper-rustls = { default-features = false, version = "0.27.9" }
hyper-util = { version = "0.1.20" } hyper-util = { version = "0.1.20" }
http = "1.5.0" http = "1.5.0"
http-body = "1.1.0" http-body = "1.1.0"
http-body-util = "0.1.5" http-body-util = "0.1.4"
minlz = "1.2.3" minlz = "1.2.3"
reqwest = "0.13.4" reqwest = "0.13.4"
rustfs-kafka-async = { version = "1.2.0" } rustfs-kafka-async = { version = "1.2.0" }
@@ -182,7 +182,6 @@ quick-xml = "0.41.0"
rmp = { version = "0.8.15" } rmp = { version = "0.8.15" }
rmp-serde = { version = "1.3.1" } rmp-serde = { version = "1.3.1" }
serde = { version = "1.0.229" } serde = { version = "1.0.229" }
serde_ignored = { version = "0.1" }
serde_json = { version = "1.0.151" } serde_json = { version = "1.0.151" }
serde_urlencoded = "0.7.1" serde_urlencoded = "0.7.1"
@@ -231,9 +230,9 @@ aws-credential-types = { version = "1.3.0" }
aws-sdk-kms = { default-features = false, version = "1.114.0" } aws-sdk-kms = { default-features = false, version = "1.114.0" }
aws-sdk-s3 = { default-features = false, version = "1.141.0" } aws-sdk-s3 = { default-features = false, version = "1.141.0" }
aws-sdk-sts = { default-features = false, version = "1.110.0" } aws-sdk-sts = { default-features = false, version = "1.110.0" }
aws-smithy-http-client = { default-features = false, version = "1.3.0" } aws-smithy-http-client = { default-features = false, version = "1.2.0" }
aws-smithy-runtime-api = { version = "1.14.0" } aws-smithy-runtime-api = { version = "1.14.0" }
aws-smithy-types = { version = "1.6.2" } aws-smithy-types = { version = "1.6.1" }
base64 = "0.23.1" base64 = "0.23.1"
base64-simd = "0.8.0" base64-simd = "0.8.0"
brotli = "8.0.4" brotli = "8.0.4"
@@ -348,8 +347,8 @@ russh-sftp = "2.4.0"
dav-server = "0.11.0" dav-server = "0.11.0"
# Performance Analysis and Memory Profiling # Performance Analysis and Memory Profiling
mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" } mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "ce6338661179c8be22e516b00af7483f151485a7" }
libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11", features = ["extended"] } libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "ce6338661179c8be22e516b00af7483f151485a7", features = ["extended"] }
hotpath = { version = "0.23.2", default-features = false } hotpath = { version = "0.23.2", default-features = false }
# Snapshot testing for output format regression detection # Snapshot testing for output format regression detection
insta = { version = "1.48" } insta = { version = "1.48" }
-7
View File
@@ -21,13 +21,6 @@ use crate::{
Xxhash3, Xxhash64, Xxhash128, Xxhash3, Xxhash64, Xxhash128,
}; };
// DELIBERATE DUPLICATION of the x-amz-checksum-* names that also exist as
// AMZ_CHECKSUM_* in rustfs-utils' headers module (crates/utils/src/http/
// headers.rs): this crate is a zero-internal-dependency leaf, so it cannot
// import them, and it additionally owns the RustFS extension names
// (sha512/xxhash*) that utils does not carry. Values are pinned by the S3
// wire protocol; do not merge without a maintainer decision on the leaf
// boundary (backlog#1833).
pub const CRC_32_HEADER_NAME: &str = "x-amz-checksum-crc32"; pub const CRC_32_HEADER_NAME: &str = "x-amz-checksum-crc32";
pub const CRC_32_C_HEADER_NAME: &str = "x-amz-checksum-crc32c"; pub const CRC_32_C_HEADER_NAME: &str = "x-amz-checksum-crc32c";
pub const SHA_1_HEADER_NAME: &str = "x-amz-checksum-sha1"; pub const SHA_1_HEADER_NAME: &str = "x-amz-checksum-sha1";
-8
View File
@@ -41,14 +41,6 @@ pub const XXHASH_64_NAME: &str = "xxhash64";
pub const XXHASH_128_NAME: &str = "xxhash128"; pub const XXHASH_128_NAME: &str = "xxhash128";
pub const MD5_NAME: &str = "md5"; pub const MD5_NAME: &str = "md5";
/// One of three deliberately separate checksum registries (backlog#1833):
/// this enum owns the **streaming-hash algorithm registry**, including the
/// RustFS extensions (sha512, xxhash3/64/128). The on-disk xl.meta bitset
/// lives in `rustfs_rio::ChecksumType` (crates/rio/src/checksum.rs, varint
/// bits are append-only), and the MinIO-port client keeps its own
/// `ChecksumMode` (crates/ecstore/src/client/checksum.rs). When adding an
/// algorithm, extend all three (or record why not) — they do not derive from
/// each other.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] #[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
#[non_exhaustive] #[non_exhaustive]
pub enum ChecksumAlgorithm { pub enum ChecksumAlgorithm {
+87
View File
@@ -0,0 +1,87 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use crate::last_minute::{self};
use std::collections::HashMap;
pub struct ReplicationLatency {
// Delays for single and multipart PUT requests
upload_histogram: last_minute::LastMinuteHistogram,
}
impl ReplicationLatency {
// Merge two ReplicationLatency
pub fn merge(&mut self, other: &mut ReplicationLatency) -> &ReplicationLatency {
self.upload_histogram.merge(&other.upload_histogram);
self
}
// Get upload delay (categorized by object size interval)
pub fn get_upload_latency(&mut self) -> HashMap<String, u64> {
let mut ret = HashMap::new();
let avg = self.upload_histogram.get_avg_data();
for (i, v) in avg.iter().enumerate() {
let avg_duration = v.avg();
ret.insert(self.size_tag_to_string(i), avg_duration.as_millis() as u64);
}
ret
}
pub fn update(&mut self, size: i64, during: std::time::Duration) {
self.upload_histogram.add(size, during);
}
// Simulate the conversion from size tag to string
fn size_tag_to_string(&self, tag: usize) -> String {
match tag {
0 => String::from("Size < 1 KiB"),
1 => String::from("Size < 1 MiB"),
2 => String::from("Size < 10 MiB"),
3 => String::from("Size < 100 MiB"),
4 => String::from("Size < 1 GiB"),
_ => String::from("Size > 1 GiB"),
}
}
}
// #[derive(Debug, Clone, Default)]
// pub struct ReplicationLastMinute {
// pub last_minute: LastMinuteLatency,
// }
// impl ReplicationLastMinute {
// pub fn merge(&mut self, other: ReplicationLastMinute) -> ReplicationLastMinute {
// let mut nl = ReplicationLastMinute::default();
// nl.last_minute = self.last_minute.merge(&mut other.last_minute);
// nl
// }
// pub fn add_size(&mut self, n: i64) {
// let t = SystemTime::now()
// .duration_since(UNIX_EPOCH)
// .expect("Time went backwards")
// .as_secs();
// self.last_minute.add_all(t - 1, &AccElem { total: t - 1, size: n as u64, n: 1 });
// }
// pub fn get_total(&self) -> AccElem {
// self.last_minute.get_total()
// }
// }
// impl fmt::Display for ReplicationLastMinute {
// fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
// let t = self.last_minute.get_total();
// write!(f, "ReplicationLastMinute sz= {}, n= {}, dur= {}", t.size, t.n, t.total)
// }
// }
+41
View File
@@ -572,3 +572,44 @@ mod tests {
assert_eq!(total.n, 6); assert_eq!(total.n, 6);
} }
} }
const SIZE_LAST_ELEM_MARKER: usize = 10; // Assumed marker size is 10, modify according to actual situation
#[allow(dead_code)]
#[derive(Debug, Default)]
pub struct LastMinuteHistogram {
histogram: Vec<LastMinuteLatency>,
size: u32,
}
impl LastMinuteHistogram {
pub fn merge(&mut self, other: &LastMinuteHistogram) {
for i in 0..self.histogram.len() {
self.histogram[i].merge(&other.histogram[i]);
}
}
pub fn add(&mut self, size: i64, t: Duration) {
let index = size_to_tag(size);
self.histogram[index].add(&t);
}
pub fn get_avg_data(&mut self) -> [AccElem; SIZE_LAST_ELEM_MARKER] {
let mut res = [AccElem::default(); SIZE_LAST_ELEM_MARKER];
for (i, elem) in self.histogram.iter_mut().enumerate() {
res[i] = elem.get_total();
}
res
}
}
fn size_to_tag(size: i64) -> usize {
match size {
_ if size < 1024 => 0, // sizeLessThan1KiB
_ if size < 1024 * 1024 => 1, // sizeLessThan1MiB
_ if size < 10 * 1024 * 1024 => 2, // sizeLessThan10MiB
_ if size < 100 * 1024 * 1024 => 3, // sizeLessThan100MiB
_ if size < 1024 * 1024 * 1024 => 4, // sizeLessThan1GiB
_ => 5, // sizeGreaterThan1GiB
}
}
+1
View File
@@ -12,6 +12,7 @@
// See the License for the specific language governing permissions and // See the License for the specific language governing permissions and
// limitations under the License. // limitations under the License.
pub mod bucket_stats;
// pub mod error; // pub mod error;
pub mod globals; pub mod globals;
pub mod heal_channel; pub mod heal_channel;
-5
View File
@@ -353,11 +353,6 @@ pub const DEFAULT_OBS_TRACES_EXPORT_ENABLED: bool = true;
/// Environment variable: RUSTFS_OBS_METRICS_EXPORT_ENABLED /// Environment variable: RUSTFS_OBS_METRICS_EXPORT_ENABLED
pub const DEFAULT_OBS_METRICS_EXPORT_ENABLED: bool = true; pub const DEFAULT_OBS_METRICS_EXPORT_ENABLED: bool = true;
/// Default detailed PUT stage metrics enabled
/// Default value: false
/// Environment variable: RUSTFS_OBS_PUT_STAGE_METRICS_ENABLED
pub const DEFAULT_OBS_PUT_STAGE_METRICS_ENABLED: bool = false;
/// Default logs export enabled /// Default logs export enabled
/// It is used to enable or disable exporting logs /// It is used to enable or disable exporting logs
/// Default value: true /// Default value: true
-49
View File
@@ -137,37 +137,6 @@ pub const DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED: bool = false;
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE); const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE);
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED); const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED);
/// Request the object-transaction fencing contract used by storage-owned
/// cleanup receipts and lock-window optimizations.
///
/// This is fail-closed: enabling the writer without a live fleet proof rejects
/// the commit rather than silently using a legacy-safe path.
pub const ENV_OBJECT_TRANSACTION_FENCING_WRITE: &str = "RUSTFS_OBJECT_TRANSACTION_FENCING_WRITE";
pub const DEFAULT_OBJECT_TRANSACTION_FENCING_WRITE: bool = false;
/// Operator-attested confirmation that every serving node understands the
/// object transaction fencing contract.
pub const ENV_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED: &str = "RUSTFS_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED";
pub const DEFAULT_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED: bool = false;
const _: () = assert!(!DEFAULT_OBJECT_TRANSACTION_FENCING_WRITE);
const _: () = assert!(!DEFAULT_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED);
/// Request preserving legacy per-part checksum metadata during data movement.
///
/// This remains ineffective until
/// [`ENV_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED`] is also enabled.
pub const ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE: &str = "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE";
pub const DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_WRITE: bool = false;
/// Operator-attested confirmation that every serving node understands the
/// data-movement per-part checksum sidecar.
pub const ENV_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED: &str = "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED";
pub const DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED: bool = false;
const _: () = assert!(!DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_WRITE);
const _: () = assert!(!DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED);
// ============================================================================= // =============================================================================
// Concurrent Request Fix - Timeout and Backpressure Configuration // Concurrent Request Fix - Timeout and Backpressure Configuration
// ============================================================================= // =============================================================================
@@ -680,22 +649,4 @@ mod remote_version_state_tests {
"RUSTFS_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED" "RUSTFS_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED"
); );
} }
#[test]
fn data_movement_part_checksum_gate_uses_stable_environment_names() {
assert_eq!(super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE, "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE");
assert_eq!(
super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED,
"RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED"
);
}
#[test]
fn object_transaction_fencing_gate_uses_stable_environment_names() {
assert_eq!(super::ENV_OBJECT_TRANSACTION_FENCING_WRITE, "RUSTFS_OBJECT_TRANSACTION_FENCING_WRITE");
assert_eq!(
super::ENV_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED,
"RUSTFS_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED"
);
}
} }
+1 -2
View File
@@ -81,8 +81,7 @@ pub const ENV_TEST_IAM_FAIL_INIT_ATTEMPTS: &str = "RUSTFS_TEST_IAM_FAIL_INIT_ATT
pub const ENV_TEST_IAM_RETRY_INTERVAL_MS: &str = "RUSTFS_TEST_IAM_RETRY_INTERVAL_MS"; pub const ENV_TEST_IAM_RETRY_INTERVAL_MS: &str = "RUSTFS_TEST_IAM_RETRY_INTERVAL_MS";
/// Runtime env var controlling the transition worker count. /// Runtime env var controlling the transition worker count.
pub const ENV_TRANSITION_WORKERS: &str = "RUSTFS_MAX_TRANSITION_WORKERS"; pub const ENV_TRANSITION_WORKERS: &str = "RUSTFS_MAX_TRANSITION_WORKERS";
/// Runtime env var controlling the ILM expiry worker count. A set, parsable, /// Runtime env var controlling the expiry worker count.
/// non-zero value wins; anything else falls back to `min(cpus, 16)`.
pub const ENV_MAX_EXPIRY_WORKERS: &str = "RUSTFS_MAX_EXPIRY_WORKERS"; pub const ENV_MAX_EXPIRY_WORKERS: &str = "RUSTFS_MAX_EXPIRY_WORKERS";
/// Runtime env var controlling the absolute maximum transition workers. /// Runtime env var controlling the absolute maximum transition workers.
pub const ENV_TRANSITION_WORKERS_ABSOLUTE_MAX: &str = "RUSTFS_ABSOLUTE_MAX_WORKERS"; pub const ENV_TRANSITION_WORKERS_ABSOLUTE_MAX: &str = "RUSTFS_ABSOLUTE_MAX_WORKERS";
-5
View File
@@ -44,10 +44,6 @@ pub const ENV_OBS_METRICS_EXPORT_ENABLED: &str = "RUSTFS_OBS_METRICS_EXPORT_ENAB
pub const ENV_OBS_LOGS_EXPORT_ENABLED: &str = "RUSTFS_OBS_LOGS_EXPORT_ENABLED"; pub const ENV_OBS_LOGS_EXPORT_ENABLED: &str = "RUSTFS_OBS_LOGS_EXPORT_ENABLED";
pub const ENV_OBS_PROFILING_EXPORT_ENABLED: &str = "RUSTFS_OBS_PROFILING_EXPORT_ENABLED"; pub const ENV_OBS_PROFILING_EXPORT_ENABLED: &str = "RUSTFS_OBS_PROFILING_EXPORT_ENABLED";
/// Enables detailed per-stage PUT metrics. Disabled by default because each
/// PUT records multiple timers and histograms when attribution is active.
pub const ENV_OBS_PUT_STAGE_METRICS_ENABLED: &str = "RUSTFS_OBS_PUT_STAGE_METRICS_ENABLED";
pub const ENV_OBS_LOGGER_LEVEL: &str = "RUSTFS_OBS_LOGGER_LEVEL"; pub const ENV_OBS_LOGGER_LEVEL: &str = "RUSTFS_OBS_LOGGER_LEVEL";
pub const ENV_OBS_LOG_STDOUT_ENABLED: &str = "RUSTFS_OBS_LOG_STDOUT_ENABLED"; pub const ENV_OBS_LOG_STDOUT_ENABLED: &str = "RUSTFS_OBS_LOG_STDOUT_ENABLED";
pub const ENV_OBS_LOG_DIRECTORY: &str = "RUSTFS_OBS_LOG_DIRECTORY"; pub const ENV_OBS_LOG_DIRECTORY: &str = "RUSTFS_OBS_LOG_DIRECTORY";
@@ -145,7 +141,6 @@ mod tests {
assert_eq!(ENV_OBS_METRICS_EXPORT_ENABLED, "RUSTFS_OBS_METRICS_EXPORT_ENABLED"); assert_eq!(ENV_OBS_METRICS_EXPORT_ENABLED, "RUSTFS_OBS_METRICS_EXPORT_ENABLED");
assert_eq!(ENV_OBS_LOGS_EXPORT_ENABLED, "RUSTFS_OBS_LOGS_EXPORT_ENABLED"); assert_eq!(ENV_OBS_LOGS_EXPORT_ENABLED, "RUSTFS_OBS_LOGS_EXPORT_ENABLED");
assert_eq!(ENV_OBS_PROFILING_EXPORT_ENABLED, "RUSTFS_OBS_PROFILING_EXPORT_ENABLED"); assert_eq!(ENV_OBS_PROFILING_EXPORT_ENABLED, "RUSTFS_OBS_PROFILING_EXPORT_ENABLED");
assert_eq!(ENV_OBS_PUT_STAGE_METRICS_ENABLED, "RUSTFS_OBS_PUT_STAGE_METRICS_ENABLED");
// Test log cleanup related env keys // Test log cleanup related env keys
assert_eq!(ENV_OBS_LOG_MAX_TOTAL_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_TOTAL_SIZE_BYTES"); assert_eq!(ENV_OBS_LOG_MAX_TOTAL_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_TOTAL_SIZE_BYTES");
assert_eq!(ENV_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES"); assert_eq!(ENV_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES");
@@ -1,612 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! ILM on SSE-KMS buckets while per-key SSE authorization is enforced (backlog#1582).
//!
//! Per-key KMS authorization (`RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY=true`) scopes the
//! SSE-KMS data path to the requesting principal's `kms:GenerateDataKey` /
//! `kms:Decrypt` grants. Internal callers — the lifecycle scanner's expiry deletes
//! and the tier transition worker's reads — carry no request principal, and
//! `authorize_sse_kms_key` (rustfs/src/storage/sse.rs) exempts a `None` principal
//! so background maintenance keeps working on encrypted buckets.
//!
//! These tests pin that exemption end to end. If enforcement ever starts applying
//! to the scanner's internal operations, expiry stops happening on SSE-KMS buckets
//! and [`ilm_expiration_on_sse_kms_bucket_under_enforcement`] times out; if it
//! starts applying to the transition worker or the read-through path,
//! [`ilm_transition_on_sse_kms_bucket_under_enforcement_reads_back`] fails at the
//! transition wait or the plaintext round-trip.
//!
//! The replication half of the same acceptance item lives in
//! `crates/e2e_test/src/replication_extension_test.rs`
//! (`test_bucket_replication_sse_kms_failure_contract`); ILM had no coverage
//! before this file.
//!
//! Deployment constraint pinned by the transition test's setup: the RustFS warm
//! backend forwards the object's stored `x-amz-server-side-encryption*` metadata
//! as raw headers on the tier data PUT (`build_transition_put_options` +
//! `api_put_object.rs` header mapping), so a RustFS tier target must itself have
//! KMS enabled and hold the named key or it rejects every transition upload with
//! 400 InvalidRequest. That rejection is independent of the enforcement switch;
//! the cold server here therefore runs its own Local KMS with the same key id.
use super::common::{LocalKMSTestEnvironment, create_key_with_specific_id};
use crate::common::{RustFSTestEnvironment, admin_request, init_logging};
use aws_sdk_s3::Client;
use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::{
BucketLifecycleConfiguration, ExpirationStatus, LifecycleExpiration, LifecycleRule, LifecycleRuleFilter, RestoreRequest,
ServerSideEncryption, ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Transition,
TransitionStorageClass,
};
use serde::Deserialize;
use serial_test::serial;
use std::time::{Duration as StdDuration, Instant};
use tracing::info;
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
const SSE_KEY: &str = "kms-ilm-sse-key";
const PAYLOAD: &[u8] = b"kms ilm sse payload: survives enforcement, expires and transitions on schedule";
const EXPIRY_BUCKET: &str = "kms-ilm-expiry";
const EXPIRE_KEY: &str = "expire/object.bin";
const SURVIVOR_KEY: &str = "keep/object.bin";
const TIER_NAME: &str = "KMSCOLD";
const TIER_BUCKET: &str = "kms-ilm-cold-tier";
const TIER_PREFIX: &str = "tiered";
const TRANSITION_BUCKET: &str = "kms-ilm-transition";
const TRANSITION_KEY: &str = "tier/object.bin";
/// Generous CI safety net; with a 1s scanner cycle and 2s lifecycle days the
/// terminal state normally lands within a few seconds.
const ILM_DEADLINE: StdDuration = StdDuration::from_secs(90);
/// Start a Local-KMS server with per-key SSE authorization enforced and the
/// lifecycle clock accelerated.
///
/// KMS wiring matches `kms_authorization_negative_matrix_test.rs` (local backend,
/// `--kms-default-key-id`, insecure dev defaults). The lifecycle env matches
/// `reliant/lifecycle.rs::fast_lifecycle_env` plus `RUSTFS_ILM_DEBUG_DAY_SECS=2`,
/// so a `Days=1` rule is due about two seconds after the write.
async fn start_enforcing_ilm_server(env: &mut LocalKMSTestEnvironment) -> TestResult {
create_key_with_specific_id(&env.kms_keys_dir, SSE_KEY).await?;
let key_dir = env.kms_keys_dir.clone();
let args = vec![
"--kms-enable",
"--kms-backend",
"local",
"--kms-key-dir",
key_dir.as_str(),
"--kms-default-key-id",
SSE_KEY,
];
let envs = [
("RUSTFS_KMS_ALLOW_INSECURE_DEV_DEFAULTS", "true"),
("RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY", "false"),
("RUSTFS_SCANNER_CYCLE", "1"),
("RUSTFS_ILM_PROCESS_TIME", "1"),
("RUSTFS_ILM_DEBUG_DAY_SECS", "2"),
];
env.base_env.start_rustfs_server_with_env(args, &envs).await?;
Ok(())
}
/// Set the bucket's default encryption to SSE-KMS under [`SSE_KEY`], so plain
/// PUTs (and internal rewrites) are encrypted without per-request SSE headers.
async fn set_bucket_default_sse_kms(client: &Client, bucket: &str) -> TestResult {
let encryption_config = ServerSideEncryptionConfiguration::builder()
.rules(
ServerSideEncryptionRule::builder()
.apply_server_side_encryption_by_default(
ServerSideEncryptionByDefault::builder()
.sse_algorithm(ServerSideEncryption::AwsKms)
.kms_master_key_id(SSE_KEY)
.build()?,
)
.build(),
)
.build()?;
client
.put_bucket_encryption()
.bucket(bucket)
.server_side_encryption_configuration(encryption_config)
.send()
.await?;
Ok(())
}
/// Assert via `HeadObject` that the stored object is SSE-KMS encrypted under
/// [`SSE_KEY`]. Without this, a bucket-default misconfiguration would let the
/// tests pass on an unencrypted object and prove nothing about KMS.
async fn assert_head_sse_kms(client: &Client, bucket: &str, key: &str) -> TestResult {
let head = client.head_object().bucket(bucket).key(key).send().await?;
assert_eq!(
head.server_side_encryption(),
Some(&ServerSideEncryption::AwsKms),
"{bucket}/{key} must be SSE-KMS encrypted via the bucket default"
);
assert_eq!(
head.ssekms_key_id(),
Some(SSE_KEY),
"{bucket}/{key} must be wrapped under the configured KMS key"
);
Ok(())
}
/// Returns `true` once `GET bucket/key` fails with `NoSuchKey`, `false` while it
/// still succeeds. Any other error is surfaced. (Copied from
/// `reliant/lifecycle.rs`; that helper is private to the reliant module.)
async fn object_is_gone(client: &Client, bucket: &str, key: &str) -> Result<bool, Box<dyn std::error::Error + Send + Sync>> {
match client.get_object().bucket(bucket).key(key).send().await {
Ok(output) => {
output.body.collect().await?;
Ok(false)
}
Err(e) => {
if let Some(service_error) = e.as_service_error() {
if service_error.is_no_such_key() {
return Ok(true);
}
return Err(format!("expected NoSuchKey, got: {e:?}").into());
}
Err(format!("expected a service error, got: {e:?}").into())
}
}
}
/// Poll until `GET bucket/key` returns `NoSuchKey`, or fail after `deadline`.
async fn wait_for_object_expired(client: &Client, bucket: &str, key: &str, deadline: StdDuration) -> TestResult {
let start = Instant::now();
loop {
if object_is_gone(client, bucket, key).await? {
return Ok(());
}
if start.elapsed() >= deadline {
return Err(format!(
"object {bucket}/{key} was not expired by the lifecycle scanner within {}s; \
SSE key-policy enforcement may have started blocking the scanner's internal deletes",
deadline.as_secs()
)
.into());
}
tokio::time::sleep(StdDuration::from_millis(500)).await;
}
}
/// Install a prefix-scoped `Days`-based expiration rule.
async fn put_expiration_rule(client: &Client, bucket: &str, id: &str, prefix: &str, days: i32) -> TestResult {
let rule = LifecycleRule::builder()
.id(id)
.filter(LifecycleRuleFilter::builder().prefix(prefix).build())
.expiration(LifecycleExpiration::builder().days(days).build())
.status(ExpirationStatus::Enabled)
.build()?;
let lifecycle = BucketLifecycleConfiguration::builder().rules(rule).build()?;
client
.put_bucket_lifecycle_configuration()
.bucket(bucket)
.lifecycle_configuration(lifecycle)
.send()
.await?;
Ok(())
}
/// Install a prefix-scoped `Days`-based transition rule targeting [`TIER_NAME`].
async fn put_transition_rule(client: &Client, bucket: &str, id: &str, prefix: &str, days: i32) -> TestResult {
let rule = LifecycleRule::builder()
.id(id)
.filter(LifecycleRuleFilter::builder().prefix(prefix).build())
.transitions(
Transition::builder()
.days(days)
.storage_class(TransitionStorageClass::from(TIER_NAME))
.build(),
)
.status(ExpirationStatus::Enabled)
.build()?;
let lifecycle = BucketLifecycleConfiguration::builder().rules(rule).build()?;
client
.put_bucket_lifecycle_configuration()
.bucket(bucket)
.lifecycle_configuration(lifecycle)
.send()
.await?;
Ok(())
}
/// Start a plain Local-KMS server (no enforcement, no lifecycle acceleration)
/// holding [`SSE_KEY`], to serve as the cold tier target.
///
/// The RustFS warm backend forwards the stored SSE-KMS headers on the tier data
/// PUT, so the target re-applies managed SSE-KMS under the named key and must
/// be able to resolve it; without KMS it answers 400 InvalidRequest and the
/// transition can never complete. Enforcement stays off here: the tier writes
/// arrive under `cold`'s root credentials, and one enforcing side is enough to
/// pin the exemption.
async fn start_cold_tier_kms_server(env: &mut LocalKMSTestEnvironment) -> TestResult {
create_key_with_specific_id(&env.kms_keys_dir, SSE_KEY).await?;
let key_dir = env.kms_keys_dir.clone();
let args = vec![
"--kms-enable",
"--kms-backend",
"local",
"--kms-key-dir",
key_dir.as_str(),
"--kms-default-key-id",
SSE_KEY,
];
env.base_env
.start_rustfs_server_with_env(args, &[("RUSTFS_KMS_ALLOW_INSECURE_DEV_DEFAULTS", "true")])
.await?;
Ok(())
}
/// The subset of the manual transition run report these tests assert on.
///
/// Unknown fields are ignored, so this stays compatible with report growth; the
/// full shape is pinned by `reliant/tiering.rs`.
#[derive(Debug, Deserialize)]
struct ManualTransitionRunReport {
#[serde(default)]
scanned: u64,
#[serde(default)]
enqueued: u64,
#[serde(default)]
skipped_already_in_flight: u64,
#[serde(default)]
skipped_tier: u64,
}
#[derive(Debug, Deserialize)]
struct ManualTransitionRunResponse {
state: String,
report: ManualTransitionRunReport,
}
/// One synchronous (enqueue-only) manual transition run over `bucket/prefix`,
/// via the same admin endpoint `reliant/tiering.rs` drives.
async fn manual_transition_run(
hot: &RustFSTestEnvironment,
bucket: &str,
prefix: &str,
) -> Result<ManualTransitionRunResponse, Box<dyn std::error::Error + Send + Sync>> {
let bucket = urlencoding::encode(bucket);
let prefix = urlencoding::encode(prefix);
let tier = urlencoding::encode(TIER_NAME);
let path =
format!("/rustfs/admin/v3/ilm/transition/run?bucket={bucket}&prefix={prefix}&tier={tier}&dryRun=false&maxObjects=10");
let (status, body) = admin_request(&hot.url, http::Method::POST, &path, None, &hot.access_key, &hot.secret_key).await?;
if !status.is_success() {
return Err(format!("manual transition run failed: status={status}, body={body}").into());
}
Ok(serde_json::from_str(&body)?)
}
/// Drive manual transition runs until one reports the object as processed.
///
/// The `Days=1` rule becomes due about two seconds after the write
/// (`RUSTFS_ILM_DEBUG_DAY_SECS=2`), so early runs may legitimately report the
/// object as not yet eligible; the loop keeps running the endpoint until it
/// either enqueues the transition, sees it already in flight (the 1s scanner
/// backstop got there first), or finds it already on the tier.
async fn run_manual_transition_until_processed(
hot: &RustFSTestEnvironment,
bucket: &str,
prefix: &str,
deadline: StdDuration,
) -> TestResult {
let start = Instant::now();
loop {
let run = manual_transition_run(hot, bucket, prefix).await?;
assert_eq!(run.report.scanned, 1, "manual transition run must scan the object: {run:#?}");
if run.report.enqueued + run.report.skipped_already_in_flight + run.report.skipped_tier >= 1 {
info!(state = %run.state, report = ?run.report, "manual transition run processed the SSE-KMS object");
return Ok(());
}
if start.elapsed() >= deadline {
return Err(format!(
"manual transition runs never processed {bucket}/{prefix} within {}s; last report: {run:#?}",
deadline.as_secs()
)
.into());
}
tokio::time::sleep(StdDuration::from_millis(500)).await;
}
}
/// Wire `hot` -> `cold` as a `TierType::RustFS` remote tier via `AddTier`.
///
/// No `force`, so the server runs the real connectivity probe against `cold`
/// (the tier bucket must already exist there). Mirrors
/// `reliant/tiering.rs::add_rustfs_tier`, which is private to that module.
async fn add_rustfs_tier(hot: &RustFSTestEnvironment, cold: &RustFSTestEnvironment) -> TestResult {
let body = serde_json::json!({
"type": "rustfs",
"rustfs": {
"name": TIER_NAME,
"endpoint": cold.url.as_str(),
"accessKey": cold.access_key.as_str(),
"secretKey": cold.secret_key.as_str(),
"bucket": TIER_BUCKET,
"prefix": TIER_PREFIX,
"region": "us-east-1",
"storageClass": ""
}
})
.to_string();
let (status, resp) = admin_request(
&hot.url,
http::Method::PUT,
"/rustfs/admin/v3/tier",
Some(body),
&hot.access_key,
&hot.secret_key,
)
.await?;
if !status.is_success() {
return Err(format!("AddTier(RustFS) failed: status={status}, body={resp}").into());
}
Ok(())
}
/// Poll `HEAD` until the object's storage class is the tier name (transition
/// complete), or fail after `deadline`. (From `reliant/tiering.rs`.)
async fn wait_for_transition(client: &Client, bucket: &str, key: &str, deadline: StdDuration) -> TestResult {
let start = Instant::now();
loop {
let head = client.head_object().bucket(bucket).key(key).send().await?;
if head.storage_class().map(|sc| sc.as_str()) == Some(TIER_NAME) {
return Ok(());
}
if start.elapsed() >= deadline {
return Err(format!(
"object {bucket}/{key} was not transitioned to {TIER_NAME} within {}s (storage_class={:?}); \
SSE key-policy enforcement may have started blocking the transition worker's internal reads",
deadline.as_secs(),
head.storage_class()
)
.into());
}
tokio::time::sleep(StdDuration::from_millis(500)).await;
}
}
/// Poll `HEAD` until `x-amz-restore` reports a finished restore
/// (`ongoing-request="false"`), or fail after `deadline`.
async fn wait_for_restore_complete(client: &Client, bucket: &str, key: &str, deadline: StdDuration) -> TestResult {
let start = Instant::now();
loop {
let head = client.head_object().bucket(bucket).key(key).send().await?;
if head.restore().is_some_and(|r| r.contains("ongoing-request=\"false\"")) {
return Ok(());
}
if start.elapsed() >= deadline {
return Err(format!(
"object {bucket}/{key} restore did not complete within {}s (restore={:?}); \
SSE key-policy enforcement may have started blocking the restore copy-back's internal reads",
deadline.as_secs(),
head.restore()
)
.into());
}
tokio::time::sleep(StdDuration::from_millis(500)).await;
}
}
/// ILM expiration keeps working on an SSE-KMS bucket while per-key SSE
/// authorization is enforced.
///
/// The lifecycle scanner deletes expired objects with an internal (no-principal)
/// identity that holds no `kms` grant. If enforcement ever starts applying to
/// those internal deletes (or to the scanner's metadata reads) on encrypted
/// buckets, expiry stops happening and this test times out.
///
/// A survivor object under a non-matching prefix isolates the rule's prefix
/// filter as the cause of the deletion and proves the encrypted bucket stays
/// readable end to end after the scanner has run.
#[tokio::test]
#[serial]
async fn ilm_expiration_on_sse_kms_bucket_under_enforcement() -> TestResult {
init_logging();
let mut env = LocalKMSTestEnvironment::new().await?;
start_enforcing_ilm_server(&mut env).await?;
env.base_env.create_test_bucket(EXPIRY_BUCKET).await?;
let client = env.base_env.create_s3_client();
set_bucket_default_sse_kms(&client, EXPIRY_BUCKET).await?;
for key in [EXPIRE_KEY, SURVIVOR_KEY] {
client
.put_object()
.bucket(EXPIRY_BUCKET)
.key(key)
.body(ByteStream::from_static(PAYLOAD))
.send()
.await?;
assert_head_sse_kms(&client, EXPIRY_BUCKET, key).await?;
}
info!("both objects stored SSE-KMS encrypted under enforcement");
put_expiration_rule(&client, EXPIRY_BUCKET, "kms-ilm-expire", "expire/", 1).await?;
// The regression this pins: the scanner's internal delete must stay exempt
// from per-key SSE authorization, so the encrypted object actually expires.
wait_for_object_expired(&client, EXPIRY_BUCKET, EXPIRE_KEY, ILM_DEADLINE).await?;
info!("SSE-KMS object expired by the lifecycle scanner under enforcement");
// Negative control: same bucket, same encryption, non-matching prefix. It
// must survive the scanner and still decrypt for the requesting principal.
assert!(
!object_is_gone(&client, EXPIRY_BUCKET, SURVIVOR_KEY).await?,
"non-matching-prefix object must not be expired by a prefix-scoped rule"
);
let survivor = client.get_object().bucket(EXPIRY_BUCKET).key(SURVIVOR_KEY).send().await?;
assert_eq!(
survivor.body.collect().await?.into_bytes().as_ref(),
PAYLOAD,
"surviving SSE-KMS object must still decrypt after the scanner has run"
);
Ok(())
}
/// ILM transition to a remote tier keeps working on an SSE-KMS bucket while
/// per-key SSE authorization is enforced, and the transitioned object reads
/// back as plaintext.
///
/// The transition worker moves the stored (encrypted) bytes to the cold tier
/// with an internal (no-principal) identity; the read-through `GET` then
/// decrypts the envelope for the requesting principal. If enforcement ever
/// starts applying to the worker's internal reads, the transition wait times
/// out; if the stored envelope is mishandled across the tier round trip, the
/// plaintext comparison fails.
///
/// The transition is driven through the manual transition-run admin endpoint
/// (the mechanism `reliant/tiering.rs` established), so the test does not
/// depend on scanner scheduling; the 1s scanner cycle stays on as a backstop.
#[tokio::test]
#[serial]
#[ignore = "pins rustfs/rustfs#6025: GET on a transitioned managed-SSE object silently returns corrupt bytes (fails with enforcement on AND off, so it is not an authorization regression); un-ignore with the fix"]
async fn ilm_transition_on_sse_kms_bucket_under_enforcement_reads_back() -> TestResult {
init_logging();
// Cold-tier server: independent credentials, its own Local KMS holding the
// same key id (see the module docs for why the tier target needs KMS).
// Started first; each server's startup cleanup only matches its own unique
// address and temp dir, so the two instances coexist.
let mut cold = LocalKMSTestEnvironment::new().await?;
cold.base_env.access_key = "kmscoldtieradmin".to_string();
cold.base_env.secret_key = "kmscoldtiersecret".to_string();
start_cold_tier_kms_server(&mut cold).await?;
let cold_client = cold.base_env.create_s3_client();
cold_client.create_bucket().bucket(TIER_BUCKET).send().await?;
// Hot server: Local KMS + enforcement + accelerated lifecycle clock.
let mut env = LocalKMSTestEnvironment::new().await?;
start_enforcing_ilm_server(&mut env).await?;
let hot_client = env.base_env.create_s3_client();
add_rustfs_tier(&env.base_env, &cold.base_env).await?;
env.base_env.create_test_bucket(TRANSITION_BUCKET).await?;
set_bucket_default_sse_kms(&hot_client, TRANSITION_BUCKET).await?;
hot_client
.put_object()
.bucket(TRANSITION_BUCKET)
.key(TRANSITION_KEY)
.body(ByteStream::from_static(PAYLOAD))
.send()
.await?;
assert_head_sse_kms(&hot_client, TRANSITION_BUCKET, TRANSITION_KEY).await?;
info!("object stored SSE-KMS encrypted under enforcement");
// Days=1 is due ~2s after the write with RUSTFS_ILM_DEBUG_DAY_SECS=2.
put_transition_rule(&hot_client, TRANSITION_BUCKET, "kms-ilm-transition", "tier/", 1).await?;
// Drive the transition deterministically via the manual run endpoint, then
// wait for HEAD to report the tier as the object's storage class.
run_manual_transition_until_processed(&env.base_env, TRANSITION_BUCKET, "tier/", ILM_DEADLINE).await?;
wait_for_transition(&hot_client, TRANSITION_BUCKET, TRANSITION_KEY, ILM_DEADLINE).await?;
info!("SSE-KMS object transitioned to the remote tier under enforcement");
let head = hot_client
.head_object()
.bucket(TRANSITION_BUCKET)
.key(TRANSITION_KEY)
.send()
.await?;
assert!(
head.restore().is_none(),
"a freshly transitioned object must not advertise x-amz-restore, got {:?}",
head.restore()
);
// The remote copy exists on the cold tier. The payload the tier holds is the
// hot server's stored ciphertext, wrapped once more under the cold server's
// own managed SSE-KMS layer (the forwarded headers re-request encryption).
let remote = cold_client.list_objects_v2().bucket(TIER_BUCKET).send().await?;
assert!(!remote.contents().is_empty(), "cold-tier bucket must hold the transitioned object's data");
// Read-through GET under enforcement must succeed (not AccessDenied) and
// keep advertising SSE-KMS. Its BODY is deliberately not compared here:
// the transitioned read path skips managed-SSE decryption — a product gap
// unrelated to enforcement — so a direct GET streams the stored ciphertext
// (`new_getobjectreader` in crates/ecstore/src/client/object_api_utils.rs
// hardcodes `is_encrypted = false` and never applies the
// `ReadTransform::Encrypted` wrapping the hot-read path builds in
// crates/ecstore/src/object_api/readers.rs). Plaintext recovery is pinned
// through restore semantics below; when the read-through gap is fixed, a
// byte assertion can be added here too.
let read_through = hot_client
.get_object()
.bucket(TRANSITION_BUCKET)
.key(TRANSITION_KEY)
.send()
.await?;
assert_eq!(
read_through.server_side_encryption(),
Some(&ServerSideEncryption::AwsKms),
"transitioned object must still report SSE-KMS on read-through"
);
let read_through_body = read_through.body.collect().await?.into_bytes();
assert_eq!(
read_through_body.len(),
PAYLOAD.len(),
"read-through GET must stream the object's full logical size under enforcement"
);
// RestoreObject copies the ciphertext back from the tier under the original
// envelope metadata; the restored copy is then served by the normal
// decrypting read path. The copy-back runs with an internal (no-principal)
// identity, so this also pins the exemption on the restore path. Days=300
// because RUSTFS_ILM_DEBUG_DAY_SECS=2 accelerates the restored copy's
// expiry as well (300 accelerated days == 600s of validity).
hot_client
.restore_object()
.bucket(TRANSITION_BUCKET)
.key(TRANSITION_KEY)
.restore_request(RestoreRequest::builder().days(300).build())
.send()
.await?;
wait_for_restore_complete(&hot_client, TRANSITION_BUCKET, TRANSITION_KEY, ILM_DEADLINE).await?;
info!("SSE-KMS object restored from the remote tier under enforcement");
// The KMS-relevant half: the restored envelope decrypts back to the exact
// plaintext for the requesting principal.
let restored = hot_client
.get_object()
.bucket(TRANSITION_BUCKET)
.key(TRANSITION_KEY)
.send()
.await?;
assert_eq!(
restored.server_side_encryption(),
Some(&ServerSideEncryption::AwsKms),
"restored object must still report SSE-KMS"
);
let body = restored.body.collect().await?.into_bytes();
assert_eq!(body.as_ref(), PAYLOAD, "restored SSE-KMS object must round-trip byte-identical plaintext");
Ok(())
}
-3
View File
@@ -59,6 +59,3 @@ mod configured_roundtrip_test;
#[cfg(test)] #[cfg(test)]
mod kms_authorization_negative_matrix_test; mod kms_authorization_negative_matrix_test;
#[cfg(test)]
mod kms_ilm_sse_kms_test;
File diff suppressed because it is too large Load Diff
@@ -2854,7 +2854,7 @@ pub(crate) mod cmptst_30 {
result result
} }
#[ignore = "timing-sensitive backend-pressure latency probe; run explicitly with --ignored"] #[ignore]
#[tokio::test] #[tokio::test]
async fn regression() -> Result<(), Box<dyn std::error::Error + Send + Sync>> { async fn regression() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
crate::common::init_logging(); crate::common::init_logging();
@@ -2401,20 +2401,15 @@ async fn wait_for_site_replication_info<F>(
where where
F: Fn(&SiteReplicationInfo) -> bool, F: Fn(&SiteReplicationInfo) -> bool,
{ {
// 30s to match wait_for_replication_state: the three-node site tests run for _ in 0..40 {
// several full rustfs processes on one runner, so peer-state propagation
// can take well over 10s under CI load.
let deadline = tokio::time::Instant::now() + Duration::from_secs(30);
loop {
let info = site_replication_info(env).await?; let info = site_replication_info(env).await?;
if predicate(&info) { if predicate(&info) {
return Ok(info); return Ok(info);
} }
if tokio::time::Instant::now() >= deadline {
return Err(format!("site replication info did not reach expected state on {}", env.address).into());
}
sleep(Duration::from_millis(250)).await; sleep(Duration::from_millis(250)).await;
} }
Err(format!("site replication info did not reach expected state on {}", env.address).into())
} }
async fn wait_for_site_replication_status<F>( async fn wait_for_site_replication_status<F>(
@@ -2425,19 +2420,15 @@ async fn wait_for_site_replication_status<F>(
where where
F: Fn(&SRStatusInfo) -> bool, F: Fn(&SRStatusInfo) -> bool,
{ {
// Same 30s ceiling as wait_for_site_replication_info: the status probes for _ in 0..40 {
// fan out to every peer, so they see the same multi-process CI load.
let deadline = tokio::time::Instant::now() + Duration::from_secs(30);
loop {
let status = site_replication_status(env, query).await?; let status = site_replication_status(env, query).await?;
if predicate(&status) { if predicate(&status) {
return Ok(status); return Ok(status);
} }
if tokio::time::Instant::now() >= deadline {
return Err(format!("site replication status did not reach expected state on {}", env.address).into());
}
sleep(Duration::from_millis(250)).await; sleep(Duration::from_millis(250)).await;
} }
Err(format!("site replication status did not reach expected state on {}", env.address).into())
} }
async fn wait_for_replication_reset_target<F>( async fn wait_for_replication_reset_target<F>(
@@ -4244,49 +4235,37 @@ async fn test_bucket_replication_acceptance_matrix_local_dual_targets() -> TestR
"tag rule with disabled delete-marker replication created a marker: {tagged_state:?}" "tag rule with disabled delete-marker replication created a marker: {tagged_state:?}"
); );
// AWS S3 and MinIO both reject suspending versioning on a bucket that set_bucket_versioning(&source_env, source_bucket, BucketVersioningStatus::Suspended).await?;
// carries a replication configuration (InvalidBucketState): suspension set_bucket_versioning(&target_env_a, target_bucket_a, BucketVersioningStatus::Suspended).await?;
// would mint null versions that versioned replication can never converge. let null_put = source_client
let suspend_err = source_client
.put_bucket_versioning()
.bucket(source_bucket)
.versioning_configuration(
VersioningConfiguration::builder()
.status(BucketVersioningStatus::Suspended)
.build(),
)
.send()
.await
.expect_err("suspending versioning on a replication source must be rejected");
assert_eq!(
suspend_err.as_service_error().and_then(|error| error.code()),
Some("InvalidBucketState"),
"suspension on a replication source must fail with InvalidBucketState: {suspend_err:?}"
);
// The rejected suspension must leave the versioning + replication state
// fully intact: a fresh matched PUT still replicates with a real version.
let post_reject_put = source_client
.put_object() .put_object()
.bucket(source_bucket) .bucket(source_bucket)
.key("prefix/after-rejected-suspend.txt") .key("prefix/null.txt")
.body(ByteStream::from_static(b"still replicating")) .body(ByteStream::from_static(b"null version"))
.send() .send()
.await?; .await?;
let post_reject_version_id = post_reject_put assert!(null_put.version_id().is_none(), "suspended source PUT must create a null version");
.version_id() wait_for_replication_state(&target_client_a, target_bucket_a, "null version did not replicate", |state| {
.ok_or("PUT after rejected suspension omitted version ID")? state
.to_string(); .iter()
wait_for_replication_state( .any(|entry| entry.key == "prefix/null.txt" && entry.version_id == "null" && !entry.delete_marker)
&target_client_a, })
target_bucket_a, .await?;
"replication stopped after rejected versioning suspension", let null_delete = source_client
|state| { .delete_object()
state .bucket(source_bucket)
.iter() .key("prefix/null.txt")
.any(|entry| entry.key == "prefix/after-rejected-suspend.txt" && entry.version_id == post_reject_version_id) .send()
}, .await?;
) assert!(
null_delete.version_id().is_none(),
"suspended source DELETE must create a null delete marker"
);
wait_for_replication_state(&target_client_a, target_bucket_a, "null delete marker did not replicate", |state| {
state
.iter()
.any(|entry| entry.key == "prefix/null.txt" && entry.version_id == "null" && entry.delete_marker)
})
.await?; .await?;
Ok(()) Ok(())
-5
View File
@@ -32,11 +32,6 @@ workspace = true
[features] [features]
default = [] default = []
# Compiles the controlled list-objects namespace-journal chaos injector into a
# production binary (it is always available to tests). Off by default so the
# RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_* env vars cannot rewrite journal
# state in a stock build (backlog#1832).
list-chaos = []
rio-v2 = ["dep:rustfs-rio-v2"] rio-v2 = ["dep:rustfs-rio-v2"]
hotpath = [ hotpath = [
"hotpath/hotpath", "hotpath/hotpath",
+4 -6
View File
@@ -61,11 +61,9 @@ pub mod bucket {
delete_manual_transition_scope_admission_if_current, load_manual_transition_job_record, delete_manual_transition_scope_admission_if_current, load_manual_transition_job_record,
load_manual_transition_job_record_with_etag, load_manual_transition_scope_admission, load_manual_transition_job_record_with_etag, load_manual_transition_scope_admission,
manual_transition_job_lease_expired, manual_transition_scope_admission_lease_expired, manual_transition_job_lease_expired, manual_transition_scope_admission_lease_expired,
manual_transition_scope_key, persist_manual_transition_job_progress, manual_transition_scope_key, persist_manual_transition_job_progress, renew_manual_transition_job_lease,
persist_manual_transition_job_progress_if_owned, renew_manual_transition_job_lease, request_manual_transition_job_cancel, save_manual_transition_job_record,
renew_manual_transition_job_lease_if_owned, request_manual_transition_job_cancel, save_manual_transition_job_record_if_current, save_manual_transition_scope_admission_if_absent,
save_manual_transition_job_record, save_manual_transition_job_record_if_current,
save_manual_transition_scope_admission_if_absent, update_manual_transition_job_record,
}; };
} }
@@ -346,7 +344,7 @@ pub mod disk {
} }
pub mod error { pub mod error {
pub use crate::disk::error::{DiskError, Error, FileAccessDeniedWithContext, Result}; pub use crate::disk::error::{BitrotErrorType, DiskError, Error, FileAccessDeniedWithContext, Result};
} }
pub mod error_reduce { pub mod error_reduce {
File diff suppressed because it is too large Load Diff
@@ -86,21 +86,6 @@ where
com::save_config_with_opts(api, file, data, opts).await com::save_config_with_opts(api, file, data, opts).await
} }
pub(crate) async fn save_config_with_opts_quiet<S>(api: Arc<S>, file: &str, data: Vec<u8>, opts: &ObjectOptions) -> Result<()>
where
S: ObjectIO<
Error = Error,
RangeSpec = HTTPRangeSpec,
HeaderMap = HeaderMap,
ObjectOptions = ObjectOptions,
ObjectInfo = ObjectInfo,
GetObjectReader = GetObjectReader,
PutObjectReader = PutObjReader,
>,
{
com::save_config_with_opts_quiet(api, file, data, opts).await
}
pub(crate) async fn delete_config<S>(api: Arc<S>, file: &str) -> Result<()> pub(crate) async fn delete_config<S>(api: Arc<S>, file: &str) -> Result<()>
where where
S: ObjectOperations< S: ObjectOperations<
@@ -45,104 +45,6 @@ const MANUAL_TRANSITION_JOB_LEASE_SECONDS: i128 = 60;
const MANUAL_TRANSITION_LEGACY_SCOPE_SCAN_LIMIT: i32 = 1000; const MANUAL_TRANSITION_LEGACY_SCOPE_SCAN_LIMIT: i32 = 1000;
const MANUAL_TRANSITION_TASK_SCAN_LIMIT: i32 = 1000; const MANUAL_TRANSITION_TASK_SCAN_LIMIT: i32 = 1000;
const MANUAL_TRANSITION_WORKER_RESULT_SCAN_LIMIT: i32 = 1000; const MANUAL_TRANSITION_WORKER_RESULT_SCAN_LIMIT: i32 = 1000;
const MANUAL_TRANSITION_JOB_CAS_RETRIES: usize = 4;
#[cfg(test)]
struct ManualTransitionJobCasBarrierState {
job_id: Uuid,
paused: std::sync::atomic::AtomicBool,
arrived: tokio::sync::Notify,
release: tokio::sync::Semaphore,
}
#[cfg(test)]
pub(crate) struct ManualTransitionJobCasBarrier {
state: Arc<ManualTransitionJobCasBarrierState>,
}
#[cfg(test)]
static MANUAL_TRANSITION_JOB_CAS_BARRIER: std::sync::OnceLock<std::sync::Mutex<Option<Arc<ManualTransitionJobCasBarrierState>>>> =
std::sync::OnceLock::new();
#[cfg(test)]
impl ManualTransitionJobCasBarrier {
pub(crate) fn install(job_id: Uuid) -> Self {
let state = Arc::new(ManualTransitionJobCasBarrierState {
job_id,
paused: std::sync::atomic::AtomicBool::new(false),
arrived: tokio::sync::Notify::new(),
release: tokio::sync::Semaphore::new(0),
});
let mut slot = MANUAL_TRANSITION_JOB_CAS_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("manual transition progress CAS barrier mutex should not poison");
assert!(
slot.is_none(),
"manual transition job CAS barrier must be installed by one test at a time"
);
*slot = Some(Arc::clone(&state));
drop(slot);
Self { state }
}
pub(crate) async fn wait_until_paused(&self) {
tokio::time::timeout(std::time::Duration::from_secs(30), async {
loop {
let arrived = self.state.arrived.notified();
if self.state.paused.load(std::sync::atomic::Ordering::Acquire) {
return;
}
arrived.await;
}
})
.await
.expect("manual transition job update should reach the deterministic CAS barrier");
}
pub(crate) fn release(&self) {
self.state.release.add_permits(1);
}
}
#[cfg(test)]
impl Drop for ManualTransitionJobCasBarrier {
fn drop(&mut self) {
self.release();
let mut slot = MANUAL_TRANSITION_JOB_CAS_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("manual transition progress CAS barrier mutex should not poison");
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
*slot = None;
}
}
}
#[cfg(test)]
async fn pause_manual_transition_job_before_first_cas(job_id: Uuid) {
let barrier = MANUAL_TRANSITION_JOB_CAS_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("manual transition progress CAS barrier mutex should not poison")
.as_ref()
.filter(|barrier| barrier.job_id == job_id)
.cloned();
if let Some(barrier) = barrier
&& barrier
.paused
.compare_exchange(false, true, std::sync::atomic::Ordering::AcqRel, std::sync::atomic::Ordering::Acquire)
.is_ok()
{
barrier.arrived.notify_one();
barrier
.release
.acquire()
.await
.expect("manual transition job CAS barrier should remain open")
.forget();
}
}
fn is_false(value: &bool) -> bool { fn is_false(value: &bool) -> bool {
!*value !*value
@@ -246,6 +148,7 @@ impl ManualTransitionJobRecord {
pub fn fail(&mut self, error: impl Into<String>) { pub fn fail(&mut self, error: impl Into<String>) {
self.state = ManualTransitionJobState::Failed; self.state = ManualTransitionJobState::Failed;
self.report.tier_failure = self.report.tier_failure.saturating_add(1);
self.error = Some(error.into()); self.error = Some(error.into());
self.mark_updated_terminal(); self.mark_updated_terminal();
} }
@@ -1137,7 +1040,7 @@ pub async fn save_manual_transition_job_record_if_current(
} }
let object = manual_transition_job_record_object_name(job.job_id).map_err(manual_transition_job_store_error)?; let object = manual_transition_job_record_object_name(job.job_id).map_err(manual_transition_job_store_error)?;
let data = job.encode().map_err(manual_transition_job_store_error)?; let data = job.encode().map_err(manual_transition_job_store_error)?;
config_boundary::save_config_with_opts_quiet( config_boundary::save_config_with_opts(
api, api,
&object, &object,
data, data,
@@ -1153,54 +1056,6 @@ pub async fn save_manual_transition_job_record_if_current(
.await .await
} }
/// Applies a job-record mutation with optimistic concurrency control.
///
/// The mutation returns whether the record needs to be persisted. When a lease
/// is supplied, ownership is checked again after every conflicting write.
pub async fn update_manual_transition_job_record<F>(
api: Arc<ECStore>,
job_id: Uuid,
expected_lease_id: Option<Uuid>,
update: F,
) -> EcstoreResult<ManualTransitionJobRecord>
where
F: FnMut(&mut ManualTransitionJobRecord) -> bool,
{
update_manual_transition_job_record_from(api, job_id, expected_lease_id, None, update).await
}
async fn update_manual_transition_job_record_from<F>(
api: Arc<ECStore>,
job_id: Uuid,
expected_lease_id: Option<Uuid>,
mut current: Option<(ManualTransitionJobRecord, String)>,
mut update: F,
) -> EcstoreResult<ManualTransitionJobRecord>
where
F: FnMut(&mut ManualTransitionJobRecord) -> bool,
{
for _ in 0..MANUAL_TRANSITION_JOB_CAS_RETRIES {
let (mut record, etag) = match current.take() {
Some(current) => current,
None => load_manual_transition_job_record_with_etag(api.clone(), job_id).await?,
};
if expected_lease_id.is_some_and(|lease_id| record.lease_id != lease_id) {
return Err(Error::PreconditionFailed);
}
if !update(&mut record) {
return Ok(record);
}
#[cfg(test)]
pause_manual_transition_job_before_first_cas(job_id).await;
match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
Ok(()) => return Ok(record),
Err(Error::PreconditionFailed) => continue,
Err(err) => return Err(err),
}
}
Err(Error::PreconditionFailed)
}
pub(crate) async fn save_manual_transition_worker_result_if_absent( pub(crate) async fn save_manual_transition_worker_result_if_absent(
api: Arc<ECStore>, api: Arc<ECStore>,
record: &ManualTransitionWorkerResultRecord, record: &ManualTransitionWorkerResultRecord,
@@ -1459,113 +1314,99 @@ pub async fn reconcile_manual_transition_worker_results(
api: Arc<ECStore>, api: Arc<ECStore>,
job_id: Uuid, job_id: Uuid,
queue_snapshot: ManualTransitionQueueSnapshot, queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> {
reconcile_manual_transition_worker_results_inner(api, job_id, None, queue_snapshot, false).await
}
pub(crate) async fn reconcile_manual_transition_worker_results_if_owned(
api: Arc<ECStore>,
job_id: Uuid,
expected_lease_id: Uuid,
queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> {
reconcile_manual_transition_worker_results_inner(api, job_id, Some(expected_lease_id), queue_snapshot, false).await
}
async fn reconcile_manual_transition_worker_results_inner(
api: Arc<ECStore>,
job_id: Uuid,
expected_lease_id: Option<Uuid>,
queue_snapshot: ManualTransitionQueueSnapshot,
mark_missing_results_unknown: bool,
) -> EcstoreResult<ManualTransitionJobRecord> { ) -> EcstoreResult<ManualTransitionJobRecord> {
let task_stats = match scan_manual_transition_task_journal(api.clone(), job_id).await? { let task_stats = match scan_manual_transition_task_journal(api.clone(), job_id).await? {
ManualTransitionTaskJournal::Stats(stats) => stats, ManualTransitionTaskJournal::Stats(stats) => stats,
ManualTransitionTaskJournal::Corrupt(error) => { ManualTransitionTaskJournal::Corrupt(error) => {
return mark_manual_transition_job_unknown_for_task_journal_error( return mark_manual_transition_job_unknown_for_task_journal_error(api, job_id, error, queue_snapshot).await;
api,
job_id,
expected_lease_id,
error,
queue_snapshot,
)
.await;
} }
}; };
let stats = match scan_manual_transition_worker_result_journal(api.clone(), job_id).await? { let stats = match scan_manual_transition_worker_result_journal(api.clone(), job_id).await? {
ManualTransitionWorkerResultJournal::Stats(stats) => stats, ManualTransitionWorkerResultJournal::Stats(stats) => stats,
ManualTransitionWorkerResultJournal::Corrupt(error) => { ManualTransitionWorkerResultJournal::Corrupt(error) => {
return mark_manual_transition_job_unknown_for_worker_result_journal_error( return mark_manual_transition_job_unknown_for_worker_result_journal_error(api, job_id, error, queue_snapshot).await;
api,
job_id,
expected_lease_id,
error,
queue_snapshot,
)
.await;
} }
}; };
let mut changed = false; for _ in 0..4 {
let record = update_manual_transition_job_record(api.clone(), job_id, expected_lease_id, |record| { let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
let counts_changed = record.apply_worker_result_counts( let changed = record.apply_worker_result_counts(
stats.stats.completed, stats.stats.completed,
stats.stats.failed, stats.stats.failed,
&stats.stats.tier_failure_by_reason, &stats.stats.tier_failure_by_reason,
task_stats.queued, task_stats.queued,
queue_snapshot, queue_snapshot,
); );
let became_unknown = mark_missing_results_unknown && record.mark_unknown_if_worker_results_lost(queue_snapshot); if !changed {
changed = counts_changed || became_unknown; return Ok(record);
changed }
}) match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
.await?; Ok(()) => {
if !changed { if record.is_terminal() {
return Ok(record); delete_manual_transition_scope_admission_if_current(
api.clone(),
&record.scope_key,
record.job_id,
record.lease_id,
)
.await?;
} else {
renew_manual_transition_scope_admission_from_job(api, &record).await?;
}
return Ok(record);
}
Err(Error::PreconditionFailed) => continue,
Err(err) => return Err(err),
}
} }
if record.is_terminal() { Err(Error::PreconditionFailed)
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?;
} else {
renew_manual_transition_scope_admission_from_job(api, &record).await?;
}
Ok(record)
} }
async fn mark_manual_transition_job_unknown_for_task_journal_error( async fn mark_manual_transition_job_unknown_for_task_journal_error(
api: Arc<ECStore>, api: Arc<ECStore>,
job_id: Uuid, job_id: Uuid,
expected_lease_id: Option<Uuid>,
error: String, error: String,
queue_snapshot: ManualTransitionQueueSnapshot, queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> { ) -> EcstoreResult<ManualTransitionJobRecord> {
let mut changed = false; for _ in 0..4 {
let record = update_manual_transition_job_record(api.clone(), job_id, expected_lease_id, |record| { let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
changed = record.mark_unknown_for_task_journal_error(error.clone(), queue_snapshot); if !record.mark_unknown_for_task_journal_error(error.clone(), queue_snapshot) {
changed return Ok(record);
}) }
.await?; match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
if changed && record.is_terminal() { Ok(()) => {
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?; delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id)
.await?;
return Ok(record);
}
Err(Error::PreconditionFailed) => continue,
Err(err) => return Err(err),
}
} }
Ok(record) Err(Error::PreconditionFailed)
} }
async fn mark_manual_transition_job_unknown_for_worker_result_journal_error( async fn mark_manual_transition_job_unknown_for_worker_result_journal_error(
api: Arc<ECStore>, api: Arc<ECStore>,
job_id: Uuid, job_id: Uuid,
expected_lease_id: Option<Uuid>,
error: String, error: String,
queue_snapshot: ManualTransitionQueueSnapshot, queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> { ) -> EcstoreResult<ManualTransitionJobRecord> {
let mut changed = false; for _ in 0..4 {
let record = update_manual_transition_job_record(api.clone(), job_id, expected_lease_id, |record| { let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
changed = record.mark_unknown_for_worker_result_journal_error(error.clone(), queue_snapshot); if !record.mark_unknown_for_worker_result_journal_error(error.clone(), queue_snapshot) {
changed return Ok(record);
}) }
.await?; match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
if changed && record.is_terminal() { Ok(()) => {
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?; delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id)
.await?;
return Ok(record);
}
Err(Error::PreconditionFailed) => continue,
Err(err) => return Err(err),
}
} }
Ok(record) Err(Error::PreconditionFailed)
} }
pub async fn save_manual_transition_scope_admission_if_absent( pub async fn save_manual_transition_scope_admission_if_absent(
@@ -1762,14 +1603,19 @@ async fn find_active_legacy_manual_transition_scope_conflict(
} }
pub async fn request_manual_transition_job_cancel(api: Arc<ECStore>, job_id: Uuid) -> EcstoreResult<ManualTransitionJobRecord> { pub async fn request_manual_transition_job_cancel(api: Arc<ECStore>, job_id: Uuid) -> EcstoreResult<ManualTransitionJobRecord> {
update_manual_transition_job_record(api, job_id, None, |record| { for _ in 0..4 {
let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
if record.is_terminal() || record.cancel_requested { if record.is_terminal() || record.cancel_requested {
return false; return Ok(record);
} }
record.mark_cancel_requested(); record.mark_cancel_requested();
true match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
}) Ok(()) => return Ok(record),
.await Err(Error::PreconditionFailed) => continue,
Err(err) => return Err(err),
}
}
Err(Error::PreconditionFailed)
} }
pub async fn persist_manual_transition_job_progress( pub async fn persist_manual_transition_job_progress(
@@ -1778,39 +1624,10 @@ pub async fn persist_manual_transition_job_progress(
report: &ManualTransitionRunReport, report: &ManualTransitionRunReport,
queue_snapshot: ManualTransitionQueueSnapshot, queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> { ) -> EcstoreResult<ManualTransitionJobRecord> {
let current = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?; let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
persist_manual_transition_job_progress_inner(api, job_id, current.0.lease_id, Some(current), report, queue_snapshot).await record.update_running_progress(report.clone(), queue_snapshot);
} save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await?;
renew_manual_transition_scope_admission_from_job(api, &record).await?;
pub async fn persist_manual_transition_job_progress_if_owned(
api: Arc<ECStore>,
job_id: Uuid,
expected_lease_id: Uuid,
report: &ManualTransitionRunReport,
queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> {
persist_manual_transition_job_progress_inner(api, job_id, expected_lease_id, None, report, queue_snapshot).await
}
async fn persist_manual_transition_job_progress_inner(
api: Arc<ECStore>,
job_id: Uuid,
expected_lease_id: Uuid,
current: Option<(ManualTransitionJobRecord, String)>,
report: &ManualTransitionRunReport,
queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> {
let record = update_manual_transition_job_record_from(api.clone(), job_id, Some(expected_lease_id), current, |record| {
if record.state != ManualTransitionJobState::Running {
return false;
}
record.update_running_progress(report.clone(), queue_snapshot);
true
})
.await?;
if record.state == ManualTransitionJobState::Running {
renew_manual_transition_scope_admission_from_job(api, &record).await?;
}
Ok(record) Ok(record)
} }
@@ -1844,58 +1661,25 @@ pub async fn renew_manual_transition_job_lease(
job_id: Uuid, job_id: Uuid,
queue_snapshot: ManualTransitionQueueSnapshot, queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> { ) -> EcstoreResult<ManualTransitionJobRecord> {
let current = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?; let (mut record, mut etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
renew_manual_transition_job_lease_inner(api, job_id, current.0.lease_id, Some(current), queue_snapshot).await if record.state == ManualTransitionJobState::Running {
} if record.scan_completed && queue_snapshot.queued == 0 && queue_snapshot.active == 0 {
record = reconcile_manual_transition_worker_results(api.clone(), job_id, queue_snapshot).await?;
pub async fn renew_manual_transition_job_lease_if_owned( if record.is_terminal() || !record.report.worker_transition_pending() {
api: Arc<ECStore>, return Ok(record);
job_id: Uuid,
expected_lease_id: Uuid,
queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> {
renew_manual_transition_job_lease_inner(api, job_id, expected_lease_id, None, queue_snapshot).await
}
async fn renew_manual_transition_job_lease_inner(
api: Arc<ECStore>,
job_id: Uuid,
expected_lease_id: Uuid,
current: Option<(ManualTransitionJobRecord, String)>,
queue_snapshot: ManualTransitionQueueSnapshot,
) -> EcstoreResult<ManualTransitionJobRecord> {
let (current, current_etag) = match current {
Some(current) => current,
None => load_manual_transition_job_record_with_etag(api.clone(), job_id).await?,
};
if current.lease_id != expected_lease_id {
return Err(Error::PreconditionFailed);
}
if current.state != ManualTransitionJobState::Running {
return Ok(current);
}
if current.scan_completed && queue_snapshot.queued == 0 && queue_snapshot.active == 0 {
return reconcile_manual_transition_worker_results_inner(api, job_id, Some(expected_lease_id), queue_snapshot, true)
.await;
}
let record = update_manual_transition_job_record_from(
api.clone(),
job_id,
Some(expected_lease_id),
Some((current, current_etag)),
|record| {
if record.state != ManualTransitionJobState::Running {
return false;
} }
(record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
}
let became_terminal = record.mark_unknown_if_worker_results_lost(queue_snapshot);
if !became_terminal {
record.renew_lease(queue_snapshot); record.renew_lease(queue_snapshot);
true }
}, save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await?;
) if became_terminal {
.await?; delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?;
if record.is_terminal() { } else {
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?; renew_manual_transition_scope_admission_from_job(api, &record).await?;
} else if record.state == ManualTransitionJobState::Running { }
renew_manual_transition_scope_admission_from_job(api, &record).await?;
} }
Ok(record) Ok(record)
} }
@@ -1904,31 +1688,15 @@ async fn renew_manual_transition_scope_admission_from_job(
api: Arc<ECStore>, api: Arc<ECStore>,
record: &ManualTransitionJobRecord, record: &ManualTransitionJobRecord,
) -> EcstoreResult<()> { ) -> EcstoreResult<()> {
for _ in 0..MANUAL_TRANSITION_JOB_CAS_RETRIES { if let Ok((admission, admission_etag)) =
let (admission, admission_etag) = load_manual_transition_scope_admission_with_etag(api.clone(), &record.scope_key).await
match load_manual_transition_scope_admission_with_etag(api.clone(), &record.scope_key).await { && admission.job_id == record.job_id
Ok(admission) => admission, && admission.lease_id == record.lease_id
Err(Error::ConfigNotFound) => return Ok(()), {
Err(err) => return Err(err), let renewed_admission = ManualTransitionScopeAdmission::from_job(record);
}; save_manual_transition_scope_admission_if_current(api, &renewed_admission, &admission_etag).await?;
if admission.job_id != record.job_id || admission.lease_id != record.lease_id {
return Err(Error::PreconditionFailed);
}
let mut renewed_admission = ManualTransitionScopeAdmission::from_job(record);
renewed_admission.lease_expires_at_unix_nanos = renewed_admission
.lease_expires_at_unix_nanos
.max(admission.lease_expires_at_unix_nanos);
renewed_admission.updated_at_unix_nanos = renewed_admission.updated_at_unix_nanos.max(admission.updated_at_unix_nanos);
if renewed_admission == admission {
return Ok(());
}
match save_manual_transition_scope_admission_if_current(api.clone(), &renewed_admission, &admission_etag).await {
Ok(()) => return Ok(()),
Err(Error::PreconditionFailed) => continue,
Err(err) => return Err(err),
}
} }
Err(Error::PreconditionFailed) Ok(())
} }
pub async fn delete_manual_transition_scope_admission_if_current( pub async fn delete_manual_transition_scope_admission_if_current(
@@ -2618,14 +2386,14 @@ mod tests {
} }
#[test] #[test]
fn manual_transition_job_record_control_plane_failure_does_not_count_tier_failure() { fn manual_transition_job_record_failure_counts_tier_failure() {
let options = ManualTransitionRunOptions::default(); let options = ManualTransitionRunOptions::default();
let mut record = ManualTransitionJobRecord::new(Uuid::new_v4(), "bucket", &options, TEST_OWNER); let mut record = ManualTransitionJobRecord::new(Uuid::new_v4(), "bucket", &options, TEST_OWNER);
record.fail("missing tier"); record.fail("missing tier");
assert_eq!(record.state, ManualTransitionJobState::Failed); assert_eq!(record.state, ManualTransitionJobState::Failed);
assert_eq!(record.report.tier_failure, 0); assert_eq!(record.report.tier_failure, 1);
assert_eq!(record.error.as_deref(), Some("missing tier")); assert_eq!(record.error.as_deref(), Some("missing tier"));
} }
+2 -12
View File
@@ -27,24 +27,12 @@ use crate::client::utils::base64_decode;
use crate::client::utils::base64_encode; use crate::client::utils::base64_encode;
use crate::client::{api_put_object::PutObjectOptions, api_s3_datatypes::ObjectPart}; use crate::client::{api_put_object::PutObjectOptions, api_s3_datatypes::ObjectPart};
use crate::{disk::DiskAPI, object_api::GetObjectReader}; use crate::{disk::DiskAPI, object_api::GetObjectReader};
// s3s::header has no CRC64NVME constant yet; the canonical RustFS copy lives
// in rustfs-utils' headers module.
use rustfs_utils::http::headers::AMZ_CHECKSUM_CRC64NVME;
use s3s::header::{ use s3s::header::{
X_AMZ_CHECKSUM_ALGORITHM, X_AMZ_CHECKSUM_CRC32, X_AMZ_CHECKSUM_CRC32C, X_AMZ_CHECKSUM_SHA1, X_AMZ_CHECKSUM_SHA256, X_AMZ_CHECKSUM_ALGORITHM, X_AMZ_CHECKSUM_CRC32, X_AMZ_CHECKSUM_CRC32C, X_AMZ_CHECKSUM_SHA1, X_AMZ_CHECKSUM_SHA256,
}; };
use enumset::{EnumSet, EnumSetType, enum_set}; use enumset::{EnumSet, EnumSetType, enum_set};
/// One of three deliberately separate checksum registries (backlog#1833):
/// this enum is the MinIO-port client's wire vocabulary and stops at the
/// standard S3 set (CRC64NVME is its newest member; the RustFS extensions do
/// not exist on this client path). The streaming-hash registry lives in
/// `rustfs_checksums::ChecksumAlgorithm` (crates/checksums/src/lib.rs) and
/// the on-disk xl.meta bitset in `rustfs_rio::ChecksumType`
/// (crates/rio/src/checksum.rs, varint bits are append-only). When adding an
/// algorithm, extend all three (or record why not) — they do not derive from
/// each other.
#[derive(Debug, EnumSetType, Default)] #[derive(Debug, EnumSetType, Default)]
#[enumset(repr = "u8")] #[enumset(repr = "u8")]
pub enum ChecksumMode { pub enum ChecksumMode {
@@ -69,6 +57,8 @@ lazy_static! {
static ref C_ChecksumFullObjectCRC32C: EnumSet<ChecksumMode> = static ref C_ChecksumFullObjectCRC32C: EnumSet<ChecksumMode> =
enum_set!(ChecksumMode::ChecksumCRC32C | ChecksumMode::ChecksumFullObject); enum_set!(ChecksumMode::ChecksumCRC32C | ChecksumMode::ChecksumFullObject);
} }
const AMZ_CHECKSUM_CRC64NVME: &str = "x-amz-checksum-crc64nvme";
impl ChecksumMode { impl ChecksumMode {
//pub const CRC64_NVME_POLYNOMIAL: i64 = 0xad93d23594c93659; //pub const CRC64_NVME_POLYNOMIAL: i64 = 0xad93d23594c93659;
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: cluster/RPC migration leaves transport capabilities staged for upcoming owners. // #730: cluster/RPC migration leaves transport capabilities staged for upcoming owners.
#![allow(dead_code)]
mod control_plane; mod control_plane;
pub(crate) mod rpc; pub(crate) mod rpc;
-1
View File
@@ -256,7 +256,6 @@ impl<S> ReplayScopeChannel<S> {
} }
} }
#[allow(dead_code, reason = "replay-state probe asserted by this file's tests (backlog#1823)")]
fn peer_replay_state(audience: &str) -> PeerReplayState { fn peer_replay_state(audience: &str) -> PeerReplayState {
PEER_REPLAY_STATES PEER_REPLAY_STATES
.lock() .lock()
@@ -31,7 +31,7 @@ use rustfs_config::{
DEFAULT_INTERNODE_DATA_TRANSPORT, ENV_RUSTFS_INTERNODE_DATA_TRANSPORT, INTERNODE_DATA_TRANSPORT_TCP, DEFAULT_INTERNODE_DATA_TRANSPORT, ENV_RUSTFS_INTERNODE_DATA_TRANSPORT, INTERNODE_DATA_TRANSPORT_TCP,
KNOWN_INTERNODE_DATA_TRANSPORT_BACKENDS, KNOWN_INTERNODE_DATA_TRANSPORT_BACKENDS,
}; };
use rustfs_rio::{ChunkReaderBox, HttpChunkReader, HttpReader, HttpWriter}; use rustfs_rio::{HttpReader, HttpWriter};
use sha2::{Digest, Sha256}; use sha2::{Digest, Sha256};
use std::collections::HashMap; use std::collections::HashMap;
use std::future::Future; use std::future::Future;
@@ -43,10 +43,6 @@ use tokio::io::{AsyncReadExt, AsyncWrite};
use tokio::sync::OnceCell; use tokio::sync::OnceCell;
use uuid::Uuid; use uuid::Uuid;
#[allow(
dead_code,
reason = "live in the cfg(not(test)) half of build_internode_data_transport_from_env (backlog#1823)"
)]
static INTERNODE_DATA_TRANSPORT: OnceLock<std::result::Result<Arc<dyn InternodeDataTransport>, String>> = OnceLock::new(); static INTERNODE_DATA_TRANSPORT: OnceLock<std::result::Result<Arc<dyn InternodeDataTransport>, String>> = OnceLock::new();
const READ_FILE_STREAM_PATH: &str = "/rustfs/rpc/read_file_stream"; const READ_FILE_STREAM_PATH: &str = "/rustfs/rpc/read_file_stream";
@@ -138,10 +134,6 @@ fn put_file_capability_status_is_legacy(status: u16) -> bool {
} }
#[derive(Debug, Clone, Copy, Eq, PartialEq)] #[derive(Debug, Clone, Copy, Eq, PartialEq)]
#[allow(
dead_code,
reason = "capability-negotiation seam; constructed only by transport test doubles (backlog#1823)"
)]
pub struct InternodeDataTransportCapabilities { pub struct InternodeDataTransportCapabilities {
/// Backend can open a streaming remote disk reader. /// Backend can open a streaming remote disk reader.
pub streaming_read: bool, pub streaming_read: bool,
@@ -158,10 +150,6 @@ pub struct InternodeDataTransportCapabilities {
} }
impl InternodeDataTransportCapabilities { impl InternodeDataTransportCapabilities {
#[allow(
dead_code,
reason = "capability-negotiation seam; used by transport test doubles (backlog#1823)"
)]
pub const fn tcp_http() -> Self { pub const fn tcp_http() -> Self {
Self { Self {
streaming_read: true, streaming_read: true,
@@ -233,17 +221,6 @@ pub struct NsScannerCapabilityRequest {
#[async_trait] #[async_trait]
pub trait InternodeDataTransport: Send + Sync + std::fmt::Debug { pub trait InternodeDataTransport: Send + Sync + std::fmt::Debug {
async fn open_read(&self, request: ReadStreamRequest) -> Result<FileReader>; async fn open_read(&self, request: ReadStreamRequest) -> Result<FileReader>;
async fn open_read_fresh(&self, request: ReadStreamRequest) -> Result<FileReader> {
self.open_read(request).await
}
/// Opens an owned-chunk stream when this transport can retain receive-buffer
/// ownership. `None` preserves the established `open_read` fallback.
async fn open_read_chunks(&self, _request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
Ok(None)
}
async fn open_read_chunks_fresh(&self, request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
self.open_read_chunks(request).await
}
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter>; async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter>;
async fn open_walk_dir(&self, request: WalkDirStreamRequest) -> Result<FileReader>; async fn open_walk_dir(&self, request: WalkDirStreamRequest) -> Result<FileReader>;
async fn open_ns_scanner(&self, _request: NsScannerStreamRequest) -> Result<FileReader> { async fn open_ns_scanner(&self, _request: NsScannerStreamRequest) -> Result<FileReader> {
@@ -252,12 +229,7 @@ pub trait InternodeDataTransport: Send + Sync + std::fmt::Debug {
async fn probe_ns_scanner(&self, _request: NsScannerCapabilityRequest) -> Result<Uuid> { async fn probe_ns_scanner(&self, _request: NsScannerCapabilityRequest) -> Result<Uuid> {
Err(Error::MethodNotAllowed) Err(Error::MethodNotAllowed)
} }
// Interface facet nobody calls yet: every transport implements both, but no
// caller negotiates on them. Kept for the internode transport split
// (backlog#1350); deleting them would delete the seam and six impls.
#[allow(dead_code, reason = "unused capability-negotiation facet (backlog#1823)")]
fn name(&self) -> &'static str; fn name(&self) -> &'static str;
#[allow(dead_code, reason = "unused capability-negotiation facet (backlog#1823)")]
fn capabilities(&self) -> InternodeDataTransportCapabilities; fn capabilities(&self) -> InternodeDataTransportCapabilities;
} }
@@ -275,34 +247,6 @@ impl InternodeDataTransport for TcpHttpInternodeDataTransport {
)) ))
} }
async fn open_read_fresh(&self, request: ReadStreamRequest) -> Result<FileReader> {
let url = build_read_file_stream_url(&request);
let mut headers = json_headers();
build_auth_headers(&url, &Method::GET, &mut headers)?;
Ok(Box::new(
HttpReader::new_fresh_connection_with_stall_timeout(url, Method::GET, headers, None, request.stall_timeout).await?,
))
}
async fn open_read_chunks(&self, request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
let url = build_read_file_stream_url(&request);
let mut headers = json_headers();
build_auth_headers(&url, &Method::GET, &mut headers)?;
Ok(Some(Box::new(
HttpChunkReader::new_with_stall_timeout(url, Method::GET, headers, None, request.stall_timeout).await?,
)))
}
async fn open_read_chunks_fresh(&self, request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
let url = build_read_file_stream_url(&request);
let mut headers = json_headers();
build_auth_headers(&url, &Method::GET, &mut headers)?;
Ok(Some(Box::new(
HttpChunkReader::new_fresh_connection_with_stall_timeout(url, Method::GET, headers, None, request.stall_timeout)
.await?,
)))
}
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter> { async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter> {
let server_epoch = self.put_file_auth_capability(&request.endpoint).await?; let server_epoch = self.put_file_auth_capability(&request.endpoint).await?;
let nonce = server_epoch.map(|_| Uuid::new_v4()); let nonce = server_epoch.map(|_| Uuid::new_v4());
@@ -712,10 +656,6 @@ fn build_internode_data_transport_result(
} }
} }
#[allow(
dead_code,
reason = "live in the cfg(test) half of build_internode_data_transport_from_env, which bypasses the process static (backlog#1823)"
)]
pub fn build_internode_data_transport(configured_transport: Option<&str>) -> Result<Arc<dyn InternodeDataTransport>> { pub fn build_internode_data_transport(configured_transport: Option<&str>) -> Result<Arc<dyn InternodeDataTransport>> {
build_internode_data_transport_result(configured_transport).map_err(Error::other) build_internode_data_transport_result(configured_transport).map_err(Error::other)
} }
@@ -854,6 +854,7 @@ impl PeerS3Client for LocalPeerS3Client {
#[derive(Debug)] #[derive(Debug)]
pub struct RemotePeerS3Client { pub struct RemotePeerS3Client {
pub node: Option<Node>,
pub pools: Option<Vec<usize>>, pub pools: Option<Vec<usize>>,
addr: String, addr: String,
/// Health tracker for connection monitoring /// Health tracker for connection monitoring
@@ -885,6 +886,7 @@ impl RemotePeerS3Client {
pub fn new(node: Option<Node>, pools: Option<Vec<usize>>) -> Self { pub fn new(node: Option<Node>, pools: Option<Vec<usize>>) -> Self {
let addr = node.as_ref().map(|v| v.url.to_string()).unwrap_or_default(); let addr = node.as_ref().map(|v| v.url.to_string()).unwrap_or_default();
let client = Self { let client = Self {
node,
pools, pools,
addr, addr,
health: Arc::new(DiskHealthTracker::new()), health: Arc::new(DiskHealthTracker::new()),
@@ -903,6 +905,10 @@ impl RemotePeerS3Client {
.map_err(|err| Error::other(format!("can not get client, err: {err}"))) .map_err(|err| Error::other(format!("can not get client, err: {err}")))
} }
pub fn get_addr(&self) -> String {
self.addr.clone()
}
/// Start health monitoring for the remote peer /// Start health monitoring for the remote peer
fn start_health_monitoring(&self) { fn start_health_monitoring(&self) {
let health = Arc::clone(&self.health); let health = Arc::clone(&self.health);
@@ -1202,10 +1208,6 @@ impl PeerS3Client for RemotePeerS3Client {
} }
} }
#[allow(
dead_code,
reason = "local bucket-heal path reached only by this file's tests (backlog#1823)"
)]
pub async fn heal_bucket_local(bucket: &str, opts: &HealOpts) -> Result<HealResultItem> { pub async fn heal_bucket_local(bucket: &str, opts: &HealOpts) -> Result<HealResultItem> {
let disks = clone_drives().await; let disks = clone_drives().await;
heal_bucket_local_on_disks(bucket, opts, disks).await heal_bucket_local_on_disks(bucket, opts, disks).await
@@ -1402,10 +1404,6 @@ pub(crate) async fn heal_bucket_local_on_disks(
} }
} }
#[allow(
dead_code,
reason = "reached only through heal_bucket_local, which only tests call (backlog#1823)"
)]
async fn clone_drives() -> Vec<Option<DiskStore>> { async fn clone_drives() -> Vec<Option<DiskStore>> {
runtime_sources::local_disk_entries().await runtime_sources::local_disk_entries().await
} }
@@ -1587,7 +1585,15 @@ mod tests {
} }
fn test_remote_peer(addr: &str) -> RemotePeerS3Client { fn test_remote_peer(addr: &str) -> RemotePeerS3Client {
let node = Node {
url: url::Url::parse(addr).expect("test peer URL should parse"),
pools: vec![0],
is_local: false,
grid_host: addr.to_string(),
};
RemotePeerS3Client { RemotePeerS3Client {
node: Some(node),
pools: Some(vec![0]), pools: Some(vec![0]),
addr: addr.to_string(), addr: addr.to_string(),
health: Arc::new(DiskHealthTracker::new()), health: Arc::new(DiskHealthTracker::new()),
File diff suppressed because it is too large Load Diff
@@ -48,6 +48,10 @@ impl RemoteClient {
Self { addr: endpoint } Self { addr: endpoint }
} }
pub fn from_url(url: url::Url) -> Self {
Self { addr: url.to_string() }
}
fn build_ping_request() -> PingRequest { fn build_ping_request() -> PingRequest {
let mut fbb = flatbuffers::FlatBufferBuilder::new(); let mut fbb = flatbuffers::FlatBufferBuilder::new();
let payload = fbb.create_vector(b"health-check"); let payload = fbb.create_vector(b"health-check");
+3
View File
@@ -46,6 +46,7 @@ use rustfs_config::{
SCANNER_SUB_SYS, SCANNER_SUB_SYS,
}; };
use rustfs_filemeta::FileInfo; use rustfs_filemeta::FileInfo;
use rustfs_utils::path::SLASH_SEPARATOR;
use serde_json::{Map, Value}; use serde_json::{Map, Value};
use std::collections::{HashMap, HashSet}; use std::collections::{HashMap, HashSet};
use std::sync::LazyLock; use std::sync::LazyLock;
@@ -199,6 +200,8 @@ pub const STORAGE_CLASS_SUB_SYS: &str = "storage_class";
pub const COMMA_SEPARATED_LISTS: &[&str] = &[rustfs_config::oidc::OIDC_SCOPES, rustfs_config::oidc::OIDC_OTHER_AUDIENCES]; pub const COMMA_SEPARATED_LISTS: &[&str] = &[rustfs_config::oidc::OIDC_SCOPES, rustfs_config::oidc::OIDC_OTHER_AUDIENCES];
static CONFIG_BUCKET: LazyLock<String> = LazyLock::new(|| format!("{RUSTFS_META_BUCKET}{SLASH_SEPARATOR}{CONFIG_PREFIX}"));
type ServerConfigDecryptFn = crate::bucket::migration::LegacyBlobDecryptFn; type ServerConfigDecryptFn = crate::bucket::migration::LegacyBlobDecryptFn;
static SERVER_CONFIG_DECRYPT_FN: LazyLock<RwLock<Option<ServerConfigDecryptFn>>> = LazyLock::new(|| RwLock::new(None)); static SERVER_CONFIG_DECRYPT_FN: LazyLock<RwLock<Option<ServerConfigDecryptFn>>> = LazyLock::new(|| RwLock::new(None));
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: configuration migration keeps legacy subsystem definitions available behind this module. // #730: configuration migration keeps legacy subsystem definitions available behind this module.
#![allow(dead_code)]
mod audit; mod audit;
pub mod com; pub mod com;
+20 -70
View File
@@ -101,7 +101,6 @@ const DEFAULT_RRS_STORAGE_CLASS: &str = "EC:1";
const ZERO_SET_DRIVE_COUNT_ERROR: &str = "set drive count must be greater than zero"; const ZERO_SET_DRIVE_COUNT_ERROR: &str = "set drive count must be greater than zero";
pub static DEFAULT_INLINE_BLOCK: usize = 128 * 1024; pub static DEFAULT_INLINE_BLOCK: usize = 128 * 1024;
const DEFAULT_INLINE_OBJECT_BUDGET: usize = 2 * DEFAULT_INLINE_BLOCK;
pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| { pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
let kvs = vec![ let kvs = vec![
@@ -151,8 +150,6 @@ pub struct Config {
optimize: Option<String>, optimize: Option<String>,
inline_block: usize, inline_block: usize,
initialized: bool, initialized: bool,
#[serde(default, skip_serializing_if = "std::ops::Not::not")]
inline_block_explicit: bool,
#[serde(skip)] #[serde(skip)]
standard_parities: Vec<PoolParity>, standard_parities: Vec<PoolParity>,
#[serde(skip)] #[serde(skip)]
@@ -189,10 +186,6 @@ impl Config {
/// A topology-bound lookup fails closed for unknown drive counts and for /// A topology-bound lookup fails closed for unknown drive counts and for
/// deserialized legacy configurations that have no pool topology. Legacy /// deserialized legacy configurations that have no pool topology. Legacy
/// callers retain scalar compatibility through [`Self::get_parity_for_sc`]. /// callers retain scalar compatibility through [`Self::get_parity_for_sc`].
#[allow(
dead_code,
reason = "per-set parity resolution asserted by this file's tests (backlog#1823)"
)]
pub(crate) fn parity_for_sc(&self, sc: &str, drives_per_set: usize) -> Option<usize> { pub(crate) fn parity_for_sc(&self, sc: &str, drives_per_set: usize) -> Option<usize> {
if !self.initialized { if !self.initialized {
return None; return None;
@@ -240,19 +233,17 @@ impl Config {
.map(|(pool_index, pool)| (pool_index, pool.drives_per_set)) .map(|(pool_index, pool)| (pool_index, pool.drives_per_set))
} }
pub fn should_inline(&self, shard_size: i64, data_shards: usize, versioned: bool) -> bool { pub fn should_inline(&self, shard_size: i64, versioned: bool) -> bool {
if shard_size < 0 || data_shards == 0 { if shard_size < 0 {
return false; return false;
} }
let shard_size = shard_size as usize; let shard_size = shard_size as usize;
// Keep the historical two-data-shard object budget while preventing
// wider EC layouts from multiplying the maximum inline object size. let mut inline_block = DEFAULT_INLINE_BLOCK;
let inline_block = if self.initialized && self.inline_block_explicit { if self.initialized {
self.inline_block inline_block = self.inline_block;
} else { }
(DEFAULT_INLINE_OBJECT_BUDGET / data_shards).min(DEFAULT_INLINE_BLOCK)
};
if versioned { if versioned {
shard_size <= inline_block / 8 shard_size <= inline_block / 8
@@ -401,7 +392,6 @@ fn lookup_config_for_pools_with_env(
} }
let optimize = overrides.optimize; let optimize = overrides.optimize;
let inline_block_explicit = overrides.inline_block.is_some();
let inline_block = if let Some(value) = overrides.inline_block { let inline_block = if let Some(value) = overrides.inline_block {
let block = value let block = value
.parse::<bytesize::ByteSize>() .parse::<bytesize::ByteSize>()
@@ -434,7 +424,6 @@ fn lookup_config_for_pools_with_env(
optimize, optimize,
inline_block, inline_block,
initialized: true, initialized: true,
inline_block_explicit,
standard_parities, standard_parities,
rrs_parities, rrs_parities,
}) })
@@ -552,26 +541,22 @@ mod tests {
} }
#[test] #[test]
fn should_inline_scales_default_threshold_by_data_shards() { fn should_inline_preserves_exact_default_shard_boundaries() {
let config = lookup_config_for_pools_with_env(&KVS::new(), &[3, 12], no_env_overrides()) let config = Config::default();
.expect("default inline policy should resolve for EC2+1 and EC8+4");
for (case, shard_size, data_shards, versioned, expected) in [ for (case, shard_size, versioned, expected) in [
("EC2+1 unversioned exact", 128 * 1024, 2, false, true), ("unversioned below", 128 * 1024 - 1, false, true),
("EC2+1 unversioned above", 128 * 1024 + 1, 2, false, false), ("unversioned exact", 128 * 1024, false, true),
("EC2+1 versioned exact", 16 * 1024, 2, true, true), ("unversioned above", 128 * 1024 + 1, false, false),
("EC2+1 versioned above", 16 * 1024 + 1, 2, true, false), ("versioned below", 16 * 1024 - 1, true, true),
("EC8+4 unversioned exact", 32 * 1024, 8, false, true), ("versioned exact", 16 * 1024, true, true),
("EC8+4 unversioned above", 32 * 1024 + 1, 8, false, false), ("versioned above", 16 * 1024 + 1, true, false),
("EC8+4 versioned exact", 4 * 1024, 8, true, true), ("negative", -1, false, false),
("EC8+4 versioned above", 4 * 1024 + 1, 8, true, false),
("negative", -1, 2, false, false),
("zero data shards", 0, 0, false, false),
] { ] {
assert_eq!( assert_eq!(
config.should_inline(shard_size, data_shards, versioned), config.should_inline(shard_size, versioned),
expected, expected,
"{case}: shard_size={shard_size}, data_shards={data_shards}, versioned={versioned}" "{case}: shard_size={shard_size}, versioned={versioned}"
); );
} }
} }
@@ -592,28 +577,13 @@ mod tests {
let shard_size = erasure.shard_file_size(object_size); let shard_size = erasure.shard_file_size(object_size);
assert_eq!(shard_size, expected_shard_size, "{case}: object_size={object_size}"); assert_eq!(shard_size, expected_shard_size, "{case}: object_size={object_size}");
assert_eq!( assert_eq!(
config.should_inline(shard_size, erasure.data_shards, versioned), config.should_inline(shard_size, versioned),
expected, expected,
"{case}: object_size={object_size}, shard_size={shard_size}, versioned={versioned}" "{case}: object_size={object_size}, shard_size={shard_size}, versioned={versioned}"
); );
} }
} }
#[test]
fn explicit_inline_block_preserves_fixed_per_shard_rollback() {
let overrides = StorageClassEnvOverrides {
inline_block: Some("128KiB".to_string()),
..Default::default()
};
let config = lookup_config_for_pools_with_env(&KVS::new(), &[12], overrides)
.expect("explicit inline block should resolve for EC8+4");
assert!(config.should_inline(128 * 1024, 8, false));
assert!(!config.should_inline(128 * 1024 + 1, 8, false));
assert!(config.should_inline(16 * 1024, 8, true));
assert!(!config.should_inline(16 * 1024 + 1, 8, true));
}
#[test] #[test]
fn write_capability_contract_only_accepts_implemented_layouts() { fn write_capability_contract_only_accepts_implemented_layouts() {
assert_eq!(SUPPORTED_WRITE_CLASSES, [STANDARD, RRS]); assert_eq!(SUPPORTED_WRITE_CLASSES, [STANDARD, RRS]);
@@ -807,7 +777,6 @@ mod tests {
let encoded = serde_json::to_string(&cfg).expect("config should serialize"); let encoded = serde_json::to_string(&cfg).expect("config should serialize");
assert!(!encoded.contains("standard_parities")); assert!(!encoded.contains("standard_parities"));
assert!(!encoded.contains("rrs_parities")); assert!(!encoded.contains("rrs_parities"));
assert!(!encoded.contains("inline_block_explicit"));
let decoded: Config = serde_json::from_str(&encoded).expect("legacy scalar config should deserialize"); let decoded: Config = serde_json::from_str(&encoded).expect("legacy scalar config should deserialize");
assert_eq!(decoded.get_parity_for_sc(STANDARD), Some(2)); assert_eq!(decoded.get_parity_for_sc(STANDARD), Some(2));
@@ -817,25 +786,6 @@ mod tests {
assert!(validate_parity(0, 0).is_err()); assert!(validate_parity(0, 0).is_err());
} }
#[test]
fn explicit_inline_block_survives_config_round_trip() {
let cfg = lookup_config_for_pools_with_env(
&KVS::new(),
&[12],
StorageClassEnvOverrides {
inline_block: Some("128KiB".to_string()),
..Default::default()
},
)
.expect("explicit inline block should resolve");
assert!(cfg.should_inline(100 * 1024, 8, false));
let encoded = serde_json::to_string(&cfg).expect("config should serialize");
assert!(encoded.contains("\"inline_block_explicit\":true"));
let decoded: Config = serde_json::from_str(&encoded).expect("explicit inline config should deserialize");
assert!(decoded.should_inline(100 * 1024, 8, false));
}
#[test] #[test]
fn lookup_config_reads_rrs_from_class_rrs_key() { fn lookup_config_reads_rrs_from_class_rrs_key() {
// Regression: kvs.get(RRS) used RRS="REDUCED_REDUNDANCY" instead of // Regression: kvs.get(RRS) used RRS="REDUCED_REDUNDANCY" instead of
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: pool coordination helpers are being migrated behind runtime owners. // #730: pool coordination helpers are being migrated behind runtime owners.
#![allow(dead_code)]
pub(crate) mod pools; pub(crate) mod pools;
pub(crate) mod sets; pub(crate) mod sets;
+9 -85
View File
@@ -226,7 +226,6 @@ fn ensure_decommission_start_rebalance_meta_allowed(meta: Option<&RebalanceMeta>
ensure_decommission_not_rebalancing(meta.is_some_and(is_rebalance_conflicting_with_decommission)) ensure_decommission_not_rebalancing(meta.is_some_and(is_rebalance_conflicting_with_decommission))
} }
#[allow(dead_code, reason = "leader precondition asserted by this file's tests (backlog#1823)")]
fn ensure_local_decommission_pool_leaders(endpoints: &EndpointServerPools, indices: &[usize]) -> Result<()> { fn ensure_local_decommission_pool_leaders(endpoints: &EndpointServerPools, indices: &[usize]) -> Result<()> {
for idx in indices { for idx in indices {
ensure_local_decommission_pool_leader(endpoints, *idx)?; ensure_local_decommission_pool_leader(endpoints, *idx)?;
@@ -1059,19 +1058,11 @@ fn should_cleanup_decommission_source_entry(decommissioned: usize, total_version
} }
#[derive(Debug, Clone, Copy, PartialEq, Eq)] #[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[allow(
dead_code,
reason = "terminal-state classification asserted by this file's tests (backlog#1823)"
)]
enum DecommissionTerminalState { enum DecommissionTerminalState {
Completed, Completed,
Failed, Failed,
} }
#[allow(
dead_code,
reason = "terminal-state classification asserted by this file's tests (backlog#1823)"
)]
fn classify_decommission_terminal_state(failed_items_present: bool) -> DecommissionTerminalState { fn classify_decommission_terminal_state(failed_items_present: bool) -> DecommissionTerminalState {
if failed_items_present { if failed_items_present {
DecommissionTerminalState::Failed DecommissionTerminalState::Failed
@@ -2275,19 +2266,15 @@ fn decommission_delete_marker_opts(
version: &rustfs_filemeta::FileInfo, version: &rustfs_filemeta::FileInfo,
version_id: Option<String>, version_id: Option<String>,
src_pool_idx: usize, src_pool_idx: usize,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
) -> ObjectOptions { ) -> ObjectOptions {
let version_suspended = version.version_id.is_none() && version_id.is_none();
ObjectOptions { ObjectOptions {
versioned: !version_suspended, versioned: true,
version_suspended, version_id,
version_id: version_id.or_else(|| version_suspended.then(|| uuid::Uuid::nil().to_string())),
mod_time: version.mod_time, mod_time: version.mod_time,
src_pool_idx, src_pool_idx,
data_movement: true, data_movement: true,
delete_marker: true, delete_marker: true,
skip_decommissioned: true, skip_decommissioned: true,
expected_bucket_incarnation_id,
delete_replication: version delete_replication: version
.replication_state_internal .replication_state_internal
.as_ref() .as_ref()
@@ -2312,7 +2299,6 @@ fn decommission_remote_tiered_opts(
version: &rustfs_filemeta::FileInfo, version: &rustfs_filemeta::FileInfo,
version_id: Option<String>, version_id: Option<String>,
src_pool_idx: usize, src_pool_idx: usize,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
) -> ObjectOptions { ) -> ObjectOptions {
ObjectOptions { ObjectOptions {
versioned: version_id.is_some(), versioned: version_id.is_some(),
@@ -2321,9 +2307,6 @@ fn decommission_remote_tiered_opts(
user_defined: version.metadata.clone(), user_defined: version.metadata.clone(),
src_pool_idx, src_pool_idx,
data_movement: true, data_movement: true,
include_part_checksums: true,
http_preconditions: Some(crate::data_movement::data_movement_target_precondition()),
expected_bucket_incarnation_id,
..Default::default() ..Default::default()
} }
} }
@@ -2822,7 +2805,6 @@ impl ECStore {
lifecycle_config: Option<BucketLifecycleConfiguration>, lifecycle_config: Option<BucketLifecycleConfiguration>,
object_lock_config: Option<ObjectLockConfiguration>, object_lock_config: Option<ObjectLockConfiguration>,
replication_config: Option<(ReplicationConfiguration, OffsetDateTime)>, replication_config: Option<(ReplicationConfiguration, OffsetDateTime)>,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
) -> Result<()> { ) -> Result<()> {
debug!( debug!(
event = EVENT_DECOMMISSION_ENTRY, event = EVENT_DECOMMISSION_ENTRY,
@@ -2852,11 +2834,6 @@ impl ECStore {
} }
decommission_cancel_signal_result(rx.is_cancelled())?; decommission_cancel_signal_result(rx.is_cancelled())?;
let bucket_incarnation_fence = match expected_bucket_incarnation_id {
Some(expected) => Some(self.acquire_bucket_incarnation_fence(&bucket, expected).await?),
None => None,
};
let mut fivs = load_decommission_entry_exact_versions(&set, &entry, &bucket, "file_info_versions").await?; let mut fivs = load_decommission_entry_exact_versions(&set, &entry, &bucket, "file_info_versions").await?;
fivs.versions fivs.versions
@@ -2917,7 +2894,7 @@ impl ECStore {
.delete_object( .delete_object(
bucket.as_str(), bucket.as_str(),
&version.name, &version.name,
decommission_delete_marker_opts(version, version_id.clone(), idx, expected_bucket_incarnation_id), decommission_delete_marker_opts(version, version_id.clone(), idx),
) )
.await .await
{ {
@@ -3007,7 +2984,7 @@ impl ECStore {
bucket.as_str(), bucket.as_str(),
&version.name, &version.name,
version, version,
&decommission_remote_tiered_opts(version, version_id.clone(), idx, expected_bucket_incarnation_id), &decommission_remote_tiered_opts(version, version_id.clone(), idx),
) )
.await .await
{ {
@@ -3079,11 +3056,7 @@ impl ECStore {
) )
.await?; .await?;
if let Err(err) = self if let Err(err) = self.clone().decommission_object(idx, bucket, rd).await {
.clone()
.decommission_object(idx, bucket, rd, expected_bucket_incarnation_id)
.await
{
if is_decommission_copy_cleanup_safe_error(&err) { if is_decommission_copy_cleanup_safe_error(&err) {
ignore = true; ignore = true;
cleanup_ignored = true; cleanup_ignored = true;
@@ -3160,9 +3133,6 @@ impl ECStore {
} }
if should_cleanup_decommission_source_entry(decommissioned, fivs.versions.len(), expired) { if should_cleanup_decommission_source_entry(decommissioned, fivs.versions.len(), expired) {
if bucket_incarnation_fence.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
return Err(Error::other("decommission bucket incarnation fence was lost before source cleanup"));
}
decommission_cancel_signal_result(rx.is_cancelled())?; decommission_cancel_signal_result(rx.is_cancelled())?;
self.save_decommission_entry_progress_stage( self.save_decommission_entry_progress_stage(
@@ -3187,12 +3157,6 @@ impl ECStore {
entry.name.as_str(), entry.name.as_str(),
&fivs, &fivs,
&cleanup_preflight_allowed_missing, &cleanup_preflight_allowed_missing,
data_movement::SourceCleanupBucketFence {
expected_incarnation_id: expected_bucket_incarnation_id,
lifecycle_guard: bucket_incarnation_fence
.as_ref()
.and_then(|guard| guard.namespace_lock_guard()),
},
"decommission", "decommission",
) )
.await .await
@@ -3304,11 +3268,6 @@ impl ECStore {
let mut lifecycle_config = None; let mut lifecycle_config = None;
let mut object_lock_config = None; let mut object_lock_config = None;
let mut replication_config = None; let mut replication_config = None;
let expected_bucket_incarnation_id = if bi.name == RUSTFS_META_BUCKET {
None
} else {
Some(self.bucket_incarnation_id_from_disk(&bi.name).await?)
};
if bi.name != RUSTFS_META_BUCKET { if bi.name != RUSTFS_META_BUCKET {
let _ = resolve_decommission_optional_bucket_config_result( let _ = resolve_decommission_optional_bucket_config_result(
@@ -3362,7 +3321,6 @@ impl ECStore {
let lifecycle_config = lifecycle_config.clone(); let lifecycle_config = lifecycle_config.clone();
let object_lock_config = object_lock_config.clone(); let object_lock_config = object_lock_config.clone();
let replication_config = replication_config.clone(); let replication_config = replication_config.clone();
let expected_bucket_incarnation_id = expected_bucket_incarnation_id;
let entry_error = entry_error.clone(); let entry_error = entry_error.clone();
let callback_rx = callback_rx.clone(); let callback_rx = callback_rx.clone();
@@ -3425,7 +3383,6 @@ impl ECStore {
lifecycle_config, lifecycle_config,
object_lock_config, object_lock_config,
replication_config, replication_config,
expected_bucket_incarnation_id,
) )
.await .await
{ {
@@ -4211,24 +4168,10 @@ impl ECStore {
} }
#[tracing::instrument(skip(self, rd))] #[tracing::instrument(skip(self, rd))]
async fn decommission_object( async fn decommission_object(self: Arc<Self>, pool_idx: usize, bucket: String, rd: GetObjectReader) -> Result<()> {
self: Arc<Self>,
pool_idx: usize,
bucket: String,
rd: GetObjectReader,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
) -> Result<()> {
warn!("decommission_object: start {} {}", &bucket, &rd.object_info.name); warn!("decommission_object: start {} {}", &bucket, &rd.object_info.name);
let object_name = rd.object_info.name.clone(); let object_name = rd.object_info.name.clone();
let result = data_movement::migrate_object( let result = data_movement::migrate_object(self, pool_idx, bucket.clone(), rd, "decommission_object").await;
self,
pool_idx,
bucket.clone(),
rd,
expected_bucket_incarnation_id,
"decommission_object",
)
.await;
if result.is_ok() { if result.is_ok() {
warn!("decommission_object: migrated {} {}", &bucket, &object_name); warn!("decommission_object: migrated {} {}", &bucket, &object_name);
} }
@@ -4404,8 +4347,7 @@ mod tests {
..Default::default() ..Default::default()
}; };
let incarnation = uuid::Uuid::new_v4(); let opts = decommission_delete_marker_opts(&version, Some("version-id".to_string()), 7);
let opts = decommission_delete_marker_opts(&version, Some("version-id".to_string()), 7, Some(incarnation));
let replication = opts.delete_replication.expect("replication state should be preserved"); let replication = opts.delete_replication.expect("replication state should be preserved");
assert!(opts.versioned); assert!(opts.versioned);
@@ -4415,25 +4357,11 @@ mod tests {
assert_eq!(opts.src_pool_idx, 7); assert_eq!(opts.src_pool_idx, 7);
assert_eq!(opts.version_id.as_deref(), Some("version-id")); assert_eq!(opts.version_id.as_deref(), Some("version-id"));
assert_eq!(opts.mod_time, Some(mod_time)); assert_eq!(opts.mod_time, Some(mod_time));
assert_eq!(opts.expected_bucket_incarnation_id, Some(incarnation));
assert_eq!(replication.replica_status, ReplicationStatusType::Replica); assert_eq!(replication.replica_status, ReplicationStatusType::Replica);
assert!(replication.delete_marker); assert!(replication.delete_marker);
assert_eq!(replication.replicate_decision_str, "existing"); assert_eq!(replication.replicate_decision_str, "existing");
} }
#[test]
fn decommission_delete_marker_opts_preserves_suspended_null_version() {
let version = rustfs_filemeta::FileInfo {
deleted: true,
..Default::default()
};
let opts = decommission_delete_marker_opts(&version, None, 7, None);
assert!(!opts.versioned);
assert!(opts.version_suspended);
assert_eq!(opts.version_id.as_deref(), Some(uuid::Uuid::nil().to_string().as_str()));
}
#[test] #[test]
fn test_decommission_object_migration_read_opts_are_raw_data_movement() { fn test_decommission_object_migration_read_opts_are_raw_data_movement() {
let opts = decommission_object_migration_read_opts(Some("vid-1".to_string())); let opts = decommission_object_migration_read_opts(Some("vid-1".to_string()));
@@ -4455,8 +4383,7 @@ mod tests {
..Default::default() ..Default::default()
}; };
let incarnation = uuid::Uuid::new_v4(); let opts = decommission_remote_tiered_opts(&version, Some("version-id".to_string()), 9);
let opts = decommission_remote_tiered_opts(&version, Some("version-id".to_string()), 9, Some(incarnation));
assert!(opts.versioned); assert!(opts.versioned);
assert!(opts.data_movement); assert!(opts.data_movement);
@@ -4464,9 +4391,6 @@ mod tests {
assert_eq!(opts.version_id.as_deref(), Some("version-id")); assert_eq!(opts.version_id.as_deref(), Some("version-id"));
assert_eq!(opts.mod_time, Some(mod_time)); assert_eq!(opts.mod_time, Some(mod_time));
assert_eq!(opts.user_defined.get("x-amz-meta-key").map(String::as_str), Some("value")); assert_eq!(opts.user_defined.get("x-amz-meta-key").map(String::as_str), Some("value"));
assert!(opts.include_part_checksums);
assert!(opts.http_preconditions.is_some());
assert_eq!(opts.expected_bucket_incarnation_id, Some(incarnation));
} }
#[test] #[test]
File diff suppressed because it is too large Load Diff
@@ -653,7 +653,6 @@ fn reconcile_servers_with_endpoint_topology(
(added, report) (added, report)
} }
#[allow(dead_code, reason = "exercised by this file's topology tests (backlog#1823)")]
fn server_topology_completeness_report( fn server_topology_completeness_report(
servers: &[ServerProperties], servers: &[ServerProperties],
endpoints: &EndpointServerPools, endpoints: &EndpointServerPools,
-62
View File
@@ -46,49 +46,21 @@ pub(crate) const GET_CODEC_STREAMING_OBJECT_CLASS_MULTIPART: &str = "multipart";
pub(crate) const GET_STAGE_DECODE: &str = "decode"; pub(crate) const GET_STAGE_DECODE: &str = "decode";
pub(crate) const GET_STAGE_EMIT: &str = "emit"; pub(crate) const GET_STAGE_EMIT: &str = "emit";
pub(crate) const GET_STAGE_FILL: &str = "fill"; pub(crate) const GET_STAGE_FILL: &str = "fill";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_FIRST_BYTE: &str = "first_byte"; pub(crate) const GET_STAGE_FIRST_BYTE: &str = "first_byte";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_FIRST_METADATA_RESPONSE: &str = "first_metadata_response"; pub(crate) const GET_STAGE_FIRST_METADATA_RESPONSE: &str = "first_metadata_response";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_FIRST_VALID_METADATA_RESPONSE: &str = "first_valid_metadata_response"; pub(crate) const GET_STAGE_FIRST_VALID_METADATA_RESPONSE: &str = "first_valid_metadata_response";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_FIRST_SHARD_READ: &str = "first_shard_read"; pub(crate) const GET_STAGE_FIRST_SHARD_READ: &str = "first_shard_read";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_FULL_BODY: &str = "full_body"; pub(crate) const GET_STAGE_FULL_BODY: &str = "full_body";
pub(crate) const GET_STAGE_INLINE_PREPARE: &str = "inline_prepare"; pub(crate) const GET_STAGE_INLINE_PREPARE: &str = "inline_prepare";
pub(crate) const GET_STAGE_LOCK_ACQUIRE: &str = "lock_acquire"; pub(crate) const GET_STAGE_LOCK_ACQUIRE: &str = "lock_acquire";
pub(crate) const GET_STAGE_METADATA: &str = "metadata"; pub(crate) const GET_STAGE_METADATA: &str = "metadata";
pub(crate) const GET_STAGE_METADATA_CACHE_LOOKUP: &str = "metadata_cache_lookup"; pub(crate) const GET_STAGE_METADATA_CACHE_LOOKUP: &str = "metadata_cache_lookup";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_METADATA_FANOUT: &str = "metadata_fanout"; pub(crate) const GET_STAGE_METADATA_FANOUT: &str = "metadata_fanout";
pub(crate) const GET_STAGE_METADATA_RESOLVE: &str = "metadata_resolve"; pub(crate) const GET_STAGE_METADATA_RESOLVE: &str = "metadata_resolve";
pub(crate) const GET_STAGE_OBJECT_INFO: &str = "object_info"; pub(crate) const GET_STAGE_OBJECT_INFO: &str = "object_info";
pub(crate) const GET_STAGE_OUTPUT_LOCK_WAIT: &str = "output_lock_wait"; pub(crate) const GET_STAGE_OUTPUT_LOCK_WAIT: &str = "output_lock_wait";
pub(crate) const GET_STAGE_OUTPUT_POLL: &str = "output_poll"; pub(crate) const GET_STAGE_OUTPUT_POLL: &str = "output_poll";
pub(crate) const GET_STAGE_PATH_DECISION: &str = "path_decision"; pub(crate) const GET_STAGE_PATH_DECISION: &str = "path_decision";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_QUORUM_REACHED: &str = "quorum_reached"; pub(crate) const GET_STAGE_QUORUM_REACHED: &str = "quorum_reached";
pub(crate) const GET_STAGE_RANGE: &str = "range"; pub(crate) const GET_STAGE_RANGE: &str = "range";
pub(crate) const GET_STAGE_READER_SETUP: &str = "reader_setup"; pub(crate) const GET_STAGE_READER_SETUP: &str = "reader_setup";
@@ -112,28 +84,12 @@ pub(crate) const GET_STAGE_READER_STREAM_FIRST_READ: &str = "reader_stream_first
pub(crate) const GET_STAGE_READER_TASK_BITROT_READER_INIT: &str = "reader_task_bitrot_reader_init"; pub(crate) const GET_STAGE_READER_TASK_BITROT_READER_INIT: &str = "reader_task_bitrot_reader_init";
pub(crate) const GET_STAGE_READER_TASK_FILE_OPEN: &str = "reader_task_file_open"; pub(crate) const GET_STAGE_READER_TASK_FILE_OPEN: &str = "reader_task_file_open";
pub(crate) const GET_STAGE_READER_TASK_READER_CONSTRUCTION: &str = "reader_task_reader_construction"; pub(crate) const GET_STAGE_READER_TASK_READER_CONSTRUCTION: &str = "reader_task_reader_construction";
pub(crate) const GET_STAGE_READ_VERSION_DECODE: &str = "read_version_decode";
pub(crate) const GET_STAGE_READ_VERSION_PATH_CHECK: &str = "read_version_path_check";
pub(crate) const GET_STAGE_READ_VERSION_PATH_RESOLVE: &str = "read_version_path_resolve";
pub(crate) const GET_STAGE_READ_VERSION_XLMETA_READ: &str = "read_version_xlmeta_read";
pub(crate) const GET_STAGE_RECONSTRUCT: &str = "reconstruct"; pub(crate) const GET_STAGE_RECONSTRUCT: &str = "reconstruct";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_RESPONSE_HANDOFF: &str = "response_handoff"; pub(crate) const GET_STAGE_RESPONSE_HANDOFF: &str = "response_handoff";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_SLOWEST_METADATA_RESPONSE: &str = "slowest_metadata_response"; pub(crate) const GET_STAGE_SLOWEST_METADATA_RESPONSE: &str = "slowest_metadata_response";
pub(crate) const GET_STAGE_STRIPE_READ: &str = "stripe_read"; pub(crate) const GET_STAGE_STRIPE_READ: &str = "stripe_read";
pub(crate) const GET_STAGE_STRIPE_READ_FIRST_SHARD: &str = "stripe_read_first_shard"; pub(crate) const GET_STAGE_STRIPE_READ_FIRST_SHARD: &str = "stripe_read_first_shard";
pub(crate) const GET_STAGE_STRIPE_READ_QUORUM: &str = "stripe_read_quorum"; pub(crate) const GET_STAGE_STRIPE_READ_QUORUM: &str = "stripe_read_quorum";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const GET_STAGE_BITROT_VERIFY: &str = "bitrot_verify"; pub(crate) const GET_STAGE_BITROT_VERIFY: &str = "bitrot_verify";
pub(crate) const GET_READER_BUFFER_OUTPUT: &str = "output"; pub(crate) const GET_READER_BUFFER_OUTPUT: &str = "output";
@@ -181,7 +137,6 @@ pub(crate) const GET_METADATA_CACHE_REASON_NO_LOCK: &str = "no_lock";
pub(crate) const GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED: &str = "not_found_or_expired"; pub(crate) const GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED: &str = "not_found_or_expired";
pub(crate) const GET_METADATA_CACHE_REASON_NOT_READ_DATA: &str = "not_read_data"; pub(crate) const GET_METADATA_CACHE_REASON_NOT_READ_DATA: &str = "not_read_data";
pub(crate) const GET_METADATA_CACHE_REASON_PART_NUMBER: &str = "part_number"; pub(crate) const GET_METADATA_CACHE_REASON_PART_NUMBER: &str = "part_number";
pub(crate) const GET_METADATA_CACHE_REASON_PART_CHECKSUMS: &str = "part_checksums";
pub(crate) const GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ: &str = "raw_data_movement_read"; pub(crate) const GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ: &str = "raw_data_movement_read";
pub(crate) const GET_METADATA_CACHE_REASON_STALE_PUBLICATION: &str = "stale_publication"; pub(crate) const GET_METADATA_CACHE_REASON_STALE_PUBLICATION: &str = "stale_publication";
pub(crate) const GET_METADATA_CACHE_REASON_USABLE: &str = "usable"; pub(crate) const GET_METADATA_CACHE_REASON_USABLE: &str = "usable";
@@ -199,20 +154,8 @@ pub(crate) const GET_METADATA_EARLY_STOP_REASON_VERSION_NOT_FOUND: &str = "versi
pub(crate) const GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM: &str = "version_match_quorum"; pub(crate) const GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM: &str = "version_match_quorum";
/// Early-stop active state labels /// Early-stop active state labels
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const EARLY_STOP_ACTIVE_HIT: &str = "hit"; pub(crate) const EARLY_STOP_ACTIVE_HIT: &str = "hit";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const EARLY_STOP_ACTIVE_MISS: &str = "miss"; pub(crate) const EARLY_STOP_ACTIVE_MISS: &str = "miss";
#[allow(
dead_code,
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
)]
pub(crate) const EARLY_STOP_ACTIVE_DISABLED: &str = "disabled"; pub(crate) const EARLY_STOP_ACTIVE_DISABLED: &str = "disabled";
#[derive(Clone, Copy, Debug, Eq, PartialEq)] #[derive(Clone, Copy, Debug, Eq, PartialEq)]
@@ -498,10 +441,6 @@ mod tests {
assert_eq!(GET_STAGE_QUORUM_REACHED, "quorum_reached"); assert_eq!(GET_STAGE_QUORUM_REACHED, "quorum_reached");
assert_eq!(GET_STAGE_RANGE, "range"); assert_eq!(GET_STAGE_RANGE, "range");
assert_eq!(GET_STAGE_READER_SETUP, "reader_setup"); assert_eq!(GET_STAGE_READER_SETUP, "reader_setup");
assert_eq!(GET_STAGE_READ_VERSION_DECODE, "read_version_decode");
assert_eq!(GET_STAGE_READ_VERSION_PATH_CHECK, "read_version_path_check");
assert_eq!(GET_STAGE_READ_VERSION_PATH_RESOLVE, "read_version_path_resolve");
assert_eq!(GET_STAGE_READ_VERSION_XLMETA_READ, "read_version_xlmeta_read");
assert_eq!(GET_STAGE_RECONSTRUCT, "reconstruct"); assert_eq!(GET_STAGE_RECONSTRUCT, "reconstruct");
assert_eq!(GET_STAGE_RESPONSE_HANDOFF, "response_handoff"); assert_eq!(GET_STAGE_RESPONSE_HANDOFF, "response_handoff");
assert_eq!(GET_STAGE_SLOWEST_METADATA_RESPONSE, "slowest_metadata_response"); assert_eq!(GET_STAGE_SLOWEST_METADATA_RESPONSE, "slowest_metadata_response");
@@ -541,7 +480,6 @@ mod tests {
assert_eq!(GET_METADATA_CACHE_REASON_NO_LOCK, "no_lock"); assert_eq!(GET_METADATA_CACHE_REASON_NO_LOCK, "no_lock");
assert_eq!(GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED, "not_found_or_expired"); assert_eq!(GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED, "not_found_or_expired");
assert_eq!(GET_METADATA_CACHE_REASON_NOT_READ_DATA, "not_read_data"); assert_eq!(GET_METADATA_CACHE_REASON_NOT_READ_DATA, "not_read_data");
assert_eq!(GET_METADATA_CACHE_REASON_PART_CHECKSUMS, "part_checksums");
assert_eq!(GET_METADATA_CACHE_REASON_PART_NUMBER, "part_number"); assert_eq!(GET_METADATA_CACHE_REASON_PART_NUMBER, "part_number");
assert_eq!(GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ, "raw_data_movement_read"); assert_eq!(GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ, "raw_data_movement_read");
assert_eq!(GET_METADATA_CACHE_REASON_STALE_PUBLICATION, "stale_publication"); assert_eq!(GET_METADATA_CACHE_REASON_STALE_PUBLICATION, "stale_publication");
+2
View File
@@ -13,6 +13,8 @@
// limitations under the License. // limitations under the License.
// #730: diagnostics constants are staged for request-path telemetry migration. // #730: diagnostics constants are staged for request-path telemetry migration.
#![allow(dead_code)]
pub(crate) mod admin_server_info; pub(crate) mod admin_server_info;
pub(crate) mod get; pub(crate) mod get;
pub(crate) mod pool;
+30
View File
@@ -0,0 +1,30 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! BytesPool metric label constants.
//!
//! These constants are used when recording pool acquisition and return
//! metrics to avoid string allocations and ensure label consistency.
/// BytesPool tier labels
pub const POOL_TIER_SMALL: &str = "small";
pub const POOL_TIER_MEDIUM: &str = "medium";
pub const POOL_TIER_LARGE: &str = "large";
pub const POOL_TIER_XLARGE: &str = "xlarge";
/// BytesPool outcome labels
pub const POOL_OUTCOME_HIT: &str = "hit";
pub const POOL_OUTCOME_MISS: &str = "miss";
pub const POOL_OUTCOME_RECYCLED: &str = "recycled";
pub const POOL_OUTCOME_DROPPED: &str = "dropped";
-15
View File
@@ -2022,21 +2022,6 @@ impl DiskAPI for LocalDiskWrapper {
.await .await
} }
async fn read_file_stream_chunks(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
) -> Result<Option<rustfs_rio::ChunkReaderBox>> {
self.track_disk_health_with_op(
"read_file_stream_chunks",
|| async { self.disk.read_file_stream_chunks(volume, path, offset, length).await },
get_max_timeout_duration(),
)
.await
}
async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<bytes::Bytes> { async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<bytes::Bytes> {
self.track_disk_health_with_op( self.track_disk_health_with_op(
"read_file_mmap_copy", "read_file_mmap_copy",
+26 -6
View File
@@ -113,9 +113,6 @@ pub enum DiskError {
#[error("bit-rot hash algorithm is invalid")] #[error("bit-rot hash algorithm is invalid")]
BitrotHashAlgoInvalid, BitrotHashAlgoInvalid,
/// Never constructed locally by RustFS (only reachable through wire
/// decoding, and no current node sends it). The wire code is kept for
/// cross-version compatibility — do not renumber or remove (backlog#1831).
#[error("Rename across devices not allowed, please fix your backend configuration")] #[error("Rename across devices not allowed, please fix your backend configuration")]
CrossDeviceLink, CrossDeviceLink,
@@ -146,9 +143,6 @@ pub enum DiskError {
#[error("io error {0}")] #[error("io error {0}")]
Io(#[source] io::Error), Io(#[source] io::Error),
/// Never constructed locally by RustFS (only reachable through wire
/// decoding, and no current node sends it). The wire code is kept for
/// cross-version compatibility — do not renumber or remove (backlog#1831).
#[error("source stalled")] #[error("source stalled")]
SourceStalled, SourceStalled,
@@ -648,6 +642,19 @@ impl Hash for DiskError {
// is currently commented out to avoid complexity. These can be re-enabled // is currently commented out to avoid complexity. These can be re-enabled
// when needed for specific disk quorum checking and error aggregation logic. // when needed for specific disk quorum checking and error aggregation logic.
/// Bitrot errors
#[derive(Debug, thiserror::Error)]
pub enum BitrotErrorType {
#[error("bitrot checksum verification failed")]
BitrotChecksumMismatch { expected: String, got: String },
}
impl From<BitrotErrorType> for DiskError {
fn from(e: BitrotErrorType) -> Self {
DiskError::other(e)
}
}
/// Context wrapper for file access errors /// Context wrapper for file access errors
#[derive(Debug, thiserror::Error)] #[derive(Debug, thiserror::Error)]
pub struct FileAccessDeniedWithContext { pub struct FileAccessDeniedWithContext {
@@ -862,6 +869,19 @@ mod tests {
let _disk_error: DiskError = json_error.into(); let _disk_error: DiskError = json_error.into();
} }
#[test]
fn test_bitrot_error_type() {
let bitrot_error = BitrotErrorType::BitrotChecksumMismatch {
expected: "abc123".to_string(),
got: "def456".to_string(),
};
assert!(bitrot_error.to_string().contains("bitrot checksum verification failed"));
let disk_error: DiskError = bitrot_error.into();
assert!(matches!(disk_error, DiskError::Io(_)));
}
#[test] #[test]
fn test_file_access_denied_with_context() { fn test_file_access_denied_with_context() {
let path = PathBuf::from("/test/path"); let path = PathBuf::from("/test/path");
+25 -191
View File
@@ -15,11 +15,6 @@
use crate::config::storageclass::DEFAULT_INLINE_BLOCK; use crate::config::storageclass::DEFAULT_INLINE_BLOCK;
use crate::crash_inject::{self, CrashPoint}; use crate::crash_inject::{self, CrashPoint};
use crate::data_usage::local_snapshot::ensure_data_usage_layout; use crate::data_usage::local_snapshot::ensure_data_usage_layout;
use crate::diagnostics::get::{
GET_OBJECT_PATH_INTERNAL_META, GET_OBJECT_PATH_LEGACY_DUPLEX, GET_STAGE_READ_VERSION_DECODE,
GET_STAGE_READ_VERSION_PATH_CHECK, GET_STAGE_READ_VERSION_PATH_RESOLVE, GET_STAGE_READ_VERSION_XLMETA_READ,
get_stage_timer_if_enabled, record_get_stage_duration_if_enabled,
};
#[cfg(test)] #[cfg(test)]
use crate::disk::HEALING_MARKER_PATH; use crate::disk::HEALING_MARKER_PATH;
use crate::disk::disk_store::{get_drive_walkdir_stall_timeout, get_object_disk_read_timeout}; use crate::disk::disk_store::{get_drive_walkdir_stall_timeout, get_object_disk_read_timeout};
@@ -9179,7 +9174,7 @@ impl DiskAPI for LocalDisk {
if let Some(src_file_path_parent) = src_file_path.parent() { if let Some(src_file_path_parent) = src_file_path.parent() {
if src_volume != super::RUSTFS_META_MULTIPART_BUCKET { if src_volume != super::RUSTFS_META_MULTIPART_BUCKET {
let _ = std::fs::remove_dir(src_file_path_parent); let _ = remove_std(src_file_path_parent);
} else { } else {
let _ = self let _ = self
.delete_file(&dst_volume_dir, &src_file_path_parent.to_path_buf(), true, false) .delete_file(&dst_volume_dir, &src_file_path_parent.to_path_buf(), true, false)
@@ -9504,7 +9499,7 @@ impl DiskAPI for LocalDisk {
if let Some(ref cleanup) = cleanup_path { if let Some(ref cleanup) = cleanup_path {
let _ = self.delete_file(&dst_volume_dir, cleanup, true, false).await; let _ = self.delete_file(&dst_volume_dir, cleanup, true, false).await;
} else if let Some(parent) = src_file_path.parent() { } else if let Some(parent) = src_file_path.parent() {
let _ = std::fs::remove_dir(parent); let _ = remove_std(parent);
} }
// Heal reuses a version's `data_dir` and lands the rebuilt shard on // Heal reuses a version's `data_dir` and lands the rebuilt shard on
@@ -9845,12 +9840,6 @@ impl DiskAPI for LocalDisk {
opts: &ReadOptions, opts: &ReadOptions,
) -> Result<FileInfo> { ) -> Result<FileInfo> {
crate::hp_guard!("LocalDisk::read_version"); crate::hp_guard!("LocalDisk::read_version");
let stage_metrics_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
let metrics_path = if stage_metrics_enabled && crate::bucket::utils::is_meta_bucketname(volume) {
GET_OBJECT_PATH_INTERNAL_META
} else {
GET_OBJECT_PATH_LEGACY_DUPLEX
};
if !org_volume.is_empty() { if !org_volume.is_empty() {
let org_volume_path = self.io_get_bucket_path(org_volume)?; let org_volume_path = self.io_get_bucket_path(org_volume)?;
if !skip_access_checks(org_volume) { if !skip_access_checks(org_volume) {
@@ -9860,46 +9849,36 @@ impl DiskAPI for LocalDisk {
} }
} }
let path_resolve_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let file_path = self.io_get_object_path(volume, path)?; let file_path = self.io_get_object_path(volume, path)?;
let volume_dir = self.io_get_bucket_path(volume)?; let volume_dir = self.io_get_bucket_path(volume)?;
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_READ_VERSION_PATH_RESOLVE, path_resolve_start);
let path_check_start = get_stage_timer_if_enabled(stage_metrics_enabled);
check_path_length(file_path.to_string_lossy().as_ref())?; check_path_length(file_path.to_string_lossy().as_ref())?;
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_READ_VERSION_PATH_CHECK, path_check_start);
let read_data = opts.read_data; let read_data = opts.read_data;
let xlmeta_read_start = get_stage_timer_if_enabled(stage_metrics_enabled); let (data, _) = self
let raw_read_result = self.read_raw(volume, volume_dir.clone(), file_path, read_data).await; .read_raw(volume, volume_dir.clone(), file_path, read_data)
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_READ_VERSION_XLMETA_READ, xlmeta_read_start); .await
let (data, _) = raw_read_result.map_err(|e| { .map_err(|e| {
if e == DiskError::FileNotFound && !version_id.is_empty() { if e == DiskError::FileNotFound && !version_id.is_empty() {
DiskError::FileVersionNotFound DiskError::FileVersionNotFound
} else { } else {
e e
} }
})?; })?;
let decode_start = get_stage_timer_if_enabled(stage_metrics_enabled); let mut fi = get_file_info(
let file_info_result: Result<FileInfo> = (|| { &data,
let fi = get_file_info( volume,
&data, path,
volume, version_id,
path, FileInfoOpts {
version_id, data: read_data,
FileInfoOpts { include_free_versions: opts.incl_free_versions,
data: read_data, },
include_free_versions: opts.incl_free_versions, )?;
include_part_checksums: false,
}, fi.validate_for_metadata_read()?;
)?;
fi.validate_for_metadata_read()?;
Ok(fi)
})();
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_READ_VERSION_DECODE, decode_start);
let mut fi = file_info_result?;
if fi.is_canonical_delete_marker() { if fi.is_canonical_delete_marker() {
return Ok(fi); return Ok(fi);
} }
@@ -10582,108 +10561,6 @@ mod test {
meta.marshal_msg().expect("test metadata should encode") meta.marshal_msg().expect("test metadata should encode")
} }
#[test]
#[serial_test::serial]
fn read_version_records_local_metadata_stage_breakdown() {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("test runtime should be created");
let recorder = crate::test_metrics::CapturingRecorder::default();
let previous_gate = rustfs_io_metrics::get_stage_metrics_enabled();
rustfs_io_metrics::set_get_stage_metrics_enabled(true);
metrics::with_local_recorder(&recorder, || {
runtime.block_on(async {
let dir = tempfile::tempdir().expect("temp dir should be created");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "stage-breakdown";
ensure_test_volume(&disk, bucket).await;
let object_dir = dir.path().join(bucket).join(object);
fs::create_dir_all(&object_dir)
.await
.expect("object directory should be created");
fs::write(
object_dir.join(STORAGE_FORMAT_FILE),
test_meta(test_file_info(object, Uuid::new_v4(), None, Some(Bytes::from_static(b"inline")))),
)
.await
.expect("object metadata should be written");
disk.read_version(
"",
bucket,
object,
"",
&ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("read_version should succeed");
let meta_object = "stage-breakdown-meta";
let meta_object_dir = dir.path().join(RUSTFS_META_BUCKET).join(meta_object);
fs::create_dir_all(&meta_object_dir)
.await
.expect("internal metadata object directory should be created");
fs::write(
meta_object_dir.join(STORAGE_FORMAT_FILE),
test_meta(test_file_info(meta_object, Uuid::new_v4(), None, Some(Bytes::from_static(b"meta")))),
)
.await
.expect("internal metadata should be written");
disk.read_version(
"",
RUSTFS_META_BUCKET,
meta_object,
"",
&ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("internal metadata read_version should succeed");
});
});
rustfs_io_metrics::set_get_stage_metrics_enabled(previous_gate);
for stage in [
GET_STAGE_READ_VERSION_PATH_RESOLVE,
GET_STAGE_READ_VERSION_PATH_CHECK,
GET_STAGE_READ_VERSION_XLMETA_READ,
GET_STAGE_READ_VERSION_DECODE,
] {
assert_eq!(
recorder
.histogram_values(
"rustfs_io_get_object_stage_duration_seconds",
&[("path", GET_OBJECT_PATH_LEGACY_DUPLEX), ("stage", stage)]
)
.len(),
1,
"{stage} should be recorded once for user-bucket LocalDisk::read_version"
);
assert_eq!(
recorder
.histogram_values(
"rustfs_io_get_object_stage_duration_seconds",
&[("path", GET_OBJECT_PATH_INTERNAL_META), ("stage", stage)]
)
.len(),
1,
"{stage} should be recorded once for internal-meta LocalDisk::read_version"
);
}
}
#[test] #[test]
fn inline_metadata_rollback_dir_avoids_real_data_dir_collision() { fn inline_metadata_rollback_dir_avoids_real_data_dir_collision() {
let target_version = Uuid::parse_str("11111111-2222-3333-4444-555555555555").expect("version id should parse"); let target_version = Uuid::parse_str("11111111-2222-3333-4444-555555555555").expect("version id should parse");
@@ -12562,10 +12439,6 @@ mod test {
.join(RUSTFS_META_TMP_BUCKET) .join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object) .join(tmp_object)
.join(new_data_dir.to_string()); .join(new_data_dir.to_string());
let tmp_parent = tmp_data_dir
.parent()
.expect("tmp data dir should have a parent")
.to_path_buf();
fs::create_dir_all(&tmp_data_dir) fs::create_dir_all(&tmp_data_dir)
.await .await
.expect("new tmp data dir should be created"); .expect("new tmp data dir should be created");
@@ -12577,10 +12450,6 @@ mod test {
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object) disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await .await
.expect("rename_data should commit"); .expect("rename_data should commit");
assert!(
!tmp_parent.exists(),
"successful non-inline commit should remove the empty staging parent"
);
// The tmp xl.meta write point uses SyncMode::FileOnly: its parent dir // The tmp xl.meta write point uses SyncMode::FileOnly: its parent dir
// ({tmp}/{tmp_object}) must not be fsynced. // ({tmp}/{tmp_object}) must not be fsynced.
@@ -12785,9 +12654,6 @@ mod test {
let tmp_object = "tmp-new-inline"; let tmp_object = "tmp-new-inline";
ensure_test_volume(&disk, bucket).await; ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await; ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let tmp_parent = disk
.get_object_path(RUSTFS_META_TMP_BUCKET, tmp_object)
.expect("tmp parent should resolve");
let _mode = durability_mode_override::set(DurabilityMode::Strict); let _mode = durability_mode_override::set(DurabilityMode::Strict);
let version_id = Uuid::parse_str("99999999-9999-9999-9999-999999999999").expect("version id should parse"); let version_id = Uuid::parse_str("99999999-9999-9999-9999-999999999999").expect("version id should parse");
@@ -12796,7 +12662,6 @@ mod test {
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object) disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await .await
.expect("inline rename_data should commit the new object"); .expect("inline rename_data should commit the new object");
assert!(!tmp_parent.exists(), "successful inline commit should remove the empty staging parent");
let bucket_dir = disk.get_bucket_path(bucket).expect("bucket path should resolve"); let bucket_dir = disk.get_bucket_path(bucket).expect("bucket path should resolve");
let prefix_dir = disk.get_object_path(bucket, "prefix").expect("prefix path should resolve"); let prefix_dir = disk.get_object_path(bucket, "prefix").expect("prefix path should resolve");
@@ -12820,34 +12685,6 @@ mod test {
); );
} }
#[tokio::test]
async fn rename_data_inline_preserves_non_empty_staging_parent() {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "inline-staging-sentinel-bucket";
let object = "inline-object";
let tmp_object = "inline-stage-with-sentinel";
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let tmp_parent = disk
.get_object_path(RUSTFS_META_TMP_BUCKET, tmp_object)
.expect("tmp parent should resolve");
fs::create_dir_all(&tmp_parent).await.expect("tmp parent should be created");
let sentinel = tmp_parent.join("sentinel");
fs::write(&sentinel, b"keep").await.expect("sentinel should be written");
let fi = test_file_info(object, Uuid::new_v4(), None, Some(Bytes::from_static(b"inline-payload")));
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, fi, bucket, object)
.await
.expect("non-empty staging cleanup must not negate the committed object");
assert_eq!(fs::read(&sentinel).await.expect("sentinel should remain"), b"keep");
}
#[cfg(unix)] #[cfg(unix)]
#[tokio::test(flavor = "multi_thread", worker_threads = 2)] #[tokio::test(flavor = "multi_thread", worker_threads = 2)]
#[allow(clippy::await_holding_lock)] #[allow(clippy::await_holding_lock)]
@@ -13022,10 +12859,7 @@ mod test {
.expect("non-inline rename_data should commit"); .expect("non-inline rename_data should commit");
assert!(!replacement_dir.exists(), "the destination object directory must not be replaced"); assert!(!replacement_dir.exists(), "the destination object directory must not be replaced");
assert!( assert!(staging_parent.exists(), "the guarded staging parent must retain its identity");
!staging_parent.exists(),
"successful commit should remove the empty staging parent after releasing its guard"
);
assert!( assert!(
!replacement_staging_parent.exists(), !replacement_staging_parent.exists(),
"the staging parent must not be replaced between data and metadata publication" "the staging parent must not be replaced between data and metadata publication"
-26
View File
@@ -65,7 +65,6 @@ use error::{Error, Result};
use local::LocalDisk; use local::LocalDisk;
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo}; use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_madmin::info_commands::DiskMetrics; use rustfs_madmin::info_commands::DiskMetrics;
use rustfs_rio::ChunkReaderBox;
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::{fmt::Debug, path::PathBuf, sync::Arc, time::Duration}; use std::{fmt::Debug, path::PathBuf, sync::Arc, time::Duration};
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -428,19 +427,6 @@ impl DiskAPI for Disk {
} }
} }
async fn read_file_stream_chunks(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
) -> Result<Option<ChunkReaderBox>> {
match self {
Disk::Local(_) => Ok(None),
Disk::Remote(remote_disk) => remote_disk.read_file_stream_chunks(volume, path, offset, length).await,
}
}
#[tracing::instrument(level = "trace", skip_all)] #[tracing::instrument(level = "trace", skip_all)]
async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes> { async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes> {
match self { match self {
@@ -879,18 +865,6 @@ pub trait DiskAPI: Debug + Send + Sync + 'static {
async fn read_file(&self, volume: &str, path: &str) -> Result<FileReader>; async fn read_file(&self, volume: &str, path: &str) -> Result<FileReader>;
async fn read_file_stream(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<FileReader>; async fn read_file_stream(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<FileReader>;
/// Returns an owned-chunk stream when the backing transport can preserve
/// receive-buffer ownership. `None` retains the ordinary reader path.
async fn read_file_stream_chunks(
&self,
_volume: &str,
_path: &str,
_offset: usize,
_length: usize,
) -> Result<Option<ChunkReaderBox>> {
Ok(None)
}
/// File read using mmap-then-copy on Unix or an efficient read on non-Unix. /// File read using mmap-then-copy on Unix or an efficient read on non-Unix.
async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes>; async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes>;
@@ -26,7 +26,6 @@ pub(crate) const GET_RECONSTRUCT_OUTCOME_SKIP_DATA_COMPLETE: &str = "skip_data_c
pub(crate) const GET_RECONSTRUCT_OUTCOME_SKIP_EMPTY_PAYLOAD: &str = "skip_empty_payload"; pub(crate) const GET_RECONSTRUCT_OUTCOME_SKIP_EMPTY_PAYLOAD: &str = "skip_empty_payload";
pub(crate) trait DecodeWorkspace: Send + Sync + 'static { pub(crate) trait DecodeWorkspace: Send + Sync + 'static {
#[allow(dead_code, reason = "workspace width asserted by decode_reader tests (backlog#1823)")]
fn shard_len(&self) -> usize; fn shard_len(&self) -> usize;
} }
@@ -34,14 +33,11 @@ pub(crate) trait ErasureDecodeEngine: Send + Sync + 'static {
type Workspace: DecodeWorkspace; type Workspace: DecodeWorkspace;
fn data_shards(&self) -> usize; fn data_shards(&self) -> usize;
#[allow(dead_code, reason = "engine trait facet asserted by decode_reader tests (backlog#1823)")]
fn parity_shards(&self) -> usize; fn parity_shards(&self) -> usize;
fn block_size(&self) -> usize; fn block_size(&self) -> usize;
fn engine_name(&self) -> &'static str; fn engine_name(&self) -> &'static str;
#[allow(dead_code, reason = "engine trait facet asserted by decode_reader tests (backlog#1823)")]
fn supports_progressive_decode(&self) -> bool; fn supports_progressive_decode(&self) -> bool;
#[allow(dead_code, reason = "engine trait facet asserted by decode_reader tests (backlog#1823)")]
fn supports_aligned_shards(&self) -> bool; fn supports_aligned_shards(&self) -> bool;
fn prepare_workspace(&self, shard_len: usize) -> io::Result<Self::Workspace>; fn prepare_workspace(&self, shard_len: usize) -> io::Result<Self::Workspace>;
@@ -24,7 +24,6 @@ impl RustfsCodecDecodeWorkspace {
} }
#[inline] #[inline]
#[allow(dead_code, reason = "workspace width asserted by decode_reader tests (backlog#1823)")]
pub(crate) fn shard_len(&self) -> usize { pub(crate) fn shard_len(&self) -> usize {
self.shard_len self.shard_len
} }
@@ -77,13 +76,6 @@ impl ShardBufferPool {
self.buffers[index] = Some(buf); self.buffers[index] = Some(buf);
} }
#[cfg(test)]
pub(crate) fn stored_allocation(&self, index: usize) -> Option<(*const u8, usize)> {
self.buffers
.get(index)
.and_then(|buf| buf.as_ref().map(|buf| (buf.as_ptr(), buf.capacity())))
}
#[cfg(test)] #[cfg(test)]
fn stored_capacity(&self, index: usize) -> Option<usize> { fn stored_capacity(&self, index: usize) -> Option<usize> {
self.buffers.get(index).and_then(|buf| buf.as_ref().map(Vec::capacity)) self.buffers.get(index).and_then(|buf| buf.as_ref().map(Vec::capacity))
+9 -526
View File
@@ -14,10 +14,7 @@
use pin_project_lite::pin_project; use pin_project_lite::pin_project;
use rustfs_utils::HashAlgorithm; use rustfs_utils::HashAlgorithm;
use std::future::poll_fn;
use std::io::IoSlice; use std::io::IoSlice;
use std::pin::Pin;
use std::task::{Context, Poll};
use std::time::Duration; use std::time::Duration;
use tokio::io::{AsyncRead, AsyncReadExt, AsyncWrite, AsyncWriteExt}; use tokio::io::{AsyncRead, AsyncReadExt, AsyncWrite, AsyncWriteExt};
use tracing::error; use tracing::error;
@@ -26,18 +23,6 @@ const LOG_COMPONENT_ECSTORE: &str = "ecstore";
const LOG_SUBSYSTEM_ERASURE: &str = "erasure"; const LOG_SUBSYSTEM_ERASURE: &str = "erasure";
const EVENT_BITROT_SHORT_SHARD_READ: &str = "bitrot_short_shard_read"; const EVENT_BITROT_SHORT_SHARD_READ: &str = "bitrot_short_shard_read";
const EVENT_BITROT_HASH_MISMATCH: &str = "bitrot_hash_mismatch"; const EVENT_BITROT_HASH_MISMATCH: &str = "bitrot_hash_mismatch";
const MAX_RETAINED_CHUNKS_PER_BLOCK: usize = 64;
const MAX_CHUNK_POLLS_PER_YIELD: usize = MAX_RETAINED_CHUNKS_PER_BLOCK + 1;
/// Result of polling an optional owned-chunk handoff.
pub enum ShardChunkRead {
/// The source does not support owned-chunk handoff and remains untouched.
Unsupported,
/// The source reached EOF.
Eof,
/// A non-empty chunk containing at most the requested number of bytes.
Chunk(bytes::Bytes),
}
/// A shard source that may already hold its bytes in memory. /// A shard source that may already hold its bytes in memory.
/// ///
@@ -57,12 +42,6 @@ pub trait ShardSource: AsyncRead + Send + Sync + Unpin {
fn try_take_block(&mut self, _n: usize) -> Option<bytes::Bytes> { fn try_take_block(&mut self, _n: usize) -> Option<bytes::Bytes> {
None None
} }
/// Polls one owned chunk when the source supports chunk handoff.
/// `Unsupported` must leave the source untouched.
fn poll_read_chunk(self: Pin<&mut Self>, _cx: &mut Context<'_>, _max: usize) -> Poll<std::io::Result<ShardChunkRead>> {
Poll::Ready(Ok(ShardChunkRead::Unsupported))
}
} }
/// Borrowed and owned byte slices are ordinary streaming sources: they carry no /// Borrowed and owned byte slices are ordinary streaming sources: they carry no
@@ -96,9 +75,6 @@ pin_project! {
// contiguous on-disk `[hash][data]` block so both are pulled in a single // contiguous on-disk `[hash][data]` block so both are pulled in a single
// pass; grown lazily and never shrunk. // pass; grown lazily and never shrunk.
buf: Vec<u8>, buf: Vec<u8>,
// Reused owned chunk vector for the remote HTTP fast path. Keeping the
// allocation with the reader avoids allocating once per bitrot block.
chunks: Vec<bytes::Bytes>,
skip_verify: bool, skip_verify: bool,
last_verify_duration: Duration, last_verify_duration: Duration,
} }
@@ -115,7 +91,6 @@ where
hash_algo: algo, hash_algo: algo,
shard_size, shard_size,
buf: Vec::new(), buf: Vec::new(),
chunks: Vec::new(),
skip_verify, skip_verify,
last_verify_duration: Duration::ZERO, last_verify_duration: Duration::ZERO,
} }
@@ -125,11 +100,6 @@ where
self.last_verify_duration self.last_verify_duration
} }
#[cfg(test)]
pub(crate) fn inner_ref(&self) -> &R {
&self.inner
}
/// Read a single (hash+data) block, verify hash, and copy `out.len()` bytes /// Read a single (hash+data) block, verify hash, and copy `out.len()` bytes
/// into `out`. Returns an error if the shard is short, the hash mismatches, /// into `out`. Returns an error if the shard is short, the hash mismatches,
/// or `out` is larger than one shard. On error `out`'s contents are /// or `out` is larger than one shard. On error `out`'s contents are
@@ -290,6 +260,11 @@ where
let need = hash_size + want; let need = hash_size + want;
// In-memory fast path: the block is already resident, so slice it instead
// of copying it into the scratch buffer first (rustfs/backlog#1159). One
// copy (`extend_from_slice`) instead of two. A source that cannot serve
// `need` bytes returns `None` and falls through to the scratch path,
// keeping the short-read contract.
if let Some(block) = self.inner.try_take_block(need) { if let Some(block) = self.inner.try_take_block(need) {
let (data, verify) = split_and_verify(&self.hash_algo, self.skip_verify, &block)?; let (data, verify) = split_and_verify(&self.hash_algo, self.skip_verify, &block)?;
out.extend_from_slice(data); out.extend_from_slice(data);
@@ -297,126 +272,6 @@ where
return Ok(want); return Ok(want);
} }
self.chunks.clear();
let handed_off = {
let inner = &mut self.inner;
let chunks = &mut self.chunks;
let tail_buf = &mut self.buf;
let mut received = 0usize;
poll_fn(|cx| {
for _ in 0..MAX_CHUNK_POLLS_PER_YIELD {
let next = match Pin::new(&mut *inner).poll_read_chunk(cx, need - received) {
Poll::Ready(Ok(next)) => next,
Poll::Ready(Err(err)) => return Poll::Ready(Err(err)),
Poll::Pending => return Poll::Pending,
};
let chunk = match next {
ShardChunkRead::Unsupported if received == 0 => return Poll::Ready(Ok(false)),
ShardChunkRead::Unsupported => {
return Poll::Ready(Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"chunk handoff became unavailable after transferring data",
)));
}
ShardChunkRead::Eof => {
return Poll::Ready(Err(short_shard_read(received.saturating_sub(hash_size), want)));
}
ShardChunkRead::Chunk(chunk) => chunk,
};
if received == 0 {
tail_buf.clear();
}
if chunk.is_empty() {
return Poll::Ready(Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"chunk handoff returned an empty chunk",
)));
}
let remaining = need - received;
if chunk.len() > remaining {
return Poll::Ready(Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"chunk handoff exceeded its requested boundary",
)));
}
received += chunk.len();
if chunks.len() == MAX_RETAINED_CHUNKS_PER_BLOCK {
if tail_buf.is_empty() {
tail_buf.reserve_exact(need - (received - chunk.len()));
}
tail_buf.extend_from_slice(&chunk);
} else {
chunks.push(chunk);
}
if received == need {
return Poll::Ready(Ok(true));
}
}
cx.waker().wake_by_ref();
Poll::Pending
})
.await?
};
if handed_off {
if self.chunks.len() == 1 && self.buf.is_empty() {
let block = &self.chunks[0];
let (data, verify) = split_and_verify(&self.hash_algo, self.skip_verify, block)?;
out.extend_from_slice(data);
self.last_verify_duration = verify;
return Ok(want);
}
let block_chunks = || {
self.chunks
.iter()
.map(|chunk| chunk.as_ref())
.chain((!self.buf.is_empty()).then_some(self.buf.as_slice()))
};
if !self.skip_verify {
let verify_start = std::time::Instant::now();
let actual_hash = self
.hash_algo
.hash_encode_slices(block_chunks().scan(hash_size, |skip, chunk| {
let start = (*skip).min(chunk.len());
*skip -= start;
Some(&chunk[start..])
}));
let verify = verify_start.elapsed();
let mut hash_offset = 0;
let mut remaining = hash_size;
for chunk in block_chunks() {
let take = remaining.min(chunk.len());
if actual_hash.as_ref()[hash_offset..hash_offset + take] != chunk[..take] {
error!(
event = EVENT_BITROT_HASH_MISMATCH,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_ERASURE,
state = "failed",
data_len = want,
"bitrot hash mismatch"
);
return Err(std::io::Error::new(std::io::ErrorKind::InvalidData, "bitrot hash mismatch"));
}
hash_offset += take;
remaining -= take;
if remaining == 0 {
break;
}
}
self.last_verify_duration = verify;
}
let mut skip = hash_size;
for chunk in block_chunks() {
let start = skip.min(chunk.len());
skip -= start;
out.extend_from_slice(&chunk[start..]);
}
return Ok(want);
}
// Streaming path: same single pass and same verification as `read`; only // Streaming path: same single pass and same verification as `read`; only
// the sink differs (`extend_from_slice` into `out` instead of // the sink differs (`extend_from_slice` into `out` instead of
// `copy_from_slice` into a pre-zeroed buffer). // `copy_from_slice` into a pre-zeroed buffer).
@@ -822,167 +677,18 @@ impl BitrotWriterWrapper {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::ShardSource;
use super::{ use super::{
BitrotReader, BitrotWriter, BitrotWriterWrapper, CustomWriter, bitrot_shard_file_size, bitrot_verify, write_all_vectored, BitrotReader, BitrotWriter, BitrotWriterWrapper, CustomWriter, bitrot_shard_file_size, bitrot_verify, write_all_vectored,
}; };
use super::{MAX_RETAINED_CHUNKS_PER_BLOCK, ShardChunkRead, ShardSource};
use bytes::Bytes;
use rustfs_utils::HashAlgorithm; use rustfs_utils::HashAlgorithm;
use std::collections::VecDeque; use std::io::{Cursor, IoSlice};
use std::io::{self, Cursor, IoSlice};
use std::pin::Pin;
use std::sync::{ use std::sync::{
Arc, Arc,
atomic::{AtomicUsize, Ordering}, atomic::{AtomicUsize, Ordering},
}; };
use std::task::{Context, Poll}; use std::task::{Context, Poll};
use std::time::Duration; use tokio::io::{AsyncWrite, AsyncWriteExt};
use tokio::io::{AsyncRead, AsyncWrite, AsyncWriteExt, ReadBuf};
struct FragmentedSource {
chunks: VecDeque<Bytes>,
}
impl FragmentedSource {
fn new(bytes: Vec<u8>, fragment_sizes: &[usize]) -> Self {
let mut chunks = VecDeque::new();
let mut offset = 0;
for &size in fragment_sizes {
let end = (offset + size).min(bytes.len());
if offset < end {
chunks.push_back(Bytes::copy_from_slice(&bytes[offset..end]));
}
offset = end;
}
if offset < bytes.len() {
chunks.push_back(Bytes::copy_from_slice(&bytes[offset..]));
}
Self { chunks }
}
}
impl AsyncRead for FragmentedSource {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("fragmented source must use chunk handoff")))
}
}
impl ShardSource for FragmentedSource {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
let Some(mut chunk) = self.chunks.pop_front() else {
return Poll::Ready(Ok(ShardChunkRead::Eof));
};
if chunk.len() > max {
self.chunks.push_front(chunk.split_off(max));
chunk.truncate(max);
}
Poll::Ready(Ok(ShardChunkRead::Chunk(chunk)))
}
}
struct GeneratedChunkSource {
bytes: Bytes,
offset: usize,
fragment_size: usize,
fail_at: Option<usize>,
}
impl GeneratedChunkSource {
fn new(bytes: Vec<u8>, fragment_size: usize) -> Self {
assert!(fragment_size > 0);
Self {
bytes: Bytes::from(bytes),
offset: 0,
fragment_size,
fail_at: None,
}
}
fn failing(bytes: Vec<u8>, fragment_size: usize, fail_at: usize) -> Self {
Self {
fail_at: Some(fail_at),
..Self::new(bytes, fragment_size)
}
}
}
impl AsyncRead for GeneratedChunkSource {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("generated source must use chunk handoff")))
}
}
impl ShardSource for GeneratedChunkSource {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
if self.fail_at == Some(self.offset) {
return Poll::Ready(Err(rustfs_rio::new_test_internode_http_io_error(
rustfs_rio::InternodeHttpErrorKind::BodyStreamAborted,
)));
}
if self.offset == self.bytes.len() {
return Poll::Ready(Ok(ShardChunkRead::Eof));
}
let error_limit = self.fail_at.unwrap_or(self.bytes.len());
let take = self
.fragment_size
.min(max)
.min(error_limit - self.offset)
.min(self.bytes.len() - self.offset);
let start = self.offset;
self.offset += take;
Poll::Ready(Ok(ShardChunkRead::Chunk(self.bytes.slice(start..start + take))))
}
}
struct InvalidChunkSource {
mode: InvalidChunkMode,
}
#[derive(Clone, Copy)]
enum InvalidChunkMode {
Empty,
Oversized,
UnsupportedAfterChunk,
Unsupported,
}
impl AsyncRead for InvalidChunkSource {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("invalid source must use chunk handoff")))
}
}
impl ShardSource for InvalidChunkSource {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
match self.mode {
InvalidChunkMode::Empty => Poll::Ready(Ok(ShardChunkRead::Chunk(Bytes::new()))),
InvalidChunkMode::Oversized => Poll::Ready(Ok(ShardChunkRead::Chunk(Bytes::from(vec![0; max + 1])))),
InvalidChunkMode::UnsupportedAfterChunk => {
self.mode = InvalidChunkMode::Unsupported;
Poll::Ready(Ok(ShardChunkRead::Chunk(Bytes::from_static(b"x"))))
}
InvalidChunkMode::Unsupported => Poll::Ready(Ok(ShardChunkRead::Unsupported)),
}
}
}
struct ScratchReuseSource {
block: Option<Bytes>,
saw_reused_scratch: bool,
}
impl AsyncRead for ScratchReuseSource {
fn poll_read(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
let Some(block) = self.block.take() else {
return Poll::Ready(Ok(()));
};
self.saw_reused_scratch = buf.initialize_unfilled()[..block.len()].iter().all(|byte| *byte == 0xa5);
buf.put_slice(&block);
Poll::Ready(Ok(()))
}
}
impl ShardSource for ScratchReuseSource {}
#[derive(Default)] #[derive(Default)]
struct VectoredCountingWriter { struct VectoredCountingWriter {
@@ -1740,70 +1446,6 @@ mod tests {
assert!(out.is_empty(), "corrupt bytes must never reach the caller's buffer"); assert!(out.is_empty(), "corrupt bytes must never reach the caller's buffer");
} }
#[tokio::test]
async fn chunked_handoff_verifies_data_split_across_hash_boundaries() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let data: Vec<u8> = (0..SHARD).map(|index| (index % 251) as u8).collect();
let mut encoded = Vec::new();
BitrotWriter::new(&mut encoded, SHARD, algo.clone())
.write(&data)
.await
.expect("write shard");
let mut output = Vec::with_capacity(SHARD);
BitrotReader::new(FragmentedSource::new(encoded, &[3, 11, 19, 37, 128]), SHARD, algo, false)
.read_appending(&mut output, SHARD)
.await
.expect("fragmented shard must verify");
assert_eq!(output, data);
}
#[tokio::test]
async fn chunked_handoff_never_appends_a_corrupt_shard() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let mut encoded = Vec::new();
BitrotWriter::new(&mut encoded, SHARD, algo.clone())
.write(&vec![9u8; SHARD])
.await
.expect("write shard");
let last = encoded.len() - 1;
encoded[last] ^= 0xff;
let mut output = Vec::with_capacity(SHARD);
let err = BitrotReader::new(FragmentedSource::new(encoded, &[7, 17, 31]), SHARD, algo, false)
.read_appending(&mut output, SHARD)
.await
.expect_err("corrupt fragmented shard must fail");
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
assert!(output.is_empty());
}
#[tokio::test]
async fn chunked_handoff_does_not_hash_when_verification_is_skipped() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let mut encoded = Vec::new();
BitrotWriter::new(&mut encoded, SHARD, algo.clone())
.write(&vec![9u8; SHARD])
.await
.expect("write shard");
encoded[0] ^= 0xff;
let mut output = Vec::with_capacity(SHARD);
let mut reader = BitrotReader::new(FragmentedSource::new(encoded, &[7, 17, 31]), SHARD, algo, true);
reader
.read_appending(&mut output, SHARD)
.await
.expect("skipped verification must accept fragmented shard bytes");
assert_eq!(reader.last_verify_duration(), Duration::ZERO);
assert_eq!(output, vec![9u8; SHARD]);
}
#[tokio::test] #[tokio::test]
async fn read_appending_rejects_a_want_larger_than_the_shard() { async fn read_appending_rejects_a_want_larger_than_the_shard() {
let algo = HashAlgorithm::HighwayHash256; let algo = HashAlgorithm::HighwayHash256;
@@ -1855,21 +1497,10 @@ mod tests {
// Equivalence: same bytes out of both paths. // Equivalence: same bytes out of both paths.
let mut via_mem: Vec<u8> = Vec::with_capacity(SHARD); let mut via_mem: Vec<u8> = Vec::with_capacity(SHARD);
let mut memory_reader = BitrotReader::new(Cursor::new(Bytes::from(encoded.clone())), SHARD, algo.clone(), false); BitrotReader::new(Cursor::new(Bytes::from(encoded.clone())), SHARD, algo.clone(), false)
memory_reader
.read_appending(&mut via_mem, SHARD) .read_appending(&mut via_mem, SHARD)
.await .await
.expect("in-memory read"); .expect("in-memory read");
assert_eq!(
memory_reader.chunks.capacity(),
0,
"the synchronous fast path must not allocate chunk storage"
);
assert_eq!(
memory_reader.buf.capacity(),
0,
"the synchronous fast path must not allocate scratch storage"
);
let mut via_stream: Vec<u8> = Vec::with_capacity(SHARD); let mut via_stream: Vec<u8> = Vec::with_capacity(SHARD);
BitrotReader::new(Cursor::new(encoded), SHARD, algo, false) BitrotReader::new(Cursor::new(encoded), SHARD, algo, false)
@@ -1906,152 +1537,4 @@ mod tests {
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData); assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert!(out.is_empty(), "corrupt bytes must never reach the caller's buffer"); assert!(out.is_empty(), "corrupt bytes must never reach the caller's buffer");
} }
#[tokio::test]
async fn streaming_fallback_reuses_initialized_scratch() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![7u8; SHARD];
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let source = ScratchReuseSource {
block: Some(Bytes::copy_from_slice(&encoded)),
saw_reused_scratch: false,
};
let mut reader = BitrotReader::new(source, SHARD, algo, false);
reader.buf = vec![0xa5; encoded.len()];
let mut output = Vec::new();
reader
.read_appending(&mut output, SHARD)
.await
.expect("streaming fallback should verify");
assert!(reader.inner.saw_reused_scratch, "capability probing must not clear reusable scratch");
assert_eq!(output, data);
}
#[tokio::test]
async fn chunked_handoff_bounds_production_sized_one_byte_fragments() {
const SHARD: usize = 1024 * 1024 / 4;
let algo = HashAlgorithm::HighwayHash256S;
let data: Vec<u8> = (0..SHARD).map(|index| (index % 251) as u8).collect();
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let encoded_len = encoded.len();
let mut reader = BitrotReader::new(GeneratedChunkSource::new(encoded, 1), SHARD, algo, false);
let mut output = Vec::with_capacity(SHARD);
reader
.read_appending(&mut output, SHARD)
.await
.expect("one-byte fragments should verify with bounded retained state");
assert_eq!(output, data);
assert_eq!(reader.chunks.len(), MAX_RETAINED_CHUNKS_PER_BLOCK);
assert!(reader.chunks.capacity() <= MAX_RETAINED_CHUNKS_PER_BLOCK);
assert_eq!(reader.buf.len(), encoded_len - MAX_RETAINED_CHUNKS_PER_BLOCK);
}
#[tokio::test]
async fn chunked_handoff_keeps_sixty_four_frames_zero_copy_and_respects_poll_budget() {
const SHARD: usize = 1024 * 1024;
const FRAME: usize = 16 * 1024;
let algo = HashAlgorithm::HighwayHash256S;
let small_data = vec![3u8; 4096];
let small_encoded = encode_one_block(&small_data, 4096, algo.clone()).await;
let mut exact_reader =
BitrotReader::new(FragmentedSource::new(small_encoded.clone(), &[1; 63]), 4096, algo.clone(), false);
let mut exact_output = Vec::new();
exact_reader
.read_appending(&mut exact_output, 4096)
.await
.expect("exactly sixty-four frames should verify");
assert_eq!(exact_output, small_data);
assert_eq!(exact_reader.chunks.len(), MAX_RETAINED_CHUNKS_PER_BLOCK);
assert!(exact_reader.buf.is_empty(), "the threshold itself must remain zero-copy");
let mut yielded_reader = BitrotReader::new(FragmentedSource::new(small_encoded, &[1; 65]), 4096, algo.clone(), false);
let mut yielded_output = Vec::new();
let mut yielded_read = Box::pin(yielded_reader.read_appending(&mut yielded_output, 4096));
let mut cx = Context::from_waker(std::task::Waker::noop());
assert!(std::future::Future::poll(yielded_read.as_mut(), &mut cx).is_pending());
assert!(matches!(std::future::Future::poll(yielded_read.as_mut(), &mut cx), Poll::Ready(Ok(4096))));
drop(yielded_read);
assert_eq!(yielded_output, small_data);
let data = vec![7u8; SHARD];
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let mut reader = BitrotReader::new(FragmentedSource::new(encoded, &[FRAME; 64]), SHARD, algo, false);
let mut output = Vec::with_capacity(SHARD);
let mut read = Box::pin(reader.read_appending(&mut output, SHARD));
assert!(
matches!(std::future::Future::poll(read.as_mut(), &mut cx), Poll::Ready(Ok(SHARD))),
"sixty-five normal HTTP frames should complete without a cooperative yield"
);
drop(read);
assert_eq!(output, data);
assert_eq!(reader.chunks.len(), MAX_RETAINED_CHUNKS_PER_BLOCK);
assert_eq!(reader.buf.len(), HashAlgorithm::HighwayHash256S.size());
}
#[tokio::test]
async fn chunked_tail_failures_preserve_errors_and_output() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![7u8; SHARD];
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let sentinel = vec![1u8, 2, 3];
let mut short_output = sentinel.clone();
let short_err = BitrotReader::new(GeneratedChunkSource::new(encoded[..100].to_vec(), 1), SHARD, algo.clone(), false)
.read_appending(&mut short_output, SHARD)
.await
.expect_err("EOF after the retention threshold must stay a short read");
assert_eq!(short_err.kind(), io::ErrorKind::UnexpectedEof);
assert_eq!(short_output, sentinel);
let mut corrupt = encoded.clone();
let last = corrupt.len() - 1;
corrupt[last] ^= 0xff;
let mut corrupt_output = sentinel.clone();
let corrupt_err = BitrotReader::new(GeneratedChunkSource::new(corrupt, 1), SHARD, algo.clone(), false)
.read_appending(&mut corrupt_output, SHARD)
.await
.expect_err("corrupt coalesced tail must fail verification");
assert_eq!(corrupt_err.kind(), io::ErrorKind::InvalidData);
assert_eq!(corrupt_output, sentinel);
let mut failed_output = sentinel.clone();
let body_err = BitrotReader::new(GeneratedChunkSource::failing(encoded, 1, 65), SHARD, algo, false)
.read_appending(&mut failed_output, SHARD)
.await
.expect_err("a terminal body error must not become EOF");
let source = body_err
.get_ref()
.and_then(|source| source.downcast_ref::<rustfs_rio::InternodeHttpError>())
.expect("body error should retain internode classification");
assert_eq!(source.kind(), rustfs_rio::InternodeHttpErrorKind::BodyStreamAborted);
assert_eq!(failed_output, sentinel);
}
#[tokio::test]
async fn chunked_handoff_rejects_invalid_source_contracts() {
const SHARD: usize = 64;
for mode in [
InvalidChunkMode::Empty,
InvalidChunkMode::Oversized,
InvalidChunkMode::UnsupportedAfterChunk,
] {
let source = InvalidChunkSource { mode };
let mut output = vec![9u8];
let err = BitrotReader::new(source, SHARD, HashAlgorithm::HighwayHash256S, false)
.read_appending(&mut output, SHARD)
.await
.expect_err("invalid chunk contracts must fail closed");
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
assert_eq!(output, vec![9u8]);
}
}
} }
+39 -119
View File
@@ -25,9 +25,7 @@ use crate::disk::error_reduce::reduce_errs;
use crate::erasure::codec::workspace::ShardBufferPool; use crate::erasure::codec::workspace::ShardBufferPool;
use crate::erasure::coding::{BitrotReader, Erasure}; use crate::erasure::coding::{BitrotReader, Erasure};
use crate::io_support::bitrot::DeferredReaderStripeHandle; use crate::io_support::bitrot::DeferredReaderStripeHandle;
use crate::set_disk::shard_source::{ use crate::set_disk::shard_source::{ShardReadCost, ShardStripeSource, StripeReadState};
INLINE_SHARD_SLOTS, ShardBuffers, ShardErrors, ShardReadCost, ShardStripeSource, StripeReadState,
};
use futures::FutureExt; use futures::FutureExt;
use futures::stream::{FuturesUnordered, StreamExt}; use futures::stream::{FuturesUnordered, StreamExt};
use pin_project_lite::pin_project; use pin_project_lite::pin_project;
@@ -43,6 +41,9 @@ use tracing::{debug, error, warn};
type ShardReadFuture<'a> = Pin<Box<dyn Future<Output = (usize, ShardReadCost, Result<Vec<u8>, Error>, bool)> + Send + 'a>>; type ShardReadFuture<'a> = Pin<Box<dyn Future<Output = (usize, ShardReadCost, Result<Vec<u8>, Error>, bool)> + Send + 'a>>;
const INLINE_SHARD_SLOTS: usize = 32;
type ShardBuffers = SmallVec<[Option<Vec<u8>>; INLINE_SHARD_SLOTS]>;
type ShardErrors = SmallVec<[Option<Error>; INLINE_SHARD_SLOTS]>;
type ShardIndexes = SmallVec<[usize; INLINE_SHARD_SLOTS]>; type ShardIndexes = SmallVec<[usize; INLINE_SHARD_SLOTS]>;
type ActiveReaders = SmallVec<[bool; INLINE_SHARD_SLOTS]>; type ActiveReaders = SmallVec<[bool; INLINE_SHARD_SLOTS]>;
@@ -213,7 +214,6 @@ fn shard_read_launch_rank(cost: ShardReadCost) -> u8 {
} }
} }
#[allow(dead_code, reason = "launch ordering asserted by this file's tests (backlog#1823)")]
fn shard_read_launch_order(read_costs: &[ShardReadCost], num_readers: usize, locality_preference_enabled: bool) -> Vec<usize> { fn shard_read_launch_order(read_costs: &[ShardReadCost], num_readers: usize, locality_preference_enabled: bool) -> Vec<usize> {
let mut order: Vec<usize> = (0..num_readers).collect(); let mut order: Vec<usize> = (0..num_readers).collect();
if locality_preference_enabled { if locality_preference_enabled {
@@ -392,7 +392,6 @@ pub(crate) struct ParallelReader<R> {
// Request-scoped shard buffers keyed by shard index. Keeping ownership in // Request-scoped shard buffers keyed by shard index. Keeping ownership in
// `ParallelReader` avoids dropping unused parity/backup slot buffers between stripes. // `ParallelReader` avoids dropping unused parity/backup slot buffers between stripes.
buffers: ShardBufferPool, buffers: ShardBufferPool,
stripe_state: Option<Box<StripeReadState>>,
// Lockstep-path state (verify_reconstruction == true). `engaged[i]` marks // Lockstep-path state (verify_reconstruction == true). `engaged[i]` marks
// readers that participate in each stripe read: all data slots from the // readers that participate in each stripe read: all data slots from the
// start, parity slots only once a data shard is missing/dead. Unengaged // start, parity slots only once a data shard is missing/dead. Unengaged
@@ -409,10 +408,6 @@ where
R: crate::erasure::coding::ShardSource, R: crate::erasure::coding::ShardSource,
{ {
// Readers should handle disk errors before being passed in, ensuring each reader reaches the available number of BitrotReaders // Readers should handle disk errors before being passed in, ensuring each reader reaches the available number of BitrotReaders
#[allow(
dead_code,
reason = "ParallelReader constructor used only by this file's tests (backlog#1823)"
)]
pub fn new(readers: Vec<Option<BitrotReader<R>>>, e: Erasure, offset: usize, total_length: usize) -> Self { pub fn new(readers: Vec<Option<BitrotReader<R>>>, e: Erasure, offset: usize, total_length: usize) -> Self {
Self::new_with_metrics_path_read_timeout_and_reconstruction_verification( Self::new_with_metrics_path_read_timeout_and_reconstruction_verification(
readers, readers,
@@ -425,7 +420,6 @@ where
) )
} }
#[allow(dead_code, reason = "constructor used only by this file's tests (backlog#1823)")]
pub fn new_with_metrics_path( pub fn new_with_metrics_path(
readers: Vec<Option<BitrotReader<R>>>, readers: Vec<Option<BitrotReader<R>>>,
e: Erasure, e: Erasure,
@@ -444,7 +438,6 @@ where
) )
} }
#[allow(dead_code, reason = "constructor used only by this file's tests (backlog#1823)")]
pub fn new_with_metrics_path_and_read_costs( pub fn new_with_metrics_path_and_read_costs(
readers: Vec<Option<BitrotReader<R>>>, readers: Vec<Option<BitrotReader<R>>>,
e: Erasure, e: Erasure,
@@ -521,7 +514,6 @@ where
) )
} }
#[allow(dead_code, reason = "constructor used only by this file's tests (backlog#1823)")]
fn new_with_read_timeout( fn new_with_read_timeout(
readers: Vec<Option<BitrotReader<R>>>, readers: Vec<Option<BitrotReader<R>>>,
e: Erasure, e: Erasure,
@@ -604,7 +596,6 @@ where
verify_reconstruction, verify_reconstruction,
locality_preference_enabled: get_shard_locality_preference_enabled(), locality_preference_enabled: get_shard_locality_preference_enabled(),
buffers: ShardBufferPool::new(e.data_shards + e.parity_shards), buffers: ShardBufferPool::new(e.data_shards + e.parity_shards),
stripe_state: None,
engaged, engaged,
deferred_handles: Vec::new(), deferred_handles: Vec::new(),
stripe_index: 0, stripe_index: 0,
@@ -709,12 +700,6 @@ where
{ {
#[hotpath::measure(impl_type = "ParallelReader")] #[hotpath::measure(impl_type = "ParallelReader")]
pub async fn read(&mut self) -> StripeReadOutput { pub async fn read(&mut self) -> StripeReadOutput {
let mut state = StripeReadState::with_slot_count(self.readers.len(), self.data_shards);
self.read_into_state(&mut state).await;
state.into_parts()
}
async fn read_into_state(&mut self, state: &mut StripeReadState) {
// On the reconstruction-verifying GET path, read every live shard reader // On the reconstruction-verifying GET path, read every live shard reader
// in lockstep so all readers advance one block per stripe and stay // in lockstep so all readers advance one block per stripe and stay
// mutually aligned. The adaptive data-first path below only reads // mutually aligned. The adaptive data-first path below only reads
@@ -724,14 +709,12 @@ where
// than the data shards, producing "inconsistent read source shards" and // than the data shards, producing "inconsistent read source shards" and
// truncating large-object GETs under concurrency (backlog#832). // truncating large-object GETs under concurrency (backlog#832).
if self.verify_reconstruction { if self.verify_reconstruction {
self.read_lockstep(state).await; return self.read_lockstep().await;
return;
} }
// if self.readers.len() != self.total_shards { // if self.readers.len() != self.total_shards {
// return Err(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers")); // return Err(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers"));
// } // }
let num_readers = self.readers.len(); let num_readers = self.readers.len();
state.reset(num_readers, self.data_shards);
let shard_size = if self.offset + self.shard_size > self.shard_file_size { let shard_size = if self.offset + self.shard_size > self.shard_file_size {
self.shard_file_size - self.offset self.shard_file_size - self.offset
@@ -740,7 +723,7 @@ where
}; };
if shard_size == 0 { if shard_size == 0 {
return; return (smallvec![None; num_readers], smallvec![None; num_readers]);
} }
// Advance to the next stripe so the following read() computes the correct // Advance to the next stripe so the following read() computes the correct
@@ -751,7 +734,8 @@ where
// is only read above to derive `shard_size`, so advancing here is safe. // is only read above to derive `shard_size`, so advancing here is safe.
self.offset += shard_size; self.offset += shard_size;
let (shards, errs) = state.parts_mut(); let mut shards: ShardBuffers = smallvec![None; num_readers];
let mut errs: ShardErrors = smallvec![None; num_readers];
let read_costs = self.read_costs.as_slice(); let read_costs = self.read_costs.as_slice();
let locality_preference_enabled = self.locality_preference_enabled; let locality_preference_enabled = self.locality_preference_enabled;
let low_cost_available = self let low_cost_available = self
@@ -898,8 +882,8 @@ where
} }
let result_is_err = record_shard_read_result( let result_is_err = record_shard_read_result(
shards, &mut shards,
errs, &mut errs,
&mut retire_readers, &mut retire_readers,
&mut success, &mut success,
&mut successful_costs, &mut successful_costs,
@@ -960,8 +944,8 @@ where
active_readers[i] = false; active_readers[i] = false;
completed += 1; completed += 1;
if record_shard_read_result( if record_shard_read_result(
shards, &mut shards,
errs, &mut errs,
&mut retire_readers, &mut retire_readers,
&mut success, &mut success,
&mut successful_costs, &mut successful_costs,
@@ -973,7 +957,7 @@ where
failed += 1; failed += 1;
} }
} }
retire_abandoned_readers(errs, &mut retire_readers, &active_readers); retire_abandoned_readers(&mut errs, &mut retire_readers, &active_readers);
} }
if let Some(path) = self.metrics_path { if let Some(path) = self.metrics_path {
@@ -1017,6 +1001,8 @@ where
for i in retire_readers { for i in retire_readers {
self.readers[i] = None; self.readers[i] = None;
} }
(shards, errs)
} }
/// Lockstep stripe read for the reconstruction-verifying GET path. /// Lockstep stripe read for the reconstruction-verifying GET path.
@@ -1044,18 +1030,18 @@ where
/// stripe would reintroduce the desync. A parity reader that cannot be /// stripe would reintroduce the desync. A parity reader that cannot be
/// realigned (no pending deferred handle) is likewise retired instead of /// realigned (no pending deferred handle) is likewise retired instead of
/// being read out of position. /// being read out of position.
async fn read_lockstep(&mut self, state: &mut StripeReadState) { async fn read_lockstep(&mut self) -> StripeReadOutput {
let num_readers = self.readers.len(); let num_readers = self.readers.len();
state.reset(num_readers, self.data_shards);
let shard_size = if self.offset + self.shard_size > self.shard_file_size { let shard_size = if self.offset + self.shard_size > self.shard_file_size {
self.shard_file_size - self.offset self.shard_file_size - self.offset
} else { } else {
self.shard_size self.shard_size
}; };
let (shards, errs) = state.parts_mut(); let mut shards: ShardBuffers = smallvec![None; num_readers];
let mut errs: ShardErrors = smallvec![None; num_readers];
if shard_size == 0 { if shard_size == 0 {
return; return (shards, errs);
} }
// Advance to the next stripe (see the matching note in `read`); the // Advance to the next stripe (see the matching note in `read`); the
@@ -1293,6 +1279,8 @@ where
for i in retire_readers { for i in retire_readers {
self.readers[i] = None; self.readers[i] = None;
} }
(shards, errs)
} }
/// Attempt to bring an as-yet-unread parity reader into the lockstep read /// Attempt to bring an as-yet-unread parity reader into the lockstep read
@@ -1338,6 +1326,10 @@ where
} }
} }
} }
pub fn can_decode(&self, shards: &[Option<Vec<u8>>]) -> bool {
shards.iter().filter(|s| s.is_some()).count() >= self.data_shards
}
} }
#[async_trait::async_trait] #[async_trait::async_trait]
@@ -1345,20 +1337,10 @@ impl<R> ShardStripeSource for ParallelReader<R>
where where
R: crate::erasure::coding::ShardSource, R: crate::erasure::coding::ShardSource,
{ {
async fn read_next_stripe(&mut self) -> Box<StripeReadState> { async fn read_next_stripe(&mut self) -> StripeReadState {
let mut state = self let read_quorum = self.data_shards;
.stripe_state let (shards, errors) = ParallelReader::read(self).await;
.take() StripeReadState::from_parts_with_read_costs(shards, errors, &self.read_costs, read_quorum)
.unwrap_or_else(|| Box::new(StripeReadState::with_slot_count(self.readers.len(), self.data_shards)));
self.read_into_state(&mut state).await;
state
}
fn recycle_stripe(&mut self, mut state: Box<StripeReadState>) {
self.recycle_shards(state.shards_mut());
state.reset(0, self.data_shards);
debug_assert!(self.stripe_state.is_none(), "a stripe cannot be recycled twice");
self.stripe_state = Some(state);
} }
} }
@@ -1543,7 +1525,6 @@ impl Erasure {
.await .await
} }
#[allow(dead_code, reason = "read-cost decode path asserted by this file's tests (backlog#1823)")]
pub(crate) async fn decode_with_read_costs<W, R>( pub(crate) async fn decode_with_read_costs<W, R>(
&self, &self,
writer: &mut W, writer: &mut W,
@@ -1614,9 +1595,9 @@ impl Erasure {
*ret_err = Some(err.into()); *ret_err = Some(err.into());
} }
// Shard-availability check, written out here rather than called on the // Equivalent to `ParallelReader::can_decode`; inlined so this helper does
// reader so this helper does not need to borrow it, leaving the reader // not need to borrow the reader, leaving the reader free for the
// free for the concurrent next-stripe read under prefetch. // concurrent next-stripe read under prefetch.
let available_shards = shards.iter().filter(|shard| shard.is_some()).count(); let available_shards = shards.iter().filter(|shard| shard.is_some()).count();
if available_shards < self.data_shards { if available_shards < self.data_shards {
let reason = GetObjectFailureReason::ReadQuorum; let reason = GetObjectFailureReason::ReadQuorum;
@@ -1991,18 +1972,13 @@ mod tests {
type BoxedShardReader = crate::io_support::bitrot::ShardReader; type BoxedShardReader = crate::io_support::bitrot::ShardReader;
#[test] #[test]
fn parallel_reader_keeps_stripe_scratch_out_of_line() { fn shard_scratch_stays_inline_through_the_common_limit_and_spills_safely() {
eprintln!( let inline: ShardBuffers = smallvec![None; INLINE_SHARD_SLOTS];
"parallel_reader={} stripe_state={} cached_state={}", assert!(!inline.spilled(), "the common shard-count boundary must not allocate");
std::mem::size_of::<ParallelReader<Cursor<Vec<u8>>>>(),
std::mem::size_of::<StripeReadState>(), let spilled: ShardBuffers = smallvec![None; INLINE_SHARD_SLOTS + 1];
std::mem::size_of::<Option<Box<StripeReadState>>>() assert!(spilled.spilled(), "larger supported shard counts must fall back to the heap");
); assert_eq!(spilled.len(), INLINE_SHARD_SLOTS + 1);
assert_eq!(
std::mem::size_of::<Option<Box<StripeReadState>>>(),
std::mem::size_of::<usize>(),
"the request-scoped cache must remain pointer-sized",
);
} }
#[tokio::test] #[tokio::test]
@@ -2021,62 +1997,6 @@ mod tests {
assert_eq!(errors.len(), TOTAL_SHARDS); assert_eq!(errors.len(), TOTAL_SHARDS);
} }
#[tokio::test]
async fn codec_reader_reuses_inline_and_spilled_stripe_scratch_between_reads() {
for total_shards in [INLINE_SHARD_SLOTS, INLINE_SHARD_SLOTS + 1] {
let data_shards = total_shards - 1;
let readers = std::iter::repeat_with(|| None).take(total_shards).collect();
let erasure = Erasure::new(data_shards, 1, data_shards * 2);
let mut reader: ParallelReader<Cursor<Vec<u8>>> = ParallelReader::new(readers, erasure, 0, data_shards * 2);
let first = ShardStripeSource::read_next_stripe(&mut reader).await;
let first_state = (&*first) as *const StripeReadState;
let first_storage = first.scratch_storage();
assert_eq!(first_storage.2, total_shards > INLINE_SHARD_SLOTS);
assert_eq!(first_storage.3, total_shards > INLINE_SHARD_SLOTS);
ShardStripeSource::recycle_stripe(&mut reader, first);
let second = ShardStripeSource::read_next_stripe(&mut reader).await;
let second_storage = second.scratch_storage();
assert_eq!(
(&*second) as *const StripeReadState,
first_state,
"the request-scoped state must be reused"
);
assert_eq!(second_storage.0, first_storage.0, "shard slots must reuse their allocation");
assert_eq!(second_storage.1, first_storage.1, "error slots must reuse their allocation");
assert_eq!(second.into_parts().0.len(), total_shards);
}
}
#[tokio::test]
async fn codec_reader_returns_shard_allocations_to_the_request_pool() {
const SHARD_SIZE: usize = 16;
let hash_algo = HashAlgorithm::None;
let readers = vec![Some(create_reader(SHARD_SIZE, 2, 0x5a, &hash_algo, false).await)];
let erasure = Erasure::new(1, 0, SHARD_SIZE);
let mut reader = ParallelReader::new(readers, erasure, 0, SHARD_SIZE * 2);
let first = ShardStripeSource::read_next_stripe(&mut reader).await;
let first_allocation = first
.shard_allocation(0)
.expect("the first stripe should own its shard allocation");
ShardStripeSource::recycle_stripe(&mut reader, first);
assert_eq!(
reader.buffers.stored_allocation(0),
Some(first_allocation),
"recycling a stripe must return its shard allocation to the request pool"
);
let second = ShardStripeSource::read_next_stripe(&mut reader).await;
assert_eq!(
second.shard_allocation(0),
Some(first_allocation),
"the next stripe must reuse the pooled shard allocation"
);
}
/// Counts the raw bytes pulled from a shard stream, to prove which shards /// Counts the raw bytes pulled from a shard stream, to prove which shards
/// a decode path actually touches (backlog#923 call-count evidence). /// a decode path actually touches (backlog#923 call-count evidence).
struct CountingShardReader { struct CountingShardReader {
@@ -65,7 +65,7 @@ enum FillPolicy {
} }
impl FillPolicy { impl FillPolicy {
fn load() -> Self { fn from_env() -> Self {
match rustfs_utils::get_env_usize( match rustfs_utils::get_env_usize(
ENV_RUSTFS_GET_CODEC_STREAMING_MAX_INFLIGHT, ENV_RUSTFS_GET_CODEC_STREAMING_MAX_INFLIGHT,
DEFAULT_RUSTFS_GET_CODEC_STREAMING_MAX_INFLIGHT, DEFAULT_RUSTFS_GET_CODEC_STREAMING_MAX_INFLIGHT,
@@ -75,22 +75,6 @@ impl FillPolicy {
} }
} }
fn from_env() -> Self {
#[cfg(test)]
{
Self::load()
}
#[cfg(not(test))]
{
Self::cached_core(Self::load)
}
}
fn cached_core(load: impl FnOnce() -> Self) -> Self {
static CACHED: std::sync::OnceLock<FillPolicy> = std::sync::OnceLock::new();
*CACHED.get_or_init(load)
}
const fn max_inflight(self) -> usize { const fn max_inflight(self) -> usize {
match self { match self {
Self::SingleInFlight => 1, Self::SingleInFlight => 1,
@@ -138,10 +122,6 @@ where
S: ShardStripeSource + Send + 'static, S: ShardStripeSource + Send + 'static,
E: ErasureDecodeEngine + Clone + Send + Sync + 'static, E: ErasureDecodeEngine + Clone + Send + Sync + 'static,
{ {
#[allow(
dead_code,
reason = "default-metrics-path constructor used only by this file's tests (backlog#1823)"
)]
pub(crate) fn new(source: S, engine: E, total_length: usize) -> io::Result<Self> { pub(crate) fn new(source: S, engine: E, total_length: usize) -> io::Result<Self> {
Self::new_with_metrics_path(source, engine, total_length, GET_OBJECT_PATH_CODEC_STREAMING) Self::new_with_metrics_path(source, engine, total_length, GET_OBJECT_PATH_CODEC_STREAMING)
} }
@@ -499,30 +479,22 @@ where
let mut deferred_error = None; let mut deferred_error = None;
let fill_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled); let fill_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let stripe_read_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled); let stripe_read_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let mut state = source.read_next_stripe().await; let state = source.read_next_stripe().await;
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_STRIPE_READ, stripe_read_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_STRIPE_READ, stripe_read_stage_start);
let decode_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled); let decode_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let mut output_buf = reusable_buffers.pop().unwrap_or_default(); let mut output_buf = reusable_buffers.pop().unwrap_or_default();
let result = match decode_stripe_into( let result =
metrics_path, match decode_stripe_into(metrics_path, stage_metrics_enabled, engine, workspace, state, remaining, &mut output_buf) {
stage_metrics_enabled, Ok(true) => Ok(Some(output_buf)),
engine, Ok(false) => {
workspace, reusable_buffers.push(output_buf);
&mut state, Ok(None)
remaining, }
&mut output_buf, Err(err) => {
) { reusable_buffers.push(output_buf);
Ok(true) => Ok(Some(output_buf)), Err(err)
Ok(false) => { }
reusable_buffers.push(output_buf); };
Ok(None)
}
Err(err) => {
reusable_buffers.push(output_buf);
Err(err)
}
};
source.recycle_stripe(state);
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_DECODE, decode_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_DECODE, decode_stage_start);
if let Ok(Some(first_buf)) = result.as_ref() { if let Ok(Some(first_buf)) = result.as_ref() {
let mut remaining_after_first = remaining.saturating_sub(first_buf.len()); let mut remaining_after_first = remaining.saturating_sub(first_buf.len());
@@ -531,7 +503,7 @@ where
break; break;
} }
let stripe_read_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled); let stripe_read_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let mut state = source.read_next_stripe().await; let state = source.read_next_stripe().await;
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_STRIPE_READ, stripe_read_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_STRIPE_READ, stripe_read_stage_start);
let decode_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled); let decode_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let mut queued_buf = reusable_buffers.pop().unwrap_or_default(); let mut queued_buf = reusable_buffers.pop().unwrap_or_default();
@@ -540,11 +512,10 @@ where
stage_metrics_enabled, stage_metrics_enabled,
engine, engine,
workspace, workspace,
&mut state, state,
remaining_after_first, remaining_after_first,
&mut queued_buf, &mut queued_buf,
); );
source.recycle_stripe(state);
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_DECODE, decode_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_DECODE, decode_stage_start);
match queued_result { match queued_result {
Ok(true) => { Ok(true) => {
@@ -683,10 +654,6 @@ pub(crate) struct SyncErasureDecodeReader<R> {
} }
impl<R> SyncErasureDecodeReader<R> { impl<R> SyncErasureDecodeReader<R> {
#[allow(
dead_code,
reason = "default-metrics-path constructor used only by this file's tests (backlog#1823)"
)]
pub(crate) fn new(inner: R) -> Self { pub(crate) fn new(inner: R) -> Self {
Self::new_with_metrics_path(inner, GET_OBJECT_PATH_CODEC_STREAMING) Self::new_with_metrics_path(inner, GET_OBJECT_PATH_CODEC_STREAMING)
} }
@@ -750,7 +717,7 @@ fn decode_stripe_into<E>(
stage_metrics_enabled: bool, stage_metrics_enabled: bool,
engine: &E, engine: &E,
workspace: &mut E::Workspace, workspace: &mut E::Workspace,
state: &mut StripeReadState, state: StripeReadState,
remaining: usize, remaining: usize,
output: &mut Vec<u8>, output: &mut Vec<u8>,
) -> io::Result<bool> ) -> io::Result<bool>
@@ -758,7 +725,7 @@ where
E: ErasureDecodeEngine, E: ErasureDecodeEngine,
{ {
output.clear(); output.clear();
if state.is_empty() { if state.slots().is_empty() {
return Ok(false); return Ok(false);
} }
if !state.can_decode() { if !state.can_decode() {
@@ -774,12 +741,13 @@ where
); );
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_RECONSTRUCT, reconstruct_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_RECONSTRUCT, reconstruct_stage_start);
let emit_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled); let emit_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
emit_data_shards_into(state, engine.data_shards(), engine.block_size(), remaining, output)?; emit_data_shards_into(&state, engine.data_shards(), engine.block_size(), remaining, output)?;
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_EMIT, emit_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_EMIT, emit_stage_start);
return Ok(true); return Ok(true);
} }
let reconstruct_outcome = match engine.reconstruct_into(state.shards_mut(), workspace) { let (mut shards, _errs) = state.into_parts();
let reconstruct_outcome = match engine.reconstruct_into(&mut shards, workspace) {
Ok(outcome) => outcome, Ok(outcome) => outcome,
Err(err) => { Err(err) => {
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_RECONSTRUCT, reconstruct_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_RECONSTRUCT, reconstruct_stage_start);
@@ -789,7 +757,7 @@ where
rustfs_io_metrics::record_get_object_reconstruct_outcome(metrics_path, engine.engine_name(), reconstruct_outcome); rustfs_io_metrics::record_get_object_reconstruct_outcome(metrics_path, engine.engine_name(), reconstruct_outcome);
record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_RECONSTRUCT, reconstruct_stage_start); record_get_stage_duration_if_enabled(metrics_path, GET_STAGE_RECONSTRUCT, reconstruct_stage_start);
if state.shards_mut().len() < engine.data_shards() { if shards.len() < engine.data_shards() {
return Err(io::Error::new( return Err(io::Error::new(
ErrorKind::UnexpectedEof, ErrorKind::UnexpectedEof,
"decoded stripe has fewer shards than data shard count", "decoded stripe has fewer shards than data shard count",
@@ -798,7 +766,7 @@ where
let emit_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled); let emit_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
reserve_output_capacity(output, engine.block_size().min(remaining)); reserve_output_capacity(output, engine.block_size().min(remaining));
for shard in state.shards_mut().iter().take(engine.data_shards()) { for shard in shards.iter().take(engine.data_shards()) {
if output.len() >= remaining { if output.len() >= remaining {
break; break;
} }
@@ -813,7 +781,6 @@ where
Ok(true) Ok(true)
} }
#[allow(dead_code, reason = "shard emission asserted by this file's tests (backlog#1823)")]
fn emit_data_shards(state: &StripeReadState, data_shards: usize, block_size: usize, remaining: usize) -> io::Result<Vec<u8>> { fn emit_data_shards(state: &StripeReadState, data_shards: usize, block_size: usize, remaining: usize) -> io::Result<Vec<u8>> {
let mut output = Vec::new(); let mut output = Vec::new();
emit_data_shards_into(state, data_shards, block_size, remaining, &mut output)?; emit_data_shards_into(state, data_shards, block_size, remaining, &mut output)?;
@@ -839,7 +806,10 @@ fn emit_data_shards_into(
if output.len() >= remaining { if output.len() >= remaining {
break; break;
} }
let Some(shard) = state.data_bytes(index) else { let Some(slot) = state.slot_by_index(index) else {
return Err(io::Error::new(ErrorKind::UnexpectedEof, "decoded stripe is missing a data shard"));
};
let Some(shard) = slot.data_bytes() else {
return Err(io::Error::new(ErrorKind::UnexpectedEof, "decoded stripe is missing a data shard")); return Err(io::Error::new(ErrorKind::UnexpectedEof, "decoded stripe is missing a data shard"));
}; };
let copy_len = shard.len().min(remaining - output.len()); let copy_len = shard.len().min(remaining - output.len());
@@ -856,7 +826,7 @@ mod tests {
}; };
use crate::erasure::coding::decode::ParallelReader; use crate::erasure::coding::decode::ParallelReader;
use crate::erasure::coding::{BitrotReader, BitrotWriter, Erasure}; use crate::erasure::coding::{BitrotReader, BitrotWriter, Erasure};
use crate::set_disk::shard_source::StripeReadState; use crate::set_disk::shard_source::{ShardSlot, StripeReadState};
use rustfs_utils::HashAlgorithm; use rustfs_utils::HashAlgorithm;
use std::collections::VecDeque; use std::collections::VecDeque;
use std::future::{pending, poll_fn}; use std::future::{pending, poll_fn};
@@ -875,13 +845,6 @@ mod tests {
read_count: Option<Arc<AtomicUsize>>, read_count: Option<Arc<AtomicUsize>>,
} }
struct RecordingStripeSource {
stripes: VecDeque<StripeReadState>,
read_quorum: usize,
reads: usize,
recycles: usize,
}
struct BlockingSource { struct BlockingSource {
started: Arc<Notify>, started: Arc<Notify>,
dropped: Arc<AtomicUsize>, dropped: Arc<AtomicUsize>,
@@ -936,43 +899,25 @@ mod tests {
#[async_trait::async_trait] #[async_trait::async_trait]
impl ShardStripeSource for VecStripeSource { impl ShardStripeSource for VecStripeSource {
async fn read_next_stripe(&mut self) -> Box<StripeReadState> { async fn read_next_stripe(&mut self) -> StripeReadState {
if let Some(read_count) = &self.read_count { if let Some(read_count) = &self.read_count {
read_count.fetch_add(1, Ordering::SeqCst); read_count.fetch_add(1, Ordering::SeqCst);
} }
Box::new( self.stripes
self.stripes .pop_front()
.pop_front() .unwrap_or_else(|| StripeReadState::new(Vec::new(), self.read_quorum))
.unwrap_or_else(|| StripeReadState::from_parts(Vec::new(), Vec::new(), self.read_quorum)),
)
}
}
#[async_trait::async_trait]
impl ShardStripeSource for RecordingStripeSource {
async fn read_next_stripe(&mut self) -> Box<StripeReadState> {
self.reads += 1;
Box::new(
self.stripes
.pop_front()
.unwrap_or_else(|| StripeReadState::from_parts(Vec::new(), Vec::new(), self.read_quorum)),
)
}
fn recycle_stripe(&mut self, _state: Box<StripeReadState>) {
self.recycles += 1;
} }
} }
#[async_trait::async_trait] #[async_trait::async_trait]
impl ShardStripeSource for BlockingSource { impl ShardStripeSource for BlockingSource {
async fn read_next_stripe(&mut self) -> Box<StripeReadState> { async fn read_next_stripe(&mut self) -> StripeReadState {
let _guard = BlockingSourceDropGuard { let _guard = BlockingSourceDropGuard {
dropped: Arc::clone(&self.dropped), dropped: Arc::clone(&self.dropped),
}; };
self.started.notify_one(); self.started.notify_one();
pending::<()>().await; pending::<()>().await;
Box::new(StripeReadState::from_parts(Vec::new(), Vec::new(), self.read_quorum)) StripeReadState::new(Vec::new(), self.read_quorum)
} }
} }
@@ -1145,23 +1090,6 @@ mod tests {
}); });
} }
#[test]
fn fill_policy_production_cache_loads_once() {
use std::cell::Cell;
let loads = Cell::new(0);
for _ in 0..3 {
assert_eq!(
FillPolicy::cached_core(|| {
loads.set(loads.get() + 1);
FillPolicy::DualInFlight
}),
FillPolicy::DualInFlight
);
}
assert_eq!(loads.get(), 1, "the production fill policy must not re-read the environment per reader");
}
#[test] #[test]
fn erasure_decode_reader_rejects_invalid_engine_shape() { fn erasure_decode_reader_rejects_invalid_engine_shape() {
let source = VecStripeSource { let source = VecStripeSource {
@@ -1761,10 +1689,7 @@ mod tests {
.pop_front() .pop_front()
.expect("first stripe should exist"); .expect("first stripe should exist");
let mut source = VecStripeSource { let mut source = VecStripeSource {
stripes: VecDeque::from([ stripes: VecDeque::from([first_state, StripeReadState::new(Vec::new(), erasure.data_shards)]),
first_state,
StripeReadState::from_parts(Vec::new(), Vec::new(), erasure.data_shards),
]),
read_quorum: erasure.data_shards, read_quorum: erasure.data_shards,
read_count: None, read_count: None,
}; };
@@ -1799,14 +1724,13 @@ mod tests {
.stripes .stripes
.pop_front() .pop_front()
.expect("first stripe should exist"); .expect("first stripe should exist");
let mut source = RecordingStripeSource { let mut source = VecStripeSource {
stripes: VecDeque::from([ stripes: VecDeque::from([
first_state, first_state,
StripeReadState::from_parts(vec![Some(vec![1])], Vec::new(), erasure.data_shards), StripeReadState::new(vec![ShardSlot::data(0, vec![1])], erasure.data_shards),
]), ]),
read_quorum: erasure.data_shards, read_quorum: erasure.data_shards,
reads: 0, read_count: None,
recycles: 0,
}; };
let engine = LegacyEcDecodeEngine::new(erasure); let engine = LegacyEcDecodeEngine::new(erasure);
let mut workspace = engine.prepare_workspace(4).expect("workspace should be prepared"); let mut workspace = engine.prepare_workspace(4).expect("workspace should be prepared");
@@ -1832,8 +1756,6 @@ mod tests {
.kind(), .kind(),
ErrorKind::Other ErrorKind::Other
); );
assert_eq!(source.reads, 2, "the fill must read the primary and queued stripe");
assert_eq!(source.recycles, source.reads, "every completed stripe read must be recycled");
} }
#[tokio::test] #[tokio::test]
@@ -1846,7 +1768,7 @@ mod tests {
.stripes .stripes
.pop_front() .pop_front()
.expect("first stripe should exist"), .expect("first stripe should exist"),
StripeReadState::from_parts(Vec::new(), Vec::new(), erasure.data_shards), StripeReadState::new(Vec::new(), erasure.data_shards),
]), ]),
read_quorum: erasure.data_shards, read_quorum: erasure.data_shards,
read_count: None, read_count: None,
@@ -2106,11 +2028,17 @@ mod tests {
} }
#[test] #[test]
fn emit_data_shards_preserves_output_order() { fn emit_data_shards_preserves_output_order_for_out_of_order_slots() {
let state = let state = StripeReadState::new(
StripeReadState::from_parts(vec![Some(b"ab".to_vec()), Some(b"cd".to_vec()), Some(b"ef".to_vec())], Vec::new(), 2); vec![
ShardSlot::data(1, b"cd".to_vec()),
ShardSlot::data(0, b"ab".to_vec()),
ShardSlot::data(2, b"ef".to_vec()),
],
2,
);
let output = emit_data_shards(&state, 3, 6, 5).expect("data slots should emit by shard index"); let output = emit_data_shards(&state, 3, 6, 5).expect("out-of-order data slots should emit by shard index");
assert_eq!(output, b"abcde"); assert_eq!(output, b"abcde");
} }
@@ -2123,27 +2051,27 @@ mod tests {
}; };
let mut workspace = engine.prepare_workspace(4).expect("workspace should be prepared"); let mut workspace = engine.prepare_workspace(4).expect("workspace should be prepared");
let mut output = Vec::with_capacity(1); let mut output = Vec::with_capacity(1);
let mut short_state = StripeReadState::from_parts(vec![Some(vec![1, 2, 3, 4])], Vec::new(), 1); let short_state = StripeReadState::new(vec![ShardSlot::data(0, vec![1, 2, 3, 4])], 1);
let err = decode_stripe_into( let err = decode_stripe_into(
GET_OBJECT_PATH_CODEC_STREAMING, GET_OBJECT_PATH_CODEC_STREAMING,
false, false,
&engine, &engine,
&mut workspace, &mut workspace,
&mut short_state, short_state,
8, 8,
&mut output, &mut output,
) )
.expect_err("decoded stripe shorter than data shard count must fail"); .expect_err("decoded stripe shorter than data shard count must fail");
assert_eq!(err.kind(), ErrorKind::UnexpectedEof); assert_eq!(err.kind(), ErrorKind::UnexpectedEof);
let mut missing_state = StripeReadState::from_parts(vec![None, Some(vec![5, 6, 7, 8])], Vec::new(), 1); let missing_state = StripeReadState::from_parts(vec![None, Some(vec![5, 6, 7, 8])], Vec::new(), 1);
let err = decode_stripe_into( let err = decode_stripe_into(
GET_OBJECT_PATH_CODEC_STREAMING, GET_OBJECT_PATH_CODEC_STREAMING,
false, false,
&engine, &engine,
&mut workspace, &mut workspace,
&mut missing_state, missing_state,
8, 8,
&mut output, &mut output,
) )
@@ -2154,35 +2082,6 @@ mod tests {
assert!(output.capacity() >= 32); assert!(output.capacity() >= 32);
} }
#[test]
fn decode_stripe_reconstructs_in_place_without_replacing_slot_storage() {
let erasure = Erasure::new(2, 1, 8);
let engine = LegacyEcDecodeEngine::new(erasure.clone());
let mut workspace = engine.prepare_workspace(4).expect("workspace should be prepared");
let encoded = erasure.encode_data(b"abcdefgh").expect("test stripe should encode");
let mut shards = encoded.into_iter().map(|shard| Some(shard.to_vec())).collect::<Vec<_>>();
shards[0] = None;
let mut state = StripeReadState::from_parts(shards, vec![Some(DiskError::FileCorrupt)], 2);
let before = state.scratch_storage();
let mut output = Vec::new();
let decoded = decode_stripe_into(
GET_OBJECT_PATH_CODEC_STREAMING,
false,
&engine,
&mut workspace,
&mut state,
8,
&mut output,
)
.expect("degraded stripe should reconstruct");
assert!(decoded);
assert_eq!(output, b"abcdefgh");
assert_eq!(state.scratch_storage().0, before.0, "reconstruction must retain shard slot storage");
assert_eq!(state.scratch_storage().1, before.1, "unused error storage must not be rebuilt");
}
#[tokio::test] #[tokio::test]
async fn erasure_decode_reader_reports_short_source() { async fn erasure_decode_reader_reports_short_source() {
let erasure = Erasure::new(4, 2, 32); let erasure = Erasure::new(4, 2, 32);
+42 -197
View File
@@ -18,12 +18,10 @@ use crate::disk::error_reduce::{
}; };
use crate::erasure::coding::BitrotWriterWrapper; use crate::erasure::coding::BitrotWriterWrapper;
use crate::erasure::coding::Erasure; use crate::erasure::coding::Erasure;
use crate::erasure::coding::erasure::EncodedBlock;
use crate::runtime::sources as runtime_sources; use crate::runtime::sources as runtime_sources;
use bytes::{Bytes, BytesMut}; use bytes::{Bytes, BytesMut};
use futures::StreamExt; use futures::StreamExt;
use futures::stream::FuturesUnordered; use futures::stream::FuturesUnordered;
use rustfs_utils::HashAlgorithm;
use std::sync::Arc; use std::sync::Arc;
use std::time::Instant; use std::time::Instant;
use std::vec; use std::vec;
@@ -166,7 +164,6 @@ where
if total == 0 { Ok(None) } else { Ok(Some(total)) } if total == 0 { Ok(None) } else { Ok(Some(total)) }
} }
#[allow(dead_code, reason = "byte accounting asserted by this file's tests (backlog#1823)")]
fn queued_block_bytes(block: &[Bytes]) -> usize { fn queued_block_bytes(block: &[Bytes]) -> usize {
block.iter().map(Bytes::len).sum() block.iter().map(Bytes::len).sum()
} }
@@ -226,8 +223,8 @@ async fn send_queued<T>(
sender.send(InflightEntry::new(entry, bytes)).await sender.send(InflightEntry::new(entry, bytes)).await
} }
fn queued_batch_bytes(batch: &[EncodedBlock]) -> usize { fn queued_batch_bytes(batch: &[Vec<Bytes>]) -> usize {
batch.iter().map(EncodedBlock::queued_bytes).sum() batch.iter().map(|block| queued_block_bytes(block)).sum()
} }
fn dominant_error_summary_label(summary: &WriteQuorumFailureSummary) -> &'static str { fn dominant_error_summary_label(summary: &WriteQuorumFailureSummary) -> &'static str {
@@ -339,7 +336,7 @@ impl<'a> MultiWriter<'a> {
} }
} }
async fn write_shard(writer_opt: &mut Option<BitrotWriterWrapper>, err: &mut Option<Error>, shard: &[u8]) { async fn write_shard(writer_opt: &mut Option<BitrotWriterWrapper>, err: &mut Option<Error>, shard: &Bytes) {
match writer_opt { match writer_opt {
Some(writer) => { Some(writer) => {
match writer.write(shard).await { match writer.write(shard).await {
@@ -364,20 +361,12 @@ impl<'a> MultiWriter<'a> {
} }
pub async fn write(&mut self, data: Vec<Bytes>) -> std::io::Result<()> { pub async fn write(&mut self, data: Vec<Bytes>) -> std::io::Result<()> {
self.write_shards(data.iter().map(Bytes::as_ref)).await assert_eq!(data.len(), self.writers.len());
}
async fn write_block(&mut self, block: &EncodedBlock) -> std::io::Result<()> {
self.write_shards(block.shards()).await
}
async fn write_shards<'b>(&mut self, shards: impl ExactSizeIterator<Item = &'b [u8]>) -> std::io::Result<()> {
assert_eq!(shards.len(), self.writers.len());
let budget = self.next_progress_budget(); let budget = self.next_progress_budget();
{ {
let mut futures = FuturesUnordered::new(); let mut futures = FuturesUnordered::new();
for ((writer_opt, err), shard) in self.writers.iter_mut().zip(self.errs.iter_mut()).zip(shards) { for ((writer_opt, err), shard) in self.writers.iter_mut().zip(self.errs.iter_mut()).zip(data.iter()) {
if err.is_some() { if err.is_some() {
continue; // Skip if we already have an error for this writer continue; // Skip if we already have an error for this writer
} }
@@ -501,10 +490,10 @@ impl<'a> MultiWriter<'a> {
} }
impl Erasure { impl Erasure {
async fn encode_block(self: Arc<Self>, encode_buf: Vec<u8>, len: usize) -> std::io::Result<(EncodedBlock, Vec<u8>)> { async fn encode_block(self: Arc<Self>, encode_buf: Vec<u8>, len: usize) -> std::io::Result<(Vec<Bytes>, Vec<u8>)> {
let encode_stage_start = stage_timer_if_enabled(); let encode_stage_start = stage_timer_if_enabled();
let encode_once = move || { let encode_once = move || {
let res = self.encode_data_block(&encode_buf[..len]); let res = self.encode_data(&encode_buf[..len]);
(res, encode_buf) (res, encode_buf)
}; };
@@ -529,9 +518,9 @@ impl Erasure {
Ok((res?, returned_buf)) Ok((res?, returned_buf))
} }
async fn encode_block_bytes_mut(self: Arc<Self>, encode_buf: BytesMut, len: usize) -> std::io::Result<EncodedBlock> { async fn encode_block_bytes_mut(self: Arc<Self>, encode_buf: BytesMut, len: usize) -> std::io::Result<Vec<Bytes>> {
let encode_stage_start = stage_timer_if_enabled(); let encode_stage_start = stage_timer_if_enabled();
let encode_once = move || self.encode_data_bytes_mut_block(encode_buf, len); let encode_once = move || self.encode_data_bytes_mut(encode_buf, len);
let res = match tokio::runtime::Handle::current().runtime_flavor() { let res = match tokio::runtime::Handle::current().runtime_flavor() {
// Same rationale as encode_block: inline the short EC burst on the // Same rationale as encode_block: inline the short EC burst on the
@@ -587,46 +576,13 @@ impl Erasure {
)); ));
} }
let block = self.encode_data_owned_block(buf)?; let shards = self.encode_data_owned(buf)?;
let mut mw = MultiWriter::new(writers, quorum); let mut mw = MultiWriter::new(writers, quorum);
mw.write_block(&block).await?; mw.write(shards).await?;
mw.shutdown().await?; mw.shutdown().await?;
Ok((reader, total)) Ok((reader, total))
} }
/// Encode a small inline object directly into its per-disk bitrot payloads.
/// The returned bytes are the same `[hash][shard]` representation produced
/// by `BitrotWriter`, ready to be embedded in each disk's staged `xl.meta`.
#[hotpath::measure(impl_type = "Erasure")]
pub(crate) async fn encode_inline_shards_with_size_hint<R>(
self: Arc<Self>,
mut reader: R,
size_hint: usize,
) -> std::io::Result<(R, usize, Vec<Bytes>)>
where
R: AsyncRead + Send + Sync + Unpin,
{
use tokio::io::AsyncReadExt;
let mut buf = Vec::with_capacity(small_ingest_capacity(&self, size_hint));
let total = reader.read_to_end(&mut buf).await?;
if total == 0 {
return Ok((reader, 0, Vec::new()));
}
let block = self.encode_data_owned_block(buf)?;
let mut inline_shards = Vec::with_capacity(block.shards().len());
for shard in block.shards() {
let hash = HashAlgorithm::HighwayHash256S.hash_encode(shard);
let mut encoded = BytesMut::with_capacity(hash.as_ref().len() + shard.len());
encoded.extend_from_slice(hash.as_ref());
encoded.extend_from_slice(shard);
inline_shards.push(encoded.freeze());
}
Ok((reader, total, inline_shards))
}
#[hotpath::measure(impl_type = "Erasure")] #[hotpath::measure(impl_type = "Erasure")]
pub async fn encode<R>( pub async fn encode<R>(
self: Arc<Self>, self: Arc<Self>,
@@ -668,7 +624,7 @@ impl Erasure {
let expanded_block_bytes = self.shard_size().saturating_mul(self.total_shard_count()); let expanded_block_bytes = self.shard_size().saturating_mul(self.total_shard_count());
let max_inflight_bytes = erasure_encode_max_inflight_bytes(); let max_inflight_bytes = erasure_encode_max_inflight_bytes();
let inflight_blocks = encode_channel_capacity(expanded_block_bytes, max_inflight_bytes); let inflight_blocks = encode_channel_capacity(expanded_block_bytes, max_inflight_bytes);
let (tx, mut rx) = mpsc::channel::<InflightEntry<EncodedBlock>>(inflight_blocks); let (tx, mut rx) = mpsc::channel::<InflightEntry<Vec<Bytes>>>(inflight_blocks);
let mut task = AbortOnDropTask::new(tokio::spawn(async move { let mut task = AbortOnDropTask::new(tokio::spawn(async move {
let block_size = self.block_size; let block_size = self.block_size;
@@ -690,7 +646,7 @@ impl Erasure {
let encode_buf = buf; let encode_buf = buf;
let res = self.clone().encode_block_bytes_mut(encode_buf, n).await?; let res = self.clone().encode_block_bytes_mut(encode_buf, n).await?;
buf = BytesMut::with_capacity(ingest_capacity); buf = BytesMut::with_capacity(ingest_capacity);
let queued_bytes = res.queued_bytes(); let queued_bytes = queued_block_bytes(&res);
let _producer_stage = rustfs_io_metrics::track_ec_encode_producer_bytes(queued_bytes); let _producer_stage = rustfs_io_metrics::track_ec_encode_producer_bytes(queued_bytes);
let send_wait_stage_start = stage_timer_if_enabled(); let send_wait_stage_start = stage_timer_if_enabled();
if let Err(err) = send_queued(&tx, res, queued_bytes).await { if let Err(err) = send_queued(&tx, res, queued_bytes).await {
@@ -720,7 +676,7 @@ impl Erasure {
let encode_buf = std::mem::take(&mut buf); let encode_buf = std::mem::take(&mut buf);
let (res, returned_buf) = self.clone().encode_block(encode_buf, n).await?; let (res, returned_buf) = self.clone().encode_block(encode_buf, n).await?;
buf = returned_buf; buf = returned_buf;
let queued_bytes = res.queued_bytes(); let queued_bytes = queued_block_bytes(&res);
let _producer_stage = rustfs_io_metrics::track_ec_encode_producer_bytes(queued_bytes); let _producer_stage = rustfs_io_metrics::track_ec_encode_producer_bytes(queued_bytes);
let send_wait_stage_start = stage_timer_if_enabled(); let send_wait_stage_start = stage_timer_if_enabled();
if let Err(err) = send_queued(&tx, res, queued_bytes).await { if let Err(err) = send_queued(&tx, res, queued_bytes).await {
@@ -764,9 +720,9 @@ impl Erasure {
if block.is_empty() { if block.is_empty() {
break; break;
} }
let _writer_stage = rustfs_io_metrics::track_ec_encode_writer_bytes(block.queued_bytes()); let _writer_stage = rustfs_io_metrics::track_ec_encode_writer_bytes(queued_block_bytes(&block));
let write_stage_start = stage_timer_if_enabled(); let write_stage_start = stage_timer_if_enabled();
if let Err(err) = writers.write_block(&block).await { if let Err(err) = writers.write(block).await {
write_err = Some(err); write_err = Some(err);
break; break;
} }
@@ -813,7 +769,7 @@ impl Erasure {
let inflight_blocks = encode_channel_capacity(expanded_block_bytes, max_inflight_bytes); let inflight_blocks = encode_channel_capacity(expanded_block_bytes, max_inflight_bytes);
let batch_blocks = encode_batch_block_count().min(inflight_blocks); let batch_blocks = encode_batch_block_count().min(inflight_blocks);
let channel_capacity = inflight_blocks.div_ceil(batch_blocks).max(1); let channel_capacity = inflight_blocks.div_ceil(batch_blocks).max(1);
let (tx, mut rx) = mpsc::channel::<InflightEntry<Vec<EncodedBlock>>>(channel_capacity); let (tx, mut rx) = mpsc::channel::<InflightEntry<Vec<Vec<Bytes>>>>(channel_capacity);
let mut task = AbortOnDropTask::new(tokio::spawn(async move { let mut task = AbortOnDropTask::new(tokio::spawn(async move {
let block_size = self.block_size; let block_size = self.block_size;
@@ -830,7 +786,7 @@ impl Erasure {
let encode_buf = std::mem::take(&mut buf); let encode_buf = std::mem::take(&mut buf);
let (res, returned_buf) = self.clone().encode_block(encode_buf, n).await?; let (res, returned_buf) = self.clone().encode_block(encode_buf, n).await?;
buf = returned_buf; buf = returned_buf;
let queued_bytes = res.queued_bytes(); let queued_bytes = queued_block_bytes(&res);
pending_batch_bytes = pending_batch_bytes.saturating_add(queued_bytes); pending_batch_bytes = pending_batch_bytes.saturating_add(queued_bytes);
pending_batch.push(res); pending_batch.push(res);
drop(pending_batch_stage.take()); drop(pending_batch_stage.take());
@@ -889,7 +845,7 @@ impl Erasure {
let _writer_stage = rustfs_io_metrics::track_ec_encode_writer_bytes(queued_batch_bytes(&batch)); let _writer_stage = rustfs_io_metrics::track_ec_encode_writer_bytes(queued_batch_bytes(&batch));
let write_stage_start = stage_timer_if_enabled(); let write_stage_start = stage_timer_if_enabled();
for block in batch { for block in batch {
if let Err(err) = writers.write_block(&block).await { if let Err(err) = writers.write(block).await {
write_err = Some(err); write_err = Some(err);
break; break;
} }
@@ -1939,11 +1895,7 @@ mod tests {
let baseline = rustfs_io_metrics::current_ec_encode_inflight_bytes(); let baseline = rustfs_io_metrics::current_ec_encode_inflight_bytes();
let (tx, rx) = mpsc::channel(2); let (tx, rx) = mpsc::channel(2);
let mut rx = rx; let mut rx = rx;
let erasure = Erasure::new(1, 0, 16); let batch = vec![vec![Bytes::from_static(b"queued")], vec![Bytes::from_static(b"batch")]];
let batch = vec![
erasure.encode_data_block(b"queued").expect("first block should encode"),
erasure.encode_data_block(b"batch").expect("second block should encode"),
];
let batch_bytes = queued_batch_bytes(&batch); let batch_bytes = queued_batch_bytes(&batch);
send_queued(&tx, batch, batch_bytes).await.expect("batch should be queued"); send_queued(&tx, batch, batch_bytes).await.expect("batch should be queued");
@@ -2165,39 +2117,6 @@ mod tests {
); );
} }
#[tokio::test]
async fn cancelling_inline_small_drops_stalled_write() {
const BLOCK_SIZE: usize = 16;
let (writer_entered_tx, writer_entered) = oneshot::channel();
let writes = Arc::new(std::sync::atomic::AtomicUsize::new(0));
let mut writers = vec![Some(bitrot_writer_plain(
StallOnWriteWithSignal {
entered: Some(writer_entered_tx),
writes: writes.clone(),
},
BLOCK_SIZE,
))];
let erasure = Arc::new(Erasure::new(1, 0, BLOCK_SIZE));
let reader = tokio::io::BufReader::new(Cursor::new(vec![0xA5; BLOCK_SIZE - 1]));
let encode = tokio::spawn(async move { erasure.encode_inline_small(reader, &mut writers, 1).await });
tokio::time::timeout(Duration::from_secs(1), writer_entered)
.await
.expect("inline writer should enter before cancellation")
.expect("stalling writer should signal entry");
encode.abort();
assert!(
matches!(encode.await, Err(err) if err.is_cancelled()),
"inline encode task should be cancelled"
);
assert_eq!(
writes.load(std::sync::atomic::Ordering::SeqCst),
1,
"cancellation must drop the stalled write instead of polling it again"
);
}
#[tokio::test] #[tokio::test]
async fn encode_returns_unexpected_eof_for_truncated_limited_reader() { async fn encode_returns_unexpected_eof_for_truncated_limited_reader() {
let committed = Arc::new(Mutex::new(Vec::new())); let committed = Arc::new(Mutex::new(Vec::new()));
@@ -2317,11 +2236,11 @@ mod tests {
.expect("bytesmut encode should succeed on current-thread runtime"); .expect("bytesmut encode should succeed on current-thread runtime");
let expected_shard_size = payload.len().div_ceil(erasure.data_shards); let expected_shard_size = payload.len().div_ceil(erasure.data_shards);
assert_eq!(shards.shards().len(), erasure.total_shard_count()); assert_eq!(shards.len(), erasure.total_shard_count());
assert!(shards.shards().all(|shard| shard.len() == expected_shard_size)); assert!(shards.iter().all(|shard| shard.len() == expected_shard_size));
let mut restored = Vec::new(); let mut restored = Vec::new();
for shard in shards.shards().take(erasure.data_shards) { for shard in shards.iter().take(erasure.data_shards) {
restored.extend_from_slice(shard); restored.extend_from_slice(shard);
} }
restored.truncate(payload.len()); restored.truncate(payload.len());
@@ -2424,41 +2343,6 @@ mod tests {
assert!(committed.lock().unwrap().is_empty()); assert!(committed.lock().unwrap().is_empty());
} }
#[tokio::test]
async fn encode_inline_shards_matches_writer_bitrot_layout() {
const DATA_SHARDS: usize = 2;
const PARITY_SHARDS: usize = 2;
const BLOCK_SIZE: usize = 64;
let checksum_algo = HashAlgorithm::HighwayHash256S;
for uses_legacy in [false, true] {
let erasure = Arc::new(Erasure::new_with_options(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE, uses_legacy));
for payload in [Vec::new(), vec![0xA5], vec![0x5A; BLOCK_SIZE - 1], vec![0xC3; BLOCK_SIZE]] {
let reader = tokio::io::BufReader::new(Cursor::new(payload.clone()));
let (_reader, total, inline_shards) = erasure
.clone()
.encode_inline_shards_with_size_hint(reader, payload.len())
.await
.expect("inline shards should encode");
assert_eq!(total, payload.len());
if payload.is_empty() {
assert!(inline_shards.is_empty());
continue;
}
let raw_shards = erasure.encode_data(&payload).expect("reference shards should encode");
assert_eq!(inline_shards.len(), DATA_SHARDS + PARITY_SHARDS);
for (inline, raw) in inline_shards.iter().zip(raw_shards) {
let mut writer =
BitrotWriterWrapper::new(CustomWriter::new_inline_buffer(), raw.len(), checksum_algo.clone());
writer.write(&raw).await.expect("reference writer should accept shard");
writer.shutdown().await.expect("reference writer should shutdown");
assert_eq!(inline.as_ref(), writer.into_inline_data().expect("reference writer should retain bytes"));
}
}
}
}
/// encode_inline_small: small payload is encoded into the correct number of shards /// encode_inline_small: small payload is encoded into the correct number of shards
/// and each writer receives data after shutdown. /// and each writer receives data after shutdown.
#[tokio::test] #[tokio::test]
@@ -2622,7 +2506,7 @@ mod tests {
assert_eq!(&next[..], &data[16..]); assert_eq!(&next[..], &data[16..]);
} }
async fn committed_shards_for_pipeline(pipeline: EncodePipeline, uses_legacy: bool, payload: &[u8]) -> Vec<Vec<u8>> { async fn committed_shards_for_ingest_mode(use_bytesmut_ingest: bool, uses_legacy: bool, payload: &[u8]) -> Vec<Vec<u8>> {
const DATA_SHARDS: usize = 2; const DATA_SHARDS: usize = 2;
const PARITY_SHARDS: usize = 2; const PARITY_SHARDS: usize = 2;
const TOTAL_SHARDS: usize = DATA_SHARDS + PARITY_SHARDS; const TOTAL_SHARDS: usize = DATA_SHARDS + PARITY_SHARDS;
@@ -2636,16 +2520,10 @@ mod tests {
let erasure = Arc::new(Erasure::new_with_options(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE, uses_legacy)); let erasure = Arc::new(Erasure::new_with_options(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE, uses_legacy));
let reader = tokio::io::BufReader::new(Cursor::new(payload.to_vec())); let reader = tokio::io::BufReader::new(Cursor::new(payload.to_vec()));
let (_reader, total) = match pipeline { let (_reader, total) = erasure
EncodePipeline::Vec => { .encode_with_ingest_mode(reader, &mut writers, DATA_SHARDS, use_bytesmut_ingest)
erasure .await
.encode_with_ingest_mode(reader, &mut writers, DATA_SHARDS, false) .expect("encode should succeed");
.await
}
EncodePipeline::BytesMut => erasure.encode_with_ingest_mode(reader, &mut writers, DATA_SHARDS, true).await,
EncodePipeline::Batched => erasure.encode_batched(reader, &mut writers, DATA_SHARDS).await,
}
.expect("encode should succeed");
assert_eq!(total, payload.len()); assert_eq!(total, payload.len());
committed committed
@@ -2654,64 +2532,31 @@ mod tests {
.collect() .collect()
} }
async fn expected_committed_shards(uses_legacy: bool, payload: &[u8]) -> Vec<Vec<u8>> { /// HP-10 (rustfs/backlog#931) merge gate: the BytesMut ingest path must produce
const DATA_SHARDS: usize = 2; /// byte-for-byte identical shard streams to the default Vec ingest path, for both
const PARITY_SHARDS: usize = 2; /// legacy-aware shard-size formulas, across empty, sub-block, exactly-full-block,
const TOTAL_SHARDS: usize = DATA_SHARDS + PARITY_SHARDS; /// and multi-block-with-partial-tail payloads.
const BLOCK_SIZE: usize = 64;
let committed: Vec<Arc<Mutex<Vec<u8>>>> = (0..TOTAL_SHARDS).map(|_| Arc::new(Mutex::new(Vec::new()))).collect();
let mut writers: Vec<BitrotWriterWrapper> = committed
.iter()
.map(|c| bitrot_writer(DeferredCommitWriter::new(c.clone()), BLOCK_SIZE / DATA_SHARDS))
.collect();
let erasure = Erasure::new_with_options(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE, uses_legacy);
for block in payload.chunks(BLOCK_SIZE) {
let shards = erasure.encode_data(block).expect("reference block should encode");
for (writer, shard) in writers.iter_mut().zip(shards) {
let written = writer.write(&shard).await.expect("reference shard should write");
assert_eq!(written, shard.len());
}
}
for writer in &mut writers {
writer.shutdown().await.expect("reference writer should commit");
}
committed
.iter()
.map(|c| c.lock().expect("committed buffer should be lockable").clone())
.collect()
}
/// The streaming and batched paths must produce the same bitrot-wrapped shard
/// bytes as the public block encoder for both shard-size formulas and all block
/// boundary shapes.
#[tokio::test] #[tokio::test]
async fn bytesmut_ingest_matches_vec_ingest_byte_for_byte() { async fn bytesmut_ingest_matches_vec_ingest_byte_for_byte() {
const BLOCK_SIZE: usize = 64; const BLOCK_SIZE: usize = 64;
let payloads: Vec<Vec<u8>> = vec![ let payloads: Vec<Vec<u8>> = vec![
Vec::new(), Vec::new(),
vec![1], b"tiny".to_vec(),
vec![2; BLOCK_SIZE - 1],
(0..BLOCK_SIZE as u32).map(|i| i as u8).collect(), // exactly one full block (0..BLOCK_SIZE as u32).map(|i| i as u8).collect(), // exactly one full block
vec![4; BLOCK_SIZE + 1], vec![3u8; BLOCK_SIZE * 4], // whole number of blocks
vec![3u8; BLOCK_SIZE * 4], // whole number of blocks
(0..(BLOCK_SIZE * 3 + 7) as u32).map(|i| (i % 251) as u8).collect(), // partial tail (0..(BLOCK_SIZE * 3 + 7) as u32).map(|i| (i % 251) as u8).collect(), // partial tail
]; ];
for uses_legacy in [false, true] { for uses_legacy in [false, true] {
for payload in &payloads { for payload in &payloads {
let expected = expected_committed_shards(uses_legacy, payload).await; let vec_path = committed_shards_for_ingest_mode(false, uses_legacy, payload).await;
for pipeline in [EncodePipeline::Vec, EncodePipeline::BytesMut, EncodePipeline::Batched] { let bytesmut_path = committed_shards_for_ingest_mode(true, uses_legacy, payload).await;
let actual = committed_shards_for_pipeline(pipeline, uses_legacy, payload).await; assert_eq!(
assert_eq!( vec_path,
actual, bytesmut_path,
expected, "ingest paths must be byte-identical (legacy={uses_legacy}, payload_len={})",
"streaming shards must match the public block encoder (legacy={uses_legacy}, payload_len={})", payload.len()
payload.len() );
);
}
} }
} }
} }
+140 -338
View File
@@ -29,58 +29,12 @@ use tokio::io::AsyncRead;
use tracing::warn; use tracing::warn;
use uuid::Uuid; use uuid::Uuid;
pub(crate) struct EncodedBlock {
data: Bytes,
shard_size: usize,
}
impl EncodedBlock {
fn empty() -> Self {
Self {
data: Bytes::new(),
shard_size: 0,
}
}
pub(crate) fn is_empty(&self) -> bool {
self.data.is_empty()
}
pub(crate) fn queued_bytes(&self) -> usize {
self.data.len()
}
pub(crate) fn shards(&self) -> impl ExactSizeIterator<Item = &[u8]> {
debug_assert!(self.shard_size > 0, "only non-empty encoded blocks reach shard writers");
debug_assert_eq!(self.data.len() % self.shard_size, 0);
self.data.chunks_exact(self.shard_size)
}
fn into_shards(mut self, shard_count: usize) -> Vec<Bytes> {
if self.shard_size == 0 {
return vec![Bytes::new(); shard_count];
}
let mut shards = Vec::with_capacity(shard_count);
for _ in 0..shard_count {
shards.push(self.data.split_to(self.shard_size));
}
shards
}
}
const MODERN_MAX_TOTAL_SHARDS: usize = <reed_solomon_erasure::galois_8::Field as reed_solomon_erasure::Field>::ORDER; const MODERN_MAX_TOTAL_SHARDS: usize = <reed_solomon_erasure::galois_8::Field as reed_solomon_erasure::Field>::ORDER;
const MODERN_REED_SOLOMON_CACHE_MAX_ENTRIES: usize = 64; const MODERN_REED_SOLOMON_CACHE_MAX_ENTRIES: usize = 64;
const LEGACY_REED_SOLOMON_CACHE_MAX_ENTRIES: usize = 16;
// Vec growth may retain twice the requested logical length. Keeping the logical
// workspace at half the budget bounds each cached workspace's shard allocation to 1 MiB.
const LEGACY_REED_SOLOMON_CACHE_MAX_LOGICAL_SHARD_BYTES_PER_WORKSPACE: usize = 512 * 1024;
type ModernReedSolomonCache = RwLock<HashMap<(usize, usize), Arc<ReedSolomon>>>; type ModernReedSolomonCache = RwLock<HashMap<(usize, usize), Arc<ReedSolomon>>>;
type LegacyReedSolomonCache = RwLock<HashMap<(usize, usize), Arc<LegacyReedSolomonEncoder>>>;
static MODERN_REED_SOLOMON_CACHE: OnceLock<ModernReedSolomonCache> = OnceLock::new(); static MODERN_REED_SOLOMON_CACHE: OnceLock<ModernReedSolomonCache> = OnceLock::new();
static LEGACY_REED_SOLOMON_CACHE: OnceLock<LegacyReedSolomonCache> = OnceLock::new();
/// Errors returned when constructing an [`Erasure`] codec. /// Errors returned when constructing an [`Erasure`] codec.
#[derive(Debug, thiserror::Error)] #[derive(Debug, thiserror::Error)]
@@ -147,61 +101,43 @@ pub fn calc_shard_size_legacy(block_size: usize, data_shards: usize) -> usize {
struct LegacyReedSolomonEncoder { struct LegacyReedSolomonEncoder {
data_shards: usize, data_shards: usize,
parity_shards: usize, parity_shards: usize,
cache_workspaces: bool, encoder_cache: std::sync::RwLock<Option<reed_solomon_simd::ReedSolomonEncoder>>,
encoder_cache: RwLock<Option<reed_solomon_simd::ReedSolomonEncoder>>, decoder_cache: std::sync::RwLock<Option<reed_solomon_simd::ReedSolomonDecoder>>,
decoder_cache: RwLock<Option<reed_solomon_simd::ReedSolomonDecoder>>, }
impl Clone for LegacyReedSolomonEncoder {
fn clone(&self) -> Self {
Self {
data_shards: self.data_shards,
parity_shards: self.parity_shards,
encoder_cache: std::sync::RwLock::new(None),
decoder_cache: std::sync::RwLock::new(None),
}
}
} }
impl LegacyReedSolomonEncoder { impl LegacyReedSolomonEncoder {
fn new(data_shards: usize, parity_shards: usize) -> io::Result<Self> { fn new(_data_shards: usize, _parity_shards: usize) -> io::Result<Self> {
Self::with_workspace_cache(data_shards, parity_shards, false)
}
fn with_workspace_cache(data_shards: usize, parity_shards: usize, cache_workspaces: bool) -> io::Result<Self> {
Ok(Self { Ok(Self {
data_shards, data_shards: _data_shards,
parity_shards, parity_shards: _parity_shards,
cache_workspaces, encoder_cache: std::sync::RwLock::new(None),
encoder_cache: RwLock::new(None), decoder_cache: std::sync::RwLock::new(None),
decoder_cache: RwLock::new(None),
}) })
} }
fn logical_shard_bytes_upper_bound(&self, shard_len: usize) -> Option<usize> {
let aligned_shard_len = shard_len.checked_add(63)?.checked_div(64)?.checked_mul(64)?;
let high_rate_decoder_work_count = self
.parity_shards
.checked_next_power_of_two()?
.checked_add(self.data_shards)?
.checked_next_power_of_two()?;
let low_rate_decoder_work_count = self
.data_shards
.checked_next_power_of_two()?
.checked_add(self.parity_shards)?
.checked_next_power_of_two()?;
aligned_shard_len.checked_mul(high_rate_decoder_work_count.max(low_rate_decoder_work_count))
}
fn should_cache_workspace(&self, shard_len: usize) -> bool {
self.cache_workspaces
&& self
.logical_shard_bytes_upper_bound(shard_len)
.is_some_and(|bytes| bytes <= LEGACY_REED_SOLOMON_CACHE_MAX_LOGICAL_SHARD_BYTES_PER_WORKSPACE)
}
fn encode(&self, shards: SmallVec<[&mut [u8]; 16]>) -> io::Result<()> { fn encode(&self, shards: SmallVec<[&mut [u8]; 16]>) -> io::Result<()> {
let mut shards_vec: Vec<&mut [u8]> = shards.into_vec(); let mut shards_vec: Vec<&mut [u8]> = shards.into_vec();
if shards_vec.is_empty() { if shards_vec.is_empty() {
return Ok(()); return Ok(());
} }
let shard_len = shards_vec[0].len(); let shard_len = shards_vec[0].len();
let cached_encoder = self
.encoder_cache
.write()
.map_err(|_| io::Error::other("Failed to acquire encoder cache lock"))?
.take();
let mut encoder = { let mut encoder = {
match cached_encoder { let mut cache_guard = self
.encoder_cache
.write()
.map_err(|_| io::Error::other("Failed to acquire encoder cache lock"))?;
match cache_guard.take() {
Some(mut cached) => { Some(mut cached) => {
if cached.reset(self.data_shards, self.parity_shards, shard_len).is_err() { if cached.reset(self.data_shards, self.parity_shards, shard_len).is_err() {
reed_solomon_simd::ReedSolomonEncoder::new(self.data_shards, self.parity_shards, shard_len) reed_solomon_simd::ReedSolomonEncoder::new(self.data_shards, self.parity_shards, shard_len)
@@ -228,15 +164,10 @@ impl LegacyReedSolomonEncoder {
} }
} }
drop(result); drop(result);
if self.should_cache_workspace(shard_len) { *self
let mut cache = self .encoder_cache
.encoder_cache .write()
.write() .map_err(|_| io::Error::other("Failed to return encoder to cache"))? = Some(encoder);
.map_err(|_| io::Error::other("Failed to return encoder to cache"))?;
if cache.is_none() {
*cache = Some(encoder);
}
}
Ok(()) Ok(())
} }
@@ -250,13 +181,13 @@ impl LegacyReedSolomonEncoder {
.find_map(|s| s.as_ref().map(|v| v.len())) .find_map(|s| s.as_ref().map(|v| v.len()))
.ok_or_else(|| io::Error::other("No valid shards found for reconstruction"))?; .ok_or_else(|| io::Error::other("No valid shards found for reconstruction"))?;
let cached_decoder = self
.decoder_cache
.write()
.map_err(|_| io::Error::other("Failed to acquire decoder cache lock"))?
.take();
let mut decoder = { let mut decoder = {
match cached_decoder { let mut cache_guard = self
.decoder_cache
.write()
.map_err(|_| io::Error::other("Failed to acquire decoder cache lock"))?;
match cache_guard.take() {
Some(mut cached_decoder) => { Some(mut cached_decoder) => {
if let Err(e) = cached_decoder.reset(self.data_shards, self.parity_shards, shard_len) { if let Err(e) = cached_decoder.reset(self.data_shards, self.parity_shards, shard_len) {
warn!("Failed to reset SIMD decoder: {:?}, creating new one", e); warn!("Failed to reset SIMD decoder: {:?}, creating new one", e);
@@ -303,15 +234,10 @@ impl LegacyReedSolomonEncoder {
drop(result); drop(result);
if self.should_cache_workspace(shard_len) { *self
let mut cache = self .decoder_cache
.decoder_cache .write()
.write() .map_err(|_| io::Error::other("Failed to return decoder to cache"))? = Some(decoder);
.map_err(|_| io::Error::other("Failed to return decoder to cache"))?;
if cache.is_none() {
*cache = Some(decoder);
}
}
Ok(()) Ok(())
} }
@@ -469,39 +395,6 @@ fn cached_modern_reed_solomon(data_shards: usize, parity_shards: usize) -> Resul
Ok(encoder) Ok(encoder)
} }
fn cached_legacy_reed_solomon(data_shards: usize, parity_shards: usize) -> io::Result<Arc<LegacyReedSolomonEncoder>> {
let cache = LEGACY_REED_SOLOMON_CACHE.get_or_init(|| RwLock::new(HashMap::new()));
cached_legacy_reed_solomon_in(cache, data_shards, parity_shards)
}
fn cached_legacy_reed_solomon_in(
cache: &LegacyReedSolomonCache,
data_shards: usize,
parity_shards: usize,
) -> io::Result<Arc<LegacyReedSolomonEncoder>> {
let key = (data_shards, parity_shards);
if let Some(encoder) = cache
.read()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.get(&key)
.cloned()
{
return Ok(encoder);
}
let mut cache = cache.write().unwrap_or_else(|poisoned| poisoned.into_inner());
if let Some(existing) = cache.get(&key) {
return Ok(Arc::clone(existing));
}
if cache.len() < LEGACY_REED_SOLOMON_CACHE_MAX_ENTRIES {
let encoder = Arc::new(LegacyReedSolomonEncoder::with_workspace_cache(data_shards, parity_shards, true)?);
cache.insert(key, Arc::clone(&encoder));
return Ok(encoder);
}
drop(cache);
Ok(Arc::new(LegacyReedSolomonEncoder::new(data_shards, parity_shards)?))
}
fn encode_parity_shards<F>(shards: &mut [Option<Vec<u8>>], data_shards: usize, parity_shards: usize, encode: F) -> io::Result<()> fn encode_parity_shards<F>(shards: &mut [Option<Vec<u8>>], data_shards: usize, parity_shards: usize, encode: F) -> io::Result<()>
where where
F: FnOnce(SmallVec<[&mut [u8]; 16]>) -> io::Result<()>, F: FnOnce(SmallVec<[&mut [u8]; 16]>) -> io::Result<()>,
@@ -618,7 +511,7 @@ pub struct Erasure {
pub data_shards: usize, pub data_shards: usize,
pub parity_shards: usize, pub parity_shards: usize,
encoder: Option<ReedSolomonEncoder>, encoder: Option<ReedSolomonEncoder>,
legacy_encoder: Option<Arc<LegacyReedSolomonEncoder>>, legacy_encoder: Option<LegacyReedSolomonEncoder>,
pub block_size: usize, pub block_size: usize,
uses_legacy: bool, uses_legacy: bool,
_id: Uuid, _id: Uuid,
@@ -754,7 +647,7 @@ impl Erasure {
let legacy_encoder = if uses_legacy && parity_shards > 0 { let legacy_encoder = if uses_legacy && parity_shards > 0 {
Some( Some(
cached_legacy_reed_solomon(data_shards, parity_shards) LegacyReedSolomonEncoder::new(data_shards, parity_shards)
.map_err(|source| ErasureConstructionError::LegacyEncoder { source })?, .map_err(|source| ErasureConstructionError::LegacyEncoder { source })?,
) )
} else { } else {
@@ -782,48 +675,106 @@ impl Erasure {
#[tracing::instrument(level = "debug", skip_all, fields(data_len=data.len()))] #[tracing::instrument(level = "debug", skip_all, fields(data_len=data.len()))]
#[hotpath::measure(impl_type = "Erasure")] #[hotpath::measure(impl_type = "Erasure")]
pub fn encode_data(&self, data: &[u8]) -> io::Result<Vec<Bytes>> { pub fn encode_data(&self, data: &[u8]) -> io::Result<Vec<Bytes>> {
self.encode_data_block_inner(data) let shard_size_fn = if self.uses_legacy {
.map(|block| block.into_shards(self.total_shard_count())) calc_shard_size_legacy
} } else {
calc_shard_size
};
let per_shard_size = shard_size_fn(data.len(), self.data_shards);
if per_shard_size == 0 {
return Ok(vec![Bytes::new(); self.total_shard_count()]);
}
let need_total_size = per_shard_size * self.total_shard_count();
#[tracing::instrument(level = "debug", skip_all, fields(data_len=data.len()))] let mut data_buffer = BytesMut::with_capacity(need_total_size);
#[hotpath::measure(label = "Erasure::encode_data", impl_type = "Erasure")]
pub(crate) fn encode_data_block(&self, data: &[u8]) -> io::Result<EncodedBlock> {
self.encode_data_block_inner(data)
}
fn encode_data_block_inner(&self, data: &[u8]) -> io::Result<EncodedBlock> {
let mut data_buffer = BytesMut::with_capacity(self.encoded_capacity_for_data_len(data.len()));
data_buffer.extend_from_slice(data); data_buffer.extend_from_slice(data);
self.encode_buffer(data_buffer, data.len()) data_buffer.resize(need_total_size, 0u8);
{
let data_slices: SmallVec<[&mut [u8]; 16]> = data_buffer.chunks_exact_mut(per_shard_size).collect();
if self.parity_shards > 0 {
if self.uses_legacy {
if let Some(encoder) = self.legacy_encoder.as_ref() {
encoder.encode(data_slices)?;
} else {
warn!("parity_shards > 0, uses_legacy but legacy_encoder is None");
}
} else if let Some(encoder) = self.encoder.as_ref() {
encoder.encode(data_slices)?;
} else {
warn!("parity_shards > 0, but encoder is None");
}
}
}
// Zero-copy split, all shards reference data_buffer
let mut data_buffer = data_buffer.freeze();
let mut shards = Vec::with_capacity(self.total_shard_count());
for _ in 0..self.total_shard_count() {
let shard = data_buffer.split_to(per_shard_size);
shards.push(shard);
}
Ok(shards)
} }
/// Encode owned data, avoiding a copy when the caller already has a heap buffer. /// Encode owned data, avoiding a copy when the caller already has a heap buffer.
/// Falls back to copying into a new buffer if zero-copy conversion fails. /// Falls back to copying into a new buffer if zero-copy conversion fails.
#[hotpath::measure(impl_type = "Erasure")] #[hotpath::measure(impl_type = "Erasure")]
pub fn encode_data_owned(&self, data: Vec<u8>) -> io::Result<Vec<Bytes>> { pub fn encode_data_owned(&self, data: Vec<u8>) -> io::Result<Vec<Bytes>> {
self.encode_data_owned_block_inner(data) let shard_size_fn = if self.uses_legacy {
.map(|block| block.into_shards(self.total_shard_count())) calc_shard_size_legacy
} } else {
calc_shard_size
};
let per_shard_size = shard_size_fn(data.len(), self.data_shards);
if per_shard_size == 0 {
return Ok(vec![Bytes::new(); self.total_shard_count()]);
}
let need_total_size = per_shard_size * self.total_shard_count();
#[hotpath::measure(label = "Erasure::encode_data_owned", impl_type = "Erasure")]
pub(crate) fn encode_data_owned_block(&self, data: Vec<u8>) -> io::Result<EncodedBlock> {
self.encode_data_owned_block_inner(data)
}
fn encode_data_owned_block_inner(&self, data: Vec<u8>) -> io::Result<EncodedBlock> {
let data_len = data.len();
// Try zero-copy: Vec<u8> -> Bytes -> BytesMut (succeeds when refcount == 1) // Try zero-copy: Vec<u8> -> Bytes -> BytesMut (succeeds when refcount == 1)
let data_buffer = match Bytes::from(data).try_into_mut() { let mut data_buffer = match Bytes::from(data).try_into_mut() {
Ok(data_buffer) => data_buffer, Ok(mut bm) => {
bm.resize(need_total_size, 0u8);
bm
}
Err(b) => { Err(b) => {
// Rare path: refcount != 1, fall back to copy // Rare path: refcount != 1, fall back to copy
let mut data_buffer = BytesMut::with_capacity(self.encoded_capacity_for_data_len(data_len)); let mut bm = BytesMut::with_capacity(need_total_size);
data_buffer.extend_from_slice(&b); bm.extend_from_slice(&b);
data_buffer bm.resize(need_total_size, 0u8);
bm
} }
}; };
self.encode_buffer(data_buffer, data_len)
{
let data_slices: SmallVec<[&mut [u8]; 16]> = data_buffer.chunks_exact_mut(per_shard_size).collect();
if self.parity_shards > 0 {
if self.uses_legacy {
if let Some(encoder) = self.legacy_encoder.as_ref() {
encoder.encode(data_slices)?;
} else {
warn!("parity_shards > 0, uses_legacy but legacy_encoder is None");
}
} else if let Some(encoder) = self.encoder.as_ref() {
encoder.encode(data_slices)?;
} else {
warn!("parity_shards > 0, but encoder is None");
}
}
}
let mut data_buffer = data_buffer.freeze();
let mut shards = Vec::with_capacity(self.total_shard_count());
for _ in 0..self.total_shard_count() {
let shard = data_buffer.split_to(per_shard_size);
shards.push(shard);
}
Ok(shards)
} }
/// Encode data from an owned `BytesMut` buffer, avoiding the initial copy /// Encode data from an owned `BytesMut` buffer, avoiding the initial copy
@@ -835,17 +786,7 @@ impl Erasure {
/// `data_len <= block_size` — both shard-size formulas are monotone in /// `data_len <= block_size` — both shard-size formulas are monotone in
/// `data_len` — so this function never reallocates the buffer. /// `data_len` — so this function never reallocates the buffer.
#[hotpath::measure(impl_type = "Erasure")] #[hotpath::measure(impl_type = "Erasure")]
pub fn encode_data_bytes_mut(&self, data_buffer: BytesMut, data_len: usize) -> io::Result<Vec<Bytes>> { pub fn encode_data_bytes_mut(&self, mut data_buffer: BytesMut, data_len: usize) -> io::Result<Vec<Bytes>> {
self.encode_buffer(data_buffer, data_len)
.map(|block| block.into_shards(self.total_shard_count()))
}
#[hotpath::measure(label = "Erasure::encode_data_bytes_mut", impl_type = "Erasure")]
pub(crate) fn encode_data_bytes_mut_block(&self, data_buffer: BytesMut, data_len: usize) -> io::Result<EncodedBlock> {
self.encode_buffer(data_buffer, data_len)
}
fn encode_buffer(&self, mut data_buffer: BytesMut, data_len: usize) -> io::Result<EncodedBlock> {
let shard_size_fn = if self.uses_legacy { let shard_size_fn = if self.uses_legacy {
calc_shard_size_legacy calc_shard_size_legacy
} else { } else {
@@ -853,7 +794,7 @@ impl Erasure {
}; };
let per_shard_size = shard_size_fn(data_len, self.data_shards); let per_shard_size = shard_size_fn(data_len, self.data_shards);
if per_shard_size == 0 { if per_shard_size == 0 {
return Ok(EncodedBlock::empty()); return Ok(vec![Bytes::new(); self.total_shard_count()]);
} }
let need_total_size = per_shard_size * self.total_shard_count(); let need_total_size = per_shard_size * self.total_shard_count();
@@ -880,10 +821,14 @@ impl Erasure {
} }
} }
Ok(EncodedBlock { let mut data_buffer = data_buffer.freeze();
data: data_buffer.freeze(), let mut shards = Vec::with_capacity(self.total_shard_count());
shard_size: per_shard_size, for _ in 0..self.total_shard_count() {
}) let shard = data_buffer.split_to(per_shard_size);
shards.push(shard);
}
Ok(shards)
} }
/// Decode and reconstruct missing data shards in-place. /// Decode and reconstruct missing data shards in-place.
@@ -1110,10 +1055,6 @@ impl Erasure {
/// ///
/// # Errors /// # Errors
/// Returns error if reading from reader fails or if callback returns error /// Returns error if reading from reader fails or if callback returns error
#[allow(
dead_code,
reason = "callback encode path exercised only by this file's tests (backlog#1823)"
)]
pub(crate) async fn encode_stream_callback_async<F, Fut, E, R>( pub(crate) async fn encode_stream_callback_async<F, Fut, E, R>(
self: std::sync::Arc<Self>, self: std::sync::Arc<Self>,
reader: &mut R, reader: &mut R,
@@ -1476,7 +1417,7 @@ mod tests {
assert_eq!(cloned.block_size, legacy.block_size); assert_eq!(cloned.block_size, legacy.block_size);
assert!(cloned.uses_legacy); assert!(cloned.uses_legacy);
let data = b"legacy clone should preserve SIMD codec behavior"; let data = b"legacy clone should keep independent SIMD caches";
let encoded = cloned.encode_data(data).expect("legacy clone should encode"); let encoded = cloned.encode_data(data).expect("legacy clone should encode");
let mut shards = optional_shards(&encoded); let mut shards = optional_shards(&encoded);
shards[0] = None; shards[0] = None;
@@ -1484,93 +1425,6 @@ mod tests {
assert_eq!(recover_data(&shards, cloned.data_shards, data.len()), data); assert_eq!(recover_data(&shards, cloned.data_shards, data.len()), data);
} }
#[test]
fn legacy_codecs_share_process_cache_across_erasure_instances() {
let first = Erasure::new_with_options(6, 3, 64, true)
.legacy_encoder
.expect("legacy codec should be initialized");
let second = Erasure::new_with_options(6, 3, 128, true)
.legacy_encoder
.expect("same legacy shard layout should be initialized");
assert!(Arc::ptr_eq(&first, &second));
}
#[test]
fn legacy_workspace_cache_rejects_oversize_buffers_and_isolates_layouts() {
let four_plus_two = Erasure::new_with_options(4, 2, 64, true)
.legacy_encoder
.expect("legacy codec should be initialized");
let four_plus_one = Erasure::new_with_options(4, 1, 64, true)
.legacy_encoder
.expect("distinct parity layout should be initialized");
let three_plus_two = Erasure::new_with_options(3, 2, 64, true)
.legacy_encoder
.expect("distinct data layout should be initialized");
assert!(!Arc::ptr_eq(&four_plus_two, &four_plus_one));
assert!(!Arc::ptr_eq(&four_plus_two, &three_plus_two));
assert_eq!(four_plus_two.logical_shard_bytes_upper_bound(64 * 1024), Some(512 * 1024));
assert!(four_plus_two.should_cache_workspace(64 * 1024));
assert!(!four_plus_two.should_cache_workspace(64 * 1024 + 1));
let nine_plus_seven =
LegacyReedSolomonEncoder::with_workspace_cache(9, 7, true).expect("9+7 legacy codec should construct");
assert_eq!(nine_plus_seven.logical_shard_bytes_upper_bound(16 * 1024), Some(512 * 1024));
assert!(nine_plus_seven.should_cache_workspace(16 * 1024));
assert!(!nine_plus_seven.should_cache_workspace(16 * 1024 + 1));
let uncached = LegacyReedSolomonEncoder::new(4, 2).expect("uncached legacy codec should construct");
assert!(!uncached.should_cache_workspace(64));
}
#[test]
fn saturated_legacy_codec_cache_does_not_retain_more_workspaces() {
let cache = RwLock::new(HashMap::new());
for parity_shards in 1..=LEGACY_REED_SOLOMON_CACHE_MAX_ENTRIES {
let cached =
cached_legacy_reed_solomon_in(&cache, 32, parity_shards).expect("cacheable legacy codec should construct");
assert!(cached.cache_workspaces);
}
let uncached =
cached_legacy_reed_solomon_in(&cache, 31, 1).expect("uncached legacy codec should construct after saturation");
assert!(!uncached.cache_workspaces);
assert_eq!(
cache.read().expect("cache lock should remain healthy").len(),
LEGACY_REED_SOLOMON_CACHE_MAX_ENTRIES
);
}
#[test]
fn concurrent_legacy_codecs_preserve_byte_exact_results() {
let barrier = Arc::new(std::sync::Barrier::new(2));
let payloads = [vec![0x35; 257], vec![0xca; 1025]];
std::thread::scope(|scope| {
let handles = payloads.each_ref().map(|payload| {
let barrier = Arc::clone(&barrier);
scope.spawn(move || {
let erasure = Erasure::new_with_options(6, 3, 2048, true);
barrier.wait();
let encoded = erasure.encode_data(payload).expect("concurrent legacy encode should succeed");
barrier.wait();
let mut shards = optional_shards(&encoded);
shards[0] = None;
erasure
.decode_data(&mut shards)
.expect("concurrent legacy decode should reconstruct the missing shard");
recover_data(&shards, erasure.data_shards, payload.len())
})
});
for (handle, payload) in handles.into_iter().zip(payloads.iter()) {
assert_eq!(handle.join().expect("concurrent legacy codec worker should not panic"), *payload);
}
});
}
#[test] #[test]
fn legacy_verify_reports_invalid_empty_valid_and_corrupt_parity_sets() { fn legacy_verify_reports_invalid_empty_valid_and_corrupt_parity_sets() {
let legacy = LegacyReedSolomonEncoder::new(2, 2).expect("legacy encoder should construct"); let legacy = LegacyReedSolomonEncoder::new(2, 2).expect("legacy encoder should construct");
@@ -1644,16 +1498,10 @@ mod tests {
fn encode_data_owned_matches_borrowed_path() { fn encode_data_owned_matches_borrowed_path() {
for uses_legacy in [false, true] { for uses_legacy in [false, true] {
let erasure = Erasure::new_with_options(4, 2, 64, uses_legacy); let erasure = Erasure::new_with_options(4, 2, 64, uses_legacy);
for data in [
Vec::new(), assert_owned_encode_matches_borrowed(&erasure, Vec::new());
vec![0xA5; 1], assert_owned_encode_matches_borrowed(&erasure, b"small payload".to_vec());
b"small payload".to_vec(), assert_owned_encode_matches_borrowed(&erasure, (0_u8..37).collect());
(0_u8..37).collect(),
vec![0xA5; erasure.block_size - 1],
vec![0x5A; erasure.block_size],
] {
assert_owned_encode_matches_borrowed(&erasure, data);
}
} }
} }
@@ -1699,52 +1547,6 @@ mod tests {
} }
} }
#[test]
fn streaming_encoded_block_uses_one_contiguous_backing_buffer() {
for uses_legacy in [false, true] {
let erasure = Erasure::new_with_options(8, 8, 64, uses_legacy);
for data_len in [0, 1, 63, 64] {
let data = (0..data_len).map(|i| i as u8).collect::<Vec<_>>();
let expected = erasure.encode_data(&data).expect("public encode should succeed");
let borrowed = erasure
.encode_data_block(&data)
.expect("borrowed streaming encode should succeed");
let owned = erasure
.encode_data_owned_block(data.clone())
.expect("owned streaming encode should succeed");
let bytes_mut = erasure
.encode_data_bytes_mut_block(BytesMut::from(&data[..]), data.len())
.expect("BytesMut streaming encode should succeed");
assert_eq!(borrowed.queued_bytes(), owned.queued_bytes());
assert_eq!(borrowed.queued_bytes(), bytes_mut.queued_bytes());
if data_len == 0 {
assert!(expected.iter().all(Bytes::is_empty));
assert!(borrowed.is_empty());
assert!(owned.is_empty());
assert!(bytes_mut.is_empty());
continue;
}
assert!(borrowed.shards().eq(expected.iter().map(Bytes::as_ref)));
assert!(owned.shards().eq(expected.iter().map(Bytes::as_ref)));
assert!(bytes_mut.shards().eq(expected.iter().map(Bytes::as_ref)));
assert_eq!(borrowed.shards().len(), 16);
let first = borrowed.shards().next().expect("encoded block should have shards").as_ptr();
for (index, shard) in borrowed.shards().enumerate() {
assert_eq!(shard.as_ptr(), first.wrapping_add(index * shard.len()));
}
}
}
assert_eq!(
std::mem::size_of::<EncodedBlock>(),
std::mem::size_of::<Bytes>() + std::mem::size_of::<usize>(),
"queue entries must contain one backing buffer handle, not per-shard handles"
);
}
/// HP-10 capacity invariant: both shard-size formulas are monotone in `data_len`, /// HP-10 capacity invariant: both shard-size formulas are monotone in `data_len`,
/// so pre-reserving `shard_size(block_size) * total_shard_count` covers the /// so pre-reserving `shard_size(block_size) * total_shard_count` covers the
/// `need_total_size` of every block-or-smaller payload and the ingest buffer /// `need_total_size` of every block-or-smaller payload and the ingest buffer
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: erasure codec migration keeps staged streaming decode paths in this module. // #730: erasure codec migration keeps staged streaming decode paths in this module.
#![allow(dead_code)]
pub(crate) mod codec; pub(crate) mod codec;
pub(crate) mod coding; pub(crate) mod coding;
+42 -3
View File
@@ -13,12 +13,13 @@
// limitations under the License. // limitations under the License.
// #730: error taxonomy still exposes compatibility variants while callers move to contracts. // #730: error taxonomy still exposes compatibility variants while callers move to contracts.
#![allow(dead_code)]
use crate::bucket::error::BucketMetadataError; use crate::bucket::error::BucketMetadataError;
use crate::disk::error::DiskError; use crate::disk::error::DiskError;
use crate::storage_api_contracts::{error::StorageErrorCode, range::HTTPRangeError}; use crate::storage_api_contracts::{error::StorageErrorCode, range::HTTPRangeError};
use rustfs_utils::path::decode_dir_object; use rustfs_utils::path::decode_dir_object;
use s3s::S3ErrorCode; use s3s::{S3Error, S3ErrorCode};
pub type Error = StorageError; pub type Error = StorageError;
pub type Result<T> = core::result::Result<T, Error>; pub type Result<T> = core::result::Result<T, Error>;
@@ -901,7 +902,6 @@ pub fn is_err_decommission_running(err: &Error) -> bool {
matches!(err, &StorageError::DecommissionAlreadyRunning) matches!(err, &StorageError::DecommissionAlreadyRunning)
} }
#[allow(dead_code, reason = "predicate asserted by this file's tests (backlog#1823)")]
pub fn is_err_rebalance_running(err: &Error) -> bool { pub fn is_err_rebalance_running(err: &Error) -> bool {
matches!(err, &StorageError::RebalanceAlreadyRunning) matches!(err, &StorageError::RebalanceAlreadyRunning)
} }
@@ -910,11 +910,14 @@ pub fn is_err_operation_canceled(err: &Error) -> bool {
matches!(err, &StorageError::OperationCanceled) matches!(err, &StorageError::OperationCanceled)
} }
#[allow(dead_code, reason = "predicate asserted by this file's tests (backlog#1823)")]
pub fn is_err_not_initialized(err: &Error) -> bool { pub fn is_err_not_initialized(err: &Error) -> bool {
err.to_string().contains("errServerNotInitialized") || err.to_string().contains("ServerNotInitialized") err.to_string().contains("errServerNotInitialized") || err.to_string().contains("ServerNotInitialized")
} }
pub fn is_err_io(err: &Error) -> bool {
matches!(err, &StorageError::Io(_))
}
/// Strict "not found" predicate that only matches genuine object/version/volume /// Strict "not found" predicate that only matches genuine object/version/volume
/// absence errors: `FileNotFound`/`VolumeNotFound`/`FileVersionNotFound`/ /// absence errors: `FileNotFound`/`VolumeNotFound`/`FileVersionNotFound`/
/// `ObjectNotFound`/`VersionNotFound`. /// `ObjectNotFound`/`VersionNotFound`.
@@ -1075,9 +1078,21 @@ pub struct GenericError {
#[derive(Debug, thiserror::Error, PartialEq, Eq)] #[derive(Debug, thiserror::Error, PartialEq, Eq)]
pub enum ObjectApiError { pub enum ObjectApiError {
#[error("Operation timed out")]
OperationTimedOut,
#[error("etag of the object has changed")]
InvalidETag,
#[error("BackendDown")] #[error("BackendDown")]
BackendDown(String), BackendDown(String),
#[error("Unsupported headers in Metadata")]
UnsupportedMetadata,
#[error("Method not allowed: {}/{}", .0.bucket, .0.object)]
MethodNotAllowed(GenericError),
#[error("The operation is not valid for the current state of the object {}/{}({})", .0.bucket, .0.object, .0.version_id)] #[error("The operation is not valid for the current state of the object {}/{}({})", .0.bucket, .0.object, .0.version_id)]
InvalidObjectState(GenericError), InvalidObjectState(GenericError),
} }
@@ -1160,6 +1175,30 @@ pub fn error_resp_to_object_err(err: ErrorResponse, params: Vec<&str>) -> std::i
err err
} }
pub fn storage_to_object_err(err: Error, params: Vec<&str>) -> S3Error {
let storage_err = &err;
let mut bucket: String = "".to_string();
let mut object: String = "".to_string();
if !params.is_empty() {
bucket = params[0].to_string();
}
if params.len() >= 2 {
object = decode_dir_object(params[1]);
}
match storage_err {
StorageError::MethodNotAllowed => S3Error::with_message(
S3ErrorCode::MethodNotAllowed,
ObjectApiError::MethodNotAllowed(GenericError {
bucket,
object,
..Default::default()
})
.to_string(),
),
_ => s3s::S3Error::with_message(S3ErrorCode::Custom("err".into()), err.to_string()),
}
}
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
+2
View File
@@ -13,6 +13,8 @@
// limitations under the License. // limitations under the License.
// #730: event target types are retained for notification owner migration. // #730: event target types are retained for notification owner migration.
#![allow(dead_code)]
pub mod name; pub mod name;
pub mod targetid;
pub mod targetlist; pub mod targetlist;
+25
View File
@@ -0,0 +1,25 @@
#![allow(clippy::all)]
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
pub struct TargetID {
id: String,
name: String,
}
impl TargetID {
fn to_string(&self) -> String {
format!("{}:{}", self.id, self.name)
}
}
+18 -5
View File
@@ -12,16 +12,18 @@
// See the License for the specific language governing permissions and // See the License for the specific language governing permissions and
// limitations under the License. // limitations under the License.
use crate::event::targetid::TargetID;
use std::sync::atomic::AtomicI64; use std::sync::atomic::AtomicI64;
/// Placeholder notification target list held by `EventNotifier`.
///
/// The working notification stack lives in `rustfs-notify` / `rustfs-targets`;
/// this type never grew past its counter. `total_events` is read by the
/// notifier's log line but nothing increments it, so that field reports zero.
#[derive(Default)] #[derive(Default)]
pub struct TargetList { pub struct TargetList {
pub current_send_calls: AtomicI64,
pub total_events: AtomicI64, pub total_events: AtomicI64,
pub events_skipped: AtomicI64,
pub events_errors_total: AtomicI64,
//pub targets: HashMap<TargetID, Target>,
//pub queue: AsyncEvent,
//pub targetStats: HashMap<TargetID, TargetStat>,
} }
impl TargetList { impl TargetList {
@@ -29,3 +31,14 @@ impl TargetList {
TargetList::default() TargetList::default()
} }
} }
struct TargetStat {
current_send_calls: i64,
total_events: i64,
failed_events: i64,
}
struct TargetIDResult {
id: TargetID,
err: std::io::Error,
}
+3 -128
View File
@@ -22,13 +22,12 @@ use crate::diagnostics::get::{
#[cfg(feature = "hotpath")] #[cfg(feature = "hotpath")]
use crate::disk::FileWriter; use crate::disk::FileWriter;
use crate::disk::{self, DiskAPI as _, DiskStore, FileReader, MmapCopyStageMetrics, error::DiskError}; use crate::disk::{self, DiskAPI as _, DiskStore, FileReader, MmapCopyStageMetrics, error::DiskError};
use crate::erasure::coding::{BitrotReader, BitrotWriterWrapper, CustomWriter, ShardChunkRead}; use crate::erasure::coding::{BitrotReader, BitrotWriterWrapper, CustomWriter};
use bytes::Bytes; use bytes::Bytes;
use rustfs_config::{ use rustfs_config::{
DEFAULT_OBJECT_MMAP_READ_ENABLE, DEFAULT_OBJECT_MMAP_READ_MAX_LENGTH, ENV_OBJECT_MMAP_READ_ENABLE, DEFAULT_OBJECT_MMAP_READ_ENABLE, DEFAULT_OBJECT_MMAP_READ_MAX_LENGTH, ENV_OBJECT_MMAP_READ_ENABLE,
ENV_OBJECT_MMAP_READ_MAX_LENGTH, ENV_OBJECT_ZERO_COPY_ENABLE, ENV_OBJECT_MMAP_READ_MAX_LENGTH, ENV_OBJECT_ZERO_COPY_ENABLE,
}; };
use rustfs_rio::ChunkReaderBox;
use rustfs_utils::HashAlgorithm; use rustfs_utils::HashAlgorithm;
use std::future::Future; use std::future::Future;
use std::io::{self, Cursor}; use std::io::{self, Cursor};
@@ -52,25 +51,13 @@ tokio::task_local! {
/// (rustfs/backlog#1159). Everything else is a stream and keeps the old path. /// (rustfs/backlog#1159). Everything else is a stream and keeps the old path.
pub enum ShardReader { pub enum ShardReader {
InMemory(Cursor<Bytes>), InMemory(Cursor<Bytes>),
Chunked(ChunkReaderBox),
Stream(Box<dyn AsyncRead + Send + Sync + Unpin>), Stream(Box<dyn AsyncRead + Send + Sync + Unpin>),
} }
#[cfg(test)]
impl ShardReader {
pub(crate) fn inline_bytes(&self) -> Option<&Bytes> {
match self {
Self::InMemory(cursor) => Some(cursor.get_ref()),
Self::Chunked(_) | Self::Stream(_) => None,
}
}
}
impl AsyncRead for ShardReader { impl AsyncRead for ShardReader {
fn poll_read(self: Pin<&mut Self>, cx: &mut Context<'_>, buf: &mut tokio::io::ReadBuf<'_>) -> Poll<std::io::Result<()>> { fn poll_read(self: Pin<&mut Self>, cx: &mut Context<'_>, buf: &mut tokio::io::ReadBuf<'_>) -> Poll<std::io::Result<()>> {
match self.get_mut() { match self.get_mut() {
Self::InMemory(cursor) => Pin::new(cursor).poll_read(cx, buf), Self::InMemory(cursor) => Pin::new(cursor).poll_read(cx, buf),
Self::Chunked(reader) => Pin::new(&mut **reader).poll_read(cx, buf),
Self::Stream(reader) => Pin::new(reader).poll_read(cx, buf), Self::Stream(reader) => Pin::new(reader).poll_read(cx, buf),
} }
} }
@@ -80,19 +67,7 @@ impl crate::erasure::coding::ShardSource for ShardReader {
fn try_take_block(&mut self, n: usize) -> Option<Bytes> { fn try_take_block(&mut self, n: usize) -> Option<Bytes> {
match self { match self {
Self::InMemory(cursor) => cursor.try_take_block(n), Self::InMemory(cursor) => cursor.try_take_block(n),
Self::Chunked(_) | Self::Stream(_) => None, Self::Stream(_) => None,
}
}
fn poll_read_chunk(self: Pin<&mut Self>, cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
let Self::Chunked(reader) = self.get_mut() else {
return Poll::Ready(Ok(ShardChunkRead::Unsupported));
};
match Pin::new(&mut **reader).poll_read_chunk(cx, max) {
Poll::Ready(Ok(Some(chunk))) => Poll::Ready(Ok(ShardChunkRead::Chunk(chunk))),
Poll::Ready(Ok(None)) => Poll::Ready(Ok(ShardChunkRead::Eof)),
Poll::Ready(Err(err)) => Poll::Ready(Err(err)),
Poll::Pending => Poll::Pending,
} }
} }
} }
@@ -370,17 +345,6 @@ async fn open_disk_reader(
let metrics_path = metrics_path.filter(|_| rustfs_io_metrics::get_stage_metrics_enabled()); let metrics_path = metrics_path.filter(|_| rustfs_io_metrics::get_stage_metrics_enabled());
let stage_metrics_enabled = metrics_path.is_some(); let stage_metrics_enabled = metrics_path.is_some();
// Preserve HTTP body ownership only on healthy remote reads. Instrumented
// and local paths retain their existing AsyncRead wrappers.
if use_mmap_read
&& !disk.is_local()
&& !stage_metrics_enabled
&& !cfg!(feature = "hotpath")
&& let Some(reader) = disk.read_file_stream_chunks(bucket, path, offset, length).await?
{
return Ok(ShardReader::Chunked(reader));
}
// Mmap-copy materializes the whole `offset..offset+length` range as one // Mmap-copy materializes the whole `offset..offset+length` range as one
// owned allocation before any byte is served, and GET/heal shard reads // owned allocation before any byte is served, and GET/heal shard reads
// request the entire part span in one call. Over-cap reads (e.g. a huge // request the entire part span in one call. Over-cap reads (e.g. a huge
@@ -656,7 +620,7 @@ pub async fn create_bitrot_reader_from_bytes(
} }
#[allow(clippy::too_many_arguments)] #[allow(clippy::too_many_arguments)]
pub(crate) async fn create_bitrot_reader_from_bytes_with_stage_metrics( async fn create_bitrot_reader_from_bytes_with_stage_metrics(
inline_data: Option<Bytes>, inline_data: Option<Bytes>,
disk: Option<&DiskStore>, disk: Option<&DiskStore>,
bucket: &str, bucket: &str,
@@ -816,50 +780,6 @@ pub async fn create_bitrot_writer(
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use rustfs_rio::ChunkReader;
use std::collections::VecDeque;
struct TestChunkReader {
chunks: VecDeque<Bytes>,
}
impl TestChunkReader {
fn new(bytes: Bytes, fragment_sizes: &[usize]) -> Self {
let mut chunks = VecDeque::new();
let mut offset = 0;
for &size in fragment_sizes {
let end = (offset + size).min(bytes.len());
if offset < end {
chunks.push_back(bytes.slice(offset..end));
}
offset = end;
}
if offset < bytes.len() {
chunks.push_back(bytes.slice(offset..));
}
Self { chunks }
}
}
impl AsyncRead for TestChunkReader {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("test chunk reader must use chunk handoff")))
}
}
impl ChunkReader for TestChunkReader {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<Option<Bytes>>> {
let Some(mut chunk) = self.chunks.pop_front() else {
return Poll::Ready(Ok(None));
};
let take = chunk.len().min(max);
if take < chunk.len() {
self.chunks.push_front(chunk.split_off(take));
}
chunk.truncate(take);
Poll::Ready(Ok(Some(chunk)))
}
}
#[cfg(feature = "hotpath")] #[cfg(feature = "hotpath")]
use crate::cluster::rpc::RemoteDisk; use crate::cluster::rpc::RemoteDisk;
@@ -1749,49 +1669,4 @@ mod tests {
println!("error: {error:?}"); println!("error: {error:?}");
assert_eq!(error, DiskError::DiskNotFound); assert_eq!(error, DiskError::DiskNotFound);
} }
#[tokio::test]
async fn shard_reader_chunked_path_verifies_fragmented_remote_block() {
const SHARD_SIZE: usize = 1024;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![42u8; SHARD_SIZE];
let mut encoded = Vec::new();
crate::erasure::coding::BitrotWriter::new(&mut encoded, SHARD_SIZE, algo.clone())
.write(&data)
.await
.expect("test shard should encode");
let source = TestChunkReader::new(Bytes::from(encoded), &[3, 7, 17, 31]);
let mut reader = BitrotReader::new(ShardReader::Chunked(Box::new(source)), SHARD_SIZE, algo, false);
let mut output = Vec::with_capacity(SHARD_SIZE);
reader
.read_appending(&mut output, SHARD_SIZE)
.await
.expect("fragmented remote shard should verify");
assert_eq!(output, data);
}
#[tokio::test]
async fn shard_reader_chunked_path_handles_more_than_one_poll_budget() {
const SHARD_SIZE: usize = 1024;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![42u8; SHARD_SIZE];
let mut encoded = Vec::new();
crate::erasure::coding::BitrotWriter::new(&mut encoded, SHARD_SIZE, algo.clone())
.write(&data)
.await
.expect("test shard should encode");
let fragment_sizes = vec![1; encoded.len()];
let source = TestChunkReader::new(Bytes::from(encoded), &fragment_sizes);
let mut reader = BitrotReader::new(ShardReader::Chunked(Box::new(source)), SHARD_SIZE, algo, false);
let mut output = Vec::with_capacity(SHARD_SIZE);
reader
.read_appending(&mut output, SHARD_SIZE)
.await
.expect("fragmented remote shard should verify after multiple polls");
assert_eq!(output, data);
}
} }
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: I/O backend selection keeps test-only and staged rio helpers scoped here. // #730: I/O backend selection keeps test-only and staged rio helpers scoped here.
#![allow(dead_code)]
pub(crate) mod bitrot; pub(crate) mod bitrot;
pub(crate) mod compress; pub(crate) mod compress;
+11 -16
View File
@@ -25,20 +25,9 @@ use tokio::io::AsyncRead;
#[cfg(feature = "rio-v2")] #[cfg(feature = "rio-v2")]
const MINIO_S2_COMPRESSION_SCHEME: &str = "klauspost/compress/s2"; const MINIO_S2_COMPRESSION_SCHEME: &str = "klauspost/compress/s2";
// The S2 padding multiple rio-v2 pads compressed streams to before
// encryption. Only the padding test asserts it today, so the lib target sees
// it as unused (backlog#1823).
#[cfg(feature = "rio-v2")] #[cfg(feature = "rio-v2")]
#[allow(dead_code, reason = "on-disk contract asserted by the rio-v2 padding test (backlog#1823)")]
const ENCRYPTED_S2_PADDING_MULTIPLE: usize = 256; const ENCRYPTED_S2_PADDING_MULTIPLE: usize = 256;
/// Which rio implementation this build compiled in. Only the feature-seam
/// guard test in lib.rs reads it, so the lib target sees it as unused
/// (backlog#1823).
#[allow(
dead_code,
reason = "asserted by the rio backend feature-seam test in lib.rs (backlog#1823)"
)]
pub const fn backend_name() -> &'static str { pub const fn backend_name() -> &'static str {
#[cfg(feature = "rio-v2")] #[cfg(feature = "rio-v2")]
{ {
@@ -64,6 +53,17 @@ pub fn compression_metadata_value(algorithm: CompressionAlgorithm) -> String {
} }
} }
pub fn compression_scheme_to_algorithm(scheme: &str) -> std::io::Result<CompressionAlgorithm> {
#[cfg(feature = "rio-v2")]
if scheme.eq_ignore_ascii_case(MINIO_S2_COMPRESSION_SCHEME) {
// rio_v2 currently routes all compressed-object handling through the S2
// reader implementation, so the enum is only a placeholder token here.
return Ok(CompressionAlgorithm::default());
}
CompressionAlgorithm::from_str(scheme)
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)] #[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum ReadCompressionBackend { pub enum ReadCompressionBackend {
Legacy, Legacy,
@@ -82,11 +82,6 @@ pub fn compression_scheme_to_read_plan(scheme: &str) -> std::io::Result<(Compres
#[derive(Debug, Clone, Copy, PartialEq, Eq)] #[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum ReadEncryptionBackend { pub enum ReadEncryptionBackend {
Legacy, Legacy,
// Never constructed today — every read still selects Legacy — but the
// decrypt paths below carry live match arms for it. This is the rio-v2
// read seam (backlog#1638 / #1835), not dead code: deleting the variant
// would delete those arms with it.
#[allow(dead_code, reason = "rio-v2 read seam; match arms below are live (backlog#1823)")]
V2, V2,
} }
+2 -3
View File
@@ -21,8 +21,7 @@ use tracing::debug;
/// Supported set sizes this is used to find the optimal /// Supported set sizes this is used to find the optimal
/// single set size. /// single set size.
pub(crate) const MAX_ERASURE_SET_DRIVE_COUNT: usize = 16; const SET_SIZES: [usize; 15] = [2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16];
const SET_SIZES: [usize; 15] = [2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, MAX_ERASURE_SET_DRIVE_COUNT];
const ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT: &str = "RUSTFS_ERASURE_SET_DRIVE_COUNT"; const ENV_RUSTFS_ERASURE_SET_DRIVE_COUNT: &str = "RUSTFS_ERASURE_SET_DRIVE_COUNT";
#[derive(Deserialize, Debug, Default)] #[derive(Deserialize, Debug, Default)]
@@ -328,7 +327,7 @@ fn possible_set_counts(set_size: usize) -> Vec<usize> {
/// checks whether given count is a valid set size for erasure coding. /// checks whether given count is a valid set size for erasure coding.
fn is_valid_set_size(count: usize) -> bool { fn is_valid_set_size(count: usize) -> bool {
count >= SET_SIZES[0] && count <= MAX_ERASURE_SET_DRIVE_COUNT count >= SET_SIZES[0] && count <= SET_SIZES[SET_SIZES.len() - 1]
} }
/// Final set size with all the symmetry accounted for. /// Final set size with all the symmetry accounted for.
+9 -10
View File
@@ -209,12 +209,15 @@ impl AsMut<Vec<Endpoints>> for PoolEndpointList {
} }
impl PoolEndpointList { impl PoolEndpointList {
/// Creates a list of endpoints per pool, resolves their relevant hostnames /// creates a list of endpoints per pool, resolves their relevant
/// and discovers whether those are local or remote. /// hostnames and discovers those are local or remote.
/// async fn create_pool_endpoints(server_addr: &str, disks_layout: &DisksLayout) -> Result<Self> {
/// The policy and host overrides let tests inject an explicit startup Self::create_pool_endpoints_with(server_addr, disks_layout, None, None).await
/// topology convergence policy and local endpoint host instead of }
/// resolving them from the environment; production passes `None` for both.
/// Same as [`create_pool_endpoints`] but lets tests inject an explicit
/// startup topology convergence policy and local endpoint host instead of
/// resolving them from the environment.
async fn create_pool_endpoints_with( async fn create_pool_endpoints_with(
server_addr: &str, server_addr: &str,
disks_layout: &DisksLayout, disks_layout: &DisksLayout,
@@ -591,10 +594,6 @@ impl PoolEndpointList {
} }
const DNS_RETRY_BASE_DELAY: Duration = Duration::from_millis(500); const DNS_RETRY_BASE_DELAY: Duration = Duration::from_millis(500);
#[allow(
dead_code,
reason = "retry-cap bound asserted by this file's dns_retry_delay tests (backlog#1823)"
)]
const DNS_RETRY_MAX_DELAY: Duration = Duration::from_secs(8); const DNS_RETRY_MAX_DELAY: Duration = Duration::from_secs(8);
const DNS_RETRY_JITTER_PERCENT: u64 = 20; const DNS_RETRY_JITTER_PERCENT: u64 = 20;
/// Minimum spacing between "still retrying" warnings so a long orchestrated /// Minimum spacing between "still retrying" warnings so a long orchestrated
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: set-layout contracts are staged while ECStore ownership boundaries shrink. // #730: set-layout contracts are staged while ECStore ownership boundaries shrink.
#![allow(dead_code)]
//! Static ECStore layout boundaries. //! Static ECStore layout boundaries.
//! //!
-6
View File
@@ -4,7 +4,6 @@ use std::io::{Error, Result};
use uuid::Uuid; use uuid::Uuid;
#[derive(Debug, Clone, PartialEq, Eq)] #[derive(Debug, Clone, PartialEq, Eq)]
#[allow(dead_code, reason = "ESET-001 layout model; exercised by this file's tests (backlog#1823)")]
pub(crate) struct StaticSetLayoutSnapshot { pub(crate) struct StaticSetLayoutSnapshot {
pub(crate) deployment_id: Uuid, pub(crate) deployment_id: Uuid,
pub(crate) set_count: usize, pub(crate) set_count: usize,
@@ -13,7 +12,6 @@ pub(crate) struct StaticSetLayoutSnapshot {
pub(crate) distribution_algo: DistributionAlgoVersion, pub(crate) distribution_algo: DistributionAlgoVersion,
} }
#[allow(dead_code, reason = "ESET-001 layout model; exercised by this file's tests (backlog#1823)")]
impl StaticSetLayoutSnapshot { impl StaticSetLayoutSnapshot {
pub(crate) fn from_format(format: &FormatV3) -> Self { pub(crate) fn from_format(format: &FormatV3) -> Self {
let disk_ids = format.erasure.sets.clone(); let disk_ids = format.erasure.sets.clone();
@@ -41,20 +39,17 @@ impl StaticSetLayoutSnapshot {
} }
#[derive(Debug, Clone, Copy, PartialEq, Eq)] #[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[allow(dead_code, reason = "ESET-001 layout model; exercised by this file's tests (backlog#1823)")]
pub(crate) struct SetDiskPosition { pub(crate) struct SetDiskPosition {
pub(crate) set_index: usize, pub(crate) set_index: usize,
pub(crate) disk_index: usize, pub(crate) disk_index: usize,
} }
#[derive(Debug, Clone, PartialEq, Eq)] #[derive(Debug, Clone, PartialEq, Eq)]
#[allow(dead_code, reason = "ESET-001 layout model; exercised by this file's tests (backlog#1823)")]
pub(crate) struct RuntimeSetLayoutPlan { pub(crate) struct RuntimeSetLayoutPlan {
pub(crate) sets: Vec<Vec<RuntimeSetDrivePlan>>, pub(crate) sets: Vec<Vec<RuntimeSetDrivePlan>>,
lock_hosts_by_set: Vec<Vec<String>>, lock_hosts_by_set: Vec<Vec<String>>,
} }
#[allow(dead_code, reason = "ESET-001 layout model; exercised by this file's tests (backlog#1823)")]
impl RuntimeSetLayoutPlan { impl RuntimeSetLayoutPlan {
pub(crate) fn from_endpoint_hosts<S>(set_count: usize, drives_per_set: usize, endpoint_hosts: &[S]) -> Result<Self> pub(crate) fn from_endpoint_hosts<S>(set_count: usize, drives_per_set: usize, endpoint_hosts: &[S]) -> Result<Self>
where where
@@ -113,7 +108,6 @@ impl RuntimeSetLayoutPlan {
} }
#[derive(Debug, Clone, PartialEq, Eq)] #[derive(Debug, Clone, PartialEq, Eq)]
#[allow(dead_code, reason = "ESET-001 layout model; exercised by this file's tests (backlog#1823)")]
pub(crate) struct RuntimeSetDrivePlan { pub(crate) struct RuntimeSetDrivePlan {
pub(crate) set_index: usize, pub(crate) set_index: usize,
pub(crate) disk_index: usize, pub(crate) disk_index: usize,
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: object API readers keep staged compatibility paths during facade migration. // #730: object API readers keep staged compatibility paths during facade migration.
#![allow(dead_code)]
use crate::bucket::metadata_sys::get_versioning_config; use crate::bucket::metadata_sys::get_versioning_config;
use crate::bucket::replication::{ use crate::bucket::replication::{
+26 -88
View File
@@ -15,7 +15,6 @@
use super::*; use super::*;
use crate::io_support::rio::Index; use crate::io_support::rio::Index;
use std::mem::MaybeUninit;
#[cfg(feature = "rio-v2")] #[cfg(feature = "rio-v2")]
const DARE_PAYLOAD_SIZE: i64 = 64 * 1024; const DARE_PAYLOAD_SIZE: i64 = 64 * 1024;
@@ -449,16 +448,10 @@ impl GetObjectReader {
} }
enum ReadTransform { enum ReadTransform {
// Written but never read by production code: the enclosing struct already Plain {
// carries the same pair as `storage_offset`/`storage_length`. They survive visible_offset: usize,
// as the read plan's test-visible record — four tests assert them by visible_length: i64,
// literal pattern (`Plain { visible_offset: 6, visible_length: 4 }`), which },
// rustc does not count as a read.
#[allow(
dead_code,
reason = "asserted by literal pattern in this file's read-plan tests (backlog#1823)"
)]
Plain { visible_offset: usize, visible_length: i64 },
Compressed { Compressed {
algorithm: CompressionAlgorithm, algorithm: CompressionAlgorithm,
backend: crate::io_support::rio::ReadCompressionBackend, backend: crate::io_support::rio::ReadCompressionBackend,
@@ -929,7 +922,7 @@ struct SkipReader<R> {
inner: R, inner: R,
bytes_to_skip: usize, bytes_to_skip: usize,
bytes_skipped: usize, bytes_skipped: usize,
scratch: Box<[MaybeUninit<u8>]>, scratch: Vec<u8>,
} }
impl<R: AsyncRead + Unpin + Send + Sync> SkipReader<R> { impl<R: AsyncRead + Unpin + Send + Sync> SkipReader<R> {
@@ -938,7 +931,7 @@ impl<R: AsyncRead + Unpin + Send + Sync> SkipReader<R> {
inner, inner,
bytes_to_skip, bytes_to_skip,
bytes_skipped: 0, bytes_skipped: 0,
scratch: Box::<[u8]>::new_uninit_slice(8192), scratch: vec![0u8; 8192],
} }
} }
} }
@@ -950,7 +943,7 @@ impl<R: AsyncRead + Unpin + Send + Sync> AsyncRead for SkipReader<R> {
while this.bytes_skipped < this.bytes_to_skip { while this.bytes_skipped < this.bytes_to_skip {
let remaining = this.bytes_to_skip - this.bytes_skipped; let remaining = this.bytes_to_skip - this.bytes_skipped;
let scratch_len = remaining.min(this.scratch.len()); let scratch_len = remaining.min(this.scratch.len());
let mut scratch_buf = ReadBuf::uninit(&mut this.scratch[..scratch_len]); let mut scratch_buf = ReadBuf::new(&mut this.scratch[..scratch_len]);
match Pin::new(&mut this.inner).poll_read(cx, &mut scratch_buf) { match Pin::new(&mut this.inner).poll_read(cx, &mut scratch_buf) {
Poll::Pending => return Poll::Pending, Poll::Pending => return Poll::Pending,
Poll::Ready(Err(err)) => return Poll::Ready(Err(err)), Poll::Ready(Err(err)) => return Poll::Ready(Err(err)),
@@ -981,7 +974,7 @@ pub struct RangedDecompressReader<R: AsyncRead + Unpin + Send + Sync + 'static>
target_length: usize, target_length: usize,
current_offset: usize, current_offset: usize,
bytes_returned: usize, bytes_returned: usize,
scratch: Box<[MaybeUninit<u8>]>, scratch: Vec<u8>,
drain_on_done: bool, drain_on_done: bool,
drain_task: Option<tokio::task::JoinHandle<()>>, drain_task: Option<tokio::task::JoinHandle<()>>,
} }
@@ -1019,7 +1012,7 @@ impl<R: AsyncRead + Unpin + Send + Sync + 'static> RangedDecompressReader<R> {
target_length: actual_length, target_length: actual_length,
current_offset: 0, current_offset: 0,
bytes_returned: 0, bytes_returned: 0,
scratch: Box::<[u8]>::new_uninit_slice(8192), scratch: vec![0u8; 8192],
drain_on_done, drain_on_done,
drain_task: None, drain_task: None,
}) })
@@ -1069,7 +1062,7 @@ impl<R: AsyncRead + Unpin + Send + Sync + 'static> AsyncRead for RangedDecompres
} }
let scratch_len = std::cmp::min(this.scratch.len(), std::cmp::max(buf_capacity, 1)); let scratch_len = std::cmp::min(this.scratch.len(), std::cmp::max(buf_capacity, 1));
let mut temp_read_buf = ReadBuf::uninit(&mut this.scratch[..scratch_len]); let mut temp_read_buf = ReadBuf::new(&mut this.scratch[..scratch_len]);
let Some(inner) = this.inner.as_mut() else { let Some(inner) = this.inner.as_mut() else {
return Poll::Ready(Ok(())); return Poll::Ready(Ok(()));
@@ -1121,8 +1114,7 @@ impl<R: AsyncRead + Unpin + Send + Sync + 'static> AsyncRead for RangedDecompres
); );
if bytes_to_return > 0 { if bytes_to_return > 0 {
let data_slice = let data_slice = &this.scratch[data_start_in_buffer..data_start_in_buffer + bytes_to_return];
&temp_read_buf.filled()[data_start_in_buffer..data_start_in_buffer + bytes_to_return];
buf.put_slice(data_slice); buf.put_slice(data_slice);
this.bytes_returned += bytes_to_return; this.bytes_returned += bytes_to_return;
@@ -1141,7 +1133,7 @@ impl<R: AsyncRead + Unpin + Send + Sync + 'static> AsyncRead for RangedDecompres
std::cmp::min(n, std::cmp::min(buf.remaining(), this.target_length - this.bytes_returned)); std::cmp::min(n, std::cmp::min(buf.remaining(), this.target_length - this.bytes_returned));
if bytes_to_return > 0 { if bytes_to_return > 0 {
buf.put_slice(&temp_read_buf.filled()[..bytes_to_return]); buf.put_slice(&this.scratch[..bytes_to_return]);
this.bytes_returned += bytes_to_return; this.bytes_returned += bytes_to_return;
tracing::trace!("Returned {} bytes at offset {}", bytes_to_return, old_offset); tracing::trace!("Returned {} bytes at offset {}", bytes_to_return, old_offset);
@@ -1211,7 +1203,20 @@ impl<R: AsyncRead + Unpin + Send + 'static> AsyncRead for StreamConsumer<R> {
impl<R: AsyncRead + Unpin + Send + 'static> Drop for StreamConsumer<R> { impl<R: AsyncRead + Unpin + Send + 'static> Drop for StreamConsumer<R> {
fn drop(&mut self) { fn drop(&mut self) {
self.ensure_consumer_started(); if self.consumer_task.is_none() && self.inner.is_some() {
let mut inner = self.inner.take().unwrap();
let task = tokio::spawn(async move {
let mut buf = [0u8; 8192];
loop {
match inner.read(&mut buf).await {
Ok(0) => break, // EOF
Ok(_) => continue, // Keep consuming
Err(_) => break, // Error, stop consuming
}
}
});
self.consumer_task = Some(task);
}
} }
} }
@@ -1258,43 +1263,6 @@ mod tests {
use temp_env::async_with_vars; use temp_env::async_with_vars;
use tokio::io::AsyncReadExt; use tokio::io::AsyncReadExt;
#[derive(Debug)]
struct PendingPartialReader {
data: &'static [u8],
position: usize,
pending: bool,
}
impl PendingPartialReader {
fn new(data: &'static [u8]) -> Self {
Self {
data,
position: 0,
pending: true,
}
}
}
impl AsyncRead for PendingPartialReader {
fn poll_read(mut self: Pin<&mut Self>, cx: &mut Context<'_>, buf: &mut ReadBuf<'_>) -> Poll<std::io::Result<()>> {
if self.pending {
self.pending = false;
cx.waker().wake_by_ref();
return Poll::Pending;
}
if self.position == self.data.len() {
return Poll::Ready(Ok(()));
}
let length = buf.remaining().min(3).min(self.data.len() - self.position);
let end = self.position + length;
buf.put_slice(&self.data[self.position..end]);
self.position = end;
self.pending = true;
Poll::Ready(Ok(()))
}
}
const TEST_DIRECT_KEY_HEADER: &str = "x-rustfs-test-direct-key"; const TEST_DIRECT_KEY_HEADER: &str = "x-rustfs-test-direct-key";
const TEST_OBJECT_KEY_HEADER: &str = "x-rustfs-test-object-key"; const TEST_OBJECT_KEY_HEADER: &str = "x-rustfs-test-object-key";
const TEST_NONCE_HEADER: &str = "x-rustfs-test-nonce"; const TEST_NONCE_HEADER: &str = "x-rustfs-test-nonce";
@@ -1432,36 +1400,6 @@ mod tests {
assert_eq!(result, b"World"); assert_eq!(result, b"World");
} }
#[tokio::test]
async fn uninitialized_scratch_preserves_partial_pending_and_eof_reads() {
let mut skipped = SkipReader::new(PendingPartialReader::new(b"0123456789abcdef"), 5);
let mut skipped_output = Vec::new();
skipped
.read_to_end(&mut skipped_output)
.await
.expect("skip reader should survive partial pending reads through EOF");
assert_eq!(skipped_output, b"56789abcdef");
let mut ranged = RangedDecompressReader::new(PendingPartialReader::new(b"0123456789abcdef"), 5, 7, 16)
.expect("valid range should construct");
let mut ranged_output = Vec::new();
ranged
.read_to_end(&mut ranged_output)
.await
.expect("range reader should survive partial pending reads through EOF");
assert_eq!(ranged_output, b"56789ab");
}
#[tokio::test]
async fn uninitialized_skip_scratch_reports_early_eof() {
let mut reader = SkipReader::new(PendingPartialReader::new(b"short"), 6);
let error = reader
.read_to_end(&mut Vec::new())
.await
.expect_err("EOF before the skip boundary must remain visible");
assert_eq!(error.kind(), std::io::ErrorKind::UnexpectedEof);
}
#[tokio::test] #[tokio::test]
async fn test_ranged_decompress_reader_from_start() { async fn test_ranged_decompress_reader_from_start() {
let original_data = b"Hello, World! This is a test."; let original_data = b"Hello, World! This is a test.";
-4
View File
@@ -172,7 +172,6 @@ impl ObjectLockConfigSnapshot {
} }
} }
#[allow(dead_code, reason = "snapshot-scope predicate asserted by this file's tests (backlog#1823)")]
pub(crate) fn is_for_store_bucket( pub(crate) fn is_for_store_bucket(
&self, &self,
store_id: Uuid, store_id: Uuid,
@@ -261,9 +260,6 @@ pub struct ObjectOptions {
pub data_movement: bool, pub data_movement: bool,
pub raw_data_movement_read: bool, pub raw_data_movement_read: bool,
/// Materialize the data-movement per-part checksum sidecar for APIs that
/// return part checksums. Ordinary object reads leave it encoded.
pub include_part_checksums: bool,
pub src_pool_idx: usize, pub src_pool_idx: usize,
pub user_defined: HashMap<String, String>, pub user_defined: HashMap<String, String>,
pub preserve_etag: Option<String>, pub preserve_etag: Option<String>,
+38
View File
@@ -31,6 +31,7 @@ use std::{
use tokio::sync::{OnceCell, RwLock}; use tokio::sync::{OnceCell, RwLock};
use tokio_util::sync::CancellationToken; use tokio_util::sync::CancellationToken;
use tracing::warn; use tracing::warn;
use uuid::Uuid;
pub const DISK_ASSUME_UNKNOWN_SIZE: u64 = 1 << 30; pub const DISK_ASSUME_UNKNOWN_SIZE: u64 = 1 << 30;
pub const DISK_MIN_INODES: u64 = 1000; pub const DISK_MIN_INODES: u64 = 1000;
@@ -108,6 +109,18 @@ pub fn set_global_rustfs_port(value: u16) {
} }
} }
/// Set the global deployment id
///
/// # Arguments
/// * `id` - The Uuid to set as the global deployment id
///
/// # Returns
/// * None
///
pub fn set_global_deployment_id(id: Uuid) {
current_ctx().set_deployment_id(id);
}
/// Get the global deployment id /// Get the global deployment id
/// ///
/// # Returns /// # Returns
@@ -275,6 +288,19 @@ pub fn get_global_region() -> Option<s3s::region::Region> {
current_ctx().region() current_ctx().region()
} }
/// Initialize the global background services cancellation token
///
/// # Arguments
/// * `cancel_token` - The CancellationToken instance to set globally
///
/// # Returns
/// * `Ok(())` if successful
/// * `Err(CancellationToken)` if setting fails
///
pub fn init_background_services_cancel_token(cancel_token: CancellationToken) -> Result<(), CancellationToken> {
current_ctx().init_background_cancel_token(cancel_token)
}
/// Get the global background services cancellation token /// Get the global background services cancellation token
/// ///
/// # Returns /// # Returns
@@ -284,6 +310,18 @@ pub fn get_background_services_cancel_token() -> Option<CancellationToken> {
current_ctx().background_cancel_token() current_ctx().background_cancel_token()
} }
/// Create and initialize the global background services cancellation token
///
/// # Returns
/// * `CancellationToken` - The newly created global cancellation token
///
pub fn create_background_services_cancel_token() -> CancellationToken {
let cancel_token = CancellationToken::new();
init_background_services_cancel_token(cancel_token.clone())
.expect("background services cancel token should be initialized once during startup");
cancel_token
}
/// Shutdown all background services gracefully /// Shutdown all background services gracefully
/// ///
/// # Returns /// # Returns
-4
View File
@@ -402,10 +402,6 @@ impl InstanceContext {
} }
#[cfg(test)] #[cfg(test)]
#[allow(
dead_code,
reason = "driven by the tier-delete-journal recovery test behind `--features test-util` (backlog#1823)"
)]
pub(crate) fn wake_tier_delete_journal_recovery(&self) { pub(crate) fn wake_tier_delete_journal_recovery(&self) {
self.tier_delete_journal_recovery_wakeup.notify_one(); self.tier_delete_journal_recovery_wakeup.notify_one();
} }
+1
View File
@@ -13,6 +13,7 @@
// limitations under the License. // limitations under the License.
// #730: runtime source migration keeps fallback handles until all owners inject state. // #730: runtime source migration keeps fallback handles until all owners inject state.
#![allow(dead_code)]
pub(crate) mod global; pub(crate) mod global;
pub(crate) mod instance; pub(crate) mod instance;
+52 -11
View File
@@ -38,6 +38,7 @@ use crate::{
set_object_layer, update_erasure_type, set_object_layer, update_erasure_type,
}, },
services::batch_processor::{GlobalBatchProcessors, get_global_processors}, services::batch_processor::{GlobalBatchProcessors, get_global_processors},
services::event_notification::EventNotifier,
services::notification_sys::{NotificationSys, get_global_notification_sys}, services::notification_sys::{NotificationSys, get_global_notification_sys},
services::tier::tier::TierConfigMgr, services::tier::tier::TierConfigMgr,
store::ECStore, store::ECStore,
@@ -142,10 +143,6 @@ pub async fn setup_is_erasure_sd() -> bool {
is_erasure_sd().await is_erasure_sd().await
} }
#[allow(
dead_code,
reason = "setup-type override used only by tests across this crate (backlog#1823)"
)]
pub(crate) async fn current_setup_type() -> SetupType { pub(crate) async fn current_setup_type() -> SetupType {
if setup_is_dist_erasure().await { if setup_is_dist_erasure().await {
SetupType::DistErasure SetupType::DistErasure
@@ -158,10 +155,6 @@ pub(crate) async fn current_setup_type() -> SetupType {
} }
} }
#[allow(
dead_code,
reason = "setup-type override used only by tests across this crate (backlog#1823)"
)]
pub(crate) async fn set_setup_type(setup_type: SetupType) { pub(crate) async fn set_setup_type(setup_type: SetupType) {
update_erasure_type(setup_type).await; update_erasure_type(setup_type).await;
} }
@@ -171,9 +164,6 @@ pub(crate) async fn local_node_name() -> String {
} }
pub(crate) async fn set_local_node_name(node_name: String) { pub(crate) async fn set_local_node_name(node_name: String) {
// Also stamp the internode-metrics server label: io-metrics is a leaf
// crate and no longer resolves node identity itself (backlog#1834).
rustfs_io_metrics::internode_metrics::set_internode_server_label(node_name.as_str());
rustfs_common::set_global_local_node_name(&node_name).await; rustfs_common::set_global_local_node_name(&node_name).await;
} }
@@ -239,6 +229,14 @@ pub(crate) fn ensure_test_rpc_secret() {
let _ = rustfs_credentials::set_global_rpc_secret(TEST_RPC_SECRET.to_owned()); let _ = rustfs_credentials::set_global_rpc_secret(TEST_RPC_SECRET.to_owned());
} }
pub(crate) fn storage_class_parity(storage_class: Option<&str>) -> Option<usize> {
get_global_storage_class_snapshot().get_parity_for_sc(storage_class.unwrap_or_default())
}
pub(crate) fn storage_class_should_inline(shard_size: i64, versioned: bool) -> bool {
get_global_storage_class_snapshot().should_inline(shard_size, versioned)
}
pub(crate) fn deployment_upload_id(upload_id: &str) -> String { pub(crate) fn deployment_upload_id(upload_id: &str) -> String {
base64_simd::URL_SAFE_NO_PAD base64_simd::URL_SAFE_NO_PAD
.encode_to_string(format!("{}.{}", get_global_deployment_id().unwrap_or_default(), upload_id).as_bytes()) .encode_to_string(format!("{}.{}", get_global_deployment_id().unwrap_or_default(), upload_id).as_bytes())
@@ -331,6 +329,21 @@ pub(crate) fn storage_class_config_snapshot() -> Arc<storageclass::Config> {
get_global_storage_class_snapshot() get_global_storage_class_snapshot()
} }
/// Scalar STANDARD / RRS parity for backend-info reporting.
///
/// Retained for the rebalance/backend-info path. `get_parity_for_sc` returns
/// `None` when the runtime config is uninitialized or (post per-pool support)
/// when pools disagree, so STANDARD falls back to the caller's default and RRS
/// stays `None` — matching the pre-per-pool scalar reporting.
pub(crate) fn backend_storage_class_parities(default_standard_parity: usize) -> (Option<usize>, Option<usize>) {
let sc = get_global_storage_class_snapshot();
let standard = sc
.get_parity_for_sc(storageclass::CLASS_STANDARD)
.or(Some(default_standard_parity));
let reduced_redundancy = sc.get_parity_for_sc(storageclass::RRS);
(standard, reduced_redundancy)
}
pub(crate) fn set_storage_class_config(config: storageclass::Config) { pub(crate) fn set_storage_class_config(config: storageclass::Config) {
set_global_storage_class(config); set_global_storage_class(config);
} }
@@ -398,6 +411,10 @@ pub fn transition_state_handle() -> Arc<TransitionState> {
crate::runtime::global::current_ctx().transition_state() crate::runtime::global::current_ctx().transition_state()
} }
pub(crate) fn event_notifier_handle() -> Arc<RwLock<EventNotifier>> {
crate::runtime::global::current_ctx().event_notifier()
}
pub(crate) async fn local_disk_by_path(path: &str) -> Option<DiskStore> { pub(crate) async fn local_disk_by_path(path: &str) -> Option<DiskStore> {
local_disk_map_handle().read().await.get(path).cloned().flatten() local_disk_map_handle().read().await.get(path).cloned().flatten()
} }
@@ -491,6 +508,30 @@ pub(crate) async fn local_disk_set_drive(
instance_ctx.local_disk_set_drives().read().await[pool_idx][set_idx][disk_idx].clone() instance_ctx.local_disk_set_drives().read().await[pool_idx][set_idx][disk_idx].clone()
} }
pub(crate) async fn local_disk_for_endpoint(endpoint: &Endpoint) -> Option<DiskStore> {
let set_drives = local_disk_set_drives_handle();
let global_set_drives = set_drives.read().await;
if global_set_drives.is_empty() {
return local_disk_map_handle()
.read()
.await
.get(&endpoint.to_string())
.cloned()
.unwrap_or(None);
}
let pool_idx = usize::try_from(endpoint.pool_idx).ok()?;
let set_idx = usize::try_from(endpoint.set_idx).ok()?;
let disk_idx = usize::try_from(endpoint.disk_idx).ok()?;
global_set_drives
.get(pool_idx)
.and_then(|sets| sets.get(set_idx))
.and_then(|disks| disks.get(disk_idx))
.cloned()
.unwrap_or(None)
}
pub(crate) async fn local_disk_paths() -> Vec<String> { pub(crate) async fn local_disk_paths() -> Vec<String> {
local_disk_map_handle().read().await.keys().cloned().collect() local_disk_map_handle().read().await.keys().cloned().collect()
} }
@@ -206,38 +206,6 @@ pub(crate) fn remote_version_state_fleet_proof_matches(proof: &RemoteVersionStat
}) })
} }
#[cfg(test)]
pub(crate) struct RemoteVersionStateFleetProofGuard;
#[cfg(test)]
impl Drop for RemoteVersionStateFleetProofGuard {
fn drop(&mut self) {
replace_remote_version_state_fleet_proof(None);
}
}
#[cfg(test)]
pub(crate) fn install_remote_version_state_fleet_proof_for_test(topology_fingerprint: &str) -> RemoteVersionStateFleetProofGuard {
match REMOTE_VERSION_STATE_PROBE_TOPOLOGY.set(topology_fingerprint.to_string()) {
Ok(()) => {}
Err(_)
if REMOTE_VERSION_STATE_PROBE_TOPOLOGY
.get()
.is_some_and(|current| current == topology_fingerprint) => {}
Err(_) => panic!("remote version state test topology is already bound to another fingerprint"),
}
let peer_epochs = BTreeMap::new();
if let Some(err) = publish_remote_version_state_probe_result(
remote_version_state_fleet_proof_slot(),
topology_fingerprint,
Ok(peer_epochs),
Instant::now(),
) {
panic!("test proof installation must not fail: {err}");
}
RemoteVersionStateFleetProofGuard
}
fn remote_version_state_fleet_proof_valid_at( fn remote_version_state_fleet_proof_valid_at(
proof: Option<&RemoteVersionStateFleetProof>, proof: Option<&RemoteVersionStateFleetProof>,
expected_topology: &str, expected_topology: &str,
+5 -31
View File
@@ -16,7 +16,7 @@ use super::meta::{
clone_arc_by_index, ensure_valid_rebalance_pool_index, invalid_rebalance_pool_index_error, clone_arc_by_index, ensure_valid_rebalance_pool_index, invalid_rebalance_pool_index_error,
rebalance_metadata_not_initialized_error, should_ignore_rebalance_data_usage_cache, rebalance_metadata_not_initialized_error, should_ignore_rebalance_data_usage_cache,
}; };
use super::migration::{RebalanceMigrationBackend, migrate_entry_version}; use super::migration::migrate_entry_version;
use super::worker::{ use super::worker::{
RebalanceEntryCleanupResult, RebalanceEntryTask, load_rebalance_bucket_configs, rebalance_max_attempts, RebalanceEntryCleanupResult, RebalanceEntryTask, load_rebalance_bucket_configs, rebalance_max_attempts,
resolve_rebalance_bucket_error, resolve_rebalance_entry_cleanup_delete_result, resolve_rebalance_file_info_versions_result, resolve_rebalance_bucket_error, resolve_rebalance_entry_cleanup_delete_result, resolve_rebalance_file_info_versions_result,
@@ -144,11 +144,6 @@ impl ECStore {
return Ok(RebalanceEntryOutcome::Completed); return Ok(RebalanceEntryOutcome::Completed);
} }
let bucket_incarnation_fence = match bucket_configs.bucket_incarnation_id {
Some(expected) => Some(self.acquire_bucket_incarnation_fence(&bucket, expected).await?),
None => None,
};
let mut fivs = let mut fivs =
resolve_rebalance_file_info_versions_result(entry.file_info_versions(&bucket), bucket.as_str(), entry.name.as_str())?; resolve_rebalance_file_info_versions_result(entry.file_info_versions(&bucket), bucket.as_str(), entry.name.as_str())?;
@@ -208,14 +203,9 @@ impl ECStore {
} }
let version_id = version.version_id.map(|v| v.to_string()); let version_id = version.version_id.map(|v| v.to_string());
let expected_bucket_incarnation_id = bucket_configs.bucket_incarnation_id;
let mut transfer = |src_pool_idx: usize, bucket: String, rd: GetObjectReader| { let mut transfer = |src_pool_idx: usize, bucket: String, rd: GetObjectReader| {
let store = self.clone(); let store = self.clone();
async move { async move { store.rebalance_object(src_pool_idx, bucket, rd).await }
store
.rebalance_object(src_pool_idx, bucket, rd, expected_bucket_incarnation_id)
.await
}
}; };
// Route delete-marker migration through the store layer so it lands on the // Route delete-marker migration through the store layer so it lands on the
// cross-pool target (excluding the source pool), not back onto the source set. // cross-pool target (excluding the source pool), not back onto the source set.
@@ -224,12 +214,11 @@ impl ECStore {
async move { store.delete_object(&bucket, &object, opts).await } async move { store.delete_object(&bucket, &object, opts).await }
}; };
let result = migrate_entry_version( let result = migrate_entry_version(
&RebalanceMigrationBackend::new(set.as_ref(), self.as_ref()), set.as_ref(),
bucket.clone(), bucket.clone(),
pool_index, pool_index,
version, version,
version_id.clone(), version_id.clone(),
expected_bucket_incarnation_id,
rebalance_max_attempts(), rebalance_max_attempts(),
should_ignore_rebalance_data_usage_cache(bucket.as_str()), should_ignore_rebalance_data_usage_cache(bucket.as_str()),
&mut transfer, &mut transfer,
@@ -314,9 +303,6 @@ impl ECStore {
} }
if should_cleanup_rebalance_source_entry(rebalanced, fivs.versions.len(), expired) { if should_cleanup_rebalance_source_entry(rebalanced, fivs.versions.len(), expired) {
if bucket_incarnation_fence.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
return Err(Error::other("rebalance bucket incarnation fence was lost before source cleanup"));
}
let cleanup_result = self let cleanup_result = self
.finish_rebalance_entry_after_cleanup( .finish_rebalance_entry_after_cleanup(
pool_index, pool_index,
@@ -329,12 +315,6 @@ impl ECStore {
entry.name.as_str(), entry.name.as_str(),
&fivs, &fivs,
&cleanup_preflight_allowed_missing, &cleanup_preflight_allowed_missing,
data_movement::SourceCleanupBucketFence {
expected_incarnation_id: bucket_configs.bucket_incarnation_id,
lifecycle_guard: bucket_incarnation_fence
.as_ref()
.and_then(|guard| guard.namespace_lock_guard()),
},
"rebalance", "rebalance",
), ),
) )
@@ -409,14 +389,8 @@ impl ECStore {
} }
#[tracing::instrument(skip(self, rd))] #[tracing::instrument(skip(self, rd))]
async fn rebalance_object( async fn rebalance_object(self: Arc<Self>, pool_idx: usize, bucket: String, rd: GetObjectReader) -> Result<()> {
self: Arc<Self>, data_movement::migrate_object(self, pool_idx, bucket, rd, "rebalance_object").await
pool_idx: usize,
bucket: String,
rd: GetObjectReader,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
) -> Result<()> {
data_movement::migrate_object(self, pool_idx, bucket, rd, expected_bucket_incarnation_id, "rebalance_object").await
} }
async fn update_rebalance_last_error(&self, pool_idx: usize, message: String) -> Result<()> { async fn update_rebalance_last_error(&self, pool_idx: usize, message: String) -> Result<()> {
@@ -5,7 +5,6 @@ use crate::error::{Error, Result, is_err_object_not_found, is_err_version_not_fo
use crate::object_api::{GetObjectReader, ObjectInfo, ObjectOptions}; use crate::object_api::{GetObjectReader, ObjectInfo, ObjectOptions};
use crate::set_disk::SetDisks; use crate::set_disk::SetDisks;
use crate::storage_api_contracts::{object::ObjectIO, range::HTTPRangeSpec}; use crate::storage_api_contracts::{object::ObjectIO, range::HTTPRangeSpec};
use crate::store::ECStore;
use http::HeaderMap; use http::HeaderMap;
use rustfs_filemeta::FileInfo; use rustfs_filemeta::FileInfo;
use rustfs_utils::path::encode_dir_object; use rustfs_utils::path::encode_dir_object;
@@ -22,23 +21,15 @@ pub(crate) struct MigrationVersionResult {
pub error: Option<Error>, pub error: Option<Error>,
} }
pub(super) fn rebalance_delete_marker_opts( pub(super) fn rebalance_delete_marker_opts(version: &FileInfo, version_id: Option<String>, src_pool_idx: usize) -> ObjectOptions {
version: &FileInfo,
version_id: Option<String>,
src_pool_idx: usize,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
) -> ObjectOptions {
let version_suspended = version.version_id.is_none() && version_id.is_none();
ObjectOptions { ObjectOptions {
versioned: !version_suspended, versioned: true,
version_suspended, version_id,
version_id: version_id.or_else(|| version_suspended.then(|| uuid::Uuid::nil().to_string())),
mod_time: version.mod_time, mod_time: version.mod_time,
src_pool_idx, src_pool_idx,
data_movement: true, data_movement: true,
delete_marker: true, delete_marker: true,
skip_decommissioned: true, skip_decommissioned: true,
expected_bucket_incarnation_id,
delete_replication: version delete_replication: version
.replication_state_internal .replication_state_internal
.as_ref() .as_ref()
@@ -47,12 +38,7 @@ pub(super) fn rebalance_delete_marker_opts(
} }
} }
fn rebalance_remote_tiered_opts( fn rebalance_remote_tiered_opts(version: &FileInfo, version_id: Option<String>, src_pool_idx: usize) -> ObjectOptions {
version: &FileInfo,
version_id: Option<String>,
src_pool_idx: usize,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
) -> ObjectOptions {
ObjectOptions { ObjectOptions {
versioned: version_id.is_some(), versioned: version_id.is_some(),
version_id, version_id,
@@ -60,21 +46,6 @@ fn rebalance_remote_tiered_opts(
user_defined: version.metadata.clone(), user_defined: version.metadata.clone(),
src_pool_idx, src_pool_idx,
data_movement: true, data_movement: true,
include_part_checksums: true,
http_preconditions: Some(crate::data_movement::data_movement_target_precondition()),
expected_bucket_incarnation_id,
..Default::default()
}
}
pub(super) fn rebalance_object_migration_read_opts(version_id: Option<String>) -> ObjectOptions {
ObjectOptions {
version_id,
no_lock: true,
data_movement: true,
raw_data_movement_read: true,
skip_decommissioned: true,
skip_rebalancing: true,
..Default::default() ..Default::default()
} }
} }
@@ -99,19 +70,8 @@ pub(crate) trait MigrationBackend: Send + Sync {
) -> Result<()>; ) -> Result<()>;
} }
pub(crate) struct RebalanceMigrationBackend<'a> {
source: &'a SetDisks,
store: &'a ECStore,
}
impl<'a> RebalanceMigrationBackend<'a> {
pub(crate) fn new(source: &'a SetDisks, store: &'a ECStore) -> Self {
Self { source, store }
}
}
#[async_trait::async_trait] #[async_trait::async_trait]
impl MigrationBackend for RebalanceMigrationBackend<'_> { impl MigrationBackend for SetDisks {
async fn get_object_reader_for_migration( async fn get_object_reader_for_migration(
&self, &self,
bucket: &str, bucket: &str,
@@ -120,7 +80,7 @@ impl MigrationBackend for RebalanceMigrationBackend<'_> {
h: HeaderMap, h: HeaderMap,
opts: &ObjectOptions, opts: &ObjectOptions,
) -> Result<GetObjectReader> { ) -> Result<GetObjectReader> {
self.source.get_object_reader(bucket, object, range, h, opts).await self.get_object_reader(bucket, object, range, h, opts).await
} }
async fn move_remote_version_for_migration( async fn move_remote_version_for_migration(
@@ -130,7 +90,7 @@ impl MigrationBackend for RebalanceMigrationBackend<'_> {
fi: &FileInfo, fi: &FileInfo,
opts: &ObjectOptions, opts: &ObjectOptions,
) -> Result<()> { ) -> Result<()> {
self.store.decommission_tiered_object(bucket, object, fi, opts).await self.decommission_tiered_object(bucket, object, fi, opts).await
} }
} }
@@ -141,7 +101,6 @@ pub(crate) async fn migrate_entry_version<Backend, F, Fut, D, DFut>(
pool_index: usize, pool_index: usize,
version: &FileInfo, version: &FileInfo,
version_id: Option<String>, version_id: Option<String>,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
max_attempts: usize, max_attempts: usize,
ignore_data_usage_cache: bool, ignore_data_usage_cache: bool,
transfer: F, transfer: F,
@@ -154,13 +113,12 @@ where
D: FnMut(String, String, ObjectOptions) -> DFut + Send, D: FnMut(String, String, ObjectOptions) -> DFut + Send,
DFut: Future<Output = Result<ObjectInfo>> + Send, DFut: Future<Output = Result<ObjectInfo>> + Send,
{ {
migrate_entry_version_with_retry_wait_and_incarnation( migrate_entry_version_with_retry_wait(
set, set,
bucket, bucket,
pool_index, pool_index,
version, version,
version_id, version_id,
expected_bucket_incarnation_id,
max_attempts, max_attempts,
ignore_data_usage_cache, ignore_data_usage_cache,
transfer, transfer,
@@ -179,45 +137,6 @@ pub(super) async fn migrate_entry_version_with_retry_wait<Backend, F, Fut, D, DF
version_id: Option<String>, version_id: Option<String>,
max_attempts: usize, max_attempts: usize,
ignore_data_usage_cache: bool, ignore_data_usage_cache: bool,
transfer: F,
delete_marker: D,
wait_retry: W,
) -> MigrationVersionResult
where
Backend: MigrationBackend + ?Sized,
F: FnMut(usize, String, GetObjectReader) -> Fut + Send,
Fut: Future<Output = Result<()>> + Send,
D: FnMut(String, String, ObjectOptions) -> DFut + Send,
DFut: Future<Output = Result<ObjectInfo>> + Send,
W: FnMut(Duration) -> WFut + Send,
WFut: Future<Output = ()> + Send,
{
migrate_entry_version_with_retry_wait_and_incarnation(
set,
bucket,
pool_index,
version,
version_id,
None,
max_attempts,
ignore_data_usage_cache,
transfer,
delete_marker,
wait_retry,
)
.await
}
#[allow(clippy::too_many_arguments)]
async fn migrate_entry_version_with_retry_wait_and_incarnation<Backend, F, Fut, D, DFut, W, WFut>(
set: &Backend,
bucket: String,
pool_index: usize,
version: &FileInfo,
version_id: Option<String>,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
max_attempts: usize,
ignore_data_usage_cache: bool,
mut transfer: F, mut transfer: F,
mut delete_marker: D, mut delete_marker: D,
mut wait_retry: W, mut wait_retry: W,
@@ -250,7 +169,7 @@ where
&bucket, &bucket,
&version.name, &version.name,
version, version,
&rebalance_remote_tiered_opts(version, version_id, pool_index, expected_bucket_incarnation_id), &rebalance_remote_tiered_opts(version, version_id, pool_index),
) )
.await .await
{ {
@@ -293,7 +212,7 @@ where
if let Err(err) = delete_marker( if let Err(err) = delete_marker(
bucket.clone(), bucket.clone(),
version.name.clone(), version.name.clone(),
rebalance_delete_marker_opts(version, version_id, pool_index, expected_bucket_incarnation_id), rebalance_delete_marker_opts(version, version_id, pool_index),
) )
.await .await
{ {
@@ -336,7 +255,11 @@ where
&encode_dir_object(&version.name), &encode_dir_object(&version.name),
None, None,
HeaderMap::new(), HeaderMap::new(),
&rebalance_object_migration_read_opts(version_id.clone()), &ObjectOptions {
version_id: version_id.clone(),
no_lock: true,
..Default::default()
},
) )
.await .await
{ {
@@ -113,8 +113,6 @@ struct LegacyRebalanceMeta {
struct MigrationBackendSpy { struct MigrationBackendSpy {
get_object_reader: Mutex<Option<core::result::Result<GetObjectReader, Error>>>, get_object_reader: Mutex<Option<core::result::Result<GetObjectReader, Error>>>,
move_remote: Mutex<Option<core::result::Result<(), Error>>>, move_remote: Mutex<Option<core::result::Result<(), Error>>>,
get_opts: Mutex<Vec<ObjectOptions>>,
move_remote_opts: Mutex<Vec<ObjectOptions>>,
get_calls: AtomicUsize, get_calls: AtomicUsize,
move_remote_calls: AtomicUsize, move_remote_calls: AtomicUsize,
} }
@@ -127,8 +125,6 @@ impl MigrationBackendSpy {
Self { Self {
get_object_reader: Mutex::new(get_object_reader), get_object_reader: Mutex::new(get_object_reader),
move_remote: Mutex::new(move_remote), move_remote: Mutex::new(move_remote),
get_opts: Mutex::new(Vec::new()),
move_remote_opts: Mutex::new(Vec::new()),
get_calls: AtomicUsize::new(0), get_calls: AtomicUsize::new(0),
move_remote_calls: AtomicUsize::new(0), move_remote_calls: AtomicUsize::new(0),
} }
@@ -142,24 +138,6 @@ impl MigrationBackendSpy {
self.move_remote_calls.load(Ordering::SeqCst) self.move_remote_calls.load(Ordering::SeqCst)
} }
fn last_get_opts(&self) -> ObjectOptions {
self.get_opts
.lock()
.unwrap()
.last()
.cloned()
.expect("reader opts should be captured")
}
fn last_move_remote_opts(&self) -> ObjectOptions {
self.move_remote_opts
.lock()
.unwrap()
.last()
.cloned()
.expect("remote opts should be captured")
}
fn make_reader() -> GetObjectReader { fn make_reader() -> GetObjectReader {
GetObjectReader { GetObjectReader {
stream: Box::new(Cursor::new(vec![0_u8; 3])), stream: Box::new(Cursor::new(vec![0_u8; 3])),
@@ -178,10 +156,9 @@ impl MigrationBackend for MigrationBackendSpy {
_object: &str, _object: &str,
_range: Option<HTTPRangeSpec>, _range: Option<HTTPRangeSpec>,
_h: http::HeaderMap, _h: http::HeaderMap,
opts: &ObjectOptions, _opts: &ObjectOptions,
) -> Result<GetObjectReader> { ) -> Result<GetObjectReader> {
self.get_calls.fetch_add(1, Ordering::SeqCst); self.get_calls.fetch_add(1, Ordering::SeqCst);
self.get_opts.lock().unwrap().push(opts.clone());
if let Some(result) = self.get_object_reader.lock().unwrap().take() { if let Some(result) = self.get_object_reader.lock().unwrap().take() {
return result; return result;
} }
@@ -194,10 +171,9 @@ impl MigrationBackend for MigrationBackendSpy {
_bucket: &str, _bucket: &str,
_object: &str, _object: &str,
_fi: &FileInfo, _fi: &FileInfo,
opts: &ObjectOptions, _opts: &ObjectOptions,
) -> Result<()> { ) -> Result<()> {
self.move_remote_calls.fetch_add(1, Ordering::SeqCst); self.move_remote_calls.fetch_add(1, Ordering::SeqCst);
self.move_remote_opts.lock().unwrap().push(opts.clone());
if let Some(result) = self.move_remote.lock().unwrap().take() { if let Some(result) = self.move_remote.lock().unwrap().take() {
return result; return result;
} }
@@ -241,8 +217,7 @@ fn test_rebalance_delete_marker_opts_preserves_replication_state() {
..version_deleted() ..version_deleted()
}; };
let incarnation = uuid::Uuid::new_v4(); let opts = rebalance_delete_marker_opts(&version, Some("version-id".to_string()), 7);
let opts = rebalance_delete_marker_opts(&version, Some("version-id".to_string()), 7, Some(incarnation));
let replication = opts.delete_replication.expect("replication state should be preserved"); let replication = opts.delete_replication.expect("replication state should be preserved");
assert!(opts.versioned); assert!(opts.versioned);
@@ -252,22 +227,11 @@ fn test_rebalance_delete_marker_opts_preserves_replication_state() {
assert_eq!(opts.src_pool_idx, 7); assert_eq!(opts.src_pool_idx, 7);
assert_eq!(opts.version_id.as_deref(), Some("version-id")); assert_eq!(opts.version_id.as_deref(), Some("version-id"));
assert_eq!(opts.mod_time, Some(mod_time)); assert_eq!(opts.mod_time, Some(mod_time));
assert_eq!(opts.expected_bucket_incarnation_id, Some(incarnation));
assert_eq!(replication.replica_status, ReplicationStatusType::Replica); assert_eq!(replication.replica_status, ReplicationStatusType::Replica);
assert!(replication.delete_marker); assert!(replication.delete_marker);
assert_eq!(replication.replicate_decision_str, "existing"); assert_eq!(replication.replicate_decision_str, "existing");
} }
#[test]
fn test_rebalance_delete_marker_opts_preserves_suspended_null_version() {
let version = version_deleted();
let opts = rebalance_delete_marker_opts(&version, None, 7, None);
assert!(!opts.versioned);
assert!(opts.version_suspended);
assert_eq!(opts.version_id.as_deref(), Some(uuid::Uuid::nil().to_string().as_str()));
}
#[tokio::test] #[tokio::test]
async fn test_migrate_entry_version_remote_version_is_moved_without_transfer() { async fn test_migrate_entry_version_remote_version_is_moved_without_transfer() {
let backend = MigrationBackendSpy::new(None, Some(Ok(()))); let backend = MigrationBackendSpy::new(None, Some(Ok(())));
@@ -284,14 +248,12 @@ async fn test_migrate_entry_version_remote_version_is_moved_without_transfer() {
} }
}; };
let incarnation = uuid::Uuid::new_v4();
let result = migrate_entry_version( let result = migrate_entry_version(
&backend, &backend,
"bucket".to_string(), "bucket".to_string(),
0, 0,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
Some(incarnation),
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -307,10 +269,6 @@ async fn test_migrate_entry_version_remote_version_is_moved_without_transfer() {
assert_eq!(transfer_count.load(Ordering::SeqCst), 0); assert_eq!(transfer_count.load(Ordering::SeqCst), 0);
assert_eq!(backend.move_remote_calls(), 1); assert_eq!(backend.move_remote_calls(), 1);
assert_eq!(backend.get_calls(), 0); assert_eq!(backend.get_calls(), 0);
let remote_opts = backend.last_move_remote_opts();
assert!(remote_opts.include_part_checksums);
assert!(remote_opts.http_preconditions.is_some());
assert_eq!(remote_opts.expected_bucket_incarnation_id, Some(incarnation));
} }
#[tokio::test] #[tokio::test]
@@ -336,7 +294,6 @@ async fn test_migrate_entry_version_remote_not_found_is_cleanup_ignored() {
0, 0,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -373,7 +330,6 @@ async fn test_migrate_entry_version_remote_overwrite_is_not_ignored() {
0, 0,
&version, &version,
Some("vid-1".to_string()), Some("vid-1".to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -412,7 +368,6 @@ async fn test_migrate_entry_version_remote_failure_is_reported() {
0, 0,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -455,7 +410,6 @@ async fn test_migrate_entry_version_deleted_version_routes_delete_through_store_
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -495,7 +449,6 @@ async fn test_migrate_entry_version_deleted_version_not_found_is_ignored() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -538,7 +491,6 @@ async fn test_migrate_entry_version_deleted_version_overwrite_is_not_ignored() {
1, 1,
&version, &version,
Some("vid-1".to_string()), Some("vid-1".to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -568,7 +520,6 @@ async fn test_migrate_entry_version_reader_not_found_is_ignored() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -696,7 +647,6 @@ async fn test_migrate_entry_version_reader_fails_after_retries() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -735,7 +685,6 @@ async fn test_migrate_entry_version_zero_max_attempts_still_attempts_once() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
0, 0,
false, false,
&mut transfer, &mut transfer,
@@ -801,13 +750,6 @@ async fn test_migrate_entry_version_transfer_retries_before_success() {
assert_eq!(backend.get_calls(), 2); assert_eq!(backend.get_calls(), 2);
assert_eq!(transfer_count.load(Ordering::SeqCst), 2); assert_eq!(transfer_count.load(Ordering::SeqCst), 2);
assert_eq!(wait_count.load(Ordering::SeqCst), 1); assert_eq!(wait_count.load(Ordering::SeqCst), 1);
let read_opts = backend.last_get_opts();
assert_eq!(read_opts.version_id.as_deref(), version.version_id.map(|id| id.to_string()).as_deref());
assert!(read_opts.no_lock);
assert!(read_opts.data_movement);
assert!(read_opts.raw_data_movement_read);
assert!(read_opts.skip_decommissioned);
assert!(read_opts.skip_rebalancing);
} }
#[tokio::test] #[tokio::test]
@@ -880,7 +822,6 @@ async fn test_migrate_entry_version_transfer_fails_after_retries() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
2, 2,
false, false,
&mut transfer, &mut transfer,
@@ -919,7 +860,6 @@ async fn test_migrate_entry_version_transfer_not_found_is_ignored() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -961,7 +901,6 @@ async fn test_migrate_entry_version_transfer_overwrite_is_not_ignored() {
1, 1,
&version, &version,
Some("vid-1".to_string()), Some("vid-1".to_string()),
None,
3, 3,
false, false,
&mut transfer, &mut transfer,
@@ -1004,7 +943,6 @@ async fn test_migrate_entry_version_ignores_data_usage_cache_when_enabled() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
2, 2,
true, true,
&mut transfer, &mut transfer,
@@ -1047,7 +985,6 @@ async fn test_migrate_entry_version_data_usage_cache_moves_when_ignore_disabled(
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
2, 2,
false, false,
&mut transfer, &mut transfer,
@@ -2089,7 +2026,6 @@ async fn test_migrate_entry_version_transfer_failure_reports_write_target_stage(
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
1, 1,
false, false,
&mut transfer, &mut transfer,
@@ -2114,7 +2050,6 @@ async fn test_migrate_entry_version_reader_failure_reports_read_source_stage() {
1, 1,
&version, &version,
version.version_id.map(|v| v.to_string()), version.version_id.map(|v| v.to_string()),
None,
1, 1,
false, false,
&mut transfer, &mut transfer,
@@ -36,7 +36,6 @@ pub type RStats = Vec<Arc<RebalanceStats>>;
#[derive(Debug, Default)] #[derive(Debug, Default)]
pub(super) struct RebalanceBucketConfigs { pub(super) struct RebalanceBucketConfigs {
pub(super) bucket_incarnation_id: Option<uuid::Uuid>,
pub(super) lifecycle_config: Option<s3s::dto::BucketLifecycleConfiguration>, pub(super) lifecycle_config: Option<s3s::dto::BucketLifecycleConfiguration>,
pub(super) object_lock_config: Option<s3s::dto::ObjectLockConfiguration>, pub(super) object_lock_config: Option<s3s::dto::ObjectLockConfiguration>,
pub(super) replication_config: Option<(s3s::dto::ReplicationConfiguration, OffsetDateTime)>, pub(super) replication_config: Option<(s3s::dto::ReplicationConfiguration, OffsetDateTime)>,
@@ -406,7 +406,6 @@ pub(super) async fn load_rebalance_bucket_configs(api: &ECStore, bucket: &str) -
let expiry_configs = crate::bucket::lifecycle::get_expiry_configs(api, bucket).await?; let expiry_configs = crate::bucket::lifecycle::get_expiry_configs(api, bucket).await?;
Ok(RebalanceBucketConfigs { Ok(RebalanceBucketConfigs {
bucket_incarnation_id: Some(api.bucket_incarnation_id_from_disk(bucket).await?),
lifecycle_config: expiry_configs.lifecycle.map(|config| (*config).clone()), lifecycle_config: expiry_configs.lifecycle.map(|config| (*config).clone()),
object_lock_config: expiry_configs.object_lock.map(|config| (*config).clone()), object_lock_config: expiry_configs.object_lock.map(|config| (*config).clone()),
replication_config: resolve_rebalance_optional_bucket_config_result( replication_config: resolve_rebalance_optional_bucket_config_result(
File diff suppressed because it is too large Load Diff
-83
View File
@@ -85,15 +85,6 @@ impl SetDisks {
format!("{}/{}", Self::get_multipart_sha_dir(bucket, object), upload_uuid) format!("{}/{}", Self::get_multipart_sha_dir(bucket, object), upload_uuid)
} }
pub(super) fn get_multipart_upload_dir(bucket: &str, object: &str, upload_id: &str, data_movement: bool) -> String {
let upload_dir = Self::get_upload_id_dir(bucket, object, upload_id);
if data_movement {
format!("{DATA_MOVEMENT_MULTIPART_PREFIX}/{upload_dir}")
} else {
upload_dir
}
}
pub(super) fn get_multipart_sha_dir(bucket: &str, object: &str) -> String { pub(super) fn get_multipart_sha_dir(bucket: &str, object: &str) -> String {
let path = format!("{bucket}/{object}"); let path = format!("{bucket}/{object}");
let mut hasher = Sha256::new(); let mut hasher = Sha256::new();
@@ -475,28 +466,6 @@ impl SetDisks {
Self::find_file_info_in_quorum(metas, &mod_time, &etag, quorum) Self::find_file_info_in_quorum(metas, &mod_time, &etag, quorum)
} }
pub(crate) fn hydrate_selected_fileinfo_part_checksums(fi: &mut FileInfo) -> disk::error::Result<()> {
fi.hydrate_data_movement_part_checksums().map_err(DiskError::from)?;
for part in &fi.parts {
let Some(checksums) = part.checksums.as_ref() else {
continue;
};
let mut algorithms = HashSet::with_capacity(checksums.len());
for (name, value) in checksums {
let Some(checksum) = rustfs_rio::Checksum::new_from_string(name, value) else {
return Err(DiskError::FileCorrupt);
};
if checksum.checksum_type.is(rustfs_rio::ChecksumType::MULTIPART) {
return Err(DiskError::FileCorrupt);
}
if !algorithms.insert(checksum.checksum_type.base().0) {
return Err(DiskError::FileCorrupt);
}
}
}
Ok(())
}
fn update_hash_bytes(hasher: &mut Sha256, value: &[u8]) { fn update_hash_bytes(hasher: &mut Sha256, value: &[u8]) {
hasher.update(value.len().to_le_bytes()); hasher.update(value.len().to_le_bytes());
hasher.update(value); hasher.update(value);
@@ -1110,25 +1079,6 @@ impl SetDisks {
shuffled_disks shuffled_disks
} }
pub(super) fn shuffle_disks_owned(mut disks: Vec<Option<DiskStore>>, distribution: &[usize]) -> Vec<Option<DiskStore>> {
if distribution.is_empty() {
return disks;
}
let mut shuffled_disks = vec![None; disks.len()];
for (index, disk) in disks.iter_mut().enumerate() {
let Some(slot) = distribution
.get(index)
.and_then(|block_index| block_index.checked_sub(1))
.filter(|slot| *slot < shuffled_disks.len())
else {
continue;
};
shuffled_disks[slot] = disk.take();
}
shuffled_disks
}
pub(super) fn shuffle_check_parts(parts_errs: &[usize], distribution: &[usize]) -> Vec<usize> { pub(super) fn shuffle_check_parts(parts_errs: &[usize], distribution: &[usize]) -> Vec<usize> {
if distribution.is_empty() { if distribution.is_empty() {
return parts_errs.to_vec(); return parts_errs.to_vec();
@@ -1440,23 +1390,6 @@ mod tests {
assert_eq!(owned_slots, expected_slots, "fallback disk slots must match the borrowing variant"); assert_eq!(owned_slots, expected_slots, "fallback disk slots must match the borrowing variant");
} }
#[tokio::test]
async fn owned_shuffle_preserves_fresh_put_metadata() {
let tempdir = tempfile::tempdir().expect("tempdir should be created");
let fi = FileInfo::new("bucket/object", 2, 1);
let parts = vec![fi.clone(); fi.erasure.distribution.len()];
let disks = shuffle_test_disks(&tempdir, parts.len()).await;
let (owned_disks, owned_parts) = SetDisks::shuffle_disks_and_parts_metadata_by_index_owned(disks, parts, &fi);
assert!(owned_disks.iter().all(Option::is_some), "fresh PUT must retain every online disk");
assert_eq!(
owned_parts,
vec![fi; owned_disks.len()],
"fresh PUT metadata with pending shard indexes must survive init fallback"
);
}
// backlog#949: corrupt/adversarial distribution values (0 or > N) must not // backlog#949: corrupt/adversarial distribution values (0 or > N) must not
// trigger a `usize` underflow / out-of-bounds panic in the shuffle helpers. // trigger a `usize` underflow / out-of-bounds panic in the shuffle helpers.
#[test] #[test]
@@ -1486,22 +1419,6 @@ mod tests {
assert_eq!(result.len(), disks.len(), "output length must be preserved"); assert_eq!(result.len(), disks.len(), "output length must be preserved");
} }
#[tokio::test]
async fn owned_disk_shuffle_matches_borrowing_variant() {
let tempdir = tempfile::tempdir().expect("tempdir should be created");
let mut disks = shuffle_test_disks(&tempdir, 4).await;
disks[1] = None;
disks[3] = None;
let distribution = [3, 1, 4, 2];
let expected = SetDisks::shuffle_disks(&disks, &distribution);
let actual = SetDisks::shuffle_disks_owned(disks, &distribution);
let expected_slots = expected.iter().map(Option::is_some).collect::<Vec<_>>();
let actual_slots = actual.iter().map(Option::is_some).collect::<Vec<_>>();
assert_eq!(actual_slots, expected_slots, "owned shuffle must preserve disk placement");
}
#[tokio::test] #[tokio::test]
async fn shuffle_disks_and_parts_metadata_survives_corrupt_distribution() { async fn shuffle_disks_and_parts_metadata_survives_corrupt_distribution() {
let tempdir = tempfile::tempdir().expect("tempdir should be created"); let tempdir = tempfile::tempdir().expect("tempdir should be created");
File diff suppressed because it is too large Load Diff
+2 -31
View File
@@ -542,8 +542,7 @@ impl SetDisks {
let filter_by_etag = quorum_etag.is_some(); let filter_by_etag = quorum_etag.is_some();
match Self::pick_valid_fileinfo(&parts_metadata, quorum_mod_time, quorum_etag.clone(), read_quorum as usize) { match Self::pick_valid_fileinfo(&parts_metadata, quorum_mod_time, quorum_etag.clone(), read_quorum as usize) {
Ok(mut latest_meta) => { Ok(latest_meta) => {
Self::hydrate_selected_fileinfo_part_checksums(&mut latest_meta)?;
trace!( trace!(
event = EVENT_SET_DISK_HEAL, event = EVENT_SET_DISK_HEAL,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -1425,33 +1424,6 @@ impl SetDisks {
/// post-heal tail — reclaim identically. Never fails the heal: delete errors /// post-heal tail — reclaim identically. Never fails the heal: delete errors
/// are logged and swallowed. Callers must gate this on `!opts.dry_run`. /// are logged and swallowed. Callers must gate this on `!opts.dry_run`.
async fn reclaim_orphan_data_dirs_best_effort(&self, bucket: &str, object: &str) { async fn reclaim_orphan_data_dirs_best_effort(&self, bucket: &str, object: &str) {
match self.reconcile_old_data_cleanup_receipts(bucket, object).await {
Ok(removed) if removed > 0 => {
debug!(
event = EVENT_SET_DISK_HEAL,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_SET_DISK,
bucket,
object,
removed,
state = "old_data_cleanup_receipt_reconciled",
"Set disk old-data cleanup receipts reconciled"
);
}
Ok(_) => {}
Err(e) => {
warn!(
event = EVENT_SET_DISK_HEAL,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_SET_DISK,
bucket,
object,
error = %e,
state = "old_data_cleanup_receipt_reconcile_failed",
"Set disk old-data cleanup receipt reconcile failed"
);
}
}
match self.reclaim_orphan_data_dirs(bucket, object).await { match self.reclaim_orphan_data_dirs(bucket, object).await {
Ok(removed) if removed > 0 => { Ok(removed) if removed > 0 => {
debug!( debug!(
@@ -3198,11 +3170,10 @@ mod heal_result_report_tests {
.await .await
.expect("object should be written"); .expect("object should be written");
let snapshot = set let (fi, _, _) = set
.get_object_fileinfo(bucket, object, &opts, true, false) .get_object_fileinfo(bucket, object, &opts, true, false)
.await .await
.expect("object metadata should resolve"); .expect("object metadata should resolve");
let fi = snapshot.fi();
assert_eq!(fi.erasure.parity_blocks, 0); assert_eq!(fi.erasure.parity_blocks, 0);
let data_dir = fi.data_dir.expect("non-inline object should have a data directory"); let data_dir = fi.data_dir.expect("non-inline object should have a data directory");
let part_path = dir.path().join(bucket).join(object).join(data_dir.to_string()).join("part.1"); let part_path = dir.path().join(bucket).join(object).join(data_dir.to_string()).join("part.1");
+3 -14
View File
@@ -36,21 +36,10 @@ impl crate::storage_api_contracts::namespace::NamespaceLocking for SetDisks {
// test's transient DistErasure window) would push this set's namespace // test's transient DistErasure window) would push this set's namespace
// locking onto its own — possibly empty — dist locker list. // locking onto its own — possibly empty — dist locker list.
let set_lock = if self.ctx.is_dist_erasure().await { let set_lock = if self.ctx.is_dist_erasure().await {
let lockers = if self.lockers.len() == self.shared_lockers.len() // Calculate quorum based on lockers count (majority)
&& self let lockers_count = self.lockers.len();
.lockers
.iter()
.zip(self.shared_lockers.iter())
.all(|(current, shared)| Arc::ptr_eq(current, shared))
{
self.shared_lockers.clone()
} else {
Arc::from(self.lockers.clone())
};
// Calculate quorum from the exact client domain used by this lock.
let lockers_count = lockers.len();
let write_quorum = if lockers_count > 1 { (lockers_count / 2) + 1 } else { 1 }; let write_quorum = if lockers_count > 1 { (lockers_count / 2) + 1 } else { 1 };
NamespaceLock::with_clients_and_quorum_shared(self.set_lock_namespace.clone(), lockers, write_quorum) NamespaceLock::with_clients_and_quorum_shared(self.set_lock_namespace.clone(), self.lockers.clone(), write_quorum)
} else { } else {
NamespaceLock::with_local_manager_shared(self.set_lock_namespace.clone(), self.local_lock_manager.clone()) NamespaceLock::with_local_manager_shared(self.set_lock_namespace.clone(), self.local_lock_manager.clone())
}; };
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+68 -238
View File
@@ -19,7 +19,7 @@ use crate::diagnostics::get::{
GET_METADATA_CACHE_REASON_DATA_MOVEMENT, GET_METADATA_CACHE_REASON_DELETE_MARKER, GET_METADATA_CACHE_REASON_DIST_ERASURE, GET_METADATA_CACHE_REASON_DATA_MOVEMENT, GET_METADATA_CACHE_REASON_DELETE_MARKER, GET_METADATA_CACHE_REASON_DIST_ERASURE,
GET_METADATA_CACHE_REASON_INCL_FREE_VERSIONS, GET_METADATA_CACHE_REASON_INSUFFICIENT_CACHED_QUORUM, GET_METADATA_CACHE_REASON_INCL_FREE_VERSIONS, GET_METADATA_CACHE_REASON_INSUFFICIENT_CACHED_QUORUM,
GET_METADATA_CACHE_REASON_META_BUCKET, GET_METADATA_CACHE_REASON_NO_LOCK, GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED, GET_METADATA_CACHE_REASON_META_BUCKET, GET_METADATA_CACHE_REASON_NO_LOCK, GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED,
GET_METADATA_CACHE_REASON_NOT_READ_DATA, GET_METADATA_CACHE_REASON_PART_CHECKSUMS, GET_METADATA_CACHE_REASON_PART_NUMBER, GET_METADATA_CACHE_REASON_NOT_READ_DATA, GET_METADATA_CACHE_REASON_PART_NUMBER,
GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ, GET_METADATA_CACHE_REASON_STALE_PUBLICATION, GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ, GET_METADATA_CACHE_REASON_STALE_PUBLICATION,
GET_METADATA_CACHE_REASON_USABLE, GET_METADATA_CACHE_REASON_VERSION_ID, GET_METADATA_CACHE_REASON_VERSION_SUSPENDED, GET_METADATA_CACHE_REASON_USABLE, GET_METADATA_CACHE_REASON_VERSION_ID, GET_METADATA_CACHE_REASON_VERSION_SUSPENDED,
GET_METADATA_CACHE_REASON_VERSIONED, GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA, GET_METADATA_CACHE_REASON_VERSIONED, GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA,
@@ -180,9 +180,9 @@ impl SetDisks {
let key = GetObjectMetadataCacheKey::new(bucket, object, generation); let key = GetObjectMetadataCacheKey::new(bucket, object, generation);
let entry = Arc::new(GetObjectMetadataCacheEntry { let entry = Arc::new(GetObjectMetadataCacheEntry {
created_at: Instant::now(), created_at: Instant::now(),
fi: fi.clone(), fi: Arc::new(fi.clone()),
parts_metadata: parts_metadata.to_vec(), parts_metadata: Arc::new(parts_metadata.to_vec()),
online_disks: online_disks.to_vec(), online_disks: Arc::new(online_disks.to_vec()),
read_quorum, read_quorum,
}); });
self.insert_get_object_metadata_cache_entry_after_insert(key, generation, entry, || {}) self.insert_get_object_metadata_cache_entry_after_insert(key, generation, entry, || {})
@@ -300,7 +300,11 @@ impl SetDisks {
GET_STAGE_METADATA_CACHE_LOOKUP, GET_STAGE_METADATA_CACHE_LOOKUP,
metadata_cache_lookup_start, metadata_cache_lookup_start,
); );
return Ok(GetObjectFileInfo::shared(cached)); return Ok((
GetObjectMetadata::Shared(Arc::clone(&cached.fi)),
GetObjectMetadata::Shared(Arc::clone(&cached.parts_metadata)),
GetObjectMetadata::Shared(Arc::clone(&cached.online_disks)),
));
} }
MetadataCacheLookup::Miss => { MetadataCacheLookup::Miss => {
rustfs_io_metrics::record_get_object_metadata_cache_decision( rustfs_io_metrics::record_get_object_metadata_cache_decision(
@@ -336,7 +340,7 @@ impl SetDisks {
// read_all_fileinfo_observed (see read_all_fileinfo_early_stop in // read_all_fileinfo_observed (see read_all_fileinfo_early_stop in
// core/io_primitives.rs); unsafe requests and callers that opt out // core/io_primitives.rs); unsafe requests and callers that opt out
// (allow_early_stop=false) fall back to full-wait. // (allow_early_stop=false) fall back to full-wait.
let (mut parts_metadata, errs, metadata_fanout_diagnostics) = Self::read_all_fileinfo_observed( let (parts_metadata, errs, metadata_fanout_diagnostics) = Self::read_all_fileinfo_observed(
&disks, &disks,
"", "",
bucket, bucket,
@@ -390,17 +394,8 @@ impl SetDisks {
return Err(to_object_err(err.into(), vec![bucket, object])); return Err(to_object_err(err.into(), vec![bucket, object]));
} }
let (op_online_disks, mut fi, fileinfo_selection_quorum) = let (op_online_disks, fi, fileinfo_selection_quorum) =
Self::select_valid_fileinfo(&disks, &parts_metadata, &errs, vid.as_str(), read_quorum, write_quorum)?; Self::select_valid_fileinfo(&disks, &parts_metadata, &errs, vid.as_str(), read_quorum, write_quorum)?;
let include_part_checksums =
opts.include_part_checksums || opts.part_number.is_some() || opts.data_movement || opts.raw_data_movement_read;
if include_part_checksums {
Self::hydrate_selected_fileinfo_part_checksums(&mut fi)?;
} else {
for metadata in std::iter::once(&mut fi).chain(parts_metadata.iter_mut()) {
rustfs_utils::http::remove_str(&mut metadata.metadata, rustfs_utils::http::SUFFIX_PART_CHECKSUMS);
}
}
metadata_fanout_diagnostics.record_quorum_candidate_latency(metadata_metrics_path, fileinfo_selection_quorum); metadata_fanout_diagnostics.record_quorum_candidate_latency(metadata_metrics_path, fileinfo_selection_quorum);
if errs.iter().any(|err| err.is_some()) { if errs.iter().any(|err| err.is_some()) {
let version_id = resolved_read_repair_version_id(&fi, opts.version_id.as_deref()); let version_id = resolved_read_repair_version_id(&fi, opts.version_id.as_deref());
@@ -432,7 +427,11 @@ impl SetDisks {
// let online_disks: Vec<Option<DiskStore>> = op_online_disks.iter().filter(|v| v.is_some()).cloned().collect(); // let online_disks: Vec<Option<DiskStore>> = op_online_disks.iter().filter(|v| v.is_some()).cloned().collect();
Ok(GetObjectFileInfo::owned(fi, parts_metadata, op_online_disks)) Ok((
GetObjectMetadata::Owned(fi),
GetObjectMetadata::Owned(parts_metadata),
GetObjectMetadata::Owned(op_online_disks),
))
} }
#[hotpath::measure(impl_type = "SetDisks")] #[hotpath::measure(impl_type = "SetDisks")]
@@ -442,15 +441,14 @@ impl SetDisks {
object: &str, object: &str,
opts: &ObjectOptions, opts: &ObjectOptions,
) -> (ObjectInfo, usize, Option<StorageError>) { ) -> (ObjectInfo, usize, Option<StorageError>) {
let snapshot = match self.get_object_fileinfo(bucket, object, opts, false, false).await { let fi = match self.get_object_fileinfo(bucket, object, opts, false, false).await {
Ok(snapshot) => snapshot, Ok((fi, _, _)) => fi,
Err(e) => return (ObjectInfo::default(), 0, Some(e)), Err(e) => return (ObjectInfo::default(), 0, Some(e)),
}; };
let fi = snapshot.fi();
let write_quorum = fi.write_quorum(self.default_write_quorum()); let write_quorum = fi.write_quorum(self.default_write_quorum());
let oi = ObjectInfo::from_file_info(fi, bucket, object, opts.versioned || opts.version_suspended); let oi = ObjectInfo::from_file_info(&fi, bucket, object, opts.versioned || opts.version_suspended);
if !fi.version_purge_status().is_empty() && opts.version_id.is_some() { if !fi.version_purge_status().is_empty() && opts.version_id.is_some() {
return ( return (
@@ -482,7 +480,6 @@ impl SetDisks {
pub(super) async fn try_get_object_direct_data_shards_with_fileinfo( pub(super) async fn try_get_object_direct_data_shards_with_fileinfo(
bucket: &str, bucket: &str,
object: &str, object: &str,
erasure_cache: Arc<ErasureCache>,
fi: &FileInfo, fi: &FileInfo,
files: &[FileInfo], files: &[FileInfo],
disks: &[Option<DiskStore>], disks: &[Option<DiskStore>],
@@ -503,7 +500,13 @@ impl SetDisks {
return Ok(None); return Ok(None);
} }
let erasure = erasure_cache.get_for_file_info(fi)?; let erasure = coding::Erasure::try_new_with_options(
fi.erasure.data_blocks,
fi.erasure.parity_blocks,
fi.erasure.block_size,
fi.uses_legacy_checksum,
)
.map_err(Error::from)?;
let checksum_info = fi.erasure.get_checksum_info(part.number); let checksum_info = fi.erasure.get_checksum_info(part.number);
let checksum_algo = if fi.uses_legacy_checksum && checksum_info.algorithm == HashAlgorithm::HighwayHash256S { let checksum_algo = if fi.uses_legacy_checksum && checksum_info.algorithm == HashAlgorithm::HighwayHash256S {
@@ -631,7 +634,6 @@ impl SetDisks {
// &self, // &self,
bucket: &str, bucket: &str,
object: &str, object: &str,
erasure_cache: Arc<ErasureCache>,
offset: usize, offset: usize,
length: i64, length: i64,
writer: &mut W, writer: &mut W,
@@ -726,7 +728,13 @@ impl SetDisks {
object, offset, length, end_offset, part_index, last_part_index, last_part_relative_offset, "Multipart read bounds" object, offset, length, end_offset, part_index, last_part_index, last_part_relative_offset, "Multipart read bounds"
); );
let erasure = erasure_cache.get_for_file_info(&fi)?; let erasure = coding::Erasure::try_new_with_options(
fi.erasure.data_blocks,
fi.erasure.parity_blocks,
fi.erasure.block_size,
fi.uses_legacy_checksum,
)
.map_err(Error::from)?;
let part_indices: Vec<usize> = (part_index..=last_part_index).collect(); let part_indices: Vec<usize> = (part_index..=last_part_index).collect();
debug!(bucket, object, ?part_indices, "Multipart part indices to stream"); debug!(bucket, object, ?part_indices, "Multipart part indices to stream");
@@ -1160,7 +1168,6 @@ impl SetDisks {
pub(super) async fn get_object_decode_reader_with_fileinfo( pub(super) async fn get_object_decode_reader_with_fileinfo(
bucket: &str, bucket: &str,
object: &str, object: &str,
erasure_cache: Arc<ErasureCache>,
fi: &FileInfo, fi: &FileInfo,
files: &[FileInfo], files: &[FileInfo],
disks: &[Option<DiskStore>], disks: &[Option<DiskStore>],
@@ -1171,7 +1178,14 @@ impl SetDisks {
metrics_size_bucket: &'static str, metrics_size_bucket: &'static str,
prefer_data_blocks_first_reader_setup: bool, prefer_data_blocks_first_reader_setup: bool,
) -> Result<GetCodecStreamingReaderBuildOutcome> { ) -> Result<GetCodecStreamingReaderBuildOutcome> {
let erasure = erasure_cache.get_for_file_info(fi)?; let erasure = coding::Erasure::try_new_with_options(
fi.erasure.data_blocks,
fi.erasure.parity_blocks,
fi.erasure.block_size,
fi.uses_legacy_checksum,
)
.map_err(Error::from)?;
let (disks, files) = Self::shuffle_disks_and_parts_metadata_by_index(disks, files, fi); let (disks, files) = Self::shuffle_disks_and_parts_metadata_by_index(disks, files, fi);
if fi.parts.len() == 1 { if fi.parts.len() == 1 {
@@ -1558,7 +1572,7 @@ struct LazyCodecPartContext {
fi: FileInfo, fi: FileInfo,
files: Vec<FileInfo>, files: Vec<FileInfo>,
disks: Vec<Option<DiskStore>>, disks: Vec<Option<DiskStore>>,
erasure: Arc<coding::Erasure>, erasure: coding::Erasure,
skip_verify_bitrot: bool, skip_verify_bitrot: bool,
metrics_object_class: &'static str, metrics_object_class: &'static str,
metrics_size_bucket: &'static str, metrics_size_bucket: &'static str,
@@ -1812,9 +1826,6 @@ fn get_object_metadata_cache_request_bypass_reason(bucket: &str, opts: &ObjectOp
if opts.part_number.is_some() { if opts.part_number.is_some() {
return Some(GET_METADATA_CACHE_REASON_PART_NUMBER); return Some(GET_METADATA_CACHE_REASON_PART_NUMBER);
} }
if opts.include_part_checksums {
return Some(GET_METADATA_CACHE_REASON_PART_CHECKSUMS);
}
if opts.data_movement { if opts.data_movement {
return Some(GET_METADATA_CACHE_REASON_DATA_MOVEMENT); return Some(GET_METADATA_CACHE_REASON_DATA_MOVEMENT);
} }
@@ -2042,7 +2053,6 @@ mod metadata_cache_tests {
let err = SetDisks::get_object_with_fileinfo( let err = SetDisks::get_object_with_fileinfo(
"bucket", "bucket",
"object", "object",
Arc::new(ErasureCache::new()),
0, 0,
1, 1,
&mut output, &mut output,
@@ -2073,7 +2083,6 @@ mod metadata_cache_tests {
let err = SetDisks::get_object_with_fileinfo( let err = SetDisks::get_object_with_fileinfo(
bucket, bucket,
object, object,
Arc::new(ErasureCache::new()),
2, 2,
1, 1,
&mut output, &mut output,
@@ -2097,7 +2106,6 @@ mod metadata_cache_tests {
let err = SetDisks::get_object_with_fileinfo( let err = SetDisks::get_object_with_fileinfo(
bucket, bucket,
object, object,
Arc::new(ErasureCache::new()),
usize::MAX, usize::MAX,
1, 1,
&mut output, &mut output,
@@ -2119,7 +2127,6 @@ mod metadata_cache_tests {
let err = SetDisks::get_object_with_fileinfo( let err = SetDisks::get_object_with_fileinfo(
bucket, bucket,
object, object,
Arc::new(ErasureCache::new()),
1, 1,
1, 1,
&mut output, &mut output,
@@ -2143,7 +2150,6 @@ mod metadata_cache_tests {
let err = SetDisks::get_object_with_fileinfo( let err = SetDisks::get_object_with_fileinfo(
bucket, bucket,
object, object,
Arc::new(ErasureCache::new()),
0, 0,
1, 1,
&mut output, &mut output,
@@ -2181,7 +2187,6 @@ mod metadata_cache_tests {
SetDisks::get_object_with_fileinfo( SetDisks::get_object_with_fileinfo(
bucket, bucket,
object, object,
Arc::new(ErasureCache::new()),
0, 0,
0, 0,
&mut output, &mut output,
@@ -2214,7 +2219,6 @@ mod metadata_cache_tests {
let err = SetDisks::get_object_with_fileinfo( let err = SetDisks::get_object_with_fileinfo(
bucket, bucket,
object, object,
Arc::new(ErasureCache::new()),
0, 0,
1, 1,
&mut output, &mut output,
@@ -2493,16 +2497,6 @@ mod metadata_cache_tests {
Some(GET_METADATA_CACHE_REASON_PART_NUMBER) Some(GET_METADATA_CACHE_REASON_PART_NUMBER)
); );
opts = ObjectOptions {
include_part_checksums: true,
..Default::default()
};
assert!(!is_get_object_metadata_cache_request_eligible("bucket", &opts, true));
assert_eq!(
get_object_metadata_cache_request_bypass_reason("bucket", &opts, true),
Some(GET_METADATA_CACHE_REASON_PART_CHECKSUMS)
);
opts = ObjectOptions { opts = ObjectOptions {
data_movement: true, data_movement: true,
..Default::default() ..Default::default()
@@ -2707,14 +2701,22 @@ mod metadata_cache_tests {
.await .await
.expect("fresh cache entry should be returned"); .expect("fresh cache entry should be returned");
let returned = set let (returned_fi, returned_parts_metadata, returned_online_disks) = set
.get_object_fileinfo("bucket", "object", &ObjectOptions::default(), true, false) .get_object_fileinfo("bucket", "object", &ObjectOptions::default(), true, false)
.await .await
.expect("cache-backed metadata lookup should succeed"); .expect("cache-backed metadata lookup should succeed");
assert!( assert!(
returned.shared_entry().is_some_and(|value| Arc::ptr_eq(value, &cached)), matches!(returned_fi, GetObjectMetadata::Shared(ref value) if Arc::ptr_eq(value, &cached.fi)),
"cache hits must share the complete metadata snapshot" "cache hits must share FileInfo ownership"
);
assert!(
matches!(returned_parts_metadata, GetObjectMetadata::Shared(ref value) if Arc::ptr_eq(value, &cached.parts_metadata)),
"cache hits must share the metadata vector"
);
assert!(
matches!(returned_online_disks, GetObjectMetadata::Shared(ref value) if Arc::ptr_eq(value, &cached.online_disks)),
"cache hits must share the online-disk vector"
); );
} }
@@ -2757,9 +2759,9 @@ mod metadata_cache_tests {
), ),
Arc::new(GetObjectMetadataCacheEntry { Arc::new(GetObjectMetadataCacheEntry {
created_at: Instant::now(), created_at: Instant::now(),
fi: fi.clone(), fi: Arc::new(fi.clone()),
parts_metadata: vec![fi], parts_metadata: Arc::new(vec![fi]),
online_disks: vec![None], online_disks: Arc::new(vec![None]),
read_quorum: 1, read_quorum: 1,
}), }),
) )
@@ -2853,12 +2855,13 @@ mod metadata_cache_tests {
barrier.wait_until_paused().await; barrier.wait_until_paused().await;
set.invalidate_get_object_metadata_cache(bucket, object).await; set.invalidate_get_object_metadata_cache(bucket, object).await;
barrier.release(); barrier.release();
let snapshot = read let (fi, parts_metadata, online_disks) = read
.await .await
.expect("metadata read task should not panic") .expect("metadata read task should not panic")
.expect("metadata fanout should still return its selected FileInfo"); .expect("metadata fanout should still return its selected FileInfo");
assert!(snapshot.owned.is_some()); assert!(matches!(fi, GetObjectMetadata::Owned(_)));
assert!(snapshot.has_valid_representation()); assert!(matches!(parts_metadata, GetObjectMetadata::Owned(_)));
assert!(matches!(online_disks, GetObjectMetadata::Owned(_)));
assert!( assert!(
set.get_object_metadata_cache set.get_object_metadata_cache
@@ -2905,9 +2908,9 @@ mod metadata_cache_tests {
let key = GetObjectMetadataCacheKey::new("bucket", "object", generation); let key = GetObjectMetadataCacheKey::new("bucket", "object", generation);
let entry = Arc::new(GetObjectMetadataCacheEntry { let entry = Arc::new(GetObjectMetadataCacheEntry {
created_at: Instant::now(), created_at: Instant::now(),
fi: fi.clone(), fi: Arc::new(fi.clone()),
parts_metadata: vec![fi], parts_metadata: Arc::new(vec![fi]),
online_disks: Vec::new(), online_disks: Arc::new(Vec::new()),
read_quorum: 0, read_quorum: 0,
}); });
@@ -3015,9 +3018,9 @@ mod metadata_cache_tests {
let entry = |fi: FileInfo| { let entry = |fi: FileInfo| {
Arc::new(GetObjectMetadataCacheEntry { Arc::new(GetObjectMetadataCacheEntry {
created_at: Instant::now(), created_at: Instant::now(),
parts_metadata: vec![fi.clone()], parts_metadata: Arc::new(vec![fi.clone()]),
fi, fi: Arc::new(fi),
online_disks: Vec::new(), online_disks: Arc::new(Vec::new()),
read_quorum: 0, read_quorum: 0,
}) })
}; };
@@ -4080,8 +4083,8 @@ mod tests {
get_codec_streaming_reader_gate( get_codec_streaming_reader_gate(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
range,
None, None,
classify_get_codec_streaming_object_class(range, object_info, fi),
object_info, object_info,
fi, fi,
lock_optimization_enabled, lock_optimization_enabled,
@@ -4098,8 +4101,8 @@ mod tests {
get_codec_streaming_reader_gate( get_codec_streaming_reader_gate(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
range,
part_number, part_number,
classify_get_codec_streaming_object_class(range, object_info, fi),
object_info, object_info,
fi, fi,
lock_optimization_enabled, lock_optimization_enabled,
@@ -4119,7 +4122,6 @@ mod tests {
let result = SetDisks::get_object_decode_reader_with_fileinfo( let result = SetDisks::get_object_decode_reader_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
&fi, &fi,
&[], &[],
&[], &[],
@@ -4142,7 +4144,6 @@ mod tests {
let invalid_size = SetDisks::get_object_decode_reader_with_fileinfo( let invalid_size = SetDisks::get_object_decode_reader_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
&single_part, &single_part,
&[], &[],
&[], &[],
@@ -4163,7 +4164,6 @@ mod tests {
SetDisks::get_object_decode_reader_with_fileinfo( SetDisks::get_object_decode_reader_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
&multipart, &multipart,
&[], &[],
&[], &[],
@@ -4188,7 +4188,6 @@ mod tests {
SetDisks::get_object_decode_reader_with_fileinfo( SetDisks::get_object_decode_reader_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
&multipart, &multipart,
&[], &[],
&[], &[],
@@ -4217,7 +4216,6 @@ mod tests {
SetDisks::get_object_decode_reader_with_fileinfo( SetDisks::get_object_decode_reader_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
&multipart, &multipart,
&[], &[],
&[], &[],
@@ -4271,7 +4269,6 @@ mod tests {
SetDisks::get_object_decode_reader_with_fileinfo( SetDisks::get_object_decode_reader_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
&fi, &fi,
&files, &files,
&disks, &disks,
@@ -4325,7 +4322,6 @@ mod tests {
SetDisks::get_object_decode_reader_with_fileinfo( SetDisks::get_object_decode_reader_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
&fi, &fi,
&files, &files,
&disks, &disks,
@@ -4370,7 +4366,6 @@ mod tests {
SetDisks::get_object_with_fileinfo( SetDisks::get_object_with_fileinfo(
CODEC_STREAMING_TEST_BUCKET, CODEC_STREAMING_TEST_BUCKET,
CODEC_STREAMING_TEST_OBJECT, CODEC_STREAMING_TEST_OBJECT,
Arc::new(ErasureCache::new()),
0, 0,
part_data.len() as i64, part_data.len() as i64,
&mut output, &mut output,
@@ -4833,114 +4828,6 @@ mod tests {
.await .await
} }
async fn encoded_inline_blocks(blocks: &[&[u8]], shard_size: usize, hash_algo: HashAlgorithm) -> Bytes {
let mut writer = BitrotWriter::new(Cursor::new(Vec::new()), shard_size, hash_algo);
for block in blocks {
writer.write(block).await.expect("test block should be encoded");
}
Bytes::from(writer.into_inner().into_inner())
}
fn assert_reader_shares_inline_allocation(reader: &ObjectBitrotReader, source: &Bytes) {
let reader_bytes = reader
.inner_ref()
.inline_bytes()
.expect("inline scheduler should retain an in-memory Bytes source");
assert_eq!(
reader_bytes.as_ptr(),
source.as_ptr(),
"the scheduler must clone Bytes ownership instead of copying the inline shard payload"
);
}
#[tokio::test]
async fn inline_range_scheduler_shares_bytes_and_rejects_bitrot_mismatch() {
const SHARD_SIZE: usize = 16;
let hash_algo = HashAlgorithm::HighwayHash256S;
let first = [b'a'; SHARD_SIZE];
let second = [b'b'; SHARD_SIZE];
let mut source = encoded_inline_blocks(&[&first, &second], SHARD_SIZE, hash_algo.clone()).await;
let second_payload = hash_algo.size() * 2 + SHARD_SIZE;
source = {
let mut corrupt = source.to_vec();
corrupt[second_payload] ^= 0xff;
Bytes::from(corrupt)
};
let files = vec![encoded_reader_setup_fileinfo(Some(source.to_vec()))];
let source = files[0].data.clone().expect("inline shard should exist");
let disks = vec![None];
let mut setup = create_bitrot_readers_until_quorum_with_preference(
&files,
&disks,
"bucket",
"object",
1,
SHARD_SIZE,
SHARD_SIZE,
SHARD_SIZE,
hash_algo,
false,
false,
1,
0,
BitrotReaderSetupMode::ReadQuorum,
true,
None,
None,
)
.await;
let mut reader = setup.readers[0].take().expect("range reader should be ready");
assert_reader_shares_inline_allocation(&reader, &source);
let err = reader
.read(&mut [0; SHARD_SIZE])
.await
.expect_err("corrupt ranged inline block must fail bitrot verification");
assert_eq!(err.kind(), ErrorKind::InvalidData);
}
#[tokio::test]
async fn inline_part_scheduler_shares_bytes_and_rejects_bitrot_mismatch() {
const SHARD_SIZE: usize = 16;
let hash_algo = HashAlgorithm::HighwayHash256S;
let block = [b'p'; SHARD_SIZE];
let encoded = encoded_inline_blocks(&[&block], SHARD_SIZE, hash_algo.clone()).await;
let mut corrupt = encoded.to_vec();
corrupt[hash_algo.size()] ^= 0xff;
let files = vec![encoded_reader_setup_fileinfo(Some(corrupt))];
let source = files[0].data.clone().expect("inline shard should exist");
let disks = vec![None];
let mut setup = create_bitrot_readers_until_quorum_all_shards(
&files,
&disks,
"bucket",
"object",
7,
0,
SHARD_SIZE,
SHARD_SIZE,
hash_algo,
false,
false,
1,
0,
BitrotReaderSetupMode::VerifyReconstruction,
None,
None,
)
.await;
let mut reader = setup.readers[0].take().expect("part reader should be ready");
assert_reader_shares_inline_allocation(&reader, &source);
let err = reader
.read(&mut [0; SHARD_SIZE])
.await
.expect_err("corrupt inline part must fail bitrot verification");
assert_eq!(err.kind(), ErrorKind::InvalidData);
}
async fn decode_codec_data_blocks_first_setup( async fn decode_codec_data_blocks_first_setup(
erasure: coding::Erasure, erasure: coding::Erasure,
data: &[u8], data: &[u8],
@@ -5641,63 +5528,6 @@ mod tests {
}); });
} }
#[test]
fn codec_streaming_config_cache_loads_once() {
use std::cell::Cell;
let loads = Cell::new(0);
let expected = GetCodecStreamingConfig {
enabled: true,
rollout: GetCodecStreamingRollout::Off,
rollout_pct: 100,
body_compat_confirmed: true,
header_compat_confirmed: true,
engine: GetCodecStreamingEngine::Legacy,
min_size: DEFAULT_RUSTFS_GET_CODEC_STREAMING_MIN_SIZE,
};
for _ in 0..3 {
assert_eq!(
get_codec_streaming_config_cached_core(|| {
loads.set(loads.get() + 1);
expected
}),
expected
);
}
assert_eq!(loads.get(), 1, "production config cache must not reload env per GET");
}
#[test]
fn codec_streaming_config_loader_preserves_all_gate_env_overrides() {
temp_env::with_vars(
[
(ENV_RUSTFS_GET_CODEC_STREAMING_ENABLE, Some("false")),
(ENV_RUSTFS_GET_CODEC_STREAMING_ENGINE, Some(GET_CODEC_STREAMING_ENGINE_RUSTFS)),
(ENV_RUSTFS_GET_CODEC_STREAMING_ROLLOUT, Some("production")),
(ENV_RUSTFS_GET_CODEC_STREAMING_ROLLOUT_PCT, Some("37")),
(ENV_RUSTFS_GET_CODEC_STREAMING_BODY_COMPAT_CONFIRMED, Some("false")),
(ENV_RUSTFS_GET_CODEC_STREAMING_HEADER_COMPAT_CONFIRMED, Some("false")),
(ENV_RUSTFS_GET_CODEC_STREAMING_MIN_SIZE, None::<&str>),
(ENV_RUSTFS_GET_CODEC_STREAMING_RUSTFS_MIN_SIZE, Some("262144")),
],
|| {
assert_eq!(
load_get_codec_streaming_config(),
GetCodecStreamingConfig {
enabled: false,
rollout: GetCodecStreamingRollout::On,
rollout_pct: 37,
body_compat_confirmed: false,
header_compat_confirmed: false,
engine: GetCodecStreamingEngine::Rustfs,
min_size: 262144,
}
);
},
);
}
#[test] #[test]
fn codec_streaming_default_min_size_meets_direct_memory_ceiling() { fn codec_streaming_default_min_size_meets_direct_memory_ceiling() {
for engine in [None, Some(GET_CODEC_STREAMING_ENGINE_RUSTFS)] { for engine in [None, Some(GET_CODEC_STREAMING_ENGINE_RUSTFS)] {
+6 -27
View File
@@ -48,23 +48,6 @@ impl RestoreCleanupIdentity {
} }
} }
fn ensure_restore_metadata_lock_held(bucket: &str, object: &str, opts: &ObjectOptions, mode: &'static str) -> Result<()> {
if opts
.namespace_lock_fence
.as_ref()
.is_some_and(NamespaceLockFence::is_lock_lost)
{
return Err(StorageError::NamespaceLockQuorumUnavailable {
mode,
bucket: bucket.to_string(),
object: object.to_string(),
required: 1,
achieved: 0,
});
}
Ok(())
}
impl SetDisks { impl SetDisks {
pub(super) async fn finalize_restore_metadata( pub(super) async fn finalize_restore_metadata(
&self, &self,
@@ -92,20 +75,18 @@ impl SetDisks {
version_id, version_id,
versioned: opts.versioned, versioned: opts.versioned,
version_suspended: opts.version_suspended, version_suspended: opts.version_suspended,
include_part_checksums: true,
..Default::default() ..Default::default()
}; };
let (mut fi, _, disks) = self let (fi, _, disks) = self
.get_object_fileinfo_gated(bucket, object, &read_opts, false, false) .get_object_fileinfo_gated(bucket, object, &read_opts, false, false)
.await? .await?;
.into_owned(); let mut fi = fi.into_owned();
if let Some(expected_operation_id) = expected_operation_id { if let Some(expected_operation_id) = expected_operation_id {
require_restore_operation_id(&fi.metadata, expected_operation_id)?; require_restore_operation_id(&fi.metadata, expected_operation_id)?;
} }
if !expected.matches_file_info(&fi, &expected_etag) { if !expected.matches_file_info(&fi, &expected_etag) {
return Err(Error::other("restored object changed before restore metadata finalization")); return Err(Error::other("restored object changed before restore metadata finalization"));
} }
ensure_restore_metadata_lock_held(bucket, object, opts, "restore_finalize_metadata")?;
let restore_expiry = let restore_expiry =
lifecycle::expected_expiry_time(OffsetDateTime::now_utc(), opts.transition.restore_request.days.unwrap_or(1)); lifecycle::expected_expiry_time(OffsetDateTime::now_utc(), opts.transition.restore_request.days.unwrap_or(1));
fi.metadata.insert( fi.metadata.insert(
@@ -161,13 +142,12 @@ impl SetDisks {
version_id, version_id,
versioned: opts.versioned, versioned: opts.versioned,
version_suspended: opts.version_suspended, version_suspended: opts.version_suspended,
include_part_checksums: true,
..Default::default() ..Default::default()
}; };
let (mut fi, _, disks) = self let (fi, _, disks) = self
.get_object_fileinfo_gated(bucket, object, &read_opts, false, false) .get_object_fileinfo_gated(bucket, object, &read_opts, false, false)
.await? .await?;
.into_owned(); let mut fi = fi.into_owned();
if let Some(expected_operation_id) = expected_operation_id { if let Some(expected_operation_id) = expected_operation_id {
match restore_operation_id_from_metadata(&fi.metadata)? { match restore_operation_id_from_metadata(&fi.metadata)? {
Some(actual_operation_id) if actual_operation_id == expected_operation_id => {} Some(actual_operation_id) if actual_operation_id == expected_operation_id => {}
@@ -177,7 +157,6 @@ impl SetDisks {
if !expected.matches_file_info(&fi, &expected_etag) { if !expected.matches_file_info(&fi, &expected_etag) {
return Ok(()); return Ok(());
} }
ensure_restore_metadata_lock_held(bucket, object, opts, "restore_cleanup_metadata")?;
fi.metadata.remove(X_AMZ_RESTORE.as_str()); fi.metadata.remove(X_AMZ_RESTORE.as_str());
fi.metadata.remove(AMZ_RESTORE_EXPIRY_DAYS); fi.metadata.remove(AMZ_RESTORE_EXPIRY_DAYS);
fi.metadata.remove(AMZ_RESTORE_REQUEST_DATE); fi.metadata.remove(AMZ_RESTORE_REQUEST_DATE);
+159 -94
View File
@@ -16,14 +16,6 @@ use crate::diagnostics::get::{
GET_SHARD_READ_COST_LOCAL, GET_SHARD_READ_COST_REMOTE, GET_SHARD_READ_COST_SAME_NODE, GET_SHARD_READ_COST_UNKNOWN, GET_SHARD_READ_COST_LOCAL, GET_SHARD_READ_COST_REMOTE, GET_SHARD_READ_COST_SAME_NODE, GET_SHARD_READ_COST_UNKNOWN,
}; };
use crate::disk::error::Error; use crate::disk::error::Error;
use crate::layout::disks_layout::MAX_ERASURE_SET_DRIVE_COUNT;
use smallvec::SmallVec;
/// Generic codec callers may exceed the production set limit; `SmallVec` then
/// spills without changing slot semantics.
pub(crate) const INLINE_SHARD_SLOTS: usize = MAX_ERASURE_SET_DRIVE_COUNT;
pub(crate) type ShardBuffers = SmallVec<[Option<Vec<u8>>; INLINE_SHARD_SLOTS]>;
pub(crate) type ShardErrors = SmallVec<[Option<Error>; INLINE_SHARD_SLOTS]>;
#[derive(Debug, Clone, Copy, PartialEq, Eq)] #[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub(crate) enum ShardReadCost { pub(crate) enum ShardReadCost {
@@ -51,137 +43,202 @@ impl ShardReadCost {
} }
} }
#[derive(Debug, Clone, PartialEq, Eq)]
pub(crate) struct ShardSlot {
index: usize,
read_cost: ShardReadCost,
data: Option<Vec<u8>>,
error: Option<Error>,
}
impl ShardSlot {
pub(crate) fn new(index: usize, data: Option<Vec<u8>>, error: Option<Error>) -> Self {
Self::with_read_cost(index, ShardReadCost::Unknown, data, error)
}
pub(crate) fn with_read_cost(index: usize, read_cost: ShardReadCost, data: Option<Vec<u8>>, error: Option<Error>) -> Self {
Self {
index,
read_cost,
data,
error,
}
}
pub(crate) fn data(index: usize, data: Vec<u8>) -> Self {
Self::new(index, Some(data), None)
}
pub(crate) fn data_with_read_cost(index: usize, read_cost: ShardReadCost, data: Vec<u8>) -> Self {
Self::with_read_cost(index, read_cost, Some(data), None)
}
pub(crate) fn missing(index: usize, error: Error) -> Self {
Self::new(index, None, Some(error))
}
pub(crate) fn missing_with_read_cost(index: usize, read_cost: ShardReadCost, error: Error) -> Self {
Self::with_read_cost(index, read_cost, None, Some(error))
}
pub(crate) fn index(&self) -> usize {
self.index
}
pub(crate) fn read_cost(&self) -> ShardReadCost {
self.read_cost
}
pub(crate) fn has_data(&self) -> bool {
self.data.is_some()
}
pub(crate) fn data_bytes(&self) -> Option<&[u8]> {
self.data.as_deref()
}
pub(crate) fn error(&self) -> Option<&Error> {
self.error.as_ref()
}
}
#[derive(Debug, Clone, PartialEq, Eq)] #[derive(Debug, Clone, PartialEq, Eq)]
pub(crate) struct StripeReadState { pub(crate) struct StripeReadState {
shards: ShardBuffers, slots: Vec<ShardSlot>,
errors: ShardErrors,
read_quorum: usize, read_quorum: usize,
} }
impl StripeReadState { impl StripeReadState {
#[cfg(test)] pub(crate) fn new(slots: Vec<ShardSlot>, read_quorum: usize) -> Self {
Self { slots, read_quorum }
}
pub(crate) fn from_parts(shards: Vec<Option<Vec<u8>>>, errors: Vec<Option<Error>>, read_quorum: usize) -> Self { pub(crate) fn from_parts(shards: Vec<Option<Vec<u8>>>, errors: Vec<Option<Error>>, read_quorum: usize) -> Self {
let mut shards = SmallVec::from_vec(shards); Self::from_parts_with_read_costs(shards, errors, &[], read_quorum)
let mut errors = SmallVec::from_vec(errors); }
pub(crate) fn from_parts_with_read_costs<S, E>(shards: S, errors: E, read_costs: &[ShardReadCost], read_quorum: usize) -> Self
where
S: IntoIterator<Item = Option<Vec<u8>>>,
S::IntoIter: ExactSizeIterator,
E: IntoIterator<Item = Option<Error>>,
E::IntoIter: ExactSizeIterator,
{
let mut shards = shards.into_iter();
let mut errors = errors.into_iter();
let slot_count = shards.len().max(errors.len()); let slot_count = shards.len().max(errors.len());
shards.resize_with(slot_count, || None); let mut slots = Vec::with_capacity(slot_count);
errors.resize_with(slot_count, || None); for index in 0..slot_count {
Self { let read_cost = read_costs.get(index).copied().unwrap_or(ShardReadCost::Unknown);
shards, slots.push(ShardSlot::with_read_cost(
errors, index,
read_quorum, read_cost,
shards.next().flatten(),
errors.next().flatten(),
));
} }
} Self::new(slots, read_quorum)
pub(crate) fn with_slot_count(slot_count: usize, read_quorum: usize) -> Self {
let mut state = Self {
shards: SmallVec::new(),
errors: SmallVec::new(),
read_quorum,
};
state.reset(slot_count, read_quorum);
state
}
pub(crate) fn reset(&mut self, slot_count: usize, read_quorum: usize) {
self.shards.clear();
self.shards.resize_with(slot_count, || None);
self.errors.clear();
self.errors.resize_with(slot_count, || None);
self.read_quorum = read_quorum;
} }
pub(crate) fn available_shards(&self) -> usize { pub(crate) fn available_shards(&self) -> usize {
self.shards.iter().filter(|shard| shard.is_some()).count() self.slots.iter().filter(|slot| slot.has_data()).count()
} }
pub(crate) fn can_decode(&self) -> bool { pub(crate) fn can_decode(&self) -> bool {
self.available_shards() >= self.read_quorum self.available_shards() >= self.read_quorum
} }
pub(crate) fn is_empty(&self) -> bool { pub(crate) fn slots(&self) -> &[ShardSlot] {
self.shards.is_empty() &self.slots
} }
pub(crate) fn data_bytes(&self, index: usize) -> Option<&[u8]> { pub(crate) fn slot_by_index(&self, index: usize) -> Option<&ShardSlot> {
self.shards.get(index).and_then(Option::as_deref) if let Some(slot) = self.slots.get(index)
} && slot.index == index
{
#[cfg(test)] return Some(slot);
pub(crate) fn error(&self, index: usize) -> Option<&Error> { }
self.errors.get(index).and_then(Option::as_ref) self.slots.iter().find(|slot| slot.index == index)
} }
pub(crate) fn data_shards_complete(&self, data_shards: usize) -> bool { pub(crate) fn data_shards_complete(&self, data_shards: usize) -> bool {
self.shards.len() >= data_shards && self.shards.iter().take(data_shards).all(Option::is_some) (0..data_shards).all(|index| self.slot_by_index(index).is_some_and(ShardSlot::has_data))
} }
pub(crate) fn parts_mut(&mut self) -> (&mut ShardBuffers, &mut ShardErrors) { pub(crate) fn into_parts(self) -> (Vec<Option<Vec<u8>>>, Vec<Option<Error>>) {
(&mut self.shards, &mut self.errors) let part_count = self.slots.iter().map(|slot| slot.index).max().map_or(0, |index| index + 1);
} let mut shards = Vec::with_capacity(part_count);
shards.resize_with(part_count, || None);
pub(crate) fn shards_mut(&mut self) -> &mut ShardBuffers { let mut errors = Vec::with_capacity(part_count);
&mut self.shards errors.resize_with(part_count, || None);
} for slot in self.slots {
shards[slot.index] = slot.data;
pub(crate) fn into_parts(self) -> (ShardBuffers, ShardErrors) { errors[slot.index] = slot.error;
(self.shards, self.errors) }
} (shards, errors)
#[cfg(test)]
pub(crate) fn scratch_storage(&self) -> (*const Option<Vec<u8>>, *const Option<Error>, bool, bool) {
(self.shards.as_ptr(), self.errors.as_ptr(), self.shards.spilled(), self.errors.spilled())
}
#[cfg(test)]
pub(crate) fn shard_allocation(&self, index: usize) -> Option<(*const u8, usize)> {
self.shards
.get(index)
.and_then(|shard| shard.as_ref().map(|shard| (shard.as_ptr(), shard.capacity())))
} }
} }
#[async_trait::async_trait] #[async_trait::async_trait]
pub(crate) trait ShardStripeSource: Send { pub(crate) trait ShardStripeSource: Send {
async fn read_next_stripe(&mut self) -> Box<StripeReadState>; async fn read_next_stripe(&mut self) -> StripeReadState;
fn recycle_stripe(&mut self, _state: Box<StripeReadState>) {}
} }
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use std::mem::size_of;
#[test] #[test]
fn stripe_scratch_capacity_matches_the_production_set_limit() { fn stripe_read_state_tracks_decode_quorum() {
type OversizedShardBuffers = SmallVec<[Option<Vec<u8>>; 32]>; let state = StripeReadState::new(
type OversizedShardErrors = SmallVec<[Option<Error>; 32]>; vec![
ShardSlot::data_with_read_cost(0, ShardReadCost::Local, vec![1]),
assert_eq!(INLINE_SHARD_SLOTS, MAX_ERASURE_SET_DRIVE_COUNT); ShardSlot::missing_with_read_cost(1, ShardReadCost::Remote, Error::FileNotFound),
assert!(size_of::<ShardBuffers>() < size_of::<OversizedShardBuffers>()); ShardSlot::data_with_read_cost(2, ShardReadCost::SameNode, vec![2]),
assert!(size_of::<ShardErrors>() < size_of::<OversizedShardErrors>()); ],
} 2,
);
#[test]
fn stripe_read_state_tracks_decode_quorum_and_slot_access() {
let state =
StripeReadState::from_parts(vec![Some(vec![1]), None, Some(vec![2])], vec![None, Some(Error::FileNotFound), None], 2);
assert_eq!(state.available_shards(), 2); assert_eq!(state.available_shards(), 2);
assert!(state.can_decode()); assert!(state.can_decode());
assert_eq!(state.data_bytes(0), Some(&[1][..])); assert_eq!(state.slots()[1].index(), 1);
assert_eq!(state.error(1), Some(&Error::FileNotFound)); assert_eq!(state.slots()[0].read_cost(), ShardReadCost::Local);
assert!(state.slots()[2].read_cost().is_low_cost());
} }
#[test] #[test]
fn stripe_read_state_preserves_shards_and_errors() { fn stripe_read_state_preserves_shards_and_errors() {
let state = StripeReadState::from_parts(vec![Some(vec![1, 2, 3]), None], vec![None, Some(Error::FileCorrupt)], 2); let state = StripeReadState::new(vec![ShardSlot::missing(1, Error::FileCorrupt), ShardSlot::data(0, vec![1, 2, 3])], 2);
assert!(!state.can_decode()); assert!(!state.can_decode());
let (shards, errors) = state.into_parts(); let (shards, errors) = state.into_parts();
assert_eq!(shards.as_slice(), &[Some(vec![1, 2, 3]), None]); assert_eq!(shards, vec![Some(vec![1, 2, 3]), None]);
assert_eq!(errors.as_slice(), &[None, Some(Error::FileCorrupt)]); assert_eq!(errors, vec![None, Some(Error::FileCorrupt)]);
}
#[test]
fn stripe_read_state_builds_slots_from_parallel_reader_parts() {
let state =
StripeReadState::from_parts(vec![Some(vec![1]), None, Some(vec![3])], vec![None, Some(Error::FileNotFound)], 2);
assert!(state.can_decode());
assert_eq!(state.slots()[1].index(), 1);
assert_eq!(state.slots()[1].error(), Some(&Error::FileNotFound));
}
#[test]
fn stripe_read_state_preserves_read_cost_hints() {
let state = StripeReadState::from_parts_with_read_costs(
vec![Some(vec![1]), None, Some(vec![3])],
vec![None, Some(Error::FileNotFound)],
&[ShardReadCost::Local, ShardReadCost::Remote, ShardReadCost::Unknown],
2,
);
assert_eq!(state.slots()[0].read_cost(), ShardReadCost::Local);
assert_eq!(state.slots()[1].read_cost(), ShardReadCost::Remote);
assert_eq!(state.slots()[2].read_cost(), ShardReadCost::Unknown);
assert_eq!(ShardReadCost::SameNode.as_str(), GET_SHARD_READ_COST_SAME_NODE);
} }
#[test] #[test]
@@ -201,8 +258,8 @@ mod tests {
let state = StripeReadState::from_parts(vec![Some(vec![1]), Some(vec![2]), None], Vec::new(), 2); let state = StripeReadState::from_parts(vec![Some(vec![1]), Some(vec![2]), None], Vec::new(), 2);
assert!(state.data_shards_complete(2)); assert!(state.data_shards_complete(2));
assert_eq!(state.data_bytes(0), Some(&[1][..])); assert_eq!(state.slots()[0].data_bytes(), Some(&[1][..]));
assert_eq!(state.data_bytes(1), Some(&[2][..])); assert_eq!(state.slot_by_index(1).and_then(ShardSlot::data_bytes), Some(&[2][..]));
} }
#[test] #[test]
@@ -211,4 +268,12 @@ mod tests {
assert!(!state.data_shards_complete(2)); assert!(!state.data_shards_complete(2));
} }
#[test]
fn stripe_read_state_finds_out_of_order_slots_by_index() {
let state = StripeReadState::new(vec![ShardSlot::data(2, vec![3]), ShardSlot::data(0, vec![1])], 2);
assert_eq!(state.slot_by_index(0).and_then(ShardSlot::data_bytes), Some(&[1][..]));
assert!(state.slot_by_index(1).is_none());
}
} }
-4
View File
@@ -161,10 +161,6 @@ impl BucketIncarnationFenceGuard {
pub(crate) fn is_lock_lost(&self) -> bool { pub(crate) fn is_lock_lost(&self) -> bool {
self.inner.as_ref().is_some_and(NamespaceLockGuard::is_lock_lost) self.inner.as_ref().is_some_and(NamespaceLockGuard::is_lock_lost)
} }
pub(crate) fn namespace_lock_guard(&self) -> Option<&NamespaceLockGuard> {
self.inner.as_ref()
}
} }
impl Drop for BucketIncarnationFenceGuard { impl Drop for BucketIncarnationFenceGuard {
File diff suppressed because it is too large Load Diff
-28
View File
@@ -309,17 +309,9 @@ const ENV_API_LIST_OBJECTS_INDEX_PROVIDER: &str = "RUSTFS_LIST_OBJECTS_INDEX_PRO
const ENV_API_LIST_OBJECTS_INDEX_PROVIDER_PATH: &str = "RUSTFS_LIST_OBJECTS_INDEX_PROVIDER_PATH"; const ENV_API_LIST_OBJECTS_INDEX_PROVIDER_PATH: &str = "RUSTFS_LIST_OBJECTS_INDEX_PROVIDER_PATH";
const ENV_API_LIST_OBJECTS_INDEX_PROVIDER_GENERATION: &str = "RUSTFS_LIST_OBJECTS_INDEX_PROVIDER_GENERATION"; const ENV_API_LIST_OBJECTS_INDEX_PROVIDER_GENERATION: &str = "RUSTFS_LIST_OBJECTS_INDEX_PROVIDER_GENERATION";
const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_PATH: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_PATH"; const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_PATH: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_PATH";
// The chaos machinery below is compiled only for tests and the opt-in
// `list-chaos` feature (backlog#1832): a production binary without the
// feature carries no chaos symbols, so the two env vars cannot silently
// rewrite a bucket's namespace-journal state.
#[cfg(any(test, feature = "list-chaos"))]
const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_ENABLED: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_ENABLED"; const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_ENABLED: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_ENABLED";
#[cfg(any(test, feature = "list-chaos"))]
const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_BUCKET: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_BUCKET"; const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_BUCKET: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_BUCKET";
#[cfg(any(test, feature = "list-chaos"))]
const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_SEQUENCE: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_SEQUENCE"; const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_SEQUENCE: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_SEQUENCE";
#[cfg(any(test, feature = "list-chaos"))]
const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_STATUS: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_STATUS"; const ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_STATUS: &str = "RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_STATUS";
const ENV_API_LIST_OBJECTS_METADATA_FAST_ENABLED: &str = "RUSTFS_LIST_OBJECTS_METADATA_FAST_ENABLED"; const ENV_API_LIST_OBJECTS_METADATA_FAST_ENABLED: &str = "RUSTFS_LIST_OBJECTS_METADATA_FAST_ENABLED";
const ENV_API_LIST_OBJECTS_METADATA_FAST_STALENESS_MS: &str = "RUSTFS_LIST_OBJECTS_METADATA_FAST_STALENESS_MS"; const ENV_API_LIST_OBJECTS_METADATA_FAST_STALENESS_MS: &str = "RUSTFS_LIST_OBJECTS_METADATA_FAST_STALENESS_MS";
@@ -560,9 +552,7 @@ static LIST_OBJECTS_MUTATION_SEQUENCE: AtomicU64 = AtomicU64::new(0);
static SCANNER_NAMESPACE_MUTATION_GENERATION: AtomicU64 = AtomicU64::new(0); static SCANNER_NAMESPACE_MUTATION_GENERATION: AtomicU64 = AtomicU64::new(0);
static LIST_OBJECTS_BUCKET_MUTATION_SEQUENCE: OnceCell<RwLock<HashMap<String, u64>>> = OnceCell::const_new(); static LIST_OBJECTS_BUCKET_MUTATION_SEQUENCE: OnceCell<RwLock<HashMap<String, u64>>> = OnceCell::const_new();
static LIST_OBJECTS_NAMESPACE_JOURNAL_DEGRADED_BUCKETS: OnceCell<RwLock<HashSet<String>>> = OnceCell::const_new(); static LIST_OBJECTS_NAMESPACE_JOURNAL_DEGRADED_BUCKETS: OnceCell<RwLock<HashSet<String>>> = OnceCell::const_new();
#[cfg(any(test, feature = "list-chaos"))]
static LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_CONFIG: OnceCell<Option<NamespaceMutationJournalChaosConfig>> = OnceCell::const_new(); static LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_CONFIG: OnceCell<Option<NamespaceMutationJournalChaosConfig>> = OnceCell::const_new();
#[cfg(any(test, feature = "list-chaos"))]
static LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_APPLIED: OnceCell<RwLock<HashSet<String>>> = OnceCell::const_new(); static LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_APPLIED: OnceCell<RwLock<HashSet<String>>> = OnceCell::const_new();
async fn persistent_key_only_index_cache() -> &'static RwLock<Option<PersistentKeyOnlyIndexCache>> { async fn persistent_key_only_index_cache() -> &'static RwLock<Option<PersistentKeyOnlyIndexCache>> {
@@ -589,7 +579,6 @@ async fn list_objects_namespace_journal_degraded_buckets() -> &'static RwLock<Ha
.await .await
} }
#[cfg(any(test, feature = "list-chaos"))]
async fn list_objects_namespace_journal_chaos_config() -> Option<&'static NamespaceMutationJournalChaosConfig> { async fn list_objects_namespace_journal_chaos_config() -> Option<&'static NamespaceMutationJournalChaosConfig> {
LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_CONFIG LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_CONFIG
.get_or_init(|| async { namespace_mutation_journal_chaos_config_from_env() }) .get_or_init(|| async { namespace_mutation_journal_chaos_config_from_env() })
@@ -597,7 +586,6 @@ async fn list_objects_namespace_journal_chaos_config() -> Option<&'static Namesp
.as_ref() .as_ref()
} }
#[cfg(any(test, feature = "list-chaos"))]
async fn list_objects_namespace_journal_chaos_applied() -> &'static RwLock<HashSet<String>> { async fn list_objects_namespace_journal_chaos_applied() -> &'static RwLock<HashSet<String>> {
LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_APPLIED LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_APPLIED
.get_or_init(|| async { RwLock::new(HashSet::new()) }) .get_or_init(|| async { RwLock::new(HashSet::new()) })
@@ -693,7 +681,6 @@ enum NamespaceMutationJournalStatus {
} }
impl NamespaceMutationJournalStatus { impl NamespaceMutationJournalStatus {
#[cfg(any(test, feature = "list-chaos"))]
fn from_env_value(value: &str) -> Option<Self> { fn from_env_value(value: &str) -> Option<Self> {
if value.eq_ignore_ascii_case(LIST_OBJECTS_NAMESPACE_JOURNAL_STATUS_HEALTHY) { if value.eq_ignore_ascii_case(LIST_OBJECTS_NAMESPACE_JOURNAL_STATUS_HEALTHY) {
Some(Self::Healthy) Some(Self::Healthy)
@@ -704,7 +691,6 @@ impl NamespaceMutationJournalStatus {
} }
} }
#[cfg(any(test, feature = "list-chaos"))]
fn env_value(self) -> &'static str { fn env_value(self) -> &'static str {
match self { match self {
Self::Healthy => LIST_OBJECTS_NAMESPACE_JOURNAL_STATUS_HEALTHY, Self::Healthy => LIST_OBJECTS_NAMESPACE_JOURNAL_STATUS_HEALTHY,
@@ -726,7 +712,6 @@ struct NamespaceMutationJournalSnapshot {
degraded: bool, degraded: bool,
} }
#[cfg(any(test, feature = "list-chaos"))]
#[derive(Debug, Clone, PartialEq, Eq)] #[derive(Debug, Clone, PartialEq, Eq)]
struct NamespaceMutationJournalChaosConfig { struct NamespaceMutationJournalChaosConfig {
bucket: String, bucket: String,
@@ -810,35 +795,30 @@ fn list_objects_namespace_journal_root_from_env() -> Option<PathBuf> {
.filter(|path| !path.as_os_str().is_empty()) .filter(|path| !path.as_os_str().is_empty())
} }
#[cfg(any(test, feature = "list-chaos"))]
fn namespace_mutation_journal_chaos_enabled_from_env() -> bool { fn namespace_mutation_journal_chaos_enabled_from_env() -> bool {
std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_ENABLED) std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_ENABLED)
.ok() .ok()
.is_some_and(|value| value == "1" || value.eq_ignore_ascii_case("on") || value.eq_ignore_ascii_case("true")) .is_some_and(|value| value == "1" || value.eq_ignore_ascii_case("on") || value.eq_ignore_ascii_case("true"))
} }
#[cfg(any(test, feature = "list-chaos"))]
fn namespace_mutation_journal_chaos_bucket_from_env() -> Option<String> { fn namespace_mutation_journal_chaos_bucket_from_env() -> Option<String> {
std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_BUCKET) std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_BUCKET)
.ok() .ok()
.filter(|bucket| !bucket.is_empty()) .filter(|bucket| !bucket.is_empty())
} }
#[cfg(any(test, feature = "list-chaos"))]
fn namespace_mutation_journal_chaos_sequence_from_env() -> Option<u64> { fn namespace_mutation_journal_chaos_sequence_from_env() -> Option<u64> {
std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_SEQUENCE) std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_SEQUENCE)
.ok() .ok()
.and_then(|value| value.parse::<u64>().ok()) .and_then(|value| value.parse::<u64>().ok())
} }
#[cfg(any(test, feature = "list-chaos"))]
fn namespace_mutation_journal_chaos_status_from_env() -> Option<NamespaceMutationJournalStatus> { fn namespace_mutation_journal_chaos_status_from_env() -> Option<NamespaceMutationJournalStatus> {
std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_STATUS) std::env::var(ENV_API_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_STATUS)
.ok() .ok()
.and_then(|value| NamespaceMutationJournalStatus::from_env_value(&value)) .and_then(|value| NamespaceMutationJournalStatus::from_env_value(&value))
} }
#[cfg(any(test, feature = "list-chaos"))]
fn namespace_mutation_journal_chaos_config_from_env() -> Option<NamespaceMutationJournalChaosConfig> { fn namespace_mutation_journal_chaos_config_from_env() -> Option<NamespaceMutationJournalChaosConfig> {
if !namespace_mutation_journal_chaos_enabled_from_env() { if !namespace_mutation_journal_chaos_enabled_from_env() {
return None; return None;
@@ -866,7 +846,6 @@ fn namespace_mutation_journal_chaos_config_from_env() -> Option<NamespaceMutatio
}) })
} }
#[cfg(any(test, feature = "list-chaos"))]
fn namespace_mutation_journal_chaos_applied_key(bucket: &str, status: NamespaceMutationJournalStatus) -> String { fn namespace_mutation_journal_chaos_applied_key(bucket: &str, status: NamespaceMutationJournalStatus) -> String {
let mut key = String::with_capacity(bucket.len() + 1 + status.env_value().len()); let mut key = String::with_capacity(bucket.len() + 1 + status.env_value().len());
key.push_str(bucket); key.push_str(bucket);
@@ -875,13 +854,6 @@ fn namespace_mutation_journal_chaos_applied_key(bucket: &str, status: NamespaceM
key key
} }
/// Production no-op twin of the chaos injector: without `list-chaos` the
/// injection point compiles to nothing (backlog#1832).
#[cfg(not(any(test, feature = "list-chaos")))]
#[inline]
async fn maybe_apply_system_namespace_mutation_journal_chaos(_store: &ECStore, _bucket: &str, _default_sequence: u64) {}
#[cfg(any(test, feature = "list-chaos"))]
async fn maybe_apply_system_namespace_mutation_journal_chaos(store: &ECStore, bucket: &str, default_sequence: u64) { async fn maybe_apply_system_namespace_mutation_journal_chaos(store: &ECStore, bucket: &str, default_sequence: u64) {
let Some(config) = list_objects_namespace_journal_chaos_config().await else { let Some(config) = list_objects_namespace_journal_chaos_config().await else {
return; return;
+1 -1
View File
@@ -389,7 +389,7 @@ impl crate::storage_api_contracts::object::ObjectIO for ECStore {
type GetObjectReader = GetObjectReader; type GetObjectReader = GetObjectReader;
type PutObjectReader = PutObjReader; type PutObjectReader = PutObjReader;
#[instrument(level = "debug", skip(self, h))] #[instrument(level = "debug", skip(self))]
async fn get_object_reader( async fn get_object_reader(
&self, &self,
bucket: &str, bucket: &str,
+5 -188
View File
@@ -66,76 +66,6 @@ fn ensure_multipart_bucket_lifecycle_guard_held(
Ok(()) Ok(())
} }
#[cfg(test)]
struct DataMovementMultipartCompletionBarrierState {
bucket: String,
arrived: tokio::sync::Notify,
release: tokio::sync::Notify,
}
#[cfg(test)]
pub(crate) struct DataMovementMultipartCompletionBarrier {
state: Arc<DataMovementMultipartCompletionBarrierState>,
}
#[cfg(test)]
static DATA_MOVEMENT_MULTIPART_COMPLETION_BARRIER: std::sync::OnceLock<
std::sync::Mutex<Option<Arc<DataMovementMultipartCompletionBarrierState>>>,
> = std::sync::OnceLock::new();
#[cfg(test)]
impl DataMovementMultipartCompletionBarrier {
pub(crate) fn install(bucket: &str) -> Self {
let state = Arc::new(DataMovementMultipartCompletionBarrierState {
bucket: bucket.to_string(),
arrived: tokio::sync::Notify::new(),
release: tokio::sync::Notify::new(),
});
let mut slot = DATA_MOVEMENT_MULTIPART_COMPLETION_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("data movement multipart completion barrier mutex should not poison");
assert!(slot.is_none(), "data movement multipart completion barrier must be unique");
*slot = Some(Arc::clone(&state));
Self { state }
}
pub(crate) async fn wait_until_paused(&self) {
tokio::time::timeout(std::time::Duration::from_secs(30), self.state.arrived.notified())
.await
.expect("data movement multipart operation should reach selected completion");
}
}
#[cfg(test)]
impl Drop for DataMovementMultipartCompletionBarrier {
fn drop(&mut self) {
self.state.release.notify_one();
let mut slot = DATA_MOVEMENT_MULTIPART_COMPLETION_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("data movement multipart completion barrier mutex should not poison");
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
*slot = None;
}
}
}
#[cfg(test)]
async fn pause_data_movement_multipart_before_selected_completion(bucket: &str) {
let barrier = DATA_MOVEMENT_MULTIPART_COMPLETION_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("data movement multipart completion barrier mutex should not poison")
.as_ref()
.filter(|barrier| barrier.bucket == bucket)
.cloned();
if let Some(barrier) = barrier {
barrier.arrived.notify_one();
barrier.release.notified().await;
}
}
async fn list_pool_multipart_uploads_for_incarnation( async fn list_pool_multipart_uploads_for_incarnation(
pool: &crate::core::sets::Sets, pool: &crate::core::sets::Sets,
bucket: &str, bucket: &str,
@@ -402,7 +332,7 @@ impl ECStore {
) -> Result<MultipartUploadResult> { ) -> Result<MultipartUploadResult> {
self.handle_new_multipart_upload_with_pool_idx(bucket, object, opts) self.handle_new_multipart_upload_with_pool_idx(bucket, object, opts)
.await .await
.map(|(res, _, _)| res) .map(|(res, _)| res)
} }
pub(crate) async fn handle_new_multipart_upload_with_pool_idx( pub(crate) async fn handle_new_multipart_upload_with_pool_idx(
@@ -410,7 +340,7 @@ impl ECStore {
bucket: &str, bucket: &str,
object: &str, object: &str,
opts: &ObjectOptions, opts: &ObjectOptions,
) -> Result<(MultipartUploadResult, usize, Option<Uuid>)> { ) -> Result<(MultipartUploadResult, usize)> {
check_new_multipart_args(bucket, object)?; check_new_multipart_args(bucket, object)?;
let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?; let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
let opts = &opts; let opts = &opts;
@@ -419,20 +349,7 @@ impl ECStore {
return self.pools[0] return self.pools[0]
.new_multipart_upload(bucket, object, opts) .new_multipart_upload(bucket, object, opts)
.await .await
.map(|res| (res, 0, opts.expected_bucket_incarnation_id)); .map(|res| (res, 0));
}
if opts.data_movement && opts.version_id.is_some() {
let idx = self.select_data_movement_pool_idx(bucket, object, -1, opts, false).await?;
if idx == opts.src_pool_idx {
return Err(StorageError::DataMovementOverwriteErr(
bucket.to_owned(),
object.to_owned(),
opts.version_id.clone().unwrap_or_default(),
));
}
let res = self.pools[idx].new_multipart_upload(bucket, object, opts).await?;
return Ok((res, idx, opts.expected_bucket_incarnation_id));
} }
for (idx, pool) in self.pools.iter().enumerate() { for (idx, pool) in self.pools.iter().enumerate() {
@@ -455,7 +372,7 @@ impl ECStore {
if !res.uploads.is_empty() { if !res.uploads.is_empty() {
let res = self.pools[idx].new_multipart_upload(bucket, object, opts).await?; let res = self.pools[idx].new_multipart_upload(bucket, object, opts).await?;
return Ok((res, idx, opts.expected_bucket_incarnation_id)); return Ok((res, idx));
} }
} }
let idx = self.get_pool_idx(bucket, object, -1).await?; let idx = self.get_pool_idx(bucket, object, -1).await?;
@@ -468,7 +385,7 @@ impl ECStore {
} }
let res = self.pools[idx].new_multipart_upload(bucket, object, opts).await?; let res = self.pools[idx].new_multipart_upload(bucket, object, opts).await?;
Ok((res, idx, opts.expected_bucket_incarnation_id)) Ok((res, idx))
} }
#[instrument(skip(self))] #[instrument(skip(self))]
@@ -539,30 +456,6 @@ impl ECStore {
Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned())) Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned()))
} }
pub(crate) async fn put_object_part_for_data_movement(
&self,
target_pool_idx: usize,
bucket: &str,
object: &str,
upload_id: &str,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<PartInfo> {
let part_id = opts
.part_number
.ok_or_else(|| Error::other("targeted multipart upload requires a part number"))?;
check_put_object_part_args(bucket, object, upload_id)?;
if !opts.data_movement {
return Err(Error::other("targeted multipart upload requires data_movement options"));
}
let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
let pool = self
.pools
.get(target_pool_idx)
.ok_or_else(|| Error::other(format!("data movement target pool {target_pool_idx} is out of range")))?;
pool.put_object_part(bucket, object, upload_id, part_id, data, &opts).await
}
#[instrument(skip(self))] #[instrument(skip(self))]
pub(super) async fn handle_get_multipart_info( pub(super) async fn handle_get_multipart_info(
&self, &self,
@@ -637,26 +530,6 @@ impl ECStore {
Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned())) Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned()))
} }
pub(crate) async fn abort_multipart_upload_for_data_movement(
&self,
target_pool_idx: usize,
bucket: &str,
object: &str,
upload_id: &str,
opts: &ObjectOptions,
) -> Result<()> {
check_abort_multipart_args(bucket, object, upload_id)?;
if !opts.data_movement {
return Err(Error::other("targeted multipart abort requires data_movement options"));
}
let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
let pool = self
.pools
.get(target_pool_idx)
.ok_or_else(|| Error::other(format!("data movement target pool {target_pool_idx} is out of range")))?;
pool.abort_multipart_upload(bucket, object, upload_id, &opts).await
}
#[instrument(skip(self))] #[instrument(skip(self))]
pub(super) async fn handle_complete_multipart_upload( pub(super) async fn handle_complete_multipart_upload(
self: Arc<Self>, self: Arc<Self>,
@@ -701,62 +574,6 @@ impl ECStore {
Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned())) Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned()))
} }
pub(crate) async fn complete_multipart_upload_for_data_movement(
self: Arc<Self>,
target_pool_idx: usize,
bucket: &str,
object: &str,
upload_id: &str,
uploaded_parts: Vec<CompletePart>,
opts: &ObjectOptions,
) -> Result<ObjectInfo> {
check_complete_multipart_args(bucket, object, upload_id)?;
if !opts.data_movement {
return Err(Error::other("targeted multipart completion requires data_movement options"));
}
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
if opts.overwrites_existing_version() && !is_meta_bucketname(bucket) {
let expected_incarnation_id = opts
.expected_bucket_incarnation_id
.ok_or_else(|| Error::other("data movement completion is missing its bucket incarnation"))?;
let lifecycle_fence = opts
.bucket_lifecycle_lock_fence
.as_ref()
.ok_or_else(|| Error::other("data movement completion is missing its bucket lifecycle fence"))?;
let snapshot = match opts.object_lock_config_snapshot.as_ref() {
Some(snapshot) => Arc::clone(snapshot),
None => {
self.object_lock_config_snapshot_under_lifecycle_fence(bucket, lifecycle_fence)
.await?
}
};
if !snapshot.is_valid_for_destructive_put(self.id, bucket, expected_incarnation_id) {
return Err(Error::other(
"data movement Object Lock snapshot does not match the target bucket generation",
));
}
snapshot.add_lock_fences(&mut opts);
opts.object_lock_config_snapshot = Some(snapshot);
}
#[cfg(test)]
pause_data_movement_multipart_before_selected_completion(bucket).await;
let pool = self
.pools
.get(target_pool_idx)
.ok_or_else(|| Error::other(format!("data movement target pool {target_pool_idx} is out of range")))?
.clone();
let result = enqueue_transition_after_write(
pool.complete_multipart_upload(bucket, object, upload_id, uploaded_parts, &opts)
.await,
LcEventSrc::S3CompleteMultipartUpload,
)
.await;
if result.is_ok() {
list_objects::observe_list_objects_mutation(self.as_ref(), bucket).await;
}
result
}
} }
/// Merges per-pool `ListMultipartUploads` pages into a single globally paginated /// Merges per-pool `ListMultipartUploads` pages into a single globally paginated
+70 -627
View File
@@ -667,6 +667,18 @@ impl SelectObjectSnapshotLockLossWake {
} }
} }
fn select_object_ssec_headers(headers: &HeaderMap) -> HeaderMap {
use rustfs_utils::http::headers::{SSEC_ALGORITHM_HEADER, SSEC_KEY_HEADER, SSEC_KEY_MD5_HEADER};
let mut selected = HeaderMap::new();
for name in [SSEC_ALGORITHM_HEADER, SSEC_KEY_HEADER, SSEC_KEY_MD5_HEADER] {
if let Some(value) = headers.get(name) {
selected.insert(name, value.clone());
}
}
selected
}
// LockRegistry clones its canonical client Arc for each endpoint host, so an // LockRegistry clones its canonical client Arc for each endpoint host, so an
// exact Arc set identifies one distributed namespace-lock quorum domain. // exact Arc set identifies one distributed namespace-lock quorum domain.
fn same_distributed_lock_domain(left: &[Arc<dyn rustfs_lock::LockClient>], right: &[Arc<dyn rustfs_lock::LockClient>]) -> bool { fn same_distributed_lock_domain(left: &[Arc<dyn rustfs_lock::LockClient>], right: &[Arc<dyn rustfs_lock::LockClient>]) -> bool {
@@ -906,7 +918,7 @@ fn is_equivalent_data_movement_delete_marker(source: &ObjectInfo, target: &Objec
&& is_data_movement_delete_marker(target) && is_data_movement_delete_marker(target)
&& source.version_id == target.version_id && source.version_id == target.version_id
&& source.mod_time == target.mod_time && source.mod_time == target.mod_time
&& is_equivalent_data_movement_delete_marker_metadata(&source.user_defined, &target.user_defined) && source.user_defined == target.user_defined
&& source.user_tags == target.user_tags && source.user_tags == target.user_tags
&& source.replication_status_internal == target.replication_status_internal && source.replication_status_internal == target.replication_status_internal
&& source.replication_status == target.replication_status && source.replication_status == target.replication_status
@@ -914,185 +926,24 @@ fn is_equivalent_data_movement_delete_marker(source: &ObjectInfo, target: &Objec
&& source.version_purge_status == target.version_purge_status && source.version_purge_status == target.version_purge_status
} }
fn is_equivalent_data_movement_delete_marker_metadata(
source: &HashMap<String, String>,
target: &HashMap<String, String>,
) -> bool {
matches!(
(
data_movement_delete_marker_metadata_identity(source),
data_movement_delete_marker_metadata_identity(target)
),
(Some(source), Some(target)) if source == target
)
}
fn data_movement_delete_marker_metadata_identity(metadata: &HashMap<String, String>) -> Option<HashMap<String, String>> {
let mut identity = HashMap::with_capacity(metadata.len());
let mut local_tier_free_version_id = None;
for (key, value) in metadata {
let Some(suffix) = rustfs_utils::http::strip_internal_prefix_preserving_case(key) else {
identity.insert(key.clone(), value.clone());
continue;
};
if suffix.eq_ignore_ascii_case(rustfs_utils::http::SUFFIX_TIER_FV_ID) {
let version_id = Uuid::parse_str(value).ok().filter(|version_id| !version_id.is_nil())?;
if local_tier_free_version_id.is_some_and(|expected| expected != version_id) {
return None;
}
local_tier_free_version_id = Some(version_id);
continue;
}
let canonical_suffix = [
rustfs_utils::http::SUFFIX_REPLICA_TIMESTAMP,
rustfs_utils::http::SUFFIX_REPLICA_STATUS,
rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP,
rustfs_utils::http::SUFFIX_REPLICATION_STATUS,
rustfs_utils::http::SUFFIX_PURGESTATUS,
]
.into_iter()
.find(|candidate| suffix.eq_ignore_ascii_case(candidate))
.map(str::to_string)
.or_else(|| {
[
rustfs_utils::http::SUFFIX_REPLICATION_RESET_ARN_PREFIX,
rustfs_utils::http::SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX,
]
.into_iter()
.find_map(|prefix| {
suffix
.get(..prefix.len())
.is_some_and(|candidate| candidate.eq_ignore_ascii_case(prefix))
.then(|| format!("{prefix}{}", &suffix[prefix.len()..]))
})
})
.unwrap_or_else(|| suffix.to_string());
let canonical_value = if canonical_suffix.eq_ignore_ascii_case(rustfs_utils::http::SUFFIX_REPLICA_TIMESTAMP)
|| canonical_suffix.eq_ignore_ascii_case(rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP)
{
rustfs_filemeta::parse_replication_timestamp(value)?
.unix_timestamp_nanos()
.to_string()
} else {
value.clone()
};
let canonical_key = format!("{}{canonical_suffix}", rustfs_utils::http::RUSTFS_INTERNAL_PREFIX);
if identity
.insert(canonical_key, canonical_value.clone())
.is_some_and(|existing| existing != canonical_value)
{
return None;
}
}
for (status_suffix, timestamp_suffix) in [
(rustfs_utils::http::SUFFIX_REPLICA_STATUS, rustfs_utils::http::SUFFIX_REPLICA_TIMESTAMP),
(
rustfs_utils::http::SUFFIX_REPLICATION_STATUS,
rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP,
),
] {
let status_key = format!("{}{status_suffix}", rustfs_utils::http::RUSTFS_INTERNAL_PREFIX);
let timestamp_key = format!("{}{timestamp_suffix}", rustfs_utils::http::RUSTFS_INTERNAL_PREFIX);
match (identity.contains_key(&status_key), identity.contains_key(&timestamp_key)) {
(true, false) => {
identity.insert(timestamp_key, OffsetDateTime::UNIX_EPOCH.unix_timestamp_nanos().to_string());
}
(false, true) => return None,
_ => {}
}
}
Some(identity)
}
fn is_data_movement_delete_marker(info: &ObjectInfo) -> bool { fn is_data_movement_delete_marker(info: &ObjectInfo) -> bool {
info.delete_marker info.delete_marker
} }
fn is_expected_data_movement_delete_marker_source(source: &ObjectInfo, expected_mod_time: Option<OffsetDateTime>) -> bool {
is_data_movement_delete_marker(source)
&& source.mod_time.is_some()
&& source.mod_time == expected_mod_time
&& data_movement_delete_marker_metadata_identity(&source.user_defined).is_some()
}
fn current_data_movement_delete_marker_opts(source: &ObjectInfo, opts: &ObjectOptions) -> Option<ObjectOptions> {
let replica_status = rustfs_utils::http::get_str(&source.user_defined, rustfs_utils::http::SUFFIX_REPLICA_STATUS);
let replica_timestamp = rustfs_utils::http::get_str(&source.user_defined, rustfs_utils::http::SUFFIX_REPLICA_TIMESTAMP);
let (replica_status, replica_timestamp) = match (replica_status, replica_timestamp) {
(None, None) => Default::default(),
(Some(status), timestamp) => {
let status = crate::bucket::replication::ReplicationStatusType::from(status.as_str());
if status.is_empty() {
return None;
}
let timestamp = match timestamp {
Some(timestamp) => rustfs_filemeta::parse_replication_timestamp(&timestamp)?,
None => OffsetDateTime::UNIX_EPOCH,
};
(status, Some(timestamp))
}
(None, Some(_)) => return None,
};
let replication_status = rustfs_utils::http::get_str(&source.user_defined, rustfs_utils::http::SUFFIX_REPLICATION_STATUS);
let replication_timestamp =
rustfs_utils::http::get_str(&source.user_defined, rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP);
let (replication_status, replication_timestamp, replication_targets) = match (replication_status, replication_timestamp) {
(None, None) => Default::default(),
(Some(status), timestamp) => {
let direct_status = crate::bucket::replication::ReplicationStatusType::from(status.as_str());
let targets = crate::bucket::replication::replication_statuses_map(status.as_str());
if direct_status.is_empty() && targets.is_empty() {
return None;
}
let timestamp = match timestamp {
Some(timestamp) => rustfs_filemeta::parse_replication_timestamp(&timestamp)?,
None => OffsetDateTime::UNIX_EPOCH,
};
(Some(status), Some(timestamp), targets)
}
(None, Some(_)) => return None,
};
let mut state = source.replication_state();
if state.target_delete_marker_version_ids_corrupt {
return None;
}
state.replica_status = replica_status;
state.replica_timestamp = replica_timestamp;
state.replication_status_internal = replication_status;
state.replication_timestamp = replication_timestamp;
state.targets = replication_targets;
state.replicate_decision_str = source.replication_decision.clone();
state.delete_marker = true;
let mut target_opts = opts.clone();
target_opts.mod_time = source.mod_time;
target_opts.delete_replication = Some(state);
Some(target_opts)
}
fn expected_data_movement_tiered_object(source: &rustfs_filemeta::FileInfo) -> ObjectInfo { fn expected_data_movement_tiered_object(source: &rustfs_filemeta::FileInfo) -> ObjectInfo {
ObjectInfo::from_file_info(source, "", &source.name, source.version_id.is_some()) ObjectInfo::from_file_info(source, "", &source.name, source.version_id.is_some())
} }
fn is_equivalent_data_movement_tiered_object(source: &rustfs_filemeta::FileInfo, target: &ObjectInfo) -> bool { fn is_equivalent_data_movement_tiered_object(source: &rustfs_filemeta::FileInfo, target: &ObjectInfo) -> bool {
let expected = expected_data_movement_tiered_object(source); let expected = expected_data_movement_tiered_object(source);
let Some(source_actual_size) = effective_object_actual_size(&expected) else {
return false;
};
let Some(target_actual_size) = effective_object_actual_size(target) else {
return false;
};
source.version_id == target.version_id source.version_id == target.version_id
&& !target.delete_marker && !target.delete_marker
&& source.size == target.size && source.size == target.size
&& source.get_etag() == target.etag && source.get_etag() == target.etag
&& source.checksum == target.checksum && source.checksum == target.checksum
&& crate::data_movement::are_equivalent_data_movement_parts(&source.parts, &target.parts)
&& source.mod_time == target.mod_time && source.mod_time == target.mod_time
&& crate::data_movement::is_equivalent_data_movement_metadata(&expected, target, source_actual_size, target_actual_size) && expected.user_defined == target.user_defined
&& expected.user_tags == target.user_tags && expected.user_tags == target.user_tags
&& expected.expires == target.expires && expected.expires == target.expires
&& expected.storage_class == target.storage_class && expected.storage_class == target.storage_class
@@ -1101,12 +952,11 @@ fn is_equivalent_data_movement_tiered_object(source: &rustfs_filemeta::FileInfo,
&& expected.version_purge_status_internal == target.version_purge_status_internal && expected.version_purge_status_internal == target.version_purge_status_internal
&& expected.version_purge_status == target.version_purge_status && expected.version_purge_status == target.version_purge_status
&& expected.transitioned_object.status == target.transitioned_object.status && expected.transitioned_object.status == target.transitioned_object.status
&& expected.transition_version_state == target.transition_version_state
&& expected.transitioned_object.name == target.transitioned_object.name && expected.transitioned_object.name == target.transitioned_object.name
&& expected.transitioned_object.tier == target.transitioned_object.tier && expected.transitioned_object.tier == target.transitioned_object.tier
&& expected.transitioned_object.version_id == target.transitioned_object.version_id && expected.transitioned_object.version_id == target.transitioned_object.version_id
&& expected.transitioned_object.free_version == target.transitioned_object.free_version && expected.transitioned_object.free_version == target.transitioned_object.free_version
&& source_actual_size == target_actual_size && effective_object_actual_size(target) == Some(source.size)
} }
fn should_check_data_movement_resume_target(src_pool_idx: usize, target_pool_idx: usize) -> bool { fn should_check_data_movement_resume_target(src_pool_idx: usize, target_pool_idx: usize) -> bool {
@@ -1204,7 +1054,7 @@ impl ECStore {
))) )))
} }
pub(super) async fn object_lock_config_snapshot_under_lifecycle_fence( async fn object_lock_config_snapshot_under_lifecycle_fence(
&self, &self,
bucket: &str, bucket: &str,
lifecycle_fence: &NamespaceLockFence, lifecycle_fence: &NamespaceLockFence,
@@ -1284,7 +1134,7 @@ impl ECStore {
pool, pool,
bucket: bucket.to_owned(), bucket: bucket.to_owned(),
object, object,
headers: rustfs_utils::http::project_ssec_transport_headers(headers), headers: select_object_ssec_headers(headers),
opts, opts,
object_info, object_info,
logical_size, logical_size,
@@ -1604,8 +1454,7 @@ impl ECStore {
target_pool_idx: usize, target_pool_idx: usize,
opts: &ObjectOptions, opts: &ObjectOptions,
) -> Result<Option<ObjectInfo>> { ) -> Result<Option<ObjectInfo>> {
let mut lookup_opts = version_aware_lookup_opts(opts, true); let lookup_opts = version_aware_lookup_opts(opts, true);
lookup_opts.include_part_checksums = true;
let Some(pool) = self.pools.get(target_pool_idx) else { let Some(pool) = self.pools.get(target_pool_idx) else {
return Err(Error::other(format!( return Err(Error::other(format!(
@@ -1672,25 +1521,6 @@ impl ECStore {
) -> Result<()> { ) -> Result<()> {
check_put_object_args(bucket, object)?; check_put_object_args(bucket, object)?;
let mut opts = opts.clone();
let bucket_incarnation_fence = if is_meta_bucketname(bucket) {
None
} else {
let expected = opts
.expected_bucket_incarnation_id
.ok_or_else(|| Error::other("tiered data movement is missing its bucket incarnation snapshot"))?;
let guard = self.acquire_bucket_incarnation_fence(bucket, expected).await?;
if let Some(namespace_guard) = guard.namespace_lock_guard() {
opts.add_bucket_lifecycle_lock_guard(namespace_guard);
}
Some(guard)
};
let mut fi = fi.clone();
if opts.data_movement {
crate::data_movement::prepare_tiered_data_movement_file_info(&mut fi)?;
}
let object = encode_dir_object(object); let object = encode_dir_object(object);
if self.single_pool() { if self.single_pool() {
@@ -1703,8 +1533,7 @@ impl ECStore {
let idx = if opts.data_movement && opts.version_id.is_some() { let idx = if opts.data_movement && opts.version_id.is_some() {
Self::resolve_decommission_target_pool_idx_result( Self::resolve_decommission_target_pool_idx_result(
self.select_data_movement_pool_idx(bucket, &object, fi.size, &opts, true) self.select_data_movement_pool_idx(bucket, &object, fi.size, opts, true).await,
.await,
bucket, bucket,
&object, &object,
)? )?
@@ -1721,7 +1550,7 @@ impl ECStore {
.await; .await;
let target_pool_idx = resolve_data_movement_resume_target_pool(idx, resume_target_pool_idx, opts.src_pool_idx); let target_pool_idx = resolve_data_movement_resume_target_pool(idx, resume_target_pool_idx, opts.src_pool_idx);
if self if self
.has_equivalent_data_movement_tiered_object(bucket, &object, &fi, &opts, target_pool_idx) .has_equivalent_data_movement_tiered_object(bucket, &object, fi, opts, target_pool_idx)
.await? .await?
{ {
return Ok(()); return Ok(());
@@ -1734,30 +1563,17 @@ impl ECStore {
)); ));
} }
let result = self.pools[idx] Self::resolve_decommission_tiered_object_result(
.get_disks_by_key(&object) self.pools[idx]
.decommission_tiered_object(bucket, &object, &fi, &opts) .get_disks_by_key(&object)
.await; .decommission_tiered_object(bucket, &object, fi, opts)
if matches!(result, Err(Error::PreconditionFailed)) { .await,
if self bucket,
.has_equivalent_data_movement_tiered_object(bucket, &object, &fi, &opts, idx) &object,
.await? )
{
return Ok(());
}
return Err(StorageError::DataMovementOverwriteErr(
bucket.to_owned(),
object,
opts.version_id.clone().unwrap_or_default(),
));
}
if bucket_incarnation_fence.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
return Err(Error::other("tiered data movement bucket incarnation fence was lost during target write"));
}
Self::resolve_decommission_tiered_object_result(result, bucket, &object)
} }
#[instrument(level = "debug", skip(self, h))] #[instrument(level = "debug", skip(self))]
#[hotpath::measure(impl_type = "ECStore")] #[hotpath::measure(impl_type = "ECStore")]
pub(super) async fn handle_get_object_reader( pub(super) async fn handle_get_object_reader(
&self, &self,
@@ -1769,7 +1585,7 @@ impl ECStore {
) -> Result<GetObjectReader> { ) -> Result<GetObjectReader> {
check_get_obj_args(bucket, object)?; check_get_obj_args(bucket, object)?;
let object = rustfs_utils::path::encode_dir_object_ref(object); let object = encode_dir_object(object);
let mut opts = opts.clone(); let mut opts = opts.clone();
let read_lock_guard = self let read_lock_guard = self
.acquire_object_read_lock_if_needed("get_object", bucket, &object, &mut opts) .acquire_object_read_lock_if_needed("get_object", bucket, &object, &mut opts)
@@ -1777,21 +1593,29 @@ impl ECStore {
let reader = if self.single_pool() { let reader = if self.single_pool() {
self.pools[0] self.pools[0]
.get_object_reader(bucket, object.as_ref(), range, h, &opts) .get_object_reader(bucket, object.as_str(), range, h, &opts)
.await? .await?
} else { } else {
let (_, idx) = self let (_, idx) = self
.get_latest_accessible_object_info_with_idx(bucket, &object, &opts) .get_latest_accessible_object_info_with_idx(bucket, &object, &opts)
.await?; .await?;
self.pools[idx] self.pools[idx]
.get_object_reader(bucket, object.as_ref(), range, h, &opts) .get_object_reader(bucket, object.as_str(), range, h, &opts)
.await? .await?
}; };
Ok(Self::attach_read_lock_guard(reader, read_lock_guard)) Ok(Self::attach_read_lock_guard(reader, read_lock_guard))
} }
async fn prepare_put_object(&self, bucket: &str, object: &str, opts: &ObjectOptions) -> Result<(String, ObjectOptions)> { #[instrument(level = "debug", skip(self, data))]
#[hotpath::measure(impl_type = "ECStore")]
pub(super) async fn handle_put_object(
&self,
bucket: &str,
object: &str,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<(ObjectInfo, Option<OldCurrentSize>)> {
check_put_object_args(bucket, object)?; check_put_object_args(bucket, object)?;
let object = encode_dir_object(object); let object = encode_dir_object(object);
@@ -1818,20 +1642,22 @@ impl ECStore {
}; };
snapshot.add_lock_fences(&mut opts); snapshot.add_lock_fences(&mut opts);
} }
Ok((object, opts))
}
async fn select_put_object_pool_idx(&self, bucket: &str, object: &str, size: i64, opts: &ObjectOptions) -> Result<usize> { // Keep PUT atomic-read friendly: SetDisks takes the object write lock only
// around precondition checks and the final rename/commit.
if self.single_pool() { if self.single_pool() {
return Ok(0); return self.pools[0]
.put_object_with_old_current_size(bucket, object.as_str(), data, &opts)
.await;
} }
let idx = if opts.data_movement && opts.version_id.is_some() { let idx = if opts.data_movement && opts.version_id.is_some() {
self.select_data_movement_pool_idx(bucket, object, size, opts, false).await? self.select_data_movement_pool_idx(bucket, &object, data.size(), &opts, false)
.await?
} else if opts.no_lock { } else if opts.no_lock {
self.get_pool_idx_no_lock(bucket, object, size).await? self.get_pool_idx_no_lock(bucket, &object, data.size()).await?
} else { } else {
self.get_pool_idx(bucket, object, size).await? self.get_pool_idx(bucket, &object, data.size()).await?
}; };
if opts.data_movement && idx == opts.src_pool_idx { if opts.data_movement && idx == opts.src_pool_idx {
@@ -1841,50 +1667,7 @@ impl ECStore {
opts.version_id.clone().unwrap_or_default(), opts.version_id.clone().unwrap_or_default(),
)); ));
} }
Ok(idx)
}
pub(crate) async fn put_object_for_data_movement(
&self,
bucket: &str,
object: &str,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<(usize, Result<ObjectInfo>)> {
if !opts.data_movement {
return Err(Error::other("data movement PUT requires data_movement options"));
}
let (object, opts) = self.prepare_put_object(bucket, object, opts).await?;
let idx = self
.select_put_object_pool_idx(bucket, object.as_str(), data.size(), &opts)
.await?;
let result = self.pools[idx]
.put_object_with_old_current_size(bucket, &object, data, &opts)
.await
.map(|(object_info, _)| object_info);
let result = enqueue_transition_after_write(result, LcEventSrc::S3PutObject).await;
if result.is_ok() {
list_objects::observe_list_objects_mutation(self, bucket).await;
}
Ok((idx, result))
}
#[instrument(level = "debug", skip(self, data))]
#[hotpath::measure(impl_type = "ECStore")]
pub(super) async fn handle_put_object(
&self,
bucket: &str,
object: &str,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<(ObjectInfo, Option<OldCurrentSize>)> {
let (object, opts) = self.prepare_put_object(bucket, object, opts).await?;
let idx = self
.select_put_object_pool_idx(bucket, object.as_str(), data.size(), &opts)
.await?;
// Keep PUT atomic-read friendly: SetDisks takes the object write lock only
// around precondition checks and the final rename/commit.
self.pools[idx] self.pools[idx]
.put_object_with_old_current_size(bucket, &object, data, &opts) .put_object_with_old_current_size(bucket, &object, data, &opts)
.await .await
@@ -2327,50 +2110,6 @@ impl ECStore {
}; };
let target_pool_idx = let target_pool_idx =
resolve_data_movement_resume_target_pool(selected_target_pool_idx, resume_target_pool_idx, opts.src_pool_idx); resolve_data_movement_resume_target_pool(selected_target_pool_idx, resume_target_pool_idx, opts.src_pool_idx);
let mut delete_marker_target_opts = None;
if opts.delete_marker && should_check_data_movement_resume_target(opts.src_pool_idx, target_pool_idx) {
let source = self
.find_data_movement_target_info(bucket, object, opts.src_pool_idx, &opts)
.await?;
let Some(source) = source else {
return Err(StorageError::DataMovementOverwriteErr(
bucket.to_owned(),
object.to_owned(),
opts.version_id.unwrap_or_default(),
));
};
if !is_expected_data_movement_delete_marker_source(&source, opts.mod_time) {
return Err(StorageError::DataMovementOverwriteErr(
bucket.to_owned(),
object.to_owned(),
opts.version_id.unwrap_or_default(),
));
}
let Some(target_opts) = current_data_movement_delete_marker_opts(&source, &opts) else {
return Err(StorageError::DataMovementOverwriteErr(
bucket.to_owned(),
object.to_owned(),
opts.version_id.unwrap_or_default(),
));
};
let target = self
.find_data_movement_target_info(bucket, object, target_pool_idx, &target_opts)
.await?;
if let Some(target) = target {
if is_equivalent_data_movement_delete_marker(&source, &target) {
let mut target = target;
target.name = decode_dir_object(object);
return Ok(target);
}
return Err(StorageError::DataMovementOverwriteErr(
bucket.to_owned(),
object.to_owned(),
opts.version_id.unwrap_or_default(),
));
}
delete_marker_target_opts = Some(target_opts);
}
if !should_check_data_movement_resume_target(opts.src_pool_idx, target_pool_idx) { if !should_check_data_movement_resume_target(opts.src_pool_idx, target_pool_idx) {
if let Ok((source_pool_info, _)) = existing_pool_info if let Ok((source_pool_info, _)) = existing_pool_info
@@ -2398,8 +2137,7 @@ impl ECStore {
)); ));
} }
let target_opts = delete_marker_target_opts.unwrap_or(opts); let mut obj = self.pools[target_pool_idx].delete_object(bucket, object, opts).await?;
let mut obj = self.pools[target_pool_idx].delete_object(bucket, object, target_opts).await?;
obj.name = decode_dir_object(obj.name.as_str()); obj.name = decode_dir_object(obj.name.as_str());
return Ok(obj); return Ok(obj);
} }
@@ -3348,6 +3086,25 @@ mod tests {
assert!(second_signal.is_lost()); assert!(second_signal.is_lost());
} }
#[test]
fn select_snapshot_retains_only_ssec_headers() {
use rustfs_utils::http::headers::{SSEC_ALGORITHM_HEADER, SSEC_KEY_HEADER, SSEC_KEY_MD5_HEADER};
let mut headers = HeaderMap::new();
headers.insert(SSEC_ALGORITHM_HEADER, "AES256".parse().expect("valid SSE-C algorithm header"));
headers.insert(SSEC_KEY_HEADER, "secret-key".parse().expect("valid SSE-C key header"));
headers.insert(SSEC_KEY_MD5_HEADER, "key-md5".parse().expect("valid SSE-C key digest header"));
headers.insert("authorization", "credential".parse().expect("valid authorization header"));
let selected = select_object_ssec_headers(&headers);
assert_eq!(selected.len(), 3);
assert_eq!(selected.get(SSEC_ALGORITHM_HEADER), headers.get(SSEC_ALGORITHM_HEADER));
assert_eq!(selected.get(SSEC_KEY_HEADER), headers.get(SSEC_KEY_HEADER));
assert_eq!(selected.get(SSEC_KEY_MD5_HEADER), headers.get(SSEC_KEY_MD5_HEADER));
assert!(selected.get("authorization").is_none());
}
#[test] #[test]
fn tier_delete_entry_is_prepared_and_bound_to_source_generation() { fn tier_delete_entry_is_prepared_and_bound_to_source_generation() {
let identity = [9_u8; 32]; let identity = [9_u8; 32];
@@ -3443,247 +3200,6 @@ mod tests {
assert!(!is_equivalent_data_movement_delete_marker(&source, &mismatched)); assert!(!is_equivalent_data_movement_delete_marker(&source, &mismatched));
} }
#[test]
fn equivalent_data_movement_delete_marker_accepts_distinct_local_free_version_ids() {
let mut source = ObjectInfo {
version_id: Some(Uuid::from_u128(1)),
delete_marker: true,
mod_time: Some(OffsetDateTime::UNIX_EPOCH),
..Default::default()
};
rustfs_utils::http::insert_str(
Arc::make_mut(&mut source.user_defined),
rustfs_utils::http::SUFFIX_TIER_FV_ID,
Uuid::from_u128(2).to_string(),
);
let mut target = source.clone();
rustfs_utils::http::insert_str(
Arc::make_mut(&mut target.user_defined),
rustfs_utils::http::SUFFIX_TIER_FV_ID,
Uuid::from_u128(3).to_string(),
);
assert!(is_equivalent_data_movement_delete_marker(&source, &target));
Arc::make_mut(&mut target.user_defined).insert(
format!("{}{}", rustfs_utils::http::MINIO_INTERNAL_PREFIX, rustfs_utils::http::SUFFIX_TIER_FV_ID),
Uuid::from_u128(4).to_string(),
);
assert!(!is_equivalent_data_movement_delete_marker(&source, &target));
}
#[test]
fn equivalent_data_movement_delete_marker_accepts_replication_alias_expansion() {
let key = format!(
"{}{}",
rustfs_utils::http::MINIO_INTERNAL_PREFIX,
rustfs_utils::http::SUFFIX_REPLICATION_STATUS
);
let timestamp_key = format!(
"{}{}",
rustfs_utils::http::MINIO_INTERNAL_PREFIX,
rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP
);
let source = ObjectInfo {
version_id: Some(Uuid::from_u128(1)),
delete_marker: true,
mod_time: Some(OffsetDateTime::UNIX_EPOCH),
user_defined: Arc::new(HashMap::from([
(key.clone(), "arn=COMPLETED;".to_string()),
(timestamp_key, "1970-01-01T00:00:01Z".to_string()),
])),
..Default::default()
};
let mut target = source.clone();
rustfs_utils::http::insert_str(
Arc::make_mut(&mut target.user_defined),
rustfs_utils::http::SUFFIX_REPLICATION_STATUS,
"arn=COMPLETED;".to_string(),
);
rustfs_utils::http::insert_str(
Arc::make_mut(&mut target.user_defined),
rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP,
(OffsetDateTime::UNIX_EPOCH + time::Duration::SECOND).to_string(),
);
assert!(is_equivalent_data_movement_delete_marker(&source, &target));
Arc::make_mut(&mut target.user_defined).insert(key, "arn=FAILED;".to_string());
assert!(!is_equivalent_data_movement_delete_marker(&source, &target));
}
#[test]
fn data_movement_delete_marker_source_requires_persisted_mod_time() {
let source = ObjectInfo {
delete_marker: true,
..Default::default()
};
assert!(!is_expected_data_movement_delete_marker_source(&source, None));
let source = ObjectInfo {
mod_time: Some(OffsetDateTime::UNIX_EPOCH),
..source
};
assert!(is_expected_data_movement_delete_marker_source(&source, Some(OffsetDateTime::UNIX_EPOCH)));
assert!(!is_expected_data_movement_delete_marker_source(&source, None));
}
#[test]
fn data_movement_delete_marker_uses_current_source_replication_state() {
let expected_timestamp = OffsetDateTime::UNIX_EPOCH + time::Duration::SECOND;
let timestamp = expected_timestamp.to_string();
let mut metadata = HashMap::new();
rustfs_utils::http::insert_str(
&mut metadata,
rustfs_utils::http::SUFFIX_REPLICA_STATUS,
ReplicationStatusType::Replica.to_string(),
);
rustfs_utils::http::insert_str(&mut metadata, rustfs_utils::http::SUFFIX_REPLICA_TIMESTAMP, timestamp.clone());
rustfs_utils::http::insert_str(&mut metadata, rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP, timestamp);
rustfs_utils::http::insert_str(
&mut metadata,
rustfs_utils::http::SUFFIX_REPLICATION_STATUS,
"arn=COMPLETED;".to_string(),
);
rustfs_utils::http::insert_str(
&mut metadata,
&format!(
"{}{}",
rustfs_utils::http::SUFFIX_REPLICATION_RESET_ARN_PREFIX,
"arn:minio:replication::TenantA:bucket"
),
"reset-id".to_string(),
);
rustfs_utils::http::insert_str(
&mut metadata,
&format!(
"{}{}",
rustfs_utils::http::SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX,
"arn:minio:replication::TenantA:bucket"
),
"target-version".to_string(),
);
let source = ObjectInfo {
delete_marker: true,
mod_time: Some(OffsetDateTime::UNIX_EPOCH),
replication_status_internal: Some("arn=COMPLETED;".to_string()),
replication_decision: "arn=replicate;".to_string(),
user_defined: Arc::new(metadata),
..Default::default()
};
let opts = ObjectOptions {
mod_time: source.mod_time,
delete_replication: Some(ReplicationState {
replication_status_internal: Some("arn=PENDING;".to_string()),
..Default::default()
}),
..Default::default()
};
let target_opts = current_data_movement_delete_marker_opts(&source, &opts).expect("valid current source state");
let state = target_opts.delete_replication.as_ref().expect("current replication state");
assert_eq!(state.replication_status_internal.as_deref(), Some("arn=COMPLETED;"));
assert_eq!(state.replica_status, crate::bucket::replication::ReplicationStatusType::Replica);
assert_eq!(state.replica_timestamp, Some(expected_timestamp));
assert_eq!(state.replication_timestamp, state.replica_timestamp);
assert_eq!(state.replicate_decision_str, "arn=replicate;");
assert_eq!(
state
.reset_statuses_map
.get("arn:minio:replication::TenantA:bucket")
.map(String::as_str),
Some("reset-id")
);
assert_eq!(
state
.target_delete_marker_version_ids
.get("arn:minio:replication::TenantA:bucket")
.map(String::as_str),
Some("target-version")
);
}
#[test]
fn data_movement_delete_marker_rejects_corrupt_target_version_maps() {
let suffix = format!("{}not-an-arn", rustfs_utils::http::SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX);
let mut malformed = HashMap::new();
rustfs_utils::http::insert_str(&mut malformed, &suffix, "target-version".to_string());
let malformed_source = ObjectInfo {
delete_marker: true,
mod_time: Some(OffsetDateTime::UNIX_EPOCH),
user_defined: Arc::new(malformed),
..Default::default()
};
assert!(current_data_movement_delete_marker_opts(&malformed_source, &ObjectOptions::default()).is_none());
let mut conflicted = HashMap::new();
let suffix = format!(
"{}arn:minio:replication::target:bucket",
rustfs_utils::http::SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX
);
rustfs_utils::http::insert_str(&mut conflicted, &suffix, "target-version-a".to_string());
conflicted.insert(
format!("{}{suffix}", rustfs_utils::http::MINIO_INTERNAL_PREFIX),
"target-version-b".to_string(),
);
let conflicted_source = ObjectInfo {
user_defined: Arc::new(conflicted),
..malformed_source.clone()
};
assert!(current_data_movement_delete_marker_opts(&conflicted_source, &ObjectOptions::default()).is_none());
let mut over_cap = HashMap::new();
for index in 0..=1_000 {
let suffix = format!(
"{}arn:minio:replication::target:bucket-{index}",
rustfs_utils::http::SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX
);
rustfs_utils::http::insert_str(&mut over_cap, &suffix, format!("target-version-{index}"));
}
let over_cap_source = ObjectInfo {
user_defined: Arc::new(over_cap),
..malformed_source
};
assert!(current_data_movement_delete_marker_opts(&over_cap_source, &ObjectOptions::default()).is_none());
}
#[test]
fn data_movement_delete_marker_normalizes_legacy_missing_replication_timestamps() {
let mut source_metadata = HashMap::new();
rustfs_utils::http::insert_str(
&mut source_metadata,
rustfs_utils::http::SUFFIX_REPLICA_STATUS,
ReplicationStatusType::Replica.to_string(),
);
rustfs_utils::http::insert_str(
&mut source_metadata,
rustfs_utils::http::SUFFIX_REPLICATION_STATUS,
"arn=COMPLETED;".to_string(),
);
let source = ObjectInfo {
delete_marker: true,
mod_time: Some(OffsetDateTime::UNIX_EPOCH),
replication_status_internal: Some("arn=COMPLETED;".to_string()),
user_defined: Arc::new(source_metadata),
..Default::default()
};
let target_opts = current_data_movement_delete_marker_opts(&source, &ObjectOptions::default())
.expect("legacy status-only metadata should remain migratable");
let state = target_opts
.delete_replication
.expect("replication state should be reconstructed");
assert_eq!(state.replica_timestamp, Some(OffsetDateTime::UNIX_EPOCH));
assert_eq!(state.replication_timestamp, Some(OffsetDateTime::UNIX_EPOCH));
let mut target_metadata = (*source.user_defined).clone();
let epoch = OffsetDateTime::UNIX_EPOCH
.format(&time::format_description::well_known::Rfc3339)
.unwrap();
rustfs_utils::http::insert_str(&mut target_metadata, rustfs_utils::http::SUFFIX_REPLICA_TIMESTAMP, epoch.clone());
rustfs_utils::http::insert_str(&mut target_metadata, rustfs_utils::http::SUFFIX_REPLICATION_TIMESTAMP, epoch);
assert!(is_equivalent_data_movement_delete_marker_metadata(&source.user_defined, &target_metadata));
}
#[test] #[test]
fn equivalent_data_movement_delete_marker_rejects_metadata_and_replication_mismatch() { fn equivalent_data_movement_delete_marker_rejects_metadata_and_replication_mismatch() {
let version_id = Uuid::nil(); let version_id = Uuid::nil();
@@ -3828,67 +3344,6 @@ mod tests {
assert!(is_equivalent_data_movement_tiered_object(&source, &target)); assert!(is_equivalent_data_movement_tiered_object(&source, &target));
} }
#[test]
fn equivalent_data_movement_tiered_object_uses_logical_compressed_and_encrypted_sizes() {
let mut compressed = tiered_equivalence_source();
compressed.size = 600;
rustfs_utils::http::insert_str(&mut compressed.metadata, rustfs_utils::http::SUFFIX_COMPRESSION, "S2".to_string());
rustfs_utils::http::insert_str(&mut compressed.metadata, rustfs_utils::http::SUFFIX_ACTUAL_SIZE, "1024".to_string());
let compressed_target = tiered_equivalence_target(&compressed);
assert!(is_equivalent_data_movement_tiered_object(&compressed, &compressed_target));
let mut encrypted = tiered_equivalence_source();
encrypted.size = 640;
encrypted.metadata.insert(
rustfs_utils::http::object_encryption_keys::INTERNAL_ENCRYPTION_KEY_ID_HEADER.to_string(),
"key-id".to_string(),
);
encrypted.metadata.insert(
rustfs_utils::http::object_encryption_keys::INTERNAL_ENCRYPTION_ORIGINAL_SIZE_HEADER.to_string(),
"1024".to_string(),
);
let encrypted_target = tiered_equivalence_target(&encrypted);
assert!(is_equivalent_data_movement_tiered_object(&encrypted, &encrypted_target));
}
#[test]
fn equivalent_data_movement_tiered_object_accepts_transition_alias_expansion() {
let mut source = tiered_equivalence_source();
let suffix = rustfs_utils::http::SUFFIX_TRANSITION_TIER;
source.metadata.insert(
format!("{}{suffix}", rustfs_utils::http::MINIO_INTERNAL_PREFIX),
source.transition_tier.clone(),
);
let mut target = tiered_equivalence_target(&source);
Arc::make_mut(&mut target.user_defined)
.insert(rustfs_utils::http::internal_key_rustfs(suffix), source.transition_tier.clone());
assert!(is_equivalent_data_movement_tiered_object(&source, &target));
}
#[test]
fn equivalent_data_movement_tiered_object_requires_hydrated_part_checksums() {
let mut source = tiered_equivalence_source();
source.parts = vec![rustfs_filemeta::ObjectPartInfo {
number: 1,
mod_time: Some(OffsetDateTime::UNIX_EPOCH + time::Duration::SECOND),
checksums: Some(HashMap::from([("CRC32C".to_string(), "AAAAAA==".to_string())])),
..Default::default()
}];
rustfs_utils::http::insert_str(
&mut source.metadata,
rustfs_utils::http::SUFFIX_PART_CHECKSUMS,
r#"[[1,[["CRC32C","AAAAAA=="]]]]"#.to_string(),
);
let mut target = tiered_equivalence_target(&source);
Arc::make_mut(&mut target.parts)[0].mod_time = None;
assert!(is_equivalent_data_movement_tiered_object(&source, &target));
let mut missing = target;
Arc::make_mut(&mut missing.parts)[0].checksums = None;
assert!(!is_equivalent_data_movement_tiered_object(&source, &missing));
}
#[test] #[test]
fn equivalent_data_movement_tiered_object_rejects_transition_mismatch() { fn equivalent_data_movement_tiered_object_rejects_transition_mismatch() {
let source = tiered_equivalence_source(); let source = tiered_equivalence_source();
@@ -3898,16 +3353,6 @@ mod tests {
assert!(!is_equivalent_data_movement_tiered_object(&source, &target)); assert!(!is_equivalent_data_movement_tiered_object(&source, &target));
} }
#[test]
fn equivalent_data_movement_tiered_object_rejects_transition_version_state_mismatch() {
let mut source = tiered_equivalence_source();
source.transition_version_state = rustfs_filemeta::TransitionVersionState::Exact;
let mut target = tiered_equivalence_target(&source);
target.transition_version_state = rustfs_filemeta::TransitionVersionState::Unknown;
assert!(!is_equivalent_data_movement_tiered_object(&source, &target));
}
#[test] #[test]
fn equivalent_data_movement_tiered_object_rejects_user_metadata_mismatch() { fn equivalent_data_movement_tiered_object_rejects_user_metadata_mismatch() {
let source = tiered_equivalence_source(); let source = tiered_equivalence_source();
@@ -4577,8 +4022,6 @@ mod tests {
assert_eq!(snapshot.headers.get(SSEC_KEY_HEADER), request_headers.get(SSEC_KEY_HEADER)); assert_eq!(snapshot.headers.get(SSEC_KEY_HEADER), request_headers.get(SSEC_KEY_HEADER));
assert_eq!(snapshot.headers.get(SSEC_KEY_MD5_HEADER), request_headers.get(SSEC_KEY_MD5_HEADER)); assert_eq!(snapshot.headers.get(SSEC_KEY_MD5_HEADER), request_headers.get(SSEC_KEY_MD5_HEADER));
assert!(snapshot.headers.get("authorization").is_none()); assert!(snapshot.headers.get("authorization").is_none());
assert!(snapshot.headers.values().all(http::HeaderValue::is_sensitive));
assert!(!format!("{:?}", snapshot.headers).contains("secret-key"));
assert_eq!( assert_eq!(
snapshot.logical_size(), snapshot.logical_size(),
u64::try_from(payload.len()).expect("test payload length should fit in u64") u64::try_from(payload.len()).expect("test payload length should fit in u64")
@@ -109,7 +109,6 @@ async fn run_legacy_bitrot_test_for_object(root: &std::path::Path, disk_name: &s
FileInfoOpts { FileInfoOpts {
data: true, // need inline data for inline objects data: true, // need inline data for inline objects
include_free_versions: false, include_free_versions: false,
include_part_checksums: true,
}, },
) { ) {
Ok(f) => f, Ok(f) => f,
+1 -1
View File
@@ -37,7 +37,6 @@ crc-fast = { workspace = true }
rmp.workspace = true rmp.workspace = true
rmp-serde.workspace = true rmp-serde.workspace = true
serde = { workspace = true, features = ["derive"] } serde = { workspace = true, features = ["derive"] }
serde_json.workspace = true
time = { workspace = true, features = ["parsing", "formatting", "macros", "serde"] } time = { workspace = true, features = ["parsing", "formatting", "macros", "serde"] }
uuid = { workspace = true, features = ["v4", "fast-rng", "serde", "macro-diagnostics"] } uuid = { workspace = true, features = ["v4", "fast-rng", "serde", "macro-diagnostics"] }
tokio = { workspace = true, features = ["io-util", "macros", "sync", "fs", "rt-multi-thread"] } tokio = { workspace = true, features = ["io-util", "macros", "sync", "fs", "rt-multi-thread"] }
@@ -55,6 +54,7 @@ arc-swap.workspace = true
criterion = { workspace = true, features = ["html_reports"] } criterion = { workspace = true, features = ["html_reports"] }
tempfile = { workspace = true } tempfile = { workspace = true }
proptest = "1" proptest = "1"
serde_json.workspace = true
[[bench]] [[bench]]
name = "xl_meta_bench" name = "xl_meta_bench"
@@ -52,7 +52,6 @@ fn main() {
FileInfoOpts { FileInfoOpts {
data: false, data: false,
include_free_versions: true, include_free_versions: true,
include_part_checksums: true,
}, },
) )
.expect("decode file info"); .expect("decode file info");
+3 -78
View File
@@ -17,10 +17,9 @@ use bytes::Bytes;
use rmp_serde::Serializer; use rmp_serde::Serializer;
use rustfs_utils::HashAlgorithm; use rustfs_utils::HashAlgorithm;
use rustfs_utils::http::{ use rustfs_utils::http::{
AMZ_OBJECT_TAGGING, SUFFIX_COMPRESSION, SUFFIX_DATA_MOVED, SUFFIX_DATA_MOVED_TAGS, SUFFIX_FREE_VERSION, SUFFIX_HEALING, SUFFIX_COMPRESSION, SUFFIX_DATA_MOVED, SUFFIX_FREE_VERSION, SUFFIX_HEALING, SUFFIX_INLINE_DATA, SUFFIX_TIER_FV_ID,
SUFFIX_INLINE_DATA, SUFFIX_OBJECT_TRANSACTION_EPOCH, SUFFIX_TIER_FV_ID, SUFFIX_TIER_FV_MARKER, SUFFIX_TIER_SKIP_FV_ID, SUFFIX_TIER_FV_MARKER, SUFFIX_TIER_SKIP_FV_ID, contains_key_str, get_str, has_internal_suffix, insert_str,
contains_key_str, get_consistent_str, get_str, has_internal_suffix, insert_str, is_encryption_metadata_key, is_encryption_metadata_key, starts_with_ignore_ascii_case,
starts_with_ignore_ascii_case,
}; };
use s3s::dto::{RestoreStatus, Timestamp}; use s3s::dto::{RestoreStatus, Timestamp};
use s3s::header::X_AMZ_RESTORE; use s3s::header::X_AMZ_RESTORE;
@@ -233,17 +232,6 @@ pub enum TransitionVersionState {
Exact, Exact,
} }
impl TransitionVersionState {
pub const fn as_str(self) -> &'static str {
match self {
Self::Unknown => "unknown",
Self::KnownDisabled => "known-disabled",
Self::SuspendedNull => "suspended-null",
Self::Exact => "exact",
}
}
}
#[derive(PartialEq, Clone, Default)] #[derive(PartialEq, Clone, Default)]
pub struct FileInfo { pub struct FileInfo {
pub volume: String, pub volume: String,
@@ -1163,32 +1151,9 @@ impl FileInfo {
} }
pub fn set_data_moved(&mut self) { pub fn set_data_moved(&mut self) {
let tags_proof = format!("v1:{}", self.metadata.get(AMZ_OBJECT_TAGGING).map(String::as_str).unwrap_or_default());
insert_str(&mut self.metadata, SUFFIX_DATA_MOVED_TAGS, tags_proof);
insert_str(&mut self.metadata, SUFFIX_DATA_MOVED, "true".to_string()); insert_str(&mut self.metadata, SUFFIX_DATA_MOVED, "true".to_string());
} }
pub fn acknowledge_data_movement(&mut self) {
// Keep both empty aliases so mixed-version disks retain one metadata identity.
insert_str(&mut self.metadata, SUFFIX_DATA_MOVED, String::new());
}
pub fn set_object_transaction_epoch(&mut self, epoch: Uuid) {
insert_str(&mut self.metadata, SUFFIX_OBJECT_TRANSACTION_EPOCH, epoch.to_string());
}
pub fn object_transaction_epoch(&self) -> Result<Option<Uuid>> {
if !contains_key_str(&self.metadata, SUFFIX_OBJECT_TRANSACTION_EPOCH) {
return Ok(None);
}
let value = get_consistent_str(&self.metadata, SUFFIX_OBJECT_TRANSACTION_EPOCH).ok_or(Error::FileCorrupt)?;
let epoch = Uuid::parse_str(value).map_err(|_| Error::FileCorrupt)?;
if epoch.is_nil() {
return Err(Error::FileCorrupt);
}
Ok(Some(epoch))
}
pub fn inline_data(&self) -> bool { pub fn inline_data(&self) -> bool {
contains_key_str(&self.metadata, SUFFIX_INLINE_DATA) && !self.is_remote() contains_key_str(&self.metadata, SUFFIX_INLINE_DATA) && !self.is_remote()
} }
@@ -1501,46 +1466,6 @@ mod tests {
assert_eq!(ei.get_checksum_info(99).algorithm, HashAlgorithm::HighwayHash256S); assert_eq!(ei.get_checksum_info(99).algorithm, HashAlgorithm::HighwayHash256S);
} }
#[test]
fn object_transaction_epoch_uses_consistent_dual_internal_metadata() {
let mut fi = validation_test_fileinfo();
assert_eq!(fi.object_transaction_epoch().expect("absent epoch should decode"), None);
let epoch = Uuid::new_v4();
let epoch_text = epoch.to_string();
fi.set_object_transaction_epoch(epoch);
assert_eq!(fi.object_transaction_epoch().expect("written epoch should decode"), Some(epoch));
assert_eq!(fi.metadata.get("x-rustfs-internal-object-transaction-epoch"), Some(&epoch_text));
assert_eq!(fi.metadata.get("x-minio-internal-object-transaction-epoch"), Some(&epoch_text));
let mut rustfs_only = validation_test_fileinfo();
rustfs_only
.metadata
.insert("x-rustfs-internal-object-transaction-epoch".to_string(), epoch_text);
assert_eq!(
rustfs_only
.object_transaction_epoch()
.expect("single compatibility key should decode"),
Some(epoch)
);
let mut conflicting = fi.clone();
conflicting
.metadata
.insert("x-minio-internal-object-transaction-epoch".to_string(), Uuid::new_v4().to_string());
assert_eq!(conflicting.object_transaction_epoch(), Err(Error::FileCorrupt));
let mut malformed = validation_test_fileinfo();
malformed
.metadata
.insert("x-rustfs-internal-object-transaction-epoch".to_string(), "not-a-uuid".to_string());
assert_eq!(malformed.object_transaction_epoch(), Err(Error::FileCorrupt));
let mut nil = validation_test_fileinfo();
nil.set_object_transaction_epoch(Uuid::nil());
assert_eq!(nil.object_transaction_epoch(), Err(Error::FileCorrupt));
}
// backlog#949: distribution range/permutation validation. // backlog#949: distribution range/permutation validation.
#[test] #[test]
fn is_valid_distribution_accepts_permutation() { fn is_valid_distribution_accepts_permutation() {
+97 -156
View File
@@ -206,42 +206,6 @@ fn persist_reset_statuses(meta_sys: &mut HashMap<String, Vec<u8>>, reset_statuse
} }
} }
pub fn parse_replication_timestamp(value: &str) -> Option<OffsetDateTime> {
const DISPLAY_FORMAT: &[time::format_description::BorrowedFormatItem<'_>] = time::macros::format_description!(
"[year sign:automatic]-[month]-[day] [hour padding:none]:[minute]:[second].[subsecond] [offset_hour sign:mandatory]:[offset_minute]:[offset_second]"
);
OffsetDateTime::parse(value, &Rfc3339)
.or_else(|_| OffsetDateTime::parse(value, DISPLAY_FORMAT))
.ok()
}
fn format_replication_timestamp(value: Option<OffsetDateTime>) -> String {
let value = value.unwrap_or(OffsetDateTime::UNIX_EPOCH);
value
.to_offset(time::UtcOffset::UTC)
.format(&Rfc3339)
.unwrap_or_else(|_| value.to_string())
}
fn persist_delete_marker_replication_state(meta_sys: &mut HashMap<String, Vec<u8>>, state: &ReplicationState) {
if !state.replica_status.is_empty() {
insert_bytes(meta_sys, SUFFIX_REPLICA_STATUS, state.replica_status.as_str().as_bytes().to_vec());
insert_bytes(
meta_sys,
SUFFIX_REPLICA_TIMESTAMP,
format_replication_timestamp(state.replica_timestamp).into_bytes(),
);
}
if let Some(status) = state.replication_status_internal.as_ref().filter(|status| !status.is_empty()) {
insert_bytes(meta_sys, SUFFIX_REPLICATION_STATUS, status.as_bytes().to_vec());
insert_bytes(
meta_sys,
SUFFIX_REPLICATION_TIMESTAMP,
format_replication_timestamp(state.replication_timestamp).into_bytes(),
);
}
}
#[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)] #[derive(Clone, Debug, Default, PartialEq, Serialize, Deserialize)]
pub struct FileMeta { pub struct FileMeta {
pub versions: Vec<FileMetaShallowVersion>, pub versions: Vec<FileMetaShallowVersion>,
@@ -249,13 +213,6 @@ pub struct FileMeta {
pub meta_ver: u8, pub meta_ver: u8,
} }
struct FileInfoDecodeOptions {
read_data: bool,
include_free_versions: bool,
all_parts: bool,
include_part_checksums: bool,
}
impl FileMeta { impl FileMeta {
pub fn new() -> Self { pub fn new() -> Self {
Self { Self {
@@ -554,8 +511,53 @@ impl FileMeta {
} }
if fi.deleted { if fi.deleted {
if let (Some(delete_marker), Some(state)) = (ventry.delete_marker.as_mut(), fi.replication_state_internal.as_ref()) { if !fi.delete_marker_replication_status().is_empty()
persist_delete_marker_replication_state(&mut delete_marker.meta_sys, state); && let Some(delete_marker) = ventry.delete_marker.as_mut()
{
if fi.delete_marker_replication_status() == ReplicationStatusType::Replica {
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICA_STATUS,
fi.replication_state_internal
.as_ref()
.map(|v| v.replica_status.clone())
.unwrap_or_default()
.as_str()
.as_bytes()
.to_vec(),
);
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICA_TIMESTAMP,
fi.replication_state_internal
.as_ref()
.map(|v| v.replica_timestamp.unwrap_or(OffsetDateTime::UNIX_EPOCH).to_string())
.unwrap_or_default()
.as_bytes()
.to_vec(),
);
} else {
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICATION_STATUS,
fi.replication_state_internal
.as_ref()
.map(|v| v.replication_status_internal.clone().unwrap_or_default())
.unwrap_or_default()
.as_bytes()
.to_vec(),
);
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICATION_TIMESTAMP,
fi.replication_state_internal
.as_ref()
.map(|v| v.replication_timestamp.unwrap_or(OffsetDateTime::UNIX_EPOCH).to_string())
.unwrap_or_default()
.as_bytes()
.to_vec(),
);
}
} }
if !fi.version_purge_status().is_empty() if !fi.version_purge_status().is_empty()
@@ -607,8 +609,51 @@ impl FileMeta {
} }
if let Some(delete_marker) = v.delete_marker.as_mut() { if let Some(delete_marker) = v.delete_marker.as_mut() {
if let Some(state) = fi.replication_state_internal.as_ref() { if !fi.delete_marker_replication_status().is_empty() {
persist_delete_marker_replication_state(&mut delete_marker.meta_sys, state); if fi.delete_marker_replication_status() == ReplicationStatusType::Replica {
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICA_STATUS,
fi.replication_state_internal
.as_ref()
.map(|v| v.replica_status.clone())
.unwrap_or_default()
.as_str()
.as_bytes()
.to_vec(),
);
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICA_TIMESTAMP,
fi.replication_state_internal
.as_ref()
.map(|v| v.replica_timestamp.unwrap_or(OffsetDateTime::UNIX_EPOCH).to_string())
.unwrap_or_default()
.as_bytes()
.to_vec(),
);
} else {
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICATION_STATUS,
fi.replication_state_internal
.as_ref()
.map(|v| v.replication_status_internal.clone().unwrap_or_default())
.unwrap_or_default()
.as_bytes()
.to_vec(),
);
insert_bytes(
&mut delete_marker.meta_sys,
SUFFIX_REPLICATION_TIMESTAMP,
fi.replication_state_internal
.as_ref()
.map(|v| v.replication_timestamp.unwrap_or(OffsetDateTime::UNIX_EPOCH).to_string())
.unwrap_or_default()
.as_bytes()
.to_vec(),
);
}
} }
if let Some(state) = fi.replication_state_internal.as_ref() { if let Some(state) = fi.replication_state_internal.as_ref() {
@@ -728,47 +773,6 @@ impl FileMeta {
read_data: bool, read_data: bool,
include_free_versions: bool, include_free_versions: bool,
all_parts: bool, all_parts: bool,
) -> Result<FileInfo> {
self.to_fileinfo_with_part_checksums(
volume,
path,
version_id,
FileInfoDecodeOptions {
read_data,
include_free_versions,
all_parts,
include_part_checksums: true,
},
)
}
pub fn into_fileinfo_without_part_checksums(
&self,
volume: &str,
path: &str,
version_id: &str,
read_data: bool,
include_free_versions: bool,
) -> Result<FileInfo> {
self.to_fileinfo_with_part_checksums(
volume,
path,
version_id,
FileInfoDecodeOptions {
read_data,
include_free_versions,
all_parts: true,
include_part_checksums: false,
},
)
}
fn to_fileinfo_with_part_checksums(
&self,
volume: &str,
path: &str,
version_id: &str,
opts: FileInfoDecodeOptions,
) -> Result<FileInfo> { ) -> Result<FileInfo> {
let vid = { let vid = {
if !version_id.is_empty() { if !version_id.is_empty() {
@@ -791,7 +795,7 @@ impl FileMeta {
if header.free_version() { if header.free_version() {
non_free_versions -= 1; non_free_versions -= 1;
if opts.include_free_versions if include_free_versions
&& found_free_version.is_none() && found_free_version.is_none()
&& let Ok(found_free_fi) = ver.parse_version_meta() && let Ok(found_free_fi) = ver.parse_version_meta()
&& found_free_fi.version_type != VersionType::Invalid && found_free_fi.version_type != VersionType::Invalid
@@ -802,8 +806,7 @@ impl FileMeta {
// Known side effect: if a disk holds only free versions and they are // Known side effect: if a disk holds only free versions and they are
// corrupt, `into_fileinfo` falls through to `FileNotFound` (not // corrupt, `into_fileinfo` falls through to `FileNotFound` (not
// `FileCorrupt`), so that disk is not enqueued for heal. // `FileCorrupt`), so that disk is not enqueued for heal.
match found_free_fi.to_fileinfo_with_part_checksums(volume, path, opts.all_parts, opts.include_part_checksums) match found_free_fi.into_fileinfo(volume, path, all_parts) {
{
Ok(mut free_fi) => { Ok(mut free_fi) => {
free_fi.is_latest = true; free_fi.is_latest = true;
found_free_version = Some(free_fi); found_free_version = Some(free_fi);
@@ -831,14 +834,14 @@ impl FileMeta {
found = true; found = true;
let mut fi = ver.to_fileinfo_with_part_checksums(volume, path, opts.all_parts, opts.include_part_checksums)?; let mut fi = ver.into_fileinfo(volume, path, all_parts)?;
fi.is_latest = is_latest; fi.is_latest = is_latest;
if let Some(_d) = succ_mod_time { if let Some(_d) = succ_mod_time {
fi.successor_mod_time = succ_mod_time; fi.successor_mod_time = succ_mod_time;
} }
if opts.read_data && fi.inline_data() { if read_data && fi.inline_data() {
fi.data = self.find_inline_data_for_version(fi.version_id)?.map(bytes::Bytes::from); fi.data = self.find_inline_data_for_version(fi.version_id)?.map(bytes::Bytes::from);
} }
@@ -847,7 +850,7 @@ impl FileMeta {
if !found { if !found {
if version_id.is_empty() { if version_id.is_empty() {
if opts.include_free_versions if include_free_versions
&& non_free_versions == 0 && non_free_versions == 0
&& let Some(free_version) = found_free_version && let Some(free_version) = found_free_version
{ {
@@ -1534,68 +1537,6 @@ mod test {
assert_eq!(meta_sys2.len(), 2, "must not create a double-prefixed key"); assert_eq!(meta_sys2.len(), 2, "must not create a double-prefixed key");
} }
#[test]
fn persist_delete_marker_replication_state_keeps_replica_and_target_statuses() {
let replica_timestamp = OffsetDateTime::UNIX_EPOCH + time::Duration::SECOND;
let replication_timestamp = replica_timestamp + time::Duration::SECOND;
let state = ReplicationState {
replica_status: ReplicationStatusType::Replica,
replica_timestamp: Some(replica_timestamp),
replication_status_internal: Some("arn:target=COMPLETED;".to_string()),
replication_timestamp: Some(replication_timestamp),
..Default::default()
};
let mut meta_sys = HashMap::new();
let replica_timestamp_string = replica_timestamp
.format(&Rfc3339)
.expect("timestamp should format as RFC3339");
let replication_timestamp_string = replication_timestamp
.format(&Rfc3339)
.expect("timestamp should format as RFC3339");
persist_delete_marker_replication_state(&mut meta_sys, &state);
assert_eq!(
rustfs_utils::http::get_bytes(&meta_sys, SUFFIX_REPLICA_STATUS).as_deref(),
Some(b"REPLICA".as_slice())
);
assert_eq!(
rustfs_utils::http::get_bytes(&meta_sys, SUFFIX_REPLICA_TIMESTAMP).as_deref(),
Some(replica_timestamp_string.as_bytes())
);
assert_eq!(
rustfs_utils::http::get_bytes(&meta_sys, SUFFIX_REPLICATION_STATUS).as_deref(),
Some(b"arn:target=COMPLETED;".as_slice())
);
assert_eq!(
rustfs_utils::http::get_bytes(&meta_sys, SUFFIX_REPLICATION_TIMESTAMP).as_deref(),
Some(replication_timestamp_string.as_bytes())
);
}
#[test]
fn persist_delete_marker_replication_timestamp_normalizes_second_offset_to_utc() {
let timestamp = OffsetDateTime::UNIX_EPOCH.to_offset(time::UtcOffset::from_hms(5, 30, 15).expect("valid offset"));
assert_eq!(parse_replication_timestamp(&timestamp.to_string()), Some(timestamp));
let state = ReplicationState {
replica_status: ReplicationStatusType::Replica,
replica_timestamp: Some(timestamp),
replication_status_internal: Some("arn:target=COMPLETED;".to_string()),
replication_timestamp: Some(timestamp),
..Default::default()
};
let mut meta_sys = HashMap::new();
persist_delete_marker_replication_state(&mut meta_sys, &state);
for suffix in [SUFFIX_REPLICA_TIMESTAMP, SUFFIX_REPLICATION_TIMESTAMP] {
let persisted = rustfs_utils::http::get_bytes(&meta_sys, suffix).expect("timestamp must be persisted");
let persisted = std::str::from_utf8(&persisted).expect("timestamp must be UTF-8");
assert_eq!(parse_replication_timestamp(persisted), Some(timestamp));
assert_ne!(persisted, OffsetDateTime::UNIX_EPOCH.to_string());
}
}
/// Regression test for rustfs/rustfs#2715: a corrupted version count in /// Regression test for rustfs/rustfs#2715: a corrupted version count in
/// xl.meta must yield a decode error instead of sizing a huge allocation /// xl.meta must yield a decode error instead of sizing a huge allocation
/// from the bogus count (which aborts the whole process). /// from the bogus count (which aborts the whole process).
+120 -603
View File
@@ -29,12 +29,11 @@ use super::*;
use crate::{ChecksumInfo, TransitionVersionState}; use crate::{ChecksumInfo, TransitionVersionState};
use rustfs_utils::HashAlgorithm; use rustfs_utils::HashAlgorithm;
use rustfs_utils::http::{ use rustfs_utils::http::{
RUSTFS_INTERNAL_PREFIX, SUFFIX_CRC, SUFFIX_FREE_VERSION, SUFFIX_INLINE_DATA, SUFFIX_PART_CHECKSUMS, SUFFIX_PURGESTATUS, RUSTFS_INTERNAL_PREFIX, SUFFIX_CRC, SUFFIX_FREE_VERSION, SUFFIX_INLINE_DATA, SUFFIX_PURGESTATUS, SUFFIX_TIER_FV_ID,
SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX, SUFFIX_REPLICATION_RESET_ARN_PREFIX, SUFFIX_TIER_FV_ID,
SUFFIX_TIER_FV_MARKER, SUFFIX_TRANSITION_STATUS, SUFFIX_TRANSITION_TIER, SUFFIX_TRANSITION_TIER_DESTINATION_ID, SUFFIX_TIER_FV_MARKER, SUFFIX_TRANSITION_STATUS, SUFFIX_TRANSITION_TIER, SUFFIX_TRANSITION_TIER_DESTINATION_ID,
SUFFIX_TRANSITIONED_OBJECTNAME, SUFFIX_TRANSITIONED_VERSION_ID, SUFFIX_TRANSITIONED_VERSION_STATE, contains_key_bytes, SUFFIX_TRANSITIONED_OBJECTNAME, SUFFIX_TRANSITIONED_VERSION_ID, SUFFIX_TRANSITIONED_VERSION_STATE, contains_key_bytes,
get_bytes, get_consistent_bytes, get_str, has_internal_suffix, insert_bytes, is_internal_key, remove_bytes, get_bytes, get_consistent_bytes, get_str, has_internal_suffix, insert_bytes, is_internal_key, remove_bytes,
strip_internal_prefix, strip_internal_prefix_preserving_case, target_delete_marker_versions, strip_internal_prefix, target_delete_marker_versions,
}; };
const MSGPACK_EXT8: u8 = 0xc7; const MSGPACK_EXT8: u8 = 0xc7;
@@ -259,181 +258,63 @@ fn parse_legacy_uuid_bytes(bytes: &[u8], field: &str) -> Result<Option<Uuid>> {
/// Legacy RustFS writes used 16 raw UUID bytes. New writes and MinIO-migrated /// Legacy RustFS writes used 16 raw UUID bytes. New writes and MinIO-migrated
/// records use the provider's exact UTF-8 version text. Empty, nil UUID, and /// records use the provider's exact UTF-8 version text. Empty, nil UUID, and
/// malformed bytes are not usable remote versions. /// malformed bytes are not usable remote versions.
fn transition_version_state_from_bytes(value: Option<&[u8]>) -> Result<TransitionVersionState> { fn transitioned_version_from_meta_sys(meta_sys: &HashMap<String, Vec<u8>>) -> Result<Option<String>> {
let Some(value) = value else { if !contains_key_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_ID) {
return Ok(TransitionVersionState::Unknown); return Ok(None);
}
let Some(value) = get_consistent_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_ID) else {
return Ok(None);
}; };
match value { let value = value.to_vec();
b"known-disabled" => Ok(TransitionVersionState::KnownDisabled),
b"suspended-null" => Ok(TransitionVersionState::SuspendedNull),
b"exact" => Ok(TransitionVersionState::Exact),
b"unknown" => Ok(TransitionVersionState::Unknown),
_ => Err(Error::FileCorrupt),
}
}
fn transitioned_version_from_bytes(value: Option<&[u8]>, state: TransitionVersionState) -> Option<String> {
let value = value?;
if value.is_empty() { if value.is_empty() {
return None; return Ok(None);
} }
if state == TransitionVersionState::Unknown if let Ok(id) = Uuid::from_slice(&value) {
&& let Ok(id) = Uuid::from_slice(value) return Ok((!id.is_nil()).then(|| id.to_string()));
{
return (!id.is_nil()).then(|| id.to_string());
} }
let Ok(value) = std::str::from_utf8(value) else { let Ok(value) = String::from_utf8(value) else {
return None; return Ok(None);
}; };
if value.is_empty() if value.is_empty()
|| value.len() > MAX_TRANSITION_VERSION_LEN || value.len() > MAX_TRANSITION_VERSION_LEN
|| value.chars().any(char::is_control) || value.chars().any(char::is_control)
|| Uuid::parse_str(value).is_ok_and(|id| id.is_nil()) || Uuid::parse_str(&value).is_ok_and(|id| id.is_nil())
{ {
None Ok(None)
} else { } else {
Some(value.to_string()) Ok(Some(value))
} }
} }
fn validate_transition_version_state(state: TransitionVersionState, version: Option<&str>) -> Result<()> { fn transition_version_state_from_meta_sys(
meta_sys: &HashMap<String, Vec<u8>>,
version: Option<&str>,
) -> Result<TransitionVersionState> {
if !contains_key_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE) {
return Ok(TransitionVersionState::Unknown);
}
let value = get_consistent_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE).ok_or(Error::FileCorrupt)?;
let state = match value {
b"known-disabled" => TransitionVersionState::KnownDisabled,
b"suspended-null" => TransitionVersionState::SuspendedNull,
b"exact" => TransitionVersionState::Exact,
b"unknown" => TransitionVersionState::Unknown,
_ => return Err(Error::FileCorrupt),
};
let valid = match state { let valid = match state {
TransitionVersionState::Unknown | TransitionVersionState::KnownDisabled => version.is_none(), TransitionVersionState::Unknown | TransitionVersionState::KnownDisabled => version.is_none(),
TransitionVersionState::SuspendedNull => version == Some("null"), TransitionVersionState::SuspendedNull => version == Some("null"),
TransitionVersionState::Exact => version.is_some_and(|value| value != "null"), TransitionVersionState::Exact => version.is_some_and(|value| value != "null"),
}; };
valid.then_some(()).ok_or(Error::FileCorrupt) valid.then_some(state).ok_or(Error::FileCorrupt)
} }
#[derive(Default)] fn transition_version_state_bytes(state: TransitionVersionState) -> &'static [u8] {
struct DerivedInternalMetadata<'a> { match state {
checksum: Option<&'a [u8]>, TransitionVersionState::Unknown => b"unknown",
part_checksums: Option<&'a [u8]>, TransitionVersionState::KnownDisabled => b"known-disabled",
transition_status: Option<&'a [u8]>, TransitionVersionState::SuspendedNull => b"suspended-null",
transitioned_object: Option<&'a [u8]>, TransitionVersionState::Exact => b"exact",
transitioned_version: Option<&'a [u8]>,
transitioned_version_state: Option<&'a [u8]>,
transition_tier: Option<&'a [u8]>,
}
impl<'a> DerivedInternalMetadata<'a> {
fn from_meta_sys(meta_sys: &'a HashMap<String, Vec<u8>>) -> Result<Self> {
let mut canonical = Self::default();
let mut legacy = Self::default();
for (key, value) in meta_sys {
let Some(suffix) = rustfs_utils::http::strip_internal_prefix_preserving_case(key) else {
continue;
};
let (canonical_slot, legacy_slot, expected_suffix) = if suffix.eq_ignore_ascii_case(SUFFIX_CRC) {
(&mut canonical.checksum, &mut legacy.checksum, SUFFIX_CRC)
} else if suffix.eq_ignore_ascii_case(SUFFIX_PART_CHECKSUMS) {
(&mut canonical.part_checksums, &mut legacy.part_checksums, SUFFIX_PART_CHECKSUMS)
} else if suffix.eq_ignore_ascii_case(SUFFIX_TRANSITION_STATUS) {
(&mut canonical.transition_status, &mut legacy.transition_status, SUFFIX_TRANSITION_STATUS)
} else if suffix.eq_ignore_ascii_case(SUFFIX_TRANSITIONED_OBJECTNAME) {
(
&mut canonical.transitioned_object,
&mut legacy.transitioned_object,
SUFFIX_TRANSITIONED_OBJECTNAME,
)
} else if suffix.eq_ignore_ascii_case(SUFFIX_TRANSITIONED_VERSION_ID) {
(
&mut canonical.transitioned_version,
&mut legacy.transitioned_version,
SUFFIX_TRANSITIONED_VERSION_ID,
)
} else if suffix.eq_ignore_ascii_case(SUFFIX_TRANSITIONED_VERSION_STATE) {
(
&mut canonical.transitioned_version_state,
&mut legacy.transitioned_version_state,
SUFFIX_TRANSITIONED_VERSION_STATE,
)
} else if suffix.eq_ignore_ascii_case(SUFFIX_TRANSITION_TIER) {
(&mut canonical.transition_tier, &mut legacy.transition_tier, SUFFIX_TRANSITION_TIER)
} else {
continue;
};
let slot = if suffix == expected_suffix
&& (key.starts_with(RUSTFS_INTERNAL_PREFIX) || key.starts_with(rustfs_utils::http::MINIO_INTERNAL_PREFIX))
{
canonical_slot
} else {
legacy_slot
};
if slot.is_some_and(|current| current != value.as_slice()) {
return Err(Error::FileCorrupt);
}
*slot = Some(value.as_slice());
}
Ok(Self {
checksum: canonical.checksum.or(legacy.checksum),
part_checksums: canonical.part_checksums.or(legacy.part_checksums),
transition_status: canonical.transition_status.or(legacy.transition_status),
transitioned_object: canonical.transitioned_object.or(legacy.transitioned_object),
transitioned_version: canonical.transitioned_version.or(legacy.transitioned_version),
transitioned_version_state: canonical.transitioned_version_state.or(legacy.transitioned_version_state),
transition_tier: canonical.transition_tier.or(legacy.transition_tier),
})
}
}
struct UniquePartChecksums(HashMap<String, String>);
impl<'de> serde::Deserialize<'de> for UniquePartChecksums {
fn deserialize<D>(deserializer: D) -> std::result::Result<Self, D::Error>
where
D: serde::Deserializer<'de>,
{
struct UniquePartChecksumsVisitor;
impl<'de> serde::de::Visitor<'de> for UniquePartChecksumsVisitor {
type Value = UniquePartChecksums;
fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
formatter.write_str("an array of unique checksum name and value pairs")
}
fn visit_seq<A>(self, mut seq: A) -> std::result::Result<Self::Value, A::Error>
where
A: serde::de::SeqAccess<'de>,
{
let mut checksums = HashMap::with_capacity(seq.size_hint().unwrap_or_default());
while let Some((key, value)) = seq.next_element::<(String, String)>()? {
if checksums.insert(key, value).is_some() {
return Err(serde::de::Error::custom("duplicate part checksum name"));
}
}
Ok(UniquePartChecksums(checksums))
}
}
deserializer.deserialize_seq(UniquePartChecksumsVisitor)
}
}
impl FileInfo {
pub fn hydrate_data_movement_part_checksums(&mut self) -> Result<()> {
let present = self
.metadata
.keys()
.any(|key| has_internal_suffix(key, SUFFIX_PART_CHECKSUMS));
if !present {
return Ok(());
}
let encoded = rustfs_utils::http::get_consistent_str(&self.metadata, SUFFIX_PART_CHECKSUMS).ok_or(Error::FileCorrupt)?;
let persisted = serde_json::from_str::<Vec<(usize, UniquePartChecksums)>>(encoded).map_err(|_| Error::FileCorrupt)?;
let mut part_indices = HashMap::with_capacity(self.parts.len());
for (index, part) in self.parts.iter().enumerate() {
if part_indices.insert(part.number, index).is_some() {
return Err(Error::FileCorrupt);
}
}
for (part_number, UniquePartChecksums(checksums)) in persisted {
let index = part_indices.remove(&part_number).ok_or(Error::FileCorrupt)?;
let part = self.parts.get_mut(index).ok_or(Error::FileCorrupt)?;
part.checksums = Some(checksums);
}
Ok(())
} }
} }
@@ -441,10 +322,21 @@ fn set_transition_version_state(meta_sys: &mut HashMap<String, Vec<u8>>, state:
if state == TransitionVersionState::Unknown { if state == TransitionVersionState::Unknown {
remove_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE); remove_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE);
} else { } else {
insert_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE, state.as_str().as_bytes().to_vec()); insert_bytes(
meta_sys,
SUFFIX_TRANSITIONED_VERSION_STATE,
transition_version_state_bytes(state).to_vec(),
);
} }
} }
fn legacy_transitioned_version_id_from_meta_sys(meta_sys: &HashMap<String, Vec<u8>>) -> Option<Uuid> {
transitioned_version_from_meta_sys(meta_sys)
.ok()
.flatten()
.and_then(|value| Uuid::parse_str(&value).ok())
}
fn transitioned_version_bytes(fi: &FileInfo) -> Option<Vec<u8>> { fn transitioned_version_bytes(fi: &FileInfo) -> Option<Vec<u8>> {
fi.transition_version fi.transition_version
.as_ref() .as_ref()
@@ -556,17 +448,6 @@ impl FileMetaShallowVersion {
pub fn into_fileinfo(&self, volume: &str, path: &str, all_parts: bool) -> Result<FileInfo> { pub fn into_fileinfo(&self, volume: &str, path: &str, all_parts: bool) -> Result<FileInfo> {
self.parse_version_meta()?.into_fileinfo(volume, path, all_parts) self.parse_version_meta()?.into_fileinfo(volume, path, all_parts)
} }
pub(super) fn to_fileinfo_with_part_checksums(
&self,
volume: &str,
path: &str,
all_parts: bool,
include_part_checksums: bool,
) -> Result<FileInfo> {
self.parse_version_meta()?
.to_fileinfo_with_part_checksums(volume, path, all_parts, include_part_checksums)
}
} }
impl TryFrom<FileMetaVersion> for FileMetaShallowVersion { impl TryFrom<FileMetaVersion> for FileMetaShallowVersion {
@@ -888,16 +769,8 @@ impl FileMetaVersion {
} }
pub fn into_fileinfo(&self, volume: &str, path: &str, all_parts: bool) -> Result<FileInfo> { pub fn into_fileinfo(&self, volume: &str, path: &str, all_parts: bool) -> Result<FileInfo> {
self.to_fileinfo_with_part_checksums(volume, path, all_parts, true) // Only the Object arm carries part arrays and can fail the length guard; the
} // Legacy and Delete arms have no part arrays and stay infallible.
pub(super) fn to_fileinfo_with_part_checksums(
&self,
volume: &str,
path: &str,
all_parts: bool,
include_part_checksums: bool,
) -> Result<FileInfo> {
let mut fi = match self.version_type { let mut fi = match self.version_type {
VersionType::Invalid | VersionType::Legacy => { VersionType::Invalid | VersionType::Legacy => {
if let Some(ref legacy) = self.legacy_object { if let Some(ref legacy) = self.legacy_object {
@@ -915,14 +788,14 @@ impl FileMetaVersion {
self.object self.object
.as_ref() .as_ref()
.unwrap_or(&default_object) .unwrap_or(&default_object)
.to_fileinfo_with_part_checksums(volume, path, all_parts, include_part_checksums)? .into_fileinfo(volume, path, all_parts)?
} }
VersionType::Delete => { VersionType::Delete => {
let default_marker = MetaDeleteMarker::default(); let default_marker = MetaDeleteMarker::default();
self.delete_marker self.delete_marker
.as_ref() .as_ref()
.unwrap_or(&default_marker) .unwrap_or(&default_marker)
.into_fileinfo(volume, path, all_parts)? .into_fileinfo(volume, path, all_parts)
} }
}; };
fi.uses_legacy_checksum = self.uses_legacy_checksum; fi.uses_legacy_checksum = self.uses_legacy_checksum;
@@ -2517,18 +2390,7 @@ impl MetaObject {
} }
pub fn into_fileinfo(&self, volume: &str, path: &str, all_parts: bool) -> Result<FileInfo> { pub fn into_fileinfo(&self, volume: &str, path: &str, all_parts: bool) -> Result<FileInfo> {
self.to_fileinfo_with_part_checksums(volume, path, all_parts, true)
}
fn to_fileinfo_with_part_checksums(
&self,
volume: &str,
path: &str,
all_parts: bool,
include_part_checksums: bool,
) -> Result<FileInfo> {
let version_id = self.version_id.filter(|&vid| !vid.is_nil()); let version_id = self.version_id.filter(|&vid| !vid.is_nil());
let derived_metadata = DerivedInternalMetadata::from_meta_sys(&self.meta_sys)?;
let parts = if all_parts { let parts = if all_parts {
let n = self.part_numbers.len(); let n = self.part_numbers.len();
@@ -2612,10 +2474,7 @@ impl MetaObject {
} }
} }
let checksum = derived_metadata let checksum = get_bytes(&self.meta_sys, SUFFIX_CRC).map(Bytes::from);
.checksum
.filter(|checksum| !checksum.is_empty())
.map(Bytes::copy_from_slice);
let erasure = ErasureInfo { let erasure = ErasureInfo {
algorithm: self.erasure_algorithm.to_string(), algorithm: self.erasure_algorithm.to_string(),
@@ -2627,29 +2486,20 @@ impl MetaObject {
..Default::default() ..Default::default()
}; };
let transition_status = derived_metadata let transition_status = get_bytes(&self.meta_sys, SUFFIX_TRANSITION_STATUS)
.transition_status .map(|v| String::from_utf8_lossy(&v).to_string())
.filter(|value| !value.is_empty())
.map(|v| String::from_utf8_lossy(v).to_string())
.unwrap_or_default(); .unwrap_or_default();
let transitioned_objname = derived_metadata let transitioned_objname = get_bytes(&self.meta_sys, SUFFIX_TRANSITIONED_OBJECTNAME)
.transitioned_object .map(|v| String::from_utf8_lossy(&v).to_string())
.filter(|value| !value.is_empty())
.map(|v| String::from_utf8_lossy(v).to_string())
.unwrap_or_default(); .unwrap_or_default();
let transition_version_state = transition_version_state_from_bytes(derived_metadata.transitioned_version_state)?; let transition_version = transitioned_version_from_meta_sys(&self.meta_sys)?;
let transition_version = transitioned_version_from_bytes(derived_metadata.transitioned_version, transition_version_state); let transition_version_state = transition_version_state_from_meta_sys(&self.meta_sys, transition_version.as_deref())?;
if derived_metadata.transitioned_version_state.is_some() {
validate_transition_version_state(transition_version_state, transition_version.as_deref())?;
}
let transition_version_id = transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok()); let transition_version_id = transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok());
let transition_tier = derived_metadata let transition_tier = get_bytes(&self.meta_sys, SUFFIX_TRANSITION_TIER)
.transition_tier .map(|v| String::from_utf8_lossy(&v).to_string())
.filter(|value| !value.is_empty())
.map(|v| String::from_utf8_lossy(v).to_string())
.unwrap_or_default(); .unwrap_or_default();
let mut file_info = FileInfo { Ok(FileInfo {
version_id, version_id,
erasure, erasure,
data_dir: self.data_dir, data_dir: self.data_dir,
@@ -2669,11 +2519,7 @@ impl MetaObject {
transition_version_state, transition_version_state,
transition_tier, transition_tier,
..Default::default() ..Default::default()
}; })
if all_parts && include_part_checksums {
file_info.hydrate_data_movement_part_checksums()?;
}
Ok(file_info)
} }
pub fn set_transition(&mut self, fi: &FileInfo) { pub fn set_transition(&mut self, fi: &FileInfo) {
@@ -2864,31 +2710,39 @@ fn get_internal_replication_state(metadata: &HashMap<String, String>) -> Option<
continue; continue;
} }
if let Some(sub_key) = strip_internal_prefix_preserving_case(k) { let sub_key_opt = strip_internal_prefix(k);
if sub_key.eq_ignore_ascii_case(SUFFIX_REPLICA_TIMESTAMP) { if let Some(ref sub_key) = sub_key_opt {
has = true; match sub_key.as_str() {
rs.replica_timestamp = Some(parse_replication_timestamp(v).unwrap_or(OffsetDateTime::UNIX_EPOCH)); "replica-timestamp" => {
} else if sub_key.eq_ignore_ascii_case(SUFFIX_REPLICA_STATUS) { has = true;
has = true; rs.replica_timestamp = Some(OffsetDateTime::parse(v, &Rfc3339).unwrap_or(OffsetDateTime::UNIX_EPOCH));
rs.replica_status = ReplicationStatusType::from(v.as_str()); }
} else if sub_key.eq_ignore_ascii_case(SUFFIX_REPLICATION_TIMESTAMP) { "replica-status" => {
has = true; has = true;
rs.replication_timestamp = Some(parse_replication_timestamp(v).unwrap_or(OffsetDateTime::UNIX_EPOCH)) rs.replica_status = ReplicationStatusType::from(v.as_str());
} else if sub_key.eq_ignore_ascii_case(SUFFIX_REPLICATION_STATUS) { }
has = true; "replication-timestamp" => {
rs.replication_status_internal = Some(v.clone()); has = true;
rs.targets = replication_statuses_map(v.as_str()); rs.replication_timestamp = Some(OffsetDateTime::parse(v, &Rfc3339).unwrap_or(OffsetDateTime::UNIX_EPOCH))
} else if let Some(arn) = rustfs_utils::http::internal_key_strip_suffix_prefix(k, SUFFIX_REPLICATION_RESET_ARN_PREFIX) }
{ "replication-status" => {
has = true; has = true;
// Store the canonical full-header key so the map matches rs.replication_status_internal = Some(v.clone());
// the key `target_reset_header()` produces on the rs.targets = replication_statuses_map(v.as_str());
// write/lookup side. Storing the bare ARN keyed the map }
// inconsistently (bare on read, full on write), which _ => {
// could drop reset state across merge/reflatten cycles if let Some(arn) = sub_key.strip_prefix("replication-reset-") {
// (backlog#799 B16). has = true;
rs.reset_statuses_map // Store the canonical full-header key so the map matches
.insert(crate::replication::target_reset_header(&arn), v.clone()); // the key `target_reset_header()` produces on the
// write/lookup side. Storing the bare ARN keyed the map
// inconsistently (bare on read, full on write), which
// could drop reset state across merge/reflatten cycles
// (backlog#799 B16).
rs.reset_statuses_map
.insert(crate::replication::target_reset_header(arn), v.clone());
}
}
} }
} }
} }
@@ -2932,7 +2786,7 @@ impl MetaDeleteMarker {
contains_key_bytes(&self.meta_sys, SUFFIX_FREE_VERSION) contains_key_bytes(&self.meta_sys, SUFFIX_FREE_VERSION)
} }
pub fn into_fileinfo(&self, volume: &str, path: &str, _all_parts: bool) -> Result<FileInfo> { pub fn into_fileinfo(&self, volume: &str, path: &str, _all_parts: bool) -> FileInfo {
let metadata = self let metadata = self
.meta_sys .meta_sys
.clone() .clone()
@@ -2954,27 +2808,22 @@ impl MetaDeleteMarker {
if self.free_version() { if self.free_version() {
fi.set_tier_free_version(); fi.set_tier_free_version();
let derived_metadata = DerivedInternalMetadata::from_meta_sys(&self.meta_sys)?; fi.transition_tier = get_bytes(&self.meta_sys, SUFFIX_TRANSITION_TIER)
fi.transition_tier = derived_metadata .map(|v| String::from_utf8_lossy(&v).to_string())
.transition_tier
.filter(|value| !value.is_empty())
.map(|value| String::from_utf8_lossy(value).to_string())
.unwrap_or_default(); .unwrap_or_default();
fi.transitioned_objname = derived_metadata
.transitioned_object fi.transitioned_objname = get_bytes(&self.meta_sys, SUFFIX_TRANSITIONED_OBJECTNAME)
.filter(|value| !value.is_empty()) .map(|v| String::from_utf8_lossy(&v).to_string())
.map(|value| String::from_utf8_lossy(value).to_string())
.unwrap_or_default(); .unwrap_or_default();
fi.transition_version_state = transition_version_state_from_bytes(derived_metadata.transitioned_version_state)?;
fi.transition_version = fi.transition_version = transitioned_version_from_meta_sys(&self.meta_sys).ok().flatten();
transitioned_version_from_bytes(derived_metadata.transitioned_version, fi.transition_version_state); fi.transition_version_id = legacy_transitioned_version_id_from_meta_sys(&self.meta_sys);
fi.transition_version_id = fi.transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok()); fi.transition_version_state =
if derived_metadata.transitioned_version_state.is_some() { transition_version_state_from_meta_sys(&self.meta_sys, fi.transition_version.as_deref())
validate_transition_version_state(fi.transition_version_state, fi.transition_version.as_deref())?; .unwrap_or(TransitionVersionState::Unknown);
}
} }
Ok(fi) fi
} }
pub fn encode_to<W: std::io::Write>(&self, wr: &mut W) -> Result<()> { pub fn encode_to<W: std::io::Write>(&self, wr: &mut W) -> Result<()> {
@@ -3092,13 +2941,6 @@ impl From<FileInfo> for MetaDeleteMarker {
if !is_internal_key(key) || is_skip_meta_key(key) { if !is_internal_key(key) || is_skip_meta_key(key) {
continue; continue;
} }
if rustfs_utils::http::internal_key_strip_suffix_prefix(key, SUFFIX_REPLICATION_RESET_ARN_PREFIX).is_some()
|| rustfs_utils::http::internal_key_strip_suffix_prefix(key, SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX)
.is_some()
{
meta_sys.insert(key.clone(), metadata_value.as_bytes().to_vec());
continue;
}
let Some(suffix) = strip_internal_prefix(key) else { let Some(suffix) = strip_internal_prefix(key) else {
continue; continue;
}; };
@@ -3140,11 +2982,6 @@ impl From<FileInfo> for MetaDeleteMarker {
if !value.transition_tier.is_empty() { if !value.transition_tier.is_empty() {
insert_bytes(&mut meta_sys, SUFFIX_TRANSITION_TIER, value.transition_tier.as_bytes().to_vec()); insert_bytes(&mut meta_sys, SUFFIX_TRANSITION_TIER, value.transition_tier.as_bytes().to_vec());
} }
if let Some(state) = value.replication_state_internal.as_ref() {
persist_delete_marker_replication_state(&mut meta_sys, state);
persist_reset_statuses(&mut meta_sys, &state.reset_statuses_map);
persist_target_delete_marker_versions(&mut meta_sys, &state.target_delete_marker_version_ids, &value.metadata);
}
Self { Self {
version_id: value.version_id, version_id: value.version_id,
mod_time: value.mod_time, mod_time: value.mod_time,
@@ -3441,7 +3278,6 @@ pub fn file_info_from_raw(
FileInfoOpts { FileInfoOpts {
data: read_data, data: read_data,
include_free_versions, include_free_versions,
include_part_checksums: true,
}, },
) )
} }
@@ -3449,7 +3285,6 @@ pub fn file_info_from_raw(
pub struct FileInfoOpts { pub struct FileInfoOpts {
pub data: bool, pub data: bool,
pub include_free_versions: bool, pub include_free_versions: bool,
pub include_part_checksums: bool,
} }
pub fn get_file_info(buf: &[u8], volume: &str, path: &str, version_id: &str, opts: FileInfoOpts) -> Result<FileInfo> { pub fn get_file_info(buf: &[u8], volume: &str, path: &str, version_id: &str, opts: FileInfoOpts) -> Result<FileInfo> {
@@ -3474,11 +3309,7 @@ pub fn get_file_info(buf: &[u8], volume: &str, path: &str, version_id: &str, opt
}); });
} }
let fi = if opts.include_part_checksums { let fi = meta.into_fileinfo(volume, path, version_id, opts.data, opts.include_free_versions, true)?;
meta.into_fileinfo(volume, path, version_id, opts.data, opts.include_free_versions, true)?
} else {
meta.into_fileinfo_without_part_checksums(volume, path, version_id, opts.data, opts.include_free_versions)?
};
Ok(fi) Ok(fi)
} }
@@ -3724,56 +3555,6 @@ mod tests {
assert!(!converted.meta_sys.contains_key("content-type")); assert!(!converted.meta_sys.contains_key("content-type"));
} }
#[test]
fn delete_marker_conversion_does_not_lowercase_dynamic_replication_targets() {
let arn = "arn:rustfs:replication:us-east-1:TenantA:bucket";
let reset_suffix = format!("{SUFFIX_REPLICATION_RESET_ARN_PREFIX}{arn}");
let version_suffix = format!("{SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX}{arn}");
let mut marker = FileInfo::default();
marker.metadata.insert(
format!("{}{reset_suffix}", rustfs_utils::http::MINIO_INTERNAL_PREFIX),
"2026-08-12T00:00:00Z;COMPLETED".to_string(),
);
marker.metadata.insert(
format!("{}{version_suffix}", rustfs_utils::http::MINIO_INTERNAL_PREFIX),
"remote-version".to_string(),
);
marker.replication_state_internal = get_internal_replication_state(&marker.metadata);
let converted = MetaDeleteMarker::from(marker);
assert!(converted.meta_sys.keys().any(|key| key.ends_with(&reset_suffix)));
assert!(converted.meta_sys.keys().any(|key| key.ends_with(&version_suffix)));
assert!(!converted.meta_sys.keys().any(|key| key.contains("tenanta")));
let mut conflicting = FileInfo::default();
conflicting
.metadata
.insert(format!("{RUSTFS_INTERNAL_PREFIX}{version_suffix}"), "remote-version-a".to_string());
conflicting.metadata.insert(
format!("{}{version_suffix}", rustfs_utils::http::MINIO_INTERNAL_PREFIX),
"remote-version-b".to_string(),
);
conflicting.replication_state_internal = get_internal_replication_state(&conflicting.metadata);
assert!(
conflicting
.replication_state_internal
.as_ref()
.is_some_and(|state| state.target_delete_marker_version_ids_corrupt)
);
let roundtrip = MetaDeleteMarker::from(conflicting)
.into_fileinfo("bucket", "object", false)
.expect("dynamic replication aliases should remain decodable");
assert!(
roundtrip
.replication_state_internal
.as_ref()
.is_some_and(|state| state.target_delete_marker_version_ids_corrupt),
"conflicting dynamic aliases must remain corrupt across persistence"
);
}
#[derive(Serialize)] #[derive(Serialize)]
enum LegacyDeleteVersionTypeFixture { enum LegacyDeleteVersionTypeFixture {
#[serde(rename = "DeleteMarker")] #[serde(rename = "DeleteMarker")]
@@ -3882,138 +3663,6 @@ mod tests {
assert!(matches!(res, Err(Error::FileCorrupt)), "short part_sizes must map to FileCorrupt"); assert!(matches!(res, Err(Error::FileCorrupt)), "short part_sizes must map to FileCorrupt");
} }
#[test]
fn into_fileinfo_rejects_conflicting_derived_internal_aliases() {
for suffix in [
SUFFIX_CRC,
SUFFIX_TRANSITION_STATUS,
SUFFIX_TRANSITIONED_OBJECTNAME,
SUFFIX_TRANSITIONED_VERSION_ID,
SUFFIX_TRANSITIONED_VERSION_STATE,
SUFFIX_TRANSITION_TIER,
] {
let mut meta_sys = HashMap::from([(format!("{RUSTFS_INTERNAL_PREFIX}{suffix}"), vec![0xff, 1])]);
meta_sys.insert(format!("{}{suffix}", rustfs_utils::http::MINIO_INTERNAL_PREFIX), vec![0xfe, 2]);
let object = MetaObject {
meta_sys,
..Default::default()
};
assert_eq!(
object
.into_fileinfo("bucket", "key", false)
.expect_err("conflicting aliases must fail closed"),
Error::FileCorrupt,
"suffix {suffix}"
);
}
}
#[test]
fn into_fileinfo_recovers_noncanonical_binary_checksum_alias() {
let checksum = vec![0xff, 0x00, 0x80, 0x01];
let object = MetaObject {
meta_sys: HashMap::from([("X-Minio-Internal-crc".to_string(), checksum.clone())]),
..Default::default()
};
let file_info = object
.into_fileinfo("bucket", "key", false)
.expect("a single legacy checksum alias should remain readable");
assert_eq!(file_info.checksum.as_deref(), Some(checksum.as_slice()));
}
#[test]
fn into_fileinfo_prefers_canonical_rewrite_over_stale_mixed_case_alias() {
let checksum = vec![0xff, 0x00, 0x80, 0x01];
let mut object = MetaObject {
meta_sys: HashMap::from([("X-Minio-Internal-crc".to_string(), b"stale".to_vec())]),
..Default::default()
};
insert_bytes(&mut object.meta_sys, SUFFIX_CRC, checksum.clone());
let file_info = object
.into_fileinfo("bucket", "key", false)
.expect("canonical rewrites should supersede legacy mixed-case aliases");
assert_eq!(file_info.checksum.as_deref(), Some(checksum.as_slice()));
}
#[test]
fn into_fileinfo_recovers_data_movement_part_checksums() {
let mut object = object_with_parts(vec![1], vec![16], vec![16]);
insert_bytes(&mut object.meta_sys, SUFFIX_PART_CHECKSUMS, br#"[[1,[["CRC32C","AAAAAA=="]]]]"#.to_vec());
let file_info = object
.into_fileinfo("bucket", "key", true)
.expect("data movement part checksums should decode");
assert_eq!(
file_info.parts[0]
.checksums
.as_ref()
.and_then(|checksums| checksums.get("CRC32C"))
.map(String::as_str),
Some("AAAAAA==")
);
let mut deferred = object
.to_fileinfo_with_part_checksums("bucket", "key", true, false)
.expect("quorum candidates should retain raw checksum metadata");
assert!(deferred.parts[0].checksums.is_none());
deferred
.hydrate_data_movement_part_checksums()
.expect("the selected candidate should hydrate checksums once");
assert_eq!(deferred.parts[0].checksums, file_info.parts[0].checksums);
insert_bytes(&mut object.meta_sys, SUFFIX_PART_CHECKSUMS, b"not-json".to_vec());
assert_eq!(
object
.into_fileinfo("bucket", "key", true)
.expect_err("malformed data movement part checksums must fail closed"),
Error::FileCorrupt
);
for encoded in [
br#"[[1,[["CRC32C","AAAAAA=="]]],[1,[["CRC32C","BBBBBB=="]]]]"#.as_slice(),
br#"[[1,[["CRC32C","AAAAAA=="],["CRC32C","BBBBBB=="]]]]"#.as_slice(),
] {
insert_bytes(&mut object.meta_sys, SUFFIX_PART_CHECKSUMS, encoded.to_vec());
assert_eq!(
object
.into_fileinfo("bucket", "key", true)
.expect_err("duplicate part checksum keys must fail closed"),
Error::FileCorrupt
);
}
insert_bytes(&mut object.meta_sys, SUFFIX_PART_CHECKSUMS, br#"[[2,[["CRC32C","AAAAAA=="]]]]"#.to_vec());
assert_eq!(
object
.into_fileinfo("bucket", "key", true)
.expect_err("a sidecar for an unknown part must fail closed"),
Error::FileCorrupt
);
object.meta_sys = HashMap::from([
(
format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_PART_CHECKSUMS}"),
br#"[[1,[["CRC32C","AAAAAA=="]]]]"#.to_vec(),
),
(
format!("{}{}", rustfs_utils::http::MINIO_INTERNAL_PREFIX, SUFFIX_PART_CHECKSUMS),
br#"[[1,[["CRC32C","BBBBBB=="]]]]"#.to_vec(),
),
]);
assert_eq!(
object
.into_fileinfo("bucket", "key", true)
.expect_err("conflicting sidecar aliases must fail closed"),
Error::FileCorrupt
);
}
#[test] #[test]
fn into_fileinfo_rejects_short_part_actual_sizes_including_empty() { fn into_fileinfo_rejects_short_part_actual_sizes_including_empty() {
let obj = object_with_parts(vec![1, 2], vec![10, 20], vec![]); let obj = object_with_parts(vec![1, 2], vec![10, 20], vec![]);
@@ -4581,31 +4230,6 @@ mod tests {
assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown); assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown);
} }
#[test]
fn meta_object_transition_exact_rejects_legacy_raw_uuid_encoding() {
let mut sys = HashMap::new();
insert_bytes(&mut sys, SUFFIX_TRANSITIONED_VERSION_ID, sample_version_id().as_bytes().to_vec());
insert_bytes(&mut sys, SUFFIX_TRANSITIONED_VERSION_STATE, b"exact".to_vec());
let err = make_meta_object_with_sys(sys)
.into_fileinfo("b", "k", false)
.expect_err("exact remote versions must use their UTF-8 provider representation");
assert_eq!(err, Error::FileCorrupt);
}
#[test]
fn meta_object_transition_version_id_mixed_case_alias_is_recovered() {
let id = sample_version_id();
let sys = HashMap::from([("X-Minio-Internal-transitioned-versionID".to_string(), id.as_bytes().to_vec())]);
let fi = make_meta_object_with_sys(sys)
.into_fileinfo("b", "k", false)
.expect("a legacy mixed-case transition version alias should decode");
assert_eq!(fi.transition_version_id, Some(id));
assert_eq!(fi.transition_version, Some(id.to_string()));
}
#[test] #[test]
fn meta_object_transition_version_id_opaque_text_is_preserved() { fn meta_object_transition_version_id_opaque_text_is_preserved() {
let mut sys = HashMap::new(); let mut sys = HashMap::new();
@@ -4647,10 +4271,8 @@ mod tests {
.map(Vec::as_slice), .map(Vec::as_slice),
Some(b"exact".as_slice()) Some(b"exact".as_slice())
); );
let persisted_version = get_consistent_bytes(&object.meta_sys, SUFFIX_TRANSITIONED_VERSION_ID);
assert_eq!( assert_eq!(
transitioned_version_from_bytes(persisted_version, TransitionVersionState::Unknown) legacy_transitioned_version_id_from_meta_sys(&object.meta_sys),
.and_then(|value| Uuid::parse_str(&value).ok()),
Some(id), Some(id),
"UUID exact writes must remain readable by the legacy UUID consumer" "UUID exact writes must remain readable by the legacy UUID consumer"
); );
@@ -4659,26 +4281,6 @@ mod tests {
assert_eq!(decoded.transition_version.as_deref(), Some(expected_version.as_str())); assert_eq!(decoded.transition_version.as_deref(), Some(expected_version.as_str()));
} }
#[test]
fn meta_object_transition_version_state_exact_preserves_sixteen_byte_opaque_text() {
let expected_version = "opaque.wasabi_01";
assert_eq!(expected_version.len(), 16);
let fi = FileInfo {
transition_status: "complete".to_string(),
transition_version: Some(expected_version.to_string()),
transition_version_state: TransitionVersionState::Exact,
..Default::default()
};
let decoded = MetaObject::from(fi)
.into_fileinfo("b", "k", false)
.expect("exact opaque transition version should round trip");
assert_eq!(decoded.transition_version.as_deref(), Some(expected_version));
assert_eq!(decoded.transition_version_id, None);
assert_eq!(decoded.transition_version_state, TransitionVersionState::Exact);
}
#[test] #[test]
fn set_transition_known_disabled_removes_stale_version_dual_keys() { fn set_transition_known_disabled_removes_stale_version_dual_keys() {
let mut meta_sys = HashMap::new(); let mut meta_sys = HashMap::new();
@@ -4773,8 +4375,7 @@ mod tests {
mod_time: None, mod_time: None,
meta_sys: sys, meta_sys: sys,
} }
.into_fileinfo("b", "k", false) .into_fileinfo("b", "k", false);
.expect("nil tier version should remain an absent remote version");
assert_eq!(fi.transition_version_id, None); assert_eq!(fi.transition_version_id, None);
} }
@@ -4789,8 +4390,7 @@ mod tests {
mod_time: None, mod_time: None,
meta_sys: sys, meta_sys: sys,
} }
.into_fileinfo("b", "k", false) .into_fileinfo("b", "k", false);
.expect("legacy binary UUID tier version should decode");
assert_eq!(fi.transition_version_id, Some(id)); assert_eq!(fi.transition_version_id, Some(id));
assert_eq!(fi.transition_version, Some(id.to_string())); assert_eq!(fi.transition_version, Some(id.to_string()));
} }
@@ -4807,8 +4407,7 @@ mod tests {
mod_time: Some(sample_mod_time()), mod_time: Some(sample_mod_time()),
meta_sys: sys, meta_sys: sys,
} }
.into_fileinfo("b", "k", false) .into_fileinfo("b", "k", false);
.expect("opaque tier version should remain readable");
assert_eq!(fi.transition_version_id, None); assert_eq!(fi.transition_version_id, None);
assert_eq!(fi.transition_version.as_deref(), Some("opaque-generation-42")); assert_eq!(fi.transition_version.as_deref(), Some("opaque-generation-42"));
@@ -4830,67 +4429,12 @@ mod tests {
mod_time: Some(sample_mod_time()), mod_time: Some(sample_mod_time()),
meta_sys: sys, meta_sys: sys,
} }
.into_fileinfo("b", "k", false) .into_fileinfo("b", "k", false);
.expect("mixed-case tier aliases should decode");
assert_eq!(fi.transition_version_id, Some(id)); assert_eq!(fi.transition_version_id, Some(id));
assert_eq!(fi.transition_version, Some(id.to_string())); assert_eq!(fi.transition_version, Some(id.to_string()));
} }
#[test]
fn delete_marker_free_version_recovers_mixed_case_transition_aliases() {
let id = sample_version_id();
let id_text = id.to_string();
let mut sys = HashMap::new();
insert_bytes(&mut sys, SUFFIX_FREE_VERSION, vec![]);
for (suffix, value) in [
(SUFFIX_TRANSITIONED_VERSION_ID, id_text.as_bytes()),
(SUFFIX_TRANSITIONED_VERSION_STATE, b"exact".as_slice()),
(SUFFIX_TRANSITION_TIER, b"WARM".as_slice()),
(SUFFIX_TRANSITIONED_OBJECTNAME, b"remote-object".as_slice()),
] {
sys.insert(format!("X-Minio-Internal-{suffix}"), value.to_vec());
}
let fi = MetaDeleteMarker {
version_id: Some(sample_version_id()),
mod_time: Some(sample_mod_time()),
meta_sys: sys,
}
.into_fileinfo("b", "k", false)
.expect("mixed-case tier aliases should decode");
assert_eq!(fi.transition_version_id, Some(id));
assert_eq!(fi.transition_version, Some(id.to_string()));
assert_eq!(fi.transition_version_state, TransitionVersionState::Exact);
assert_eq!(fi.transition_tier, "WARM");
assert_eq!(fi.transitioned_objname, "remote-object");
}
#[test]
fn delete_marker_free_version_rejects_conflicting_transition_aliases() {
let mut sys = HashMap::new();
insert_bytes(&mut sys, SUFFIX_FREE_VERSION, vec![]);
sys.insert(
format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_TRANSITIONED_VERSION_ID}"),
b"source-version".to_vec(),
);
sys.insert(
format!("{}{}", rustfs_utils::http::MINIO_INTERNAL_PREFIX, SUFFIX_TRANSITIONED_VERSION_ID),
b"target-version".to_vec(),
);
let err = MetaDeleteMarker {
version_id: Some(sample_version_id()),
mod_time: Some(sample_mod_time()),
meta_sys: sys,
}
.into_fileinfo("b", "k", false)
.expect_err("conflicting transition aliases must fail closed");
assert_eq!(err, Error::FileCorrupt);
}
#[test] #[test]
fn version_header_sorts_before_prefers_object_over_delete_marker_on_equal_mod_time() { fn version_header_sorts_before_prefers_object_over_delete_marker_on_equal_mod_time() {
let object = FileMetaVersionHeader { let object = FileMetaVersionHeader {
@@ -5214,7 +4758,7 @@ mod tests {
#[test] #[test]
fn target_delete_marker_version_metadata_is_forward_and_backward_compatible() { fn target_delete_marker_version_metadata_is_forward_and_backward_compatible() {
let arn = "arn:rustfs:replication:us-east-1:TenantA:bucket"; let arn = "arn:rustfs:replication:us-east-1:target:bucket";
let suffix = format!("{}{arn}", rustfs_utils::http::SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX); let suffix = format!("{}{arn}", rustfs_utils::http::SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX);
let mut metadata = HashMap::from([(format!("{RUSTFS_INTERNAL_PREFIX}replication-status"), format!("{arn}=COMPLETED;"))]); let mut metadata = HashMap::from([(format!("{RUSTFS_INTERNAL_PREFIX}replication-status"), format!("{arn}=COMPLETED;"))]);
@@ -5259,7 +4803,7 @@ mod tests {
// must keep it keyed by `target_reset_header(arn)` (not the bare ARN) so // must keep it keyed by `target_reset_header(arn)` (not the bare ARN) so
// `ReplicationState::target_state` finds it after a round trip // `ReplicationState::target_state` finds it after a round trip
// (backlog#799 B16). // (backlog#799 B16).
let arn = "arn:rustfs:replication:us-east-1:TenantA:bucket"; let arn = "arn:rustfs:replication:us-east-1:target:bucket";
let ts = "2026-06-30T00:00:00Z;reset-1".to_string(); let ts = "2026-06-30T00:00:00Z;reset-1".to_string();
let key = crate::replication::target_reset_header(arn); let key = crate::replication::target_reset_header(arn);
let mut metadata = HashMap::new(); let mut metadata = HashMap::new();
@@ -5278,33 +4822,6 @@ mod tests {
); );
} }
#[test]
fn get_internal_replication_state_accepts_rfc3339_and_rustfs_display_timestamps() {
let replica_timestamp = OffsetDateTime::UNIX_EPOCH + time::Duration::SECOND;
let replication_timestamp = replica_timestamp + time::Duration::SECOND;
let metadata = HashMap::from([
(format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_REPLICA_STATUS}"), "REPLICA".to_string()),
(
format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_REPLICA_TIMESTAMP}"),
replica_timestamp.to_string(),
),
(
format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_REPLICATION_STATUS}"),
"arn:rustfs:replication:us-east-1:TenantA:bucket=COMPLETED;".to_string(),
),
(
format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_REPLICATION_TIMESTAMP}"),
replication_timestamp
.format(&Rfc3339)
.expect("RFC3339 timestamp should format"),
),
]);
let state = get_internal_replication_state(&metadata).expect("replication metadata should parse");
assert_eq!(state.replica_timestamp, Some(replica_timestamp));
assert_eq!(state.replication_timestamp, Some(replication_timestamp));
}
// ---- Header signature (backlog#861 / B12) ---- // ---- Header signature (backlog#861 / B12) ----
fn signed_object() -> MetaObject { fn signed_object() -> MetaObject {

Some files were not shown because too many files have changed in this diff Show More