mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-19 02:56:18 +00:00
Merge branch 'main' into overtrue/kms-1638-d2-minio-sse-read
This commit is contained in:
Generated
+111
-103
@@ -964,9 +964,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "aws-sdk-kms"
|
name = "aws-sdk-kms"
|
||||||
version = "1.114.0"
|
version = "1.115.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "c0b7d906608ee41e7ddea9983577ba82200435644d567d63dc34e822e088b453"
|
checksum = "d5b034f8b7ceadb873d0bc607c30bb4b0be68e09a84c837174e7c2c6878ff882"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"aws-credential-types",
|
"aws-credential-types",
|
||||||
@@ -990,9 +990,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "aws-sdk-s3"
|
name = "aws-sdk-s3"
|
||||||
version = "1.141.0"
|
version = "1.142.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "d9f9420d3a2467eed22ed3635ca653653162c386a0b0f65c78189f9bd3c1379e"
|
checksum = "f9e15a5c55e05f4b0b7e483160b3c85cccdf77cff02c95504f3e71d460855cd2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"aws-credential-types",
|
"aws-credential-types",
|
||||||
@@ -1027,9 +1027,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "aws-sdk-sso"
|
name = "aws-sdk-sso"
|
||||||
version = "1.105.0"
|
version = "1.106.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "6ffd0fbe7873cb548a7aa60f9573c268fff94155397fd4f14dc9f1ecaaab8516"
|
checksum = "2d0efcee834347b6705eca3eea2defd88242f43774f55d7326604222e3c86260"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"aws-credential-types",
|
"aws-credential-types",
|
||||||
@@ -1053,9 +1053,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "aws-sdk-ssooidc"
|
name = "aws-sdk-ssooidc"
|
||||||
version = "1.107.0"
|
version = "1.108.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "175763eb222a46377df7aa257a3bca980ab3e96703fefc8f4d0b8da6ad2e254c"
|
checksum = "a59312a04cf19c962cfee32b64ecfee758f8786407ff6da5b30fff46ae96f201"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"aws-credential-types",
|
"aws-credential-types",
|
||||||
@@ -1079,9 +1079,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "aws-sdk-sts"
|
name = "aws-sdk-sts"
|
||||||
version = "1.110.0"
|
version = "1.111.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "dd8b14781dfbff48984017d57167b6ea0b6471c6920ec52b44a2677c7feb3c13"
|
checksum = "120e7eb63457a9e547f9986fe3b273f77c43679da4d04f46359fa881c5e19b6e"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"aws-credential-types",
|
"aws-credential-types",
|
||||||
@@ -1598,7 +1598,7 @@ version = "0.10.4"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -1617,7 +1617,7 @@ version = "0.3.3"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "a8894febbff9f758034a5b8e12d87918f56dfc64a8e1fe757d65e29041538d93"
|
checksum = "a8894febbff9f758034a5b8e12d87918f56dfc64a8e1fe757d65e29041538d93"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -1858,9 +1858,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "cc"
|
name = "cc"
|
||||||
version = "1.4.2"
|
version = "1.4.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "5d262e149917187838d5b42777c8253bcb64500067342904e7d429499a6f277e"
|
checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"find-msvc-tools",
|
"find-msvc-tools",
|
||||||
"jobserver",
|
"jobserver",
|
||||||
@@ -1968,7 +1968,7 @@ version = "0.4.4"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad"
|
checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"crypto-common 0.1.6",
|
"crypto-common 0.1.7",
|
||||||
"inout 0.1.4",
|
"inout 0.1.4",
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -2428,7 +2428,7 @@ version = "0.5.5"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "0dc92fb57ca44df6db8059111ab3af99a63d5d0f8375d9972e319a379c6bab76"
|
checksum = "0dc92fb57ca44df6db8059111ab3af99a63d5d0f8375d9972e319a379c6bab76"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
"rand_core 0.6.4",
|
"rand_core 0.6.4",
|
||||||
"subtle",
|
"subtle",
|
||||||
"zeroize",
|
"zeroize",
|
||||||
@@ -2453,11 +2453,11 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "crypto-common"
|
name = "crypto-common"
|
||||||
version = "0.1.6"
|
version = "0.1.7"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "1bfb12502f3fc46cca1bb51ac28df9d618d813cdc3d2f25b9fe775a34af26bb3"
|
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
"typenum",
|
"typenum",
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -3664,7 +3664,7 @@ checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
|||||||
dependencies = [
|
dependencies = [
|
||||||
"block-buffer 0.10.4",
|
"block-buffer 0.10.4",
|
||||||
"const-oid 0.9.6",
|
"const-oid 0.9.6",
|
||||||
"crypto-common 0.1.6",
|
"crypto-common 0.1.7",
|
||||||
"subtle",
|
"subtle",
|
||||||
]
|
]
|
||||||
|
|
||||||
@@ -3924,7 +3924,7 @@ dependencies = [
|
|||||||
"crypto-bigint 0.5.5",
|
"crypto-bigint 0.5.5",
|
||||||
"digest 0.10.7",
|
"digest 0.10.7",
|
||||||
"ff 0.13.1",
|
"ff 0.13.1",
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
"group 0.13.0",
|
"group 0.13.0",
|
||||||
"hkdf 0.12.4",
|
"hkdf 0.12.4",
|
||||||
"pem-rfc7468 0.7.0",
|
"pem-rfc7468 0.7.0",
|
||||||
@@ -4148,9 +4148,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "find-msvc-tools"
|
name = "find-msvc-tools"
|
||||||
version = "0.1.10"
|
version = "0.1.11"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "26b73573e6edcd2af0cdf47bd6cb58f0b3839491263c314eaad1ccf24430e1de"
|
checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "findshlibs"
|
name = "findshlibs"
|
||||||
@@ -4369,9 +4369,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "generic-array"
|
name = "generic-array"
|
||||||
version = "0.14.9"
|
version = "0.14.7"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "4bb6743198531e02858aeaea5398fcc883e71851fcbcb5a2f773e2fb6cb1edf2"
|
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"typenum",
|
"typenum",
|
||||||
"version_check",
|
"version_check",
|
||||||
@@ -4380,11 +4380,11 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "generic-array"
|
name = "generic-array"
|
||||||
version = "1.4.4"
|
version = "1.4.5"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "ab4e5aa225bc56696909483320f0ff9b600f1a971b52e07a17d70f3d9b43254b"
|
checksum = "337d46834ee672ab3e48caca2cb0c78cc174fb12b3a68d0d88f99a0519a5e36e"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
"rustversion",
|
"rustversion",
|
||||||
"typenum",
|
"typenum",
|
||||||
]
|
]
|
||||||
@@ -4726,9 +4726,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "h2"
|
name = "h2"
|
||||||
version = "0.4.15"
|
version = "0.4.16"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "6cb093c84e8bd9b188d4c4a8cb6579fc016968d14c99882163cd3ff402a4f155"
|
checksum = "a9f37a958b41b3b19ee2707c06439c0e9e547e847223eb791ecb0cb821c65e27"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"atomic-waker",
|
"atomic-waker",
|
||||||
"bytes",
|
"bytes",
|
||||||
@@ -5028,9 +5028,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "hotpath"
|
name = "hotpath"
|
||||||
version = "0.23.2"
|
version = "0.23.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "62e810bedda5a467ef5c9b5c8a20763fefebc89b63ef36f7ee44a143085204a2"
|
checksum = "dce755d457a63bdd0c95e4c91511daad1b58b33209543b7f38027b676f387e5e"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"async-channel",
|
"async-channel",
|
||||||
@@ -5062,9 +5062,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "hotpath-macros"
|
name = "hotpath-macros"
|
||||||
version = "0.23.2"
|
version = "0.23.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "01bdc59bfc1a9984bee2ff5da63b2f6fccbaa57cd9a4119d709524632bddf341"
|
checksum = "a903af89a8429cb07790c3818bc15270b394f80af1bc254e5ccf9c7de2961770"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"proc-macro2",
|
"proc-macro2",
|
||||||
"quote",
|
"quote",
|
||||||
@@ -5073,15 +5073,15 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "hotpath-macros-meta"
|
name = "hotpath-macros-meta"
|
||||||
version = "0.23.2"
|
version = "0.23.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "d9216e8a01abe1e1671c376dc8736fb1bf772d7a889538d25f9e1200120ced38"
|
checksum = "bcc0ab94ffbb2ee77f4a897df02b5a137a10cf24d69bda936e59aff4dd456e61"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "hotpath-meta"
|
name = "hotpath-meta"
|
||||||
version = "0.23.2"
|
version = "0.23.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "f22a9d20435fb79511b19dae37b3607224cd98f342a410702d84657cc38fc72f"
|
checksum = "053481f6cec8f775a3276c7f6e2f21123111d28261e4edc15ea7421c445964bb"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"hotpath-macros-meta",
|
"hotpath-macros-meta",
|
||||||
]
|
]
|
||||||
@@ -5280,9 +5280,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "icu_collections"
|
name = "icu_collections"
|
||||||
version = "2.2.0"
|
version = "2.3.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "2984d1cd16c883d7935b9e07e44071dca8d917fd52ecc02c04d5fa0b5a3f191c"
|
checksum = "fa68d21081c4a05d5a901a1c62add574c77048b6a1c67be3b50ce0b60d4ca513"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"displaydoc",
|
"displaydoc",
|
||||||
"potential_utf",
|
"potential_utf",
|
||||||
@@ -5294,9 +5294,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "icu_locale_core"
|
name = "icu_locale_core"
|
||||||
version = "2.2.0"
|
version = "2.3.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "92219b62b3e2b4d88ac5119f8904c10f8f61bf7e95b640d25ba3075e6cac2c29"
|
checksum = "d56e28588da92eee5c3201a6eff33fabdd49b62269c8938d4ff050ce4d900deb"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"displaydoc",
|
"displaydoc",
|
||||||
"litemap",
|
"litemap",
|
||||||
@@ -5307,9 +5307,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "icu_normalizer"
|
name = "icu_normalizer"
|
||||||
version = "2.2.0"
|
version = "2.3.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "c56e5ee99d6e3d33bd91c5d85458b6005a22140021cc324cea84dd0e72cff3b4"
|
checksum = "12f9cf5f235641ed274641dd81c3f28d870e276763d0797aeeab72317b1c646f"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"icu_collections",
|
"icu_collections",
|
||||||
"icu_normalizer_data",
|
"icu_normalizer_data",
|
||||||
@@ -5321,16 +5321,17 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "icu_normalizer_data"
|
name = "icu_normalizer_data"
|
||||||
version = "2.2.0"
|
version = "2.3.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "da3be0ae77ea334f4da67c12f149704f19f81d1adf7c51cf482943e84a2bad38"
|
checksum = "1563da1ed3e0b3bf3d74c9b85917ac9c56464d2f57242270c09c9e752f8021a0"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "icu_properties"
|
name = "icu_properties"
|
||||||
version = "2.2.0"
|
version = "2.3.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "bee3b67d0ea5c2cca5003417989af8996f8604e34fb9ddf96208a033901e70de"
|
checksum = "7e7ca276ad3145661a65914e6daf131ca5120cd3dcee8f8f3214b8875184a148"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
|
"displaydoc",
|
||||||
"icu_collections",
|
"icu_collections",
|
||||||
"icu_locale_core",
|
"icu_locale_core",
|
||||||
"icu_properties_data",
|
"icu_properties_data",
|
||||||
@@ -5341,15 +5342,15 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "icu_properties_data"
|
name = "icu_properties_data"
|
||||||
version = "2.2.0"
|
version = "2.3.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "8e2bbb201e0c04f7b4b3e14382af113e17ba4f63e2c9d2ee626b720cbce54a14"
|
checksum = "e590f038c1464a96894fd6d10127e90a8be4509f56ff7ecef851b15cee0b7caa"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "icu_provider"
|
name = "icu_provider"
|
||||||
version = "2.2.0"
|
version = "2.3.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "139c4cf31c8b5f33d7e199446eff9c1e02decfc2f0eec2c8d71f65befa45b421"
|
checksum = "92a7ed671a6aad807a8651a2e1782a6598fda9ce5185dd8158549e95a91c6428"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"displaydoc",
|
"displaydoc",
|
||||||
"icu_locale_core",
|
"icu_locale_core",
|
||||||
@@ -5417,7 +5418,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01"
|
checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"block-padding 0.3.3",
|
"block-padding 0.3.3",
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -5968,9 +5969,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "libredox"
|
name = "libredox"
|
||||||
version = "0.1.19"
|
version = "0.1.20"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "2026a5056764a10b2bf5d56488cba40da507f5493a6a429340e2004d9ed085fa"
|
checksum = "28d0a00925a9f930d679b6789b721e3a7f9ed110f41b86d2497caa780c3a070a"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"libc",
|
"libc",
|
||||||
]
|
]
|
||||||
@@ -6033,9 +6034,9 @@ checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "litemap"
|
name = "litemap"
|
||||||
version = "0.8.2"
|
version = "0.8.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "92daf443525c4cce67b150400bc2316076100ce0b3686209eb8cf3c31612e6f0"
|
checksum = "47d9d19d1d6efa0109d2f65ff4c85cddd50bd572e5a00127ab10987290bcefae"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "local-ip-address"
|
name = "local-ip-address"
|
||||||
@@ -6482,9 +6483,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "mqttbytes-core-next"
|
name = "mqttbytes-core-next"
|
||||||
version = "0.33.3"
|
version = "0.34.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "3ff7ae19c74aba9e0ed6e4071cd52aa364e020076fa3cc6ef17e43662f756f3c"
|
checksum = "366b6ba2b4209ca4bc5ac731ccddf570d09831981eed07e5fbd63564cf0cf1aa"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"bytes",
|
"bytes",
|
||||||
"thiserror 2.0.20",
|
"thiserror 2.0.20",
|
||||||
@@ -6885,7 +6886,7 @@ version = "5.0.0"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d"
|
checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base64 0.21.7",
|
"base64 0.22.1",
|
||||||
"chrono",
|
"chrono",
|
||||||
"getrandom 0.2.17",
|
"getrandom 0.2.17",
|
||||||
"http 1.5.0",
|
"http 1.5.0",
|
||||||
@@ -7344,9 +7345,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "pageant"
|
name = "pageant"
|
||||||
version = "0.2.1"
|
version = "0.2.2"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "4f3a5ae18f65a85c67a77d18d42d3606c07948e3c17c1e5f74852b26589e88a5"
|
checksum = "3adadc44070da6f464b0918655a12f5792c156e088d8c4082d13e27d94c3e791"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"base16ct 1.0.0",
|
"base16ct 1.0.0",
|
||||||
"byteorder",
|
"byteorder",
|
||||||
@@ -7728,9 +7729,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "pkg-config"
|
name = "pkg-config"
|
||||||
version = "0.3.33"
|
version = "0.3.34"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e"
|
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "plotters"
|
name = "plotters"
|
||||||
@@ -7836,9 +7837,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "potential_utf"
|
name = "potential_utf"
|
||||||
version = "0.1.5"
|
version = "0.1.6"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "0103b1cef7ec0cf76490e969665504990193874ea05c85ff9bab8b911d0a0564"
|
checksum = "d83eb9bc6d8e5cf568e7a1101d60ee05e81ed50ea106026f3d18deeb046d7661"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"zerovec",
|
"zerovec",
|
||||||
]
|
]
|
||||||
@@ -8046,7 +8047,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
|
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"heck 0.5.0",
|
"heck 0.5.0",
|
||||||
"itertools 0.10.5",
|
"itertools 0.14.0",
|
||||||
"log",
|
"log",
|
||||||
"multimap",
|
"multimap",
|
||||||
"once_cell",
|
"once_cell",
|
||||||
@@ -8066,7 +8067,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042"
|
checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"heck 0.5.0",
|
"heck 0.5.0",
|
||||||
"itertools 0.10.5",
|
"itertools 0.14.0",
|
||||||
"log",
|
"log",
|
||||||
"multimap",
|
"multimap",
|
||||||
"petgraph 0.8.3",
|
"petgraph 0.8.3",
|
||||||
@@ -8087,7 +8088,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d"
|
checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anyhow",
|
"anyhow",
|
||||||
"itertools 0.10.5",
|
"itertools 0.14.0",
|
||||||
"proc-macro2",
|
"proc-macro2",
|
||||||
"quote",
|
"quote",
|
||||||
"syn 2.0.119",
|
"syn 2.0.119",
|
||||||
@@ -8100,7 +8101,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf"
|
checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"anyhow",
|
"anyhow",
|
||||||
"itertools 0.10.5",
|
"itertools 0.14.0",
|
||||||
"proc-macro2",
|
"proc-macro2",
|
||||||
"quote",
|
"quote",
|
||||||
"syn 2.0.119",
|
"syn 2.0.119",
|
||||||
@@ -8216,7 +8217,7 @@ dependencies = [
|
|||||||
"reqwest",
|
"reqwest",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
"smallvec",
|
"smallvec",
|
||||||
"spin 0.12.2",
|
"spin 0.12.3",
|
||||||
"symbolic-demangle",
|
"symbolic-demangle",
|
||||||
"tempfile",
|
"tempfile",
|
||||||
"thiserror 2.0.20",
|
"thiserror 2.0.20",
|
||||||
@@ -8285,9 +8286,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "quinn-proto"
|
name = "quinn-proto"
|
||||||
version = "0.11.16"
|
version = "0.11.17"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "2f4bfc015262b9df63c8845072ce59068853ff5872180c2ce2f13038b970e560"
|
checksum = "04759210543be93709136e28212294a659ef5001836ff4eab4d663e4529bba83"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"aws-lc-rs",
|
"aws-lc-rs",
|
||||||
"bytes",
|
"bytes",
|
||||||
@@ -8553,9 +8554,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "redis"
|
name = "redis"
|
||||||
version = "1.5.0"
|
version = "1.6.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "3257df217f7eab0044627a268c9cc6cdb60c0c421c88f83ac41c4e31520b6b84"
|
checksum = "e37a4ca5c6ca42aa3e6df2fd32b987a65d32a4c2159a6f3fe0fd1df306a2658f"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"arc-swap",
|
"arc-swap",
|
||||||
"arcstr",
|
"arcstr",
|
||||||
@@ -8567,7 +8568,7 @@ dependencies = [
|
|||||||
"futures-channel",
|
"futures-channel",
|
||||||
"futures-util",
|
"futures-util",
|
||||||
"itoa",
|
"itoa",
|
||||||
"num-bigint 0.4.8",
|
"num-bigint 0.5.1",
|
||||||
"percent-encoding",
|
"percent-encoding",
|
||||||
"pin-project-lite",
|
"pin-project-lite",
|
||||||
"rustls",
|
"rustls",
|
||||||
@@ -8868,9 +8869,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "rumqttc-core-next"
|
name = "rumqttc-core-next"
|
||||||
version = "0.33.3"
|
version = "0.34.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "7d7d9205738dd41a2546e82d27a634d07d8b303dcf7558565ff70caf3ceb0f9c"
|
checksum = "249896ab27ed630590971738264baa8f722f18965d2e387c706c40a3c2a572cc"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-tungstenite",
|
"async-tungstenite",
|
||||||
"futures-io",
|
"futures-io",
|
||||||
@@ -8886,18 +8887,18 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "rumqttc-next"
|
name = "rumqttc-next"
|
||||||
version = "0.33.3"
|
version = "0.34.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "ed1bad2180ff539da671da9a996152a921bc5316eb6d8a9cc3bd441653138b08"
|
checksum = "477c9bbfba8f3aecc7aad31c6de2eacb75822efaa18e7aeecb8d3d8e534fbf07"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"rumqttc-v5-next",
|
"rumqttc-v5-next",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "rumqttc-v5-next"
|
name = "rumqttc-v5-next"
|
||||||
version = "0.33.3"
|
version = "0.34.0"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "229576cbedfa9089f90c17c9454e9429ac1e89cdd223bac5cb39d837593f79bc"
|
checksum = "3dfa6ddcc7a7dd5688f9bf78d8f81cb94f367bce56c055d8d94cf81ecb0518bf"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"async-tungstenite",
|
"async-tungstenite",
|
||||||
"bytes",
|
"bytes",
|
||||||
@@ -8920,9 +8921,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "russh"
|
name = "russh"
|
||||||
version = "0.62.6"
|
version = "0.62.7"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "b41043523e0edcbd4e31d00903e26f12994f63b21bae9904f7405c1ed92752a5"
|
checksum = "9decb68e4e44e1079700e54f17c8f23806ec53d7e0db73ab1c71d9dabc666812"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"aes 0.9.2",
|
"aes 0.9.2",
|
||||||
"aws-lc-rs",
|
"aws-lc-rs",
|
||||||
@@ -8945,7 +8946,7 @@ dependencies = [
|
|||||||
"enum_dispatch",
|
"enum_dispatch",
|
||||||
"flate2",
|
"flate2",
|
||||||
"futures",
|
"futures",
|
||||||
"generic-array 1.4.4",
|
"generic-array 1.4.5",
|
||||||
"getrandom 0.4.3",
|
"getrandom 0.4.3",
|
||||||
"ghash",
|
"ghash",
|
||||||
"hex-literal",
|
"hex-literal",
|
||||||
@@ -9280,6 +9281,7 @@ dependencies = [
|
|||||||
"s3s",
|
"s3s",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
|
"smallvec",
|
||||||
"tokio",
|
"tokio",
|
||||||
"tonic",
|
"tonic",
|
||||||
"tracing",
|
"tracing",
|
||||||
@@ -9490,7 +9492,7 @@ dependencies = [
|
|||||||
"parking_lot",
|
"parking_lot",
|
||||||
"rayon",
|
"rayon",
|
||||||
"smallvec",
|
"smallvec",
|
||||||
"spin 0.12.2",
|
"spin 0.12.3",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -9825,14 +9827,19 @@ name = "rustfs-madmin"
|
|||||||
version = "1.0.0-rc.2"
|
version = "1.0.0-rc.2"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"hotpath",
|
"hotpath",
|
||||||
|
"http 1.5.0",
|
||||||
"humantime",
|
"humantime",
|
||||||
"hyper",
|
"hyper",
|
||||||
"jiff",
|
"jiff",
|
||||||
|
"reqwest",
|
||||||
"rmp-serde",
|
"rmp-serde",
|
||||||
|
"rustfs-signer",
|
||||||
|
"s3s",
|
||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
"sysinfo",
|
"sysinfo",
|
||||||
"time",
|
"time",
|
||||||
|
"tokio",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
@@ -10246,6 +10253,7 @@ dependencies = [
|
|||||||
"rustfs-ecstore",
|
"rustfs-ecstore",
|
||||||
"rustfs-filemeta",
|
"rustfs-filemeta",
|
||||||
"rustfs-lock",
|
"rustfs-lock",
|
||||||
|
"rustfs-s3-types",
|
||||||
"rustfs-storage-api",
|
"rustfs-storage-api",
|
||||||
"rustfs-utils",
|
"rustfs-utils",
|
||||||
"s3s",
|
"s3s",
|
||||||
@@ -10826,7 +10834,7 @@ checksum = "d3e97a565f76233a6003f9f5c54be1d9c5bdfa3eccfb189469f11ec4901c47dc"
|
|||||||
dependencies = [
|
dependencies = [
|
||||||
"base16ct 0.2.0",
|
"base16ct 0.2.0",
|
||||||
"der 0.7.10",
|
"der 0.7.10",
|
||||||
"generic-array 0.14.9",
|
"generic-array 0.14.7",
|
||||||
"pkcs8 0.10.2",
|
"pkcs8 0.10.2",
|
||||||
"subtle",
|
"subtle",
|
||||||
"zeroize",
|
"zeroize",
|
||||||
@@ -11390,9 +11398,9 @@ checksum = "023a211cb3138dbc438680b32560ad89f699977624c9f8dbb95a47d5b4c07dd3"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "spin"
|
name = "spin"
|
||||||
version = "0.12.2"
|
version = "0.12.3"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "8abadc99fd9c7bbb7d0ca2b31d72a067d0c0dcd7aad25ab8cac71ba91417694b"
|
checksum = "0134f9043ed38b087ac4f7d4af44c79e2c9e5094421fe3164f435ce585953b10"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"lock_api",
|
"lock_api",
|
||||||
]
|
]
|
||||||
@@ -11804,7 +11812,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
|||||||
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
|
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"fastrand",
|
"fastrand",
|
||||||
"getrandom 0.3.4",
|
"getrandom 0.4.3",
|
||||||
"once_cell",
|
"once_cell",
|
||||||
"rustix",
|
"rustix",
|
||||||
"windows-sys 0.61.2",
|
"windows-sys 0.61.2",
|
||||||
@@ -11968,9 +11976,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "tinystr"
|
name = "tinystr"
|
||||||
version = "0.8.3"
|
version = "0.8.4"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "c8323304221c2a851516f22236c5722a72eaa19749016521d6dff0824447d96d"
|
checksum = "b1e27c91459209c2986af3dcf603a5a74a4368754ce37414f59acc971167f643"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"displaydoc",
|
"displaydoc",
|
||||||
"zerovec",
|
"zerovec",
|
||||||
@@ -12644,9 +12652,9 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "uuid"
|
name = "uuid"
|
||||||
version = "1.24.0"
|
version = "1.24.1"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "bf3923a6f5c4c6382e0b653c4117f48d631ea17f38ed86e2a828e6f7412f5239"
|
checksum = "2cefc03fd367c0c6d4305de1b312cf00248c4114f4a0418ce6a6af769e3b0bd9"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"getrandom 0.4.3",
|
"getrandom 0.4.3",
|
||||||
"js-sys",
|
"js-sys",
|
||||||
@@ -13161,9 +13169,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "writeable"
|
name = "writeable"
|
||||||
version = "0.6.3"
|
version = "0.6.4"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "1ffae5123b2d3fc086436f8834ae3ab053a283cfac8fe0a0b8eaae044768a4c4"
|
checksum = "3ad82d2a33cdc9674dc7465672f271e096168fcdbe0f799d9e6db8c5892679dc"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "x509-cert"
|
name = "x509-cert"
|
||||||
@@ -13343,9 +13351,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "zerotrie"
|
name = "zerotrie"
|
||||||
version = "0.2.4"
|
version = "0.2.5"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "0f9152d31db0792fa83f70fb2f83148effb5c1f5b8c7686c3459e361d9bc20bf"
|
checksum = "4ea269c3bd32f0a32c321907a2ae912ba6f4649bb0fc764a15627e99a7095a3f"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"displaydoc",
|
"displaydoc",
|
||||||
"yoke",
|
"yoke",
|
||||||
@@ -13354,9 +13362,9 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "zerovec"
|
name = "zerovec"
|
||||||
version = "0.11.6"
|
version = "0.11.7"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "90f911cbc359ab6af17377d242225f4d75119aec87ea711a880987b18cd7b239"
|
checksum = "94b5c6b5976d66c1d703c4fd17d3f5e43c8cedaacf604961b171adc7130896d8"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"yoke",
|
"yoke",
|
||||||
"zerofrom",
|
"zerofrom",
|
||||||
@@ -13365,13 +13373,13 @@ dependencies = [
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "zerovec-derive"
|
name = "zerovec-derive"
|
||||||
version = "0.11.3"
|
version = "0.11.5"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "625dc425cab0dca6dc3c3319506e6593dcb08a9f387ea3b284dbd52a92c40555"
|
checksum = "9f212a141d820099d57ffafb9569be9617a6f27d3dc881fbee8fb56642f917a9"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"proc-macro2",
|
"proc-macro2",
|
||||||
"quote",
|
"quote",
|
||||||
"syn 2.0.119",
|
"syn 3.0.3",
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
|
|||||||
+8
-8
@@ -228,9 +228,9 @@ atoi = "3.1.0"
|
|||||||
atomic_enum = "0.3.0"
|
atomic_enum = "0.3.0"
|
||||||
aws-config = { version = "1.10.1" }
|
aws-config = { version = "1.10.1" }
|
||||||
aws-credential-types = { version = "1.3.0" }
|
aws-credential-types = { version = "1.3.0" }
|
||||||
aws-sdk-kms = { default-features = false, version = "1.114.0" }
|
aws-sdk-kms = { default-features = false, version = "1.115.0" }
|
||||||
aws-sdk-s3 = { default-features = false, version = "1.141.0" }
|
aws-sdk-s3 = { default-features = false, version = "1.142.0" }
|
||||||
aws-sdk-sts = { default-features = false, version = "1.110.0" }
|
aws-sdk-sts = { default-features = false, version = "1.111.0" }
|
||||||
aws-smithy-http-client = { default-features = false, version = "1.3.0" }
|
aws-smithy-http-client = { default-features = false, version = "1.3.0" }
|
||||||
aws-smithy-runtime-api = { version = "1.14.0" }
|
aws-smithy-runtime-api = { version = "1.14.0" }
|
||||||
aws-smithy-types = { version = "1.6.2" }
|
aws-smithy-types = { version = "1.6.2" }
|
||||||
@@ -284,8 +284,8 @@ rayon = "1.12.0"
|
|||||||
reed-solomon-erasure = { package = "rustfs-erasure-codec", version = "8.0.2" }
|
reed-solomon-erasure = { package = "rustfs-erasure-codec", version = "8.0.2" }
|
||||||
reed-solomon-simd = "3.1.0"
|
reed-solomon-simd = "3.1.0"
|
||||||
regex = { version = "1.13.1" }
|
regex = { version = "1.13.1" }
|
||||||
rumqttc = { package = "rumqttc-next", version = "0.33.3" }
|
rumqttc = { package = "rumqttc-next", version = "0.34.0" }
|
||||||
redis = { version = "1.5.0" }
|
redis = { version = "1.6.0" }
|
||||||
rustify = { version = "0.7", default-features = false }
|
rustify = { version = "0.7", default-features = false }
|
||||||
rustix = { version = "1.1.4" }
|
rustix = { version = "1.1.4" }
|
||||||
rust-embed = { version = "8.12.0" }
|
rust-embed = { version = "8.12.0" }
|
||||||
@@ -313,7 +313,7 @@ tracing-subscriber = { version = "0.3.23" }
|
|||||||
transform-stream = "0.3.1"
|
transform-stream = "0.3.1"
|
||||||
url = "2.5.8"
|
url = "2.5.8"
|
||||||
urlencoding = "2.1.3"
|
urlencoding = "2.1.3"
|
||||||
uuid = { version = "1.24.0" }
|
uuid = { version = "1.24.1" }
|
||||||
vaultrs = { version = "0.8.0" }
|
vaultrs = { version = "0.8.0" }
|
||||||
tar = "0.4.46"
|
tar = "0.4.46"
|
||||||
walkdir = "2.5.0"
|
walkdir = "2.5.0"
|
||||||
@@ -341,7 +341,7 @@ libunftp = { version = "0.23.0" }
|
|||||||
unftp-core = "0.1.0"
|
unftp-core = "0.1.0"
|
||||||
suppaftp = { version = "10.0.1" }
|
suppaftp = { version = "10.0.1" }
|
||||||
rcgen = { version = "0.14.9", default-features = false, features = ["aws_lc_rs", "crypto", "pem"] }
|
rcgen = { version = "0.14.9", default-features = false, features = ["aws_lc_rs", "crypto", "pem"] }
|
||||||
russh = { version = "0.62.6" }
|
russh = { version = "0.62.7" }
|
||||||
russh-sftp = "2.4.0"
|
russh-sftp = "2.4.0"
|
||||||
|
|
||||||
# WebDAV
|
# WebDAV
|
||||||
@@ -350,7 +350,7 @@ dav-server = "0.11.0"
|
|||||||
# Performance Analysis and Memory Profiling
|
# Performance Analysis and Memory Profiling
|
||||||
mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" }
|
mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" }
|
||||||
libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11", features = ["extended"] }
|
libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11", features = ["extended"] }
|
||||||
hotpath = { version = "0.23.2", default-features = false }
|
hotpath = { version = "0.23.3", default-features = false }
|
||||||
# Snapshot testing for output format regression detection
|
# Snapshot testing for output format regression detection
|
||||||
insta = { version = "1.48" }
|
insta = { version = "1.48" }
|
||||||
|
|
||||||
|
|||||||
@@ -40,6 +40,7 @@ mak = "mak"
|
|||||||
gae = "gae"
|
gae = "gae"
|
||||||
GAE = "GAE"
|
GAE = "GAE"
|
||||||
thr = "thr"
|
thr = "thr"
|
||||||
|
mis = "mis"
|
||||||
# s3-tests original test names (cannot be changed)
|
# s3-tests original test names (cannot be changed)
|
||||||
nonexisted = "nonexisted"
|
nonexisted = "nonexisted"
|
||||||
consts = "consts"
|
consts = "consts"
|
||||||
|
|||||||
@@ -38,7 +38,10 @@ pub const XXHASH_3_HEADER_NAME: &str = "x-amz-checksum-xxhash3";
|
|||||||
pub const XXHASH_64_HEADER_NAME: &str = "x-amz-checksum-xxhash64";
|
pub const XXHASH_64_HEADER_NAME: &str = "x-amz-checksum-xxhash64";
|
||||||
pub const XXHASH_128_HEADER_NAME: &str = "x-amz-checksum-xxhash128";
|
pub const XXHASH_128_HEADER_NAME: &str = "x-amz-checksum-xxhash128";
|
||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "Content-MD5 wire name, resolved by header_name() below and asserted by this crate's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) static MD5_HEADER_NAME: &str = "content-md5";
|
pub(crate) static MD5_HEADER_NAME: &str = "content-md5";
|
||||||
|
|
||||||
pub const CHECKSUM_ALGORITHMS_IN_PRIORITY_ORDER: [&str; 5] =
|
pub const CHECKSUM_ALGORITHMS_IN_PRIORITY_ORDER: [&str; 5] =
|
||||||
|
|||||||
@@ -476,13 +476,19 @@ impl Checksum for Xxhash64 {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
#[derive(Debug, Default)]
|
#[derive(Debug, Default)]
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "Content-MD5 is not a ChecksumAlgorithm variant and has no arm in into_impl: S3 carries it as its own header, separate from the x-amz-checksum-* family. This impl exists so the two paths share the Checksum trait, and is asserted by this crate's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
struct Md5 {
|
struct Md5 {
|
||||||
hasher: md5::Md5,
|
hasher: md5::Md5,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "Content-MD5 is not a ChecksumAlgorithm variant and has no arm in into_impl: S3 carries it as its own header, separate from the x-amz-checksum-* family. This impl exists so the two paths share the Checksum trait, and is asserted by this crate's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
impl Md5 {
|
impl Md5 {
|
||||||
fn update(&mut self, bytes: &[u8]) {
|
fn update(&mut self, bytes: &[u8]) {
|
||||||
use md5::Digest;
|
use md5::Digest;
|
||||||
|
|||||||
@@ -42,6 +42,7 @@ chrono = { workspace = true, features = ["serde"] }
|
|||||||
jiff = { workspace = true, features = ["serde"] }
|
jiff = { workspace = true, features = ["serde"] }
|
||||||
metrics = { workspace = true }
|
metrics = { workspace = true }
|
||||||
serde = { workspace = true, features = ["derive"] }
|
serde = { workspace = true, features = ["derive"] }
|
||||||
|
smallvec = { workspace = true }
|
||||||
rmp-serde = { workspace = true }
|
rmp-serde = { workspace = true }
|
||||||
s3s = { workspace = true, features = ["minio"] }
|
s3s = { workspace = true, features = ["minio"] }
|
||||||
tracing = { workspace = true }
|
tracing = { workspace = true }
|
||||||
|
|||||||
@@ -287,6 +287,9 @@ pub enum HealRequestSource {
|
|||||||
Scanner,
|
Scanner,
|
||||||
AutoHeal,
|
AutoHeal,
|
||||||
ReadRepair,
|
ReadRepair,
|
||||||
|
/// Mission Repair Feed: intents delivered by error paths and replayed
|
||||||
|
/// from the durable MRF journal.
|
||||||
|
Mrf,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl HealRequestSource {
|
impl HealRequestSource {
|
||||||
@@ -297,6 +300,7 @@ impl HealRequestSource {
|
|||||||
Self::Scanner => "scanner",
|
Self::Scanner => "scanner",
|
||||||
Self::AutoHeal => "auto_heal",
|
Self::AutoHeal => "auto_heal",
|
||||||
Self::ReadRepair => "read_repair",
|
Self::ReadRepair => "read_repair",
|
||||||
|
Self::Mrf => "mrf",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -17,8 +17,10 @@ pub mod globals;
|
|||||||
pub mod heal_channel;
|
pub mod heal_channel;
|
||||||
pub mod last_minute;
|
pub mod last_minute;
|
||||||
pub mod metrics;
|
pub mod metrics;
|
||||||
|
pub mod mrf_channel;
|
||||||
mod readiness;
|
mod readiness;
|
||||||
pub mod table_catalog;
|
pub mod table_catalog;
|
||||||
|
pub mod trace_bus;
|
||||||
|
|
||||||
pub use globals::*;
|
pub use globals::*;
|
||||||
pub use readiness::{GlobalReadiness, SystemStage};
|
pub use readiness::{GlobalReadiness, SystemStage};
|
||||||
|
|||||||
@@ -0,0 +1,203 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! Mission Repair Feed (MRF) intent channel.
|
||||||
|
//!
|
||||||
|
//! Producers on error paths (read decode failure, scanner metadata
|
||||||
|
//! corruption, partial-write recovery) hand a lightweight [`MrfIntent`] to the
|
||||||
|
//! heal crate through a global bounded channel. Delivery is strictly
|
||||||
|
//! non-blocking: `try_send_mrf_intent` never awaits and drops the intent
|
||||||
|
//! (counting it) when the channel is full or uninitialized — losing one heal
|
||||||
|
//! hint is always preferred over stalling an IO path. Durable replay of
|
||||||
|
//! unconsumed intents is the consumer's job (see `rustfs-heal`
|
||||||
|
//! `heal::mrf_queue`), mirroring MinIO's `.heal/mrf/list.bin`.
|
||||||
|
|
||||||
|
use std::sync::{
|
||||||
|
Arc, OnceLock,
|
||||||
|
atomic::{AtomicBool, Ordering},
|
||||||
|
};
|
||||||
|
use tokio::sync::mpsc;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// Bounded capacity of the global MRF channel. Backpressure is resolved by
|
||||||
|
/// dropping (and counting) intents, never by blocking the producer.
|
||||||
|
const MRF_CHANNEL_CAPACITY: usize = 8192;
|
||||||
|
|
||||||
|
/// Why an intent was produced. Drives the heal priority mapping on the
|
||||||
|
/// consumer side (DecodeFailure -> Urgent, MetadataCorruption -> High,
|
||||||
|
/// PartialWrite -> Normal).
|
||||||
|
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||||
|
pub enum MrfKind {
|
||||||
|
/// Erasure decode failed while serving a read (read path).
|
||||||
|
DecodeFailure,
|
||||||
|
/// Scanner classified object metadata as corrupt.
|
||||||
|
MetadataCorruption,
|
||||||
|
/// A write left the object with fewer committed shards than the set size.
|
||||||
|
PartialWrite,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl MrfKind {
|
||||||
|
pub const fn as_str(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
MrfKind::DecodeFailure => "decode-failure",
|
||||||
|
MrfKind::MetadataCorruption => "metadata-corruption",
|
||||||
|
MrfKind::PartialWrite => "partial-write",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One repair intent. Kept deliberately small so the in-memory queue and the
|
||||||
|
/// journal stay bounded; `bucket`/`object` are `Arc<str>` so re-arming an
|
||||||
|
/// intent never re-allocates the strings.
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub struct MrfIntent {
|
||||||
|
pub bucket: Arc<str>,
|
||||||
|
pub object: Arc<str>,
|
||||||
|
/// Version the intent targets, as raw UUID bytes.
|
||||||
|
pub version_id: Option<[u8; 16]>,
|
||||||
|
pub kind: MrfKind,
|
||||||
|
pub enqueued_at_ms: u64,
|
||||||
|
/// Times this intent has already been offered to the heal manager.
|
||||||
|
/// Dropped by the consumer once it reaches `MRF_MAX_ATTEMPTS`.
|
||||||
|
pub attempts: u8,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Consumer-side retry ceiling before an intent is given up on.
|
||||||
|
pub const MRF_MAX_ATTEMPTS: u8 = 3;
|
||||||
|
|
||||||
|
impl MrfIntent {
|
||||||
|
/// Rough in-memory footprint used by the queue's byte budget.
|
||||||
|
pub fn estimated_bytes(&self) -> usize {
|
||||||
|
// Struct + strings + version bytes; buckets and objects are usually
|
||||||
|
// far below this bound, so rounding up keeps the budget conservative.
|
||||||
|
64 + self.bucket.len() + self.object.len()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static GLOBAL_MRF_SENDER: OnceLock<mpsc::Sender<MrfIntent>> = OnceLock::new();
|
||||||
|
|
||||||
|
/// Delivery kill-switch, set from `RUSTFS_HEAL_MRF_ENABLE`. Producers check
|
||||||
|
/// this before touching the channel so the disabled path stays allocation- and
|
||||||
|
/// sync-free.
|
||||||
|
static MRF_DELIVERY_ENABLED: AtomicBool = AtomicBool::new(true);
|
||||||
|
|
||||||
|
/// Override delivery (used at heal-runtime startup from configuration).
|
||||||
|
pub fn set_mrf_delivery_enabled(enabled: bool) {
|
||||||
|
MRF_DELIVERY_ENABLED.store(enabled, Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether producers currently deliver intents.
|
||||||
|
pub fn mrf_delivery_enabled() -> bool {
|
||||||
|
MRF_DELIVERY_ENABLED.load(Ordering::Relaxed)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Create the global MRF channel and return the consumer half. Fails if the
|
||||||
|
/// channel is already initialized (the heal runtime is a singleton).
|
||||||
|
pub fn init_mrf_channel() -> Result<mpsc::Receiver<MrfIntent>, &'static str> {
|
||||||
|
let (sender, receiver) = mpsc::channel(MRF_CHANNEL_CAPACITY);
|
||||||
|
GLOBAL_MRF_SENDER
|
||||||
|
.set(sender)
|
||||||
|
.map_err(|_| "MRF channel sender already initialized")?;
|
||||||
|
Ok(receiver)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Best-effort, non-blocking intent delivery from an error path.
|
||||||
|
///
|
||||||
|
/// Returns `true` when the intent was accepted into the channel. `false`
|
||||||
|
/// means the intent was dropped (feature disabled, channel not yet
|
||||||
|
/// initialized, or channel full) — callers must not retry or await; the
|
||||||
|
/// existing read-repair / scanner heal paths remain the safety net.
|
||||||
|
///
|
||||||
|
/// This runs on IO error paths, so it stays synchronous and cheap: one
|
||||||
|
/// bounded allocation for the two `Arc<str>` handles plus the channel slot.
|
||||||
|
pub fn try_send_mrf_intent(kind: MrfKind, bucket: &str, object: &str, version_id: Option<Uuid>) -> bool {
|
||||||
|
if !mrf_delivery_enabled() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
let Some(sender) = GLOBAL_MRF_SENDER.get() else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
let intent = MrfIntent {
|
||||||
|
bucket: Arc::from(bucket),
|
||||||
|
object: Arc::from(object),
|
||||||
|
version_id: version_id.map(|vid| *vid.as_bytes()),
|
||||||
|
kind,
|
||||||
|
enqueued_at_ms: unix_now_ms(),
|
||||||
|
attempts: 0,
|
||||||
|
};
|
||||||
|
sender.try_send(intent).is_ok()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn unix_now_ms() -> u64 {
|
||||||
|
// Kept trivial: the timestamp is diagnostic metadata only; wall-clock
|
||||||
|
// failure would be a bug rather than something to handle here.
|
||||||
|
std::time::SystemTime::now()
|
||||||
|
.duration_since(std::time::UNIX_EPOCH)
|
||||||
|
.map(|d| d.as_millis() as u64)
|
||||||
|
.unwrap_or(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn intents_estimate_is_conservative() {
|
||||||
|
let intent = MrfIntent {
|
||||||
|
bucket: Arc::from("bucket"),
|
||||||
|
object: Arc::from("object"),
|
||||||
|
version_id: Some([0u8; 16]),
|
||||||
|
kind: MrfKind::DecodeFailure,
|
||||||
|
enqueued_at_ms: 0,
|
||||||
|
attempts: 0,
|
||||||
|
};
|
||||||
|
assert!(intent.estimated_bytes() >= intent.bucket.len() + intent.object.len());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn try_send_delivers_and_respects_capacity() {
|
||||||
|
let mut receiver = init_mrf_channel().expect("first initialization should succeed");
|
||||||
|
assert!(init_mrf_channel().is_err(), "double initialization must fail");
|
||||||
|
|
||||||
|
assert!(try_send_mrf_intent(MrfKind::DecodeFailure, "b", "o", Some(Uuid::nil())));
|
||||||
|
let intent = receiver.recv().await.expect("intent should arrive");
|
||||||
|
assert_eq!(intent.kind, MrfKind::DecodeFailure);
|
||||||
|
assert_eq!(intent.bucket.as_ref(), "b");
|
||||||
|
|
||||||
|
// Disable delivery: producers become no-ops.
|
||||||
|
set_mrf_delivery_enabled(false);
|
||||||
|
assert!(!try_send_mrf_intent(MrfKind::PartialWrite, "b", "o", None));
|
||||||
|
set_mrf_delivery_enabled(true);
|
||||||
|
|
||||||
|
// Fill the bounded channel past capacity: excess intents are dropped,
|
||||||
|
// never blocking.
|
||||||
|
let mut accepted = 0;
|
||||||
|
for _ in 0..(MRF_CHANNEL_CAPACITY + 64) {
|
||||||
|
if try_send_mrf_intent(MrfKind::PartialWrite, "b", "o", None) {
|
||||||
|
accepted += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert_eq!(accepted, MRF_CHANNEL_CAPACITY);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn try_send_without_channel_is_false() {
|
||||||
|
// This test may run after the tokio test above in the same process;
|
||||||
|
// the singleton semantics make a clean "uninitialized" case hard, so
|
||||||
|
// assert the flag-off behavior only.
|
||||||
|
set_mrf_delivery_enabled(false);
|
||||||
|
assert!(!try_send_mrf_intent(MrfKind::MetadataCorruption, "b", "o", None));
|
||||||
|
set_mrf_delivery_enabled(true);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,333 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
use smallvec::SmallVec;
|
||||||
|
use std::{
|
||||||
|
sync::{
|
||||||
|
Arc, OnceLock,
|
||||||
|
atomic::{AtomicUsize, Ordering},
|
||||||
|
},
|
||||||
|
time::{Duration, SystemTime},
|
||||||
|
};
|
||||||
|
use tokio::sync::broadcast;
|
||||||
|
|
||||||
|
const DEFAULT_TRACE_BUS_CAPACITY: usize = 1024;
|
||||||
|
const TRACE_ATTR_INLINE_CAPACITY: usize = 8;
|
||||||
|
|
||||||
|
static GLOBAL_TRACE_BUS: OnceLock<TraceBus> = OnceLock::new();
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum TraceKind {
|
||||||
|
Heal,
|
||||||
|
Scanner,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TraceKind {
|
||||||
|
pub const fn as_str(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
Self::Heal => "heal",
|
||||||
|
Self::Scanner => "scanner",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum TraceFunc {
|
||||||
|
HealTask,
|
||||||
|
HealBucket,
|
||||||
|
HealObject,
|
||||||
|
HealCheckAbandonedParts,
|
||||||
|
HealErasureSetPage,
|
||||||
|
ScannerFolder,
|
||||||
|
ScannerIlmAction,
|
||||||
|
ScannerHealCandidate,
|
||||||
|
Dropped,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TraceFunc {
|
||||||
|
pub const fn as_str(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
Self::HealTask => "heal.Task",
|
||||||
|
Self::HealBucket => "heal.Bucket",
|
||||||
|
Self::HealObject => "heal.Object",
|
||||||
|
Self::HealCheckAbandonedParts => "heal.CheckAbandonedParts",
|
||||||
|
Self::HealErasureSetPage => "heal.ErasureSetPage",
|
||||||
|
Self::ScannerFolder => "scanner.Folder",
|
||||||
|
Self::ScannerIlmAction => "scanner.IlmAction",
|
||||||
|
Self::ScannerHealCandidate => "scanner.HealCandidate",
|
||||||
|
Self::Dropped => "trace.Dropped",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum TraceVal {
|
||||||
|
Bool(bool),
|
||||||
|
U64(u64),
|
||||||
|
I64(i64),
|
||||||
|
Str(Arc<str>),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<bool> for TraceVal {
|
||||||
|
fn from(value: bool) -> Self {
|
||||||
|
Self::Bool(value)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<u64> for TraceVal {
|
||||||
|
fn from(value: u64) -> Self {
|
||||||
|
Self::U64(value)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<i64> for TraceVal {
|
||||||
|
fn from(value: i64) -> Self {
|
||||||
|
Self::I64(value)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<&str> for TraceVal {
|
||||||
|
fn from(value: &str) -> Self {
|
||||||
|
Self::Str(Arc::from(value))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<String> for TraceVal {
|
||||||
|
fn from(value: String) -> Self {
|
||||||
|
Self::Str(Arc::from(value))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct TraceAttr {
|
||||||
|
pub key: &'static str,
|
||||||
|
pub value: TraceVal,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct TraceEvent {
|
||||||
|
pub kind: TraceKind,
|
||||||
|
pub func: TraceFunc,
|
||||||
|
pub time: SystemTime,
|
||||||
|
pub bucket: Option<Arc<str>>,
|
||||||
|
pub object: Option<Arc<str>>,
|
||||||
|
pub duration: Duration,
|
||||||
|
pub bytes: u64,
|
||||||
|
pub attrs: SmallVec<[TraceAttr; TRACE_ATTR_INLINE_CAPACITY]>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TraceEvent {
|
||||||
|
pub fn new(kind: TraceKind, func: TraceFunc) -> Self {
|
||||||
|
Self {
|
||||||
|
kind,
|
||||||
|
func,
|
||||||
|
time: SystemTime::now(),
|
||||||
|
bucket: None,
|
||||||
|
object: None,
|
||||||
|
duration: Duration::ZERO,
|
||||||
|
bytes: 0,
|
||||||
|
attrs: SmallVec::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_bucket(mut self, bucket: impl Into<Arc<str>>) -> Self {
|
||||||
|
self.bucket = Some(bucket.into());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_object(mut self, object: impl Into<Arc<str>>) -> Self {
|
||||||
|
self.object = Some(object.into());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_duration(mut self, duration: Duration) -> Self {
|
||||||
|
self.duration = duration;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_bytes(mut self, bytes: u64) -> Self {
|
||||||
|
self.bytes = bytes;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn with_attr(mut self, key: &'static str, value: impl Into<TraceVal>) -> Self {
|
||||||
|
self.attrs.push(TraceAttr {
|
||||||
|
key,
|
||||||
|
value: value.into(),
|
||||||
|
});
|
||||||
|
self
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct TraceBus {
|
||||||
|
sender: broadcast::Sender<Arc<TraceEvent>>,
|
||||||
|
subscriber_count: Arc<AtomicUsize>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TraceBus {
|
||||||
|
pub fn new(capacity: usize) -> Self {
|
||||||
|
let capacity = capacity.max(1);
|
||||||
|
let (sender, _receiver) = broadcast::channel(capacity);
|
||||||
|
Self {
|
||||||
|
sender,
|
||||||
|
subscriber_count: Arc::new(AtomicUsize::new(0)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn subscriber_count(&self) -> usize {
|
||||||
|
self.subscriber_count.load(Ordering::Acquire)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn subscribe(&self) -> TraceSubscription {
|
||||||
|
let receiver = self.sender.subscribe();
|
||||||
|
self.subscriber_count.fetch_add(1, Ordering::AcqRel);
|
||||||
|
TraceSubscription {
|
||||||
|
receiver,
|
||||||
|
subscriber_count: Arc::clone(&self.subscriber_count),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn emit(&self, build: impl FnOnce() -> TraceEvent) -> bool {
|
||||||
|
if self.subscriber_count() == 0 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
self.sender.send(Arc::new(build())).is_ok()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Default for TraceBus {
|
||||||
|
fn default() -> Self {
|
||||||
|
Self::new(DEFAULT_TRACE_BUS_CAPACITY)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct TraceSubscription {
|
||||||
|
receiver: broadcast::Receiver<Arc<TraceEvent>>,
|
||||||
|
subscriber_count: Arc<AtomicUsize>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TraceSubscription {
|
||||||
|
pub async fn recv(&mut self) -> Result<Arc<TraceEvent>, broadcast::error::RecvError> {
|
||||||
|
self.receiver.recv().await
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn try_recv(&mut self) -> Result<Arc<TraceEvent>, broadcast::error::TryRecvError> {
|
||||||
|
self.receiver.try_recv()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for TraceSubscription {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
self.subscriber_count.fetch_sub(1, Ordering::AcqRel);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn global_trace_bus() -> &'static TraceBus {
|
||||||
|
GLOBAL_TRACE_BUS.get_or_init(TraceBus::default)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn subscribe_trace_events() -> TraceSubscription {
|
||||||
|
global_trace_bus().subscribe()
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn trace_emit(build: impl FnOnce() -> TraceEvent) -> bool {
|
||||||
|
global_trace_bus().emit(build)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn trace_subscriber_count() -> usize {
|
||||||
|
global_trace_bus().subscriber_count()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use std::sync::atomic::AtomicUsize;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn trace_emit_skips_builder_without_subscribers() {
|
||||||
|
let bus = TraceBus::new(4);
|
||||||
|
let built = AtomicUsize::new(0);
|
||||||
|
|
||||||
|
let sent = bus.emit(|| {
|
||||||
|
built.fetch_add(1, Ordering::Relaxed);
|
||||||
|
TraceEvent::new(TraceKind::Heal, TraceFunc::HealTask)
|
||||||
|
});
|
||||||
|
|
||||||
|
assert!(!sent);
|
||||||
|
assert_eq!(built.load(Ordering::Relaxed), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn trace_subscriber_receives_event() {
|
||||||
|
let bus = TraceBus::new(4);
|
||||||
|
let mut subscription = bus.subscribe();
|
||||||
|
|
||||||
|
assert!(bus.emit(|| {
|
||||||
|
TraceEvent::new(TraceKind::Heal, TraceFunc::HealObject)
|
||||||
|
.with_bucket("bucket")
|
||||||
|
.with_object("object")
|
||||||
|
.with_duration(Duration::from_millis(7))
|
||||||
|
.with_bytes(11)
|
||||||
|
.with_attr("dry", true)
|
||||||
|
}));
|
||||||
|
|
||||||
|
let event = subscription
|
||||||
|
.recv()
|
||||||
|
.await
|
||||||
|
.expect("subscriber should receive emitted trace event");
|
||||||
|
|
||||||
|
assert_eq!(event.kind, TraceKind::Heal);
|
||||||
|
assert_eq!(event.func, TraceFunc::HealObject);
|
||||||
|
assert_eq!(event.bucket.as_deref(), Some("bucket"));
|
||||||
|
assert_eq!(event.object.as_deref(), Some("object"));
|
||||||
|
assert_eq!(event.duration, Duration::from_millis(7));
|
||||||
|
assert_eq!(event.bytes, 11);
|
||||||
|
assert_eq!(
|
||||||
|
event.attrs.as_slice(),
|
||||||
|
&[TraceAttr {
|
||||||
|
key: "dry",
|
||||||
|
value: TraceVal::Bool(true)
|
||||||
|
}]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn trace_subscription_drop_decrements_count() {
|
||||||
|
let bus = TraceBus::new(4);
|
||||||
|
let subscription = bus.subscribe();
|
||||||
|
|
||||||
|
assert_eq!(bus.subscriber_count(), 1);
|
||||||
|
drop(subscription);
|
||||||
|
assert_eq!(bus.subscriber_count(), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn lagged_subscriber_drops_events_without_blocking_publishers() {
|
||||||
|
let bus = TraceBus::new(2);
|
||||||
|
let mut subscription = bus.subscribe();
|
||||||
|
|
||||||
|
for index in 0_u64..4 {
|
||||||
|
assert!(bus.emit(|| { TraceEvent::new(TraceKind::Scanner, TraceFunc::ScannerFolder).with_attr("index", index) }));
|
||||||
|
}
|
||||||
|
|
||||||
|
let err = subscription
|
||||||
|
.recv()
|
||||||
|
.await
|
||||||
|
.expect_err("receiver should observe lag instead of blocking publishers");
|
||||||
|
assert!(matches!(err, broadcast::error::RecvError::Lagged(_)));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -177,3 +177,31 @@ pub const DEFAULT_HEAL_MAINLINE_WRITE_UTILIZATION_HIGH_PERCENT: usize = 80;
|
|||||||
|
|
||||||
/// Default foreground pressure recheck delay for heal scheduler, in milliseconds.
|
/// Default foreground pressure recheck delay for heal scheduler, in milliseconds.
|
||||||
pub const DEFAULT_HEAL_MAINLINE_MAX_SLEEP_MS: u64 = 250;
|
pub const DEFAULT_HEAL_MAINLINE_MAX_SLEEP_MS: u64 = 250;
|
||||||
|
|
||||||
|
/// Environment variable that toggles the MRF (mission repair feed) intent
|
||||||
|
/// pipeline: error paths deliver repair intents to the heal runtime, and
|
||||||
|
/// unconsumed intents are replayed from the durable journal after a restart.
|
||||||
|
pub const ENV_HEAL_MRF_ENABLE: &str = "RUSTFS_HEAL_MRF_ENABLE";
|
||||||
|
|
||||||
|
/// Environment variable for the MRF in-memory queue capacity (intent count).
|
||||||
|
pub const ENV_HEAL_MRF_QUEUE_SIZE: &str = "RUSTFS_HEAL_MRF_QUEUE_SIZE";
|
||||||
|
|
||||||
|
/// Environment variable for the MRF journal byte budget. The journal is
|
||||||
|
/// compacted once its on-disk size crosses this bound.
|
||||||
|
pub const ENV_HEAL_MRF_JOURNAL_MAX_BYTES: &str = "RUSTFS_HEAL_MRF_JOURNAL_MAX_BYTES";
|
||||||
|
|
||||||
|
/// Environment variable for the MRF journal replay batch size (intents per
|
||||||
|
/// replay push round).
|
||||||
|
pub const ENV_HEAL_MRF_REPLAY_BATCH: &str = "RUSTFS_HEAL_MRF_REPLAY_BATCH";
|
||||||
|
|
||||||
|
/// Default behavior keeps the MRF intent pipeline enabled.
|
||||||
|
pub const DEFAULT_HEAL_MRF_ENABLE: bool = true;
|
||||||
|
|
||||||
|
/// Default MRF queue capacity (matches MinIO's 100k MRF list ceiling).
|
||||||
|
pub const DEFAULT_HEAL_MRF_QUEUE_SIZE: usize = 100_000;
|
||||||
|
|
||||||
|
/// Default MRF journal byte budget (8 MiB), mirroring the channel payload cap.
|
||||||
|
pub const DEFAULT_HEAL_MRF_JOURNAL_MAX_BYTES: usize = 8 * 1024 * 1024;
|
||||||
|
|
||||||
|
/// Default MRF replay batch size.
|
||||||
|
pub const DEFAULT_HEAL_MRF_REPLAY_BATCH: usize = 256;
|
||||||
|
|||||||
@@ -234,6 +234,31 @@ pub const ENV_OBJECT_DISK_WRITE_ABSOLUTE_CAP: &str = "RUSTFS_OBJECT_DISK_WRITE_A
|
|||||||
/// Default absolute per-object erasure write cap in seconds (`0` = disabled).
|
/// Default absolute per-object erasure write cap in seconds (`0` = disabled).
|
||||||
pub const DEFAULT_OBJECT_DISK_WRITE_ABSOLUTE_CAP: u64 = 0;
|
pub const DEFAULT_OBJECT_DISK_WRITE_ABSOLUTE_CAP: u64 = 0;
|
||||||
|
|
||||||
|
/// Enable foreground PutObject request admission.
|
||||||
|
///
|
||||||
|
/// This is an experimental, default-off foreground write backpressure gate for
|
||||||
|
/// strict commit tail investigations. When disabled, PUTs follow the legacy
|
||||||
|
/// path and only the existing request counters are updated.
|
||||||
|
pub const ENV_PUT_FOREGROUND_ADMISSION_ENABLE: &str = "RUSTFS_PUT_FOREGROUND_ADMISSION_ENABLE";
|
||||||
|
pub const DEFAULT_PUT_FOREGROUND_ADMISSION_ENABLE: bool = false;
|
||||||
|
|
||||||
|
/// Maximum foreground PutObject requests admitted concurrently per process.
|
||||||
|
///
|
||||||
|
/// The limit is used only when [`ENV_PUT_FOREGROUND_ADMISSION_ENABLE`] is true.
|
||||||
|
/// A value of `0` disables the gate even when the enable flag is present, so a
|
||||||
|
/// partially configured rollout cannot reject every PUT.
|
||||||
|
pub const ENV_PUT_FOREGROUND_ADMISSION_LIMIT: &str = "RUSTFS_PUT_FOREGROUND_ADMISSION_LIMIT";
|
||||||
|
pub const DEFAULT_PUT_FOREGROUND_ADMISSION_LIMIT: usize = 0;
|
||||||
|
|
||||||
|
/// Time in milliseconds a foreground PutObject waits for an admission permit.
|
||||||
|
///
|
||||||
|
/// Once this timeout expires the request fails before body ingest/storage
|
||||||
|
/// mutation with S3 `SlowDown`/503. `0` means fail fast when the limit is full.
|
||||||
|
pub const ENV_PUT_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS: &str = "RUSTFS_PUT_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS";
|
||||||
|
pub const DEFAULT_PUT_FOREGROUND_ADMISSION_WAIT_TIMEOUT_MS: u64 = 0;
|
||||||
|
|
||||||
|
const _: () = assert!(!DEFAULT_PUT_FOREGROUND_ADMISSION_ENABLE);
|
||||||
|
|
||||||
/// Environment variable for minimum GetObject timeout in seconds.
|
/// Environment variable for minimum GetObject timeout in seconds.
|
||||||
///
|
///
|
||||||
/// When dynamic timeout calculation is enabled, this is the minimum timeout
|
/// When dynamic timeout calculation is enabled, this is the minimum timeout
|
||||||
|
|||||||
@@ -870,6 +870,157 @@ pub struct DataUsageCacheInfo {
|
|||||||
pub snapshot_complete: bool,
|
pub snapshot_complete: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Prefix-level usage over a raw entry map — the shared core behind
|
||||||
|
/// [`DataUsageCache::prefix_usage`], usable by any cache-shaped reader (the
|
||||||
|
/// scanner's writer-side cache has the same map type).
|
||||||
|
///
|
||||||
|
/// Cache keys are cleaned literal paths (`bucket/pre/fix`), so sub-prefix
|
||||||
|
/// names come straight off the child keys — no reverse mapping exists or is
|
||||||
|
/// needed. A compacted prefix carries its aggregate but no children, which
|
||||||
|
/// the `compacted` flag reports so callers can say why the breakdown is
|
||||||
|
/// empty. `truncated` is set when the breakdown exceeded `max_entries` and
|
||||||
|
/// was cut (largest first).
|
||||||
|
pub fn prefix_usage_in_cache(
|
||||||
|
cache: &HashMap<String, DataUsageEntry>,
|
||||||
|
bucket: &str,
|
||||||
|
prefix: &str,
|
||||||
|
max_entries: usize,
|
||||||
|
) -> Option<PrefixUsageQuery> {
|
||||||
|
let prefix = prefix.trim_matches('/');
|
||||||
|
let root = if prefix.is_empty() {
|
||||||
|
bucket.to_string()
|
||||||
|
} else {
|
||||||
|
format!("{bucket}/{prefix}")
|
||||||
|
};
|
||||||
|
let entry = cache.get(&hash_path(&root).key())?.clone();
|
||||||
|
|
||||||
|
let usage = PrefixUsageSummary::from_entry(&flatten_entry(cache, &entry, 0)?);
|
||||||
|
|
||||||
|
let child_prefix = format!("{root}/");
|
||||||
|
let mut sub_prefixes: Vec<PrefixUsageEntry> = entry
|
||||||
|
.children
|
||||||
|
.iter()
|
||||||
|
.filter_map(|child_key| {
|
||||||
|
let child = cache.get(child_key)?;
|
||||||
|
let child_flat = flatten_entry(cache, child, 1)?;
|
||||||
|
// Child keys are literal `bucket/pre/name` paths; a trailing
|
||||||
|
// slash marks a directory object and is display-only here.
|
||||||
|
let name = child_key
|
||||||
|
.strip_prefix(child_prefix.as_str())
|
||||||
|
.unwrap_or(child_key.as_str())
|
||||||
|
.trim_end_matches('/')
|
||||||
|
.to_string();
|
||||||
|
Some(PrefixUsageEntry {
|
||||||
|
prefix: name,
|
||||||
|
usage: PrefixUsageSummary::from_entry(&child_flat),
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
sub_prefixes.sort_by(|left, right| {
|
||||||
|
right
|
||||||
|
.usage
|
||||||
|
.size
|
||||||
|
.cmp(&left.usage.size)
|
||||||
|
.then_with(|| left.prefix.cmp(&right.prefix))
|
||||||
|
});
|
||||||
|
let truncated = sub_prefixes.len() > max_entries;
|
||||||
|
sub_prefixes.truncate(max_entries);
|
||||||
|
|
||||||
|
Some(PrefixUsageQuery {
|
||||||
|
usage,
|
||||||
|
compacted: entry.compacted,
|
||||||
|
truncated,
|
||||||
|
sub_prefixes,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Maximum subtree depth [`flatten_entry`] will walk before declaring the
|
||||||
|
/// cache corrupt — the same bound the scanner's checked flatten uses.
|
||||||
|
const PREFIX_USAGE_MAX_DEPTH: usize = 1024;
|
||||||
|
|
||||||
|
/// Flatten one entry's subtree into an aggregate: the free-function twin of
|
||||||
|
/// [`DataUsageCache::flatten`], carrying the scanner checked-flatten
|
||||||
|
/// hardening so a corrupt cache (cycles, over-deep trees, overflowing
|
||||||
|
/// counters) yields `None` instead of unbounded recursion or wrapped totals.
|
||||||
|
fn flatten_entry(cache: &HashMap<String, DataUsageEntry>, root: &DataUsageEntry, depth: usize) -> Option<DataUsageEntry> {
|
||||||
|
if depth > PREFIX_USAGE_MAX_DEPTH {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let mut flattened = DataUsageEntry::default();
|
||||||
|
if !flattened.checked_merge(root) {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
flattened.compacted = root.compacted;
|
||||||
|
// The root itself is not pre-seeded: it is merged above, and a corrupt
|
||||||
|
// child edge pointing back at the root's own key is still terminated by
|
||||||
|
// the visited set on first encounter.
|
||||||
|
let mut visited: HashSet<&str> = HashSet::new();
|
||||||
|
let mut pending: Vec<(&String, usize)> = root.children.iter().map(|child| (child, depth + 1)).collect();
|
||||||
|
while let Some((key, child_depth)) = pending.pop() {
|
||||||
|
if child_depth > PREFIX_USAGE_MAX_DEPTH || !visited.insert(key.as_str()) {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let entry = cache.get(key)?;
|
||||||
|
if !flattened.checked_merge(entry) {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
pending.extend(entry.children.iter().map(|child| (child, child_depth + 1)));
|
||||||
|
}
|
||||||
|
flattened.children.clear();
|
||||||
|
Some(flattened)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Flattened counters of one prefix subtree, as returned by
|
||||||
|
/// [`DataUsageCache::prefix_usage`].
|
||||||
|
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, serde::Serialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct PrefixUsageSummary {
|
||||||
|
pub size: u64,
|
||||||
|
pub objects: u64,
|
||||||
|
pub versions: u64,
|
||||||
|
pub delete_markers: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl PrefixUsageSummary {
|
||||||
|
fn from_entry(entry: &DataUsageEntry) -> Self {
|
||||||
|
Self {
|
||||||
|
size: entry.size as u64,
|
||||||
|
objects: entry.objects as u64,
|
||||||
|
versions: entry.versions as u64,
|
||||||
|
delete_markers: entry.delete_markers as u64,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Add another set's counters into this one (entries are partitioned by
|
||||||
|
/// set, so per-set results sum).
|
||||||
|
pub fn merge(&mut self, other: &Self) {
|
||||||
|
self.size = self.size.saturating_add(other.size);
|
||||||
|
self.objects = self.objects.saturating_add(other.objects);
|
||||||
|
self.versions = self.versions.saturating_add(other.versions);
|
||||||
|
self.delete_markers = self.delete_markers.saturating_add(other.delete_markers);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One first-level sub-prefix row of a [`PrefixUsageQuery`].
|
||||||
|
#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize)]
|
||||||
|
pub struct PrefixUsageEntry {
|
||||||
|
pub prefix: String,
|
||||||
|
pub usage: PrefixUsageSummary,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Result of [`DataUsageCache::prefix_usage`].
|
||||||
|
#[derive(Clone, Debug, Default, PartialEq, Eq, serde::Serialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct PrefixUsageQuery {
|
||||||
|
pub usage: PrefixUsageSummary,
|
||||||
|
/// The prefix entry was compacted by the scanner: its aggregate is valid
|
||||||
|
/// but no sub-prefix breakdown exists on disk.
|
||||||
|
pub compacted: bool,
|
||||||
|
/// The breakdown had more entries than `max_entries`; the largest remain.
|
||||||
|
pub truncated: bool,
|
||||||
|
pub sub_prefixes: Vec<PrefixUsageEntry>,
|
||||||
|
}
|
||||||
|
|
||||||
/// Read-only projection of a scanner-written `.usage-cache.bin` file.
|
/// Read-only projection of a scanner-written `.usage-cache.bin` file.
|
||||||
///
|
///
|
||||||
/// The scanner-side `DataUsageCache` (`crates/scanner/src/data_usage_define.rs`)
|
/// The scanner-side `DataUsageCache` (`crates/scanner/src/data_usage_define.rs`)
|
||||||
@@ -997,6 +1148,21 @@ impl DataUsageCache {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Prefix-level usage for one bucket subtree, plus the one-level
|
||||||
|
/// breakdown below it (rustfs/backlog#1872, MinIO
|
||||||
|
/// `loadPrefixUsageFromBackend` parity and beyond: arbitrary prefixes and
|
||||||
|
/// full counters instead of first-level sizes only).
|
||||||
|
///
|
||||||
|
/// Cache keys are cleaned literal paths (`bucket/pre/fix`), so sub-prefix
|
||||||
|
/// names come straight off the child keys — no reverse mapping exists or
|
||||||
|
/// is needed. A compacted prefix carries its aggregate but no children,
|
||||||
|
/// which the `compacted` flag reports so callers can say why the
|
||||||
|
/// breakdown is empty. `truncated` is set when the breakdown exceeded
|
||||||
|
/// `max_entries` and was cut (largest first).
|
||||||
|
pub fn prefix_usage(&self, bucket: &str, prefix: &str, max_entries: usize) -> Option<PrefixUsageQuery> {
|
||||||
|
prefix_usage_in_cache(&self.cache, bucket, prefix, max_entries)
|
||||||
|
}
|
||||||
|
|
||||||
pub fn force_compact(&mut self, limit: usize) {
|
pub fn force_compact(&mut self, limit: usize) {
|
||||||
if self.cache.len() < limit {
|
if self.cache.len() < limit {
|
||||||
return;
|
return;
|
||||||
@@ -1898,6 +2064,126 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Build a cache shaped like `bucket/{a,b/{c,d}},bucket/loose` with
|
||||||
|
/// distinct counters so aggregation is observable.
|
||||||
|
fn prefix_usage_fixture_cache() -> DataUsageCache {
|
||||||
|
let mut cache = DataUsageCache::default();
|
||||||
|
let mut insert = |path: &str, parent: &str, size: usize, objects: usize, versions: usize, delete_markers: usize| {
|
||||||
|
cache.replace(
|
||||||
|
path,
|
||||||
|
parent,
|
||||||
|
DataUsageEntry {
|
||||||
|
size,
|
||||||
|
objects,
|
||||||
|
versions,
|
||||||
|
delete_markers,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
);
|
||||||
|
};
|
||||||
|
insert("bucket", "", 0, 0, 0, 0);
|
||||||
|
insert("bucket/a", "bucket", 100, 1, 1, 0);
|
||||||
|
insert("bucket/b", "bucket", 0, 0, 0, 0);
|
||||||
|
insert("bucket/b/c", "bucket/b", 200, 2, 2, 1);
|
||||||
|
insert("bucket/b/d", "bucket/b", 40, 1, 3, 0);
|
||||||
|
insert("bucket/loose", "bucket", 10, 1, 1, 1);
|
||||||
|
cache
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn prefix_usage_aggregates_bucket_root_and_one_level_below() {
|
||||||
|
let cache = prefix_usage_fixture_cache();
|
||||||
|
|
||||||
|
let root = cache
|
||||||
|
.prefix_usage("bucket", "", 100)
|
||||||
|
.expect("root query must find the bucket entry");
|
||||||
|
assert_eq!(root.usage.size, 350, "root aggregate flattens the whole subtree");
|
||||||
|
assert_eq!(root.usage.objects, 5);
|
||||||
|
assert_eq!(root.usage.versions, 7);
|
||||||
|
assert_eq!(root.usage.delete_markers, 2);
|
||||||
|
assert!(!root.compacted);
|
||||||
|
assert!(!root.truncated);
|
||||||
|
// Breakdown is one level: b (240) before a (100) before loose (10),
|
||||||
|
// each flattened to its own subtree total.
|
||||||
|
let names: Vec<(&str, u64)> = root
|
||||||
|
.sub_prefixes
|
||||||
|
.iter()
|
||||||
|
.map(|entry| (entry.prefix.as_str(), entry.usage.size))
|
||||||
|
.collect();
|
||||||
|
assert_eq!(names, vec![("b", 240), ("a", 100), ("loose", 10)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn prefix_usage_drills_into_arbitrary_prefixes() {
|
||||||
|
let cache = prefix_usage_fixture_cache();
|
||||||
|
|
||||||
|
let b = cache.prefix_usage("bucket", "b", 100).expect("nested prefix must resolve");
|
||||||
|
assert_eq!(b.usage.size, 240);
|
||||||
|
assert_eq!(b.usage.versions, 5);
|
||||||
|
let names: Vec<&str> = b.sub_prefixes.iter().map(|entry| entry.prefix.as_str()).collect();
|
||||||
|
assert_eq!(names, vec!["c", "d"]);
|
||||||
|
|
||||||
|
// Prefix slashes are normalized away.
|
||||||
|
let slashed = cache.prefix_usage("bucket", "/b/", 100).expect("slash-insensitive lookup");
|
||||||
|
assert_eq!(slashed.usage.size, 240);
|
||||||
|
|
||||||
|
assert!(cache.prefix_usage("bucket", "absent", 100).is_none(), "unknown prefix must be a miss");
|
||||||
|
assert!(cache.prefix_usage("other", "", 100).is_none(), "unknown bucket must be a miss");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn prefix_usage_reports_and_respects_truncation() {
|
||||||
|
let cache = prefix_usage_fixture_cache();
|
||||||
|
let capped = cache.prefix_usage("bucket", "", 2).expect("root query");
|
||||||
|
assert!(capped.truncated, "three children capped to two must flag truncation");
|
||||||
|
let names: Vec<&str> = capped.sub_prefixes.iter().map(|entry| entry.prefix.as_str()).collect();
|
||||||
|
assert_eq!(names, vec!["b", "a"], "largest prefixes survive the cut");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn prefix_usage_marks_compacted_entries() {
|
||||||
|
let mut cache = DataUsageCache::default();
|
||||||
|
cache.replace(
|
||||||
|
"bucket",
|
||||||
|
"",
|
||||||
|
DataUsageEntry {
|
||||||
|
size: 999,
|
||||||
|
objects: 9,
|
||||||
|
compacted: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
let compacted = cache.prefix_usage("bucket", "", 100).expect("compacted root resolves");
|
||||||
|
assert!(compacted.compacted, "compaction must be visible to callers");
|
||||||
|
assert_eq!(compacted.usage.size, 999);
|
||||||
|
assert!(compacted.sub_prefixes.is_empty(), "a compacted entry carries no children");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn prefix_usage_rejects_cyclic_and_dangling_caches() {
|
||||||
|
// A self-referencing child (corrupt cache) must yield a miss for the
|
||||||
|
// whole query, not unbounded recursion.
|
||||||
|
let mut cache = prefix_usage_fixture_cache();
|
||||||
|
if let Some(entry) = cache.cache.get_mut("bucket/b") {
|
||||||
|
entry.children.insert("bucket/b".to_string());
|
||||||
|
}
|
||||||
|
assert!(cache.prefix_usage("bucket", "b", 100).is_none(), "a cyclic subtree must be rejected");
|
||||||
|
// The unaffected sibling still answers.
|
||||||
|
assert!(cache.prefix_usage("bucket", "a", 100).is_some());
|
||||||
|
|
||||||
|
// A child key with no entry (dangling link) is rejected rather than
|
||||||
|
// silently dropped: half a tree would under-report usage.
|
||||||
|
let mut dangling = prefix_usage_fixture_cache();
|
||||||
|
if let Some(entry) = dangling.cache.get_mut("bucket/b") {
|
||||||
|
entry.children.insert("bucket/b/ghost".to_string());
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
dangling.prefix_usage("bucket", "b", 100).is_none(),
|
||||||
|
"a dangling child link must be rejected"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn hash_path_uses_portable_slash_semantics() {
|
fn hash_path_uses_portable_slash_semantics() {
|
||||||
for (input, expected) in [
|
for (input, expected) in [
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ use rustfs_signer::sign_v4;
|
|||||||
use s3s::Body;
|
use s3s::Body;
|
||||||
use std::ffi::OsStr;
|
use std::ffi::OsStr;
|
||||||
use std::fs as stdfs;
|
use std::fs as stdfs;
|
||||||
|
use std::io::ErrorKind;
|
||||||
use std::path::{Path, PathBuf};
|
use std::path::{Path, PathBuf};
|
||||||
use std::process::{Child, Command, Stdio};
|
use std::process::{Child, Command, Stdio};
|
||||||
use std::sync::Once;
|
use std::sync::Once;
|
||||||
@@ -51,6 +52,11 @@ pub(crate) const FAST_DATA_USAGE_SCANNER_ENV: &[(&str, &str)] =
|
|||||||
&[("RUSTFS_SCANNER_CYCLE", "1"), ("RUSTFS_SCANNER_START_DELAY_SECS", "0")];
|
&[("RUSTFS_SCANNER_CYCLE", "1"), ("RUSTFS_SCANNER_START_DELAY_SECS", "0")];
|
||||||
pub const TEST_BUCKET: &str = "e2e-test-bucket";
|
pub const TEST_BUCKET: &str = "e2e-test-bucket";
|
||||||
const RUSTFS_FULL_FEATURE: &str = "full";
|
const RUSTFS_FULL_FEATURE: &str = "full";
|
||||||
|
const TEST_PORT_MIN: u16 = 20_000;
|
||||||
|
const TEST_PORT_RANGE: u16 = 40_000;
|
||||||
|
const TEST_PORT_COUNTER_PATH: &str = "/tmp/rustfs_e2e_next_port";
|
||||||
|
const TEST_PORT_LOCK_DIR: &str = "/tmp/rustfs_e2e_port_allocator.lock";
|
||||||
|
const TEST_PORT_LOCK_STALE_AFTER: Duration = Duration::from_secs(30);
|
||||||
|
|
||||||
fn capture_log_path(log_dir: &Path, temp_dir: &str) -> Option<PathBuf> {
|
fn capture_log_path(log_dir: &Path, temp_dir: &str) -> Option<PathBuf> {
|
||||||
let temp_name = Path::new(temp_dir).file_name()?.to_string_lossy();
|
let temp_name = Path::new(temp_dir).file_name()?.to_string_lossy();
|
||||||
@@ -67,6 +73,64 @@ fn configured_capture_log_path(temp_dir: &str) -> Option<String> {
|
|||||||
capture_log_path(Path::new(&log_dir), temp_dir).map(|path| path.to_string_lossy().into_owned())
|
capture_log_path(Path::new(&log_dir), temp_dir).map(|path| path.to_string_lossy().into_owned())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
struct PortAllocatorGuard;
|
||||||
|
|
||||||
|
impl PortAllocatorGuard {
|
||||||
|
async fn acquire() -> Result<Self, Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
loop {
|
||||||
|
match stdfs::create_dir(TEST_PORT_LOCK_DIR) {
|
||||||
|
Ok(()) => return Ok(Self),
|
||||||
|
Err(err) if err.kind() == ErrorKind::AlreadyExists => {
|
||||||
|
remove_stale_port_allocator_lock();
|
||||||
|
sleep(Duration::from_millis(10)).await;
|
||||||
|
}
|
||||||
|
Err(err) => return Err(err.into()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for PortAllocatorGuard {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
let _ = stdfs::remove_dir(TEST_PORT_LOCK_DIR);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn advance_test_port(port: u16) -> u16 {
|
||||||
|
let offset = (port - TEST_PORT_MIN + 1) % TEST_PORT_RANGE;
|
||||||
|
TEST_PORT_MIN + offset
|
||||||
|
}
|
||||||
|
|
||||||
|
fn seeded_test_port() -> u16 {
|
||||||
|
let offset = (Uuid::new_v4().as_u128() % u128::from(TEST_PORT_RANGE)) as u16;
|
||||||
|
TEST_PORT_MIN + offset
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_next_test_port() -> u16 {
|
||||||
|
stdfs::read_to_string(TEST_PORT_COUNTER_PATH)
|
||||||
|
.ok()
|
||||||
|
.and_then(|value| value.trim().parse::<u16>().ok())
|
||||||
|
.filter(|port| (TEST_PORT_MIN..TEST_PORT_MIN + TEST_PORT_RANGE).contains(port))
|
||||||
|
.unwrap_or_else(seeded_test_port)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remove_stale_port_allocator_lock() {
|
||||||
|
let Ok(metadata) = stdfs::metadata(TEST_PORT_LOCK_DIR) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let Ok(modified) = metadata.modified() else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if modified.elapsed().is_ok_and(|elapsed| elapsed > TEST_PORT_LOCK_STALE_AFTER) {
|
||||||
|
let _ = stdfs::remove_dir(TEST_PORT_LOCK_DIR);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn write_next_test_port(port: u16) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
stdfs::write(TEST_PORT_COUNTER_PATH, port.to_string())?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) fn capture_command_logs(
|
pub(crate) fn capture_command_logs(
|
||||||
command: &mut Command,
|
command: &mut Command,
|
||||||
log_path: Option<&str>,
|
log_path: Option<&str>,
|
||||||
@@ -508,10 +572,21 @@ impl RustFSTestEnvironment {
|
|||||||
/// Find an available port for the test
|
/// Find an available port for the test
|
||||||
pub async fn find_available_port() -> Result<u16, Box<dyn std::error::Error + Send + Sync>> {
|
pub async fn find_available_port() -> Result<u16, Box<dyn std::error::Error + Send + Sync>> {
|
||||||
use std::net::TcpListener;
|
use std::net::TcpListener;
|
||||||
let listener = TcpListener::bind("127.0.0.1:0")?;
|
let _guard = PortAllocatorGuard::acquire().await?;
|
||||||
let port = listener.local_addr()?.port();
|
let mut next_port = read_next_test_port();
|
||||||
drop(listener);
|
|
||||||
Ok(port)
|
for _ in 0..TEST_PORT_RANGE {
|
||||||
|
let port = next_port;
|
||||||
|
next_port = advance_test_port(next_port);
|
||||||
|
write_next_test_port(next_port)?;
|
||||||
|
|
||||||
|
if let Ok(listener) = TcpListener::bind(("127.0.0.1", port)) {
|
||||||
|
drop(listener);
|
||||||
|
return Ok(port);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Err("no available E2E test port found".into())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Kill any existing RustFS processes
|
/// Kill any existing RustFS processes
|
||||||
|
|||||||
@@ -373,14 +373,14 @@ pub mod error {
|
|||||||
|
|
||||||
pub mod erasure {
|
pub mod erasure {
|
||||||
pub use crate::erasure::coding::{
|
pub use crate::erasure::coding::{
|
||||||
BitrotReader, BitrotWriter, BitrotWriterWrapper, CustomWriter, Erasure, ErasureConstructionError, ReedSolomonEncoder,
|
BitrotReader, BitrotSelfTestError, BitrotWriter, BitrotWriterWrapper, CustomWriter, Erasure, ErasureConstructionError,
|
||||||
calc_shard_size, calc_shard_size_legacy,
|
ReedSolomonEncoder, bitrot_self_test, calc_shard_size, calc_shard_size_legacy,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
pub mod event {
|
pub mod event {
|
||||||
pub use crate::event::name::EventName;
|
pub use crate::event::name::EventName;
|
||||||
pub use crate::services::event_notification::{EventArgs, register_event_dispatch_hook};
|
pub use crate::services::event_notification::{EventArgs, register_event_dispatch_hook, send_event};
|
||||||
}
|
}
|
||||||
|
|
||||||
pub mod global {
|
pub mod global {
|
||||||
@@ -483,6 +483,7 @@ pub mod store_list {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub mod storage {
|
pub mod storage {
|
||||||
|
pub use crate::core::pools::HealLifecycleExpiryContext;
|
||||||
pub use crate::store::HealWalkVersion;
|
pub use crate::store::HealWalkVersion;
|
||||||
pub use crate::store::{
|
pub use crate::store::{
|
||||||
ECStore, all_local_disk, all_local_disk_path, find_local_disk_by_ref, init_local_disks,
|
ECStore, all_local_disk, all_local_disk_path, find_local_disk_by_ref, init_local_disks,
|
||||||
|
|||||||
@@ -1549,8 +1549,8 @@ impl Default for PutObjectOptions {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl PutObjectOptions {
|
impl PutObjectOptions {
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn set_match_etag(&mut self, etag: &str) {
|
fn set_match_etag(&mut self, etag: &str) {
|
||||||
if etag == "*" {
|
if etag == "*" {
|
||||||
self.custom_header
|
self.custom_header
|
||||||
@@ -1561,6 +1561,7 @@ impl PutObjectOptions {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn set_match_etag_except(&mut self, etag: &str) {
|
fn set_match_etag_except(&mut self, etag: &str) {
|
||||||
if etag == "*" {
|
if etag == "*" {
|
||||||
self.custom_header
|
self.custom_header
|
||||||
@@ -1696,6 +1697,7 @@ impl PutObjectOptions {
|
|||||||
header
|
header
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn validate(&self, _c: Arc<TargetClient>) -> Result<(), std::io::Error> {
|
fn validate(&self, _c: Arc<TargetClient>) -> Result<(), std::io::Error> {
|
||||||
//if self.checksum.is_set() {
|
//if self.checksum.is_set() {
|
||||||
/*if !self.trailing_header_support {
|
/*if !self.trailing_header_support {
|
||||||
|
|||||||
@@ -456,16 +456,23 @@ impl<'a> LifecycleExpiryTrace<'a> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl ExpiryStats {
|
impl ExpiryStats {
|
||||||
pub fn missed_tasks(&self) -> i64 {
|
pub fn missed_tasks(&self) -> i64 {
|
||||||
self.missed_expiry_tasks.load(Ordering::SeqCst)
|
self.missed_expiry_tasks.load(Ordering::SeqCst)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "asserted by this file's tests; the lib target cannot see test-only consumers (backlog#1823)"
|
||||||
|
)]
|
||||||
fn missed_free_vers_tasks(&self) -> i64 {
|
fn missed_free_vers_tasks(&self) -> i64 {
|
||||||
self.missed_freevers_tasks.load(Ordering::SeqCst)
|
self.missed_freevers_tasks.load(Ordering::SeqCst)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "asserted by this file's tests; the lib target cannot see test-only consumers (backlog#1823)"
|
||||||
|
)]
|
||||||
fn missed_tier_journal_tasks(&self) -> i64 {
|
fn missed_tier_journal_tasks(&self) -> i64 {
|
||||||
self.missed_tier_journal_tasks.load(Ordering::SeqCst)
|
self.missed_tier_journal_tasks.load(Ordering::SeqCst)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ pub mod core;
|
|||||||
pub mod evaluator;
|
pub mod evaluator;
|
||||||
pub mod manual_transition_job;
|
pub mod manual_transition_job;
|
||||||
mod metadata_boundary;
|
mod metadata_boundary;
|
||||||
pub(crate) use metadata_boundary::get_expiry_configs;
|
pub(crate) use metadata_boundary::{LifecycleExpiryConfigs, get_expiry_configs};
|
||||||
mod object_lock_boundary;
|
mod object_lock_boundary;
|
||||||
pub use self::core as lifecycle;
|
pub use self::core as lifecycle;
|
||||||
mod replication_sink;
|
mod replication_sink;
|
||||||
|
|||||||
@@ -80,7 +80,10 @@ impl LastDayTierStats {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "asserted by this file's tests; the lib target cannot see test-only consumers (backlog#1823)"
|
||||||
|
)]
|
||||||
fn merge(&self, m: LastDayTierStats) -> LastDayTierStats {
|
fn merge(&self, m: LastDayTierStats) -> LastDayTierStats {
|
||||||
let mut cl = self.clone();
|
let mut cl = self.clone();
|
||||||
let mut cm = m;
|
let mut cm = m;
|
||||||
|
|||||||
@@ -177,9 +177,10 @@ fn should_record_remote_delete_failure(err: &std::io::Error) -> bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Default)]
|
#[derive(Default)]
|
||||||
#[allow(dead_code)]
|
|
||||||
struct ObjSweeper {
|
struct ObjSweeper {
|
||||||
|
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||||
object: String,
|
object: String,
|
||||||
|
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||||
bucket: String,
|
bucket: String,
|
||||||
version_id: Option<Uuid>,
|
version_id: Option<Uuid>,
|
||||||
versioned: bool,
|
versioned: bool,
|
||||||
@@ -191,9 +192,9 @@ struct ObjSweeper {
|
|||||||
remote_object: String,
|
remote_object: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl ObjSweeper {
|
impl ObjSweeper {
|
||||||
#[allow(clippy::new_ret_no_self)]
|
#[allow(clippy::new_ret_no_self)]
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub async fn new(bucket: &str, object: &str) -> Result<Self, std::io::Error> {
|
pub async fn new(bucket: &str, object: &str) -> Result<Self, std::io::Error> {
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
object: object.into(),
|
object: object.into(),
|
||||||
@@ -202,17 +203,20 @@ impl ObjSweeper {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub fn with_version(&mut self, vid: Option<Uuid>) -> &Self {
|
pub fn with_version(&mut self, vid: Option<Uuid>) -> &Self {
|
||||||
self.version_id = vid.clone();
|
self.version_id = vid.clone();
|
||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub fn with_versioning(&mut self, versioned: bool, suspended: bool) -> &Self {
|
pub fn with_versioning(&mut self, versioned: bool, suspended: bool) -> &Self {
|
||||||
self.versioned = versioned;
|
self.versioned = versioned;
|
||||||
self.suspended = suspended;
|
self.suspended = suspended;
|
||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub fn get_opts(&self) -> lifecycle::ObjectOpts {
|
pub fn get_opts(&self) -> lifecycle::ObjectOpts {
|
||||||
let mut opts = ObjectOpts {
|
let mut opts = ObjectOpts {
|
||||||
version_id: self.version_id.clone(),
|
version_id: self.version_id.clone(),
|
||||||
@@ -226,6 +230,7 @@ impl ObjSweeper {
|
|||||||
opts
|
opts
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub fn set_transition_state(&mut self, info: TransitionedObject) {
|
pub fn set_transition_state(&mut self, info: TransitionedObject) {
|
||||||
self.transition_tier = info.tier;
|
self.transition_tier = info.tier;
|
||||||
self.transition_status = info.status;
|
self.transition_status = info.status;
|
||||||
@@ -266,6 +271,7 @@ impl ObjSweeper {
|
|||||||
None
|
None
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub async fn sweep(&self, api: Arc<ECStore>) {
|
pub async fn sweep(&self, api: Arc<ECStore>) {
|
||||||
let Some(je) = self.should_remove_remote_object() else {
|
let Some(je) = self.should_remove_remote_object() else {
|
||||||
return;
|
return;
|
||||||
|
|||||||
@@ -312,9 +312,7 @@ mod tests {
|
|||||||
}
|
}
|
||||||
#[derive(Deserialize)]
|
#[derive(Deserialize)]
|
||||||
struct LegacyBucketQuota {
|
struct LegacyBucketQuota {
|
||||||
#[allow(dead_code)]
|
|
||||||
quota: Option<u64>,
|
quota: Option<u64>,
|
||||||
#[allow(dead_code)]
|
|
||||||
quota_type: LegacyQuotaType,
|
quota_type: LegacyQuotaType,
|
||||||
}
|
}
|
||||||
let legacy = serde_json::from_slice::<LegacyBucketQuota>(&json)
|
let legacy = serde_json::from_slice::<LegacyBucketQuota>(&json)
|
||||||
|
|||||||
@@ -95,7 +95,6 @@ impl TransitionClient {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Default)]
|
#[derive(Default)]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct GetRequest {
|
pub struct GetRequest {
|
||||||
pub buffer: Vec<u8>,
|
pub buffer: Vec<u8>,
|
||||||
pub offset: i64,
|
pub offset: i64,
|
||||||
@@ -107,11 +106,12 @@ pub struct GetRequest {
|
|||||||
pub setting_object_info: bool,
|
pub setting_object_info: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct GetResponse {
|
pub struct GetResponse {
|
||||||
pub size: i64,
|
pub size: i64,
|
||||||
//pub error: error,
|
//pub error: error,
|
||||||
|
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||||
pub did_read: bool,
|
pub did_read: bool,
|
||||||
|
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||||
pub object_info: ObjectInfo,
|
pub object_info: ObjectInfo,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -27,7 +27,6 @@ use tracing::warn;
|
|||||||
use crate::client::api_error_response::err_invalid_argument;
|
use crate::client::api_error_response::err_invalid_argument;
|
||||||
|
|
||||||
#[derive(Default)]
|
#[derive(Default)]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct AdvancedGetOptions {
|
pub struct AdvancedGetOptions {
|
||||||
pub replication_delete_marker: bool,
|
pub replication_delete_marker: bool,
|
||||||
pub is_replication_ready_for_delete_marker: bool,
|
pub is_replication_ready_for_delete_marker: bool,
|
||||||
|
|||||||
@@ -360,7 +360,6 @@ impl TransitionClient {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Default)]
|
#[derive(Default)]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct ListObjectsOptions {
|
pub struct ListObjectsOptions {
|
||||||
reverse_versions: bool,
|
reverse_versions: bool,
|
||||||
with_versions: bool,
|
with_versions: bool,
|
||||||
|
|||||||
@@ -137,8 +137,8 @@ impl Default for PutObjectOptions {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl PutObjectOptions {
|
impl PutObjectOptions {
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn set_match_etag(&mut self, etag: &str) {
|
fn set_match_etag(&mut self, etag: &str) {
|
||||||
if etag == "*" {
|
if etag == "*" {
|
||||||
self.custom_header.insert("If-Match", HeaderValue::from_static("*"));
|
self.custom_header.insert("If-Match", HeaderValue::from_static("*"));
|
||||||
@@ -149,6 +149,7 @@ impl PutObjectOptions {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn set_match_etag_except(&mut self, etag: &str) {
|
fn set_match_etag_except(&mut self, etag: &str) {
|
||||||
if etag == "*" {
|
if etag == "*" {
|
||||||
self.custom_header.insert("If-None-Match", HeaderValue::from_static("*"));
|
self.custom_header.insert("If-None-Match", HeaderValue::from_static("*"));
|
||||||
@@ -259,6 +260,7 @@ impl PutObjectOptions {
|
|||||||
header
|
header
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn validate(&self, c: TransitionClient) -> Result<(), std::io::Error> {
|
fn validate(&self, c: TransitionClient) -> Result<(), std::io::Error> {
|
||||||
//if self.checksum.is_set() {
|
//if self.checksum.is_set() {
|
||||||
/*if !self.trailing_header_support {
|
/*if !self.trailing_header_support {
|
||||||
|
|||||||
@@ -55,7 +55,6 @@ pub struct RemoveBucketOptions {
|
|||||||
const DELETE_RESPONSE_PREVIEW_LEN: usize = 1024;
|
const DELETE_RESPONSE_PREVIEW_LEN: usize = 1024;
|
||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct AdvancedRemoveOptions {
|
pub struct AdvancedRemoveOptions {
|
||||||
pub replication_delete_marker: bool,
|
pub replication_delete_marker: bool,
|
||||||
pub replication_status: ReplicationStatus,
|
pub replication_status: ReplicationStatus,
|
||||||
@@ -465,10 +464,10 @@ impl TransitionClient {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Default)]
|
#[derive(Debug, Default)]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct RemoveObjectError {
|
pub struct RemoveObjectError {
|
||||||
|
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||||
object_name: String,
|
object_name: String,
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||||
version_id: String,
|
version_id: String,
|
||||||
err: Option<std::io::Error>,
|
err: Option<std::io::Error>,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -372,8 +372,8 @@ pub struct Checksum {
|
|||||||
computed: bool,
|
computed: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl Checksum {
|
impl Checksum {
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn new(t: ChecksumMode, b: &[u8]) -> Checksum {
|
fn new(t: ChecksumMode, b: &[u8]) -> Checksum {
|
||||||
if t.is_set() && b.len() == t.raw_byte_len() {
|
if t.is_set() && b.len() == t.raw_byte_len() {
|
||||||
return Checksum {
|
return Checksum {
|
||||||
@@ -385,7 +385,7 @@ impl Checksum {
|
|||||||
Checksum::default()
|
Checksum::default()
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn new_checksum_string(t: ChecksumMode, s: &str) -> Result<Checksum, std::io::Error> {
|
fn new_checksum_string(t: ChecksumMode, s: &str) -> Result<Checksum, std::io::Error> {
|
||||||
let b = match base64_decode(s.as_bytes()) {
|
let b = match base64_decode(s.as_bytes()) {
|
||||||
Ok(b) => b,
|
Ok(b) => b,
|
||||||
@@ -412,7 +412,7 @@ impl Checksum {
|
|||||||
base64_encode(&self.r)
|
base64_encode(&self.r)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn raw(&self) -> Option<Vec<u8>> {
|
fn raw(&self) -> Option<Vec<u8>> {
|
||||||
if !self.is_set() {
|
if !self.is_set() {
|
||||||
return None;
|
return None;
|
||||||
|
|||||||
@@ -37,16 +37,17 @@ pub struct PutObjReader {
|
|||||||
//pub sealMD5Fn: SealMD5CurrFn,
|
//pub sealMD5Fn: SealMD5CurrFn,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl PutObjReader {
|
impl PutObjReader {
|
||||||
pub fn new(reader: HashReader) -> Self {
|
pub fn new(reader: HashReader) -> Self {
|
||||||
Self { reader }
|
Self { reader }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn md5_current_hex_string(&self) -> String {
|
fn md5_current_hex_string(&self) -> String {
|
||||||
self.reader.checksum().map(|v| v.encoded).unwrap_or_default()
|
self.reader.checksum().map(|v| v.encoded).unwrap_or_default()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn with_encryption(&mut self, enc_reader: HashReader) -> Result<(), std::io::Error> {
|
fn with_encryption(&mut self, enc_reader: HashReader) -> Result<(), std::io::Error> {
|
||||||
self.reader = enc_reader;
|
self.reader = enc_reader;
|
||||||
|
|
||||||
|
|||||||
@@ -214,6 +214,19 @@ fn pool_write_quorum(participant_count: usize) -> usize {
|
|||||||
(participant_count / 2) + 1
|
(participant_count / 2) + 1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Error for a peer that reported `success = false` without an error payload.
|
||||||
|
///
|
||||||
|
/// The message must stay identical across the peers of one operation: `reduce_errs`
|
||||||
|
/// buckets `Error::Io` by kind plus rendered message, so any per-peer detail (address,
|
||||||
|
/// timing) would split one shared failure into single-count buckets and downgrade a real
|
||||||
|
/// dominant error into `ErasureWriteQuorum`.
|
||||||
|
fn peer_failure_without_details(op: &str, bucket: Option<&str>) -> Error {
|
||||||
|
match bucket {
|
||||||
|
Some(bucket) => Error::other(format!("{op}({bucket}): peer returned failure without error details")),
|
||||||
|
None => Error::other(format!("{op}: peer returned failure without error details")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn reduce_pool_write_quorum_errs(per_pool_errs: &[Option<Error>]) -> Option<Error> {
|
fn reduce_pool_write_quorum_errs(per_pool_errs: &[Option<Error>]) -> Option<Error> {
|
||||||
if per_pool_errs.is_empty() {
|
if per_pool_errs.is_empty() {
|
||||||
return Some(Error::ErasureWriteQuorum);
|
return Some(Error::ErasureWriteQuorum);
|
||||||
@@ -1078,7 +1091,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
|||||||
return if let Some(err) = response.error {
|
return if let Some(err) = response.error {
|
||||||
Err(err.into())
|
Err(err.into())
|
||||||
} else {
|
} else {
|
||||||
Err(Error::other(""))
|
Err(peer_failure_without_details("heal_bucket", Some(bucket)))
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1105,7 +1118,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
|||||||
return if let Some(err) = response.error {
|
return if let Some(err) = response.error {
|
||||||
Err(err.into())
|
Err(err.into())
|
||||||
} else {
|
} else {
|
||||||
Err(Error::other(""))
|
Err(peer_failure_without_details("list_bucket", None))
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
let bucket_infos = response
|
let bucket_infos = response
|
||||||
@@ -1136,9 +1149,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
|||||||
return if let Some(err) = response.error {
|
return if let Some(err) = response.error {
|
||||||
Err(err.into())
|
Err(err.into())
|
||||||
} else {
|
} else {
|
||||||
Err(Error::other(format!(
|
Err(peer_failure_without_details("make_bucket", Some(bucket)))
|
||||||
"make_bucket({bucket}): peer returned failure without error details"
|
|
||||||
)))
|
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1162,7 +1173,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
|||||||
return if let Some(err) = response.error {
|
return if let Some(err) = response.error {
|
||||||
Err(err.into())
|
Err(err.into())
|
||||||
} else {
|
} else {
|
||||||
Err(Error::other(""))
|
Err(peer_failure_without_details("get_bucket_info", Some(bucket)))
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
let bucket_info = serde_json::from_str::<BucketInfo>(&response.bucket_info)?;
|
let bucket_info = serde_json::from_str::<BucketInfo>(&response.bucket_info)?;
|
||||||
@@ -1190,7 +1201,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
|||||||
return if let Some(err) = response.error {
|
return if let Some(err) = response.error {
|
||||||
Err(err.into())
|
Err(err.into())
|
||||||
} else {
|
} else {
|
||||||
Err(Error::other(""))
|
Err(peer_failure_without_details("delete_bucket", Some(bucket)))
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2314,4 +2325,37 @@ mod tests {
|
|||||||
.collect::<Vec<_>>();
|
.collect::<Vec<_>>();
|
||||||
assert_eq!(calls, vec![1, 1, 0, 0, 0, 0, 0, 0]);
|
assert_eq!(calls, vec![1, 1, 0, 0, 0, 0, 0, 0]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn peer_failure_without_details_names_operation_and_bucket() {
|
||||||
|
for op in ["heal_bucket", "make_bucket", "get_bucket_info", "delete_bucket"] {
|
||||||
|
let message = peer_failure_without_details(op, Some("ops-bucket")).to_string();
|
||||||
|
assert!(message.contains(op), "{op} message must name the operation: {message}");
|
||||||
|
assert!(message.contains("ops-bucket"), "{op} message must name the bucket: {message}");
|
||||||
|
}
|
||||||
|
|
||||||
|
let message = peer_failure_without_details("list_bucket", None).to_string();
|
||||||
|
assert!(message.contains("list_bucket"), "cluster-wide message must name the operation");
|
||||||
|
assert!(!message.trim().is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn peer_failure_without_details_keeps_one_reduce_errs_bucket_per_operation() {
|
||||||
|
// reduce_errs groups Io errors by kind plus rendered message: peers failing the
|
||||||
|
// same operation on the same bucket must still reach quorum as one dominant error.
|
||||||
|
let per_pool_errs = vec![
|
||||||
|
Some(peer_failure_without_details("delete_bucket", Some("shared"))),
|
||||||
|
Some(peer_failure_without_details("delete_bucket", Some("shared"))),
|
||||||
|
Some(peer_failure_without_details("delete_bucket", Some("shared"))),
|
||||||
|
];
|
||||||
|
assert_eq!(
|
||||||
|
reduce_pool_write_quorum_errs(&per_pool_errs),
|
||||||
|
Some(peer_failure_without_details("delete_bucket", Some("shared")))
|
||||||
|
);
|
||||||
|
|
||||||
|
assert_ne!(
|
||||||
|
peer_failure_without_details("delete_bucket", Some("shared")),
|
||||||
|
peer_failure_without_details("get_bucket_info", Some("shared"))
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -39,7 +39,6 @@ use rustfs_config::{
|
|||||||
};
|
};
|
||||||
use std::sync::LazyLock;
|
use std::sync::LazyLock;
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
#[allow(clippy::declare_interior_mutable_const)]
|
#[allow(clippy::declare_interior_mutable_const)]
|
||||||
/// Default KVS for audit webhook settings.
|
/// Default KVS for audit webhook settings.
|
||||||
pub static DEFAULT_AUDIT_WEBHOOK_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
pub static DEFAULT_AUDIT_WEBHOOK_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||||
@@ -117,7 +116,6 @@ pub static DEFAULT_AUDIT_WEBHOOK_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
|||||||
])
|
])
|
||||||
});
|
});
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
#[allow(clippy::declare_interior_mutable_const)]
|
#[allow(clippy::declare_interior_mutable_const)]
|
||||||
/// Default KVS for audit MQTT settings.
|
/// Default KVS for audit MQTT settings.
|
||||||
pub static DEFAULT_AUDIT_MQTT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
pub static DEFAULT_AUDIT_MQTT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||||
@@ -375,7 +373,6 @@ pub static DEFAULT_AUDIT_NATS_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
|||||||
])
|
])
|
||||||
});
|
});
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub static DEFAULT_AUDIT_PULSAR_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
pub static DEFAULT_AUDIT_PULSAR_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||||
KVS(vec![
|
KVS(vec![
|
||||||
KV {
|
KV {
|
||||||
|
|||||||
@@ -12,12 +12,9 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
use crate::error::{Error, Result};
|
|
||||||
use rustfs_config::server_config::{KV, KVS};
|
use rustfs_config::server_config::{KV, KVS};
|
||||||
use rustfs_config::{DEFAULT_HEAL_BITROT_CYCLE_SECS, HEAL_BITROT_CYCLE};
|
use rustfs_config::{DEFAULT_HEAL_BITROT_CYCLE_SECS, HEAL_BITROT_CYCLE};
|
||||||
use rustfs_utils::string::parse_bool;
|
|
||||||
use std::sync::LazyLock;
|
use std::sync::LazyLock;
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||||
KVS(vec![KV {
|
KVS(vec![KV {
|
||||||
@@ -26,59 +23,3 @@ pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
|||||||
hidden_if_empty: false,
|
hidden_if_empty: false,
|
||||||
}])
|
}])
|
||||||
});
|
});
|
||||||
|
|
||||||
#[derive(Debug, Default)]
|
|
||||||
pub struct Config {
|
|
||||||
pub bitrot: String,
|
|
||||||
pub sleep: Duration,
|
|
||||||
pub io_count: usize,
|
|
||||||
pub drive_workers: usize,
|
|
||||||
pub cache: Duration,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Config {
|
|
||||||
pub fn bitrot_scan_cycle(&self) -> Duration {
|
|
||||||
self.cache
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn get_workers(&self) -> usize {
|
|
||||||
self.drive_workers
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn update(&mut self, nopts: &Config) {
|
|
||||||
self.bitrot = nopts.bitrot.clone();
|
|
||||||
self.io_count = nopts.io_count;
|
|
||||||
self.sleep = nopts.sleep;
|
|
||||||
self.drive_workers = nopts.drive_workers;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const RUSTFS_BITROT_CYCLE_IN_MONTHS: u64 = 1;
|
|
||||||
|
|
||||||
fn parse_bitrot_config(s: &str) -> Result<Duration> {
|
|
||||||
match parse_bool(s) {
|
|
||||||
Ok(enabled) => {
|
|
||||||
if enabled {
|
|
||||||
Ok(Duration::from_secs_f64(0.0))
|
|
||||||
} else {
|
|
||||||
Ok(Duration::from_secs_f64(-1.0))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(_) => {
|
|
||||||
if !s.ends_with("m") {
|
|
||||||
return Err(Error::other("unknown format"));
|
|
||||||
}
|
|
||||||
|
|
||||||
match s.trim_end_matches('m').parse::<u64>() {
|
|
||||||
Ok(months) => {
|
|
||||||
if months < RUSTFS_BITROT_CYCLE_IN_MONTHS {
|
|
||||||
return Err(Error::other(format!("minimum bitrot cycle is {RUSTFS_BITROT_CYCLE_IN_MONTHS} month(s)")));
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(Duration::from_secs(months * 30 * 24 * 60))
|
|
||||||
}
|
|
||||||
Err(err) => Err(Error::other(err)),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -16,7 +16,6 @@
|
|||||||
|
|
||||||
mod audit;
|
mod audit;
|
||||||
pub mod com;
|
pub mod com;
|
||||||
#[allow(dead_code)]
|
|
||||||
pub mod heal;
|
pub mod heal;
|
||||||
mod notify;
|
mod notify;
|
||||||
mod oidc;
|
mod oidc;
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ use crate::bucket::replication::replication_state_from_filemeta;
|
|||||||
use crate::bucket::versioning_sys::BucketVersioningSys;
|
use crate::bucket::versioning_sys::BucketVersioningSys;
|
||||||
use crate::bucket::{
|
use crate::bucket::{
|
||||||
lifecycle::{
|
lifecycle::{
|
||||||
|
LifecycleExpiryConfigs,
|
||||||
bucket_lifecycle_audit::LcEventSrc,
|
bucket_lifecycle_audit::LcEventSrc,
|
||||||
bucket_lifecycle_ops::{
|
bucket_lifecycle_ops::{
|
||||||
LifecycleOps, apply_expiry_on_transitioned_object, apply_expiry_rule_in, eval_action_from_lifecycle,
|
LifecycleOps, apply_expiry_on_transitioned_object, apply_expiry_rule_in, eval_action_from_lifecycle,
|
||||||
@@ -1996,11 +1997,11 @@ impl PoolMeta {
|
|||||||
Ok(false)
|
Ok(false)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn validate(&self, pools: Vec<Arc<Sets>>) -> Result<bool> {
|
pub fn validate(&self, pools: Vec<Arc<Sets>>) -> Result<bool> {
|
||||||
struct PoolInfo {
|
struct PoolInfo {
|
||||||
position: usize,
|
position: usize,
|
||||||
completed: bool,
|
completed: bool,
|
||||||
|
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||||
decom_started: bool,
|
decom_started: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2335,6 +2336,10 @@ fn lifecycle_action_removes_data_movement_version(action: IlmAction) -> bool {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn lifecycle_action_skips_heal_version(action: IlmAction) -> bool {
|
||||||
|
action.delete()
|
||||||
|
}
|
||||||
|
|
||||||
fn resolve_data_movement_lifecycle_expiry_result(action: IlmAction, apply_actions: bool, applied: bool) -> Result<bool> {
|
fn resolve_data_movement_lifecycle_expiry_result(action: IlmAction, apply_actions: bool, applied: bool) -> Result<bool> {
|
||||||
if !apply_actions || applied {
|
if !apply_actions || applied {
|
||||||
return Ok(true);
|
return Ok(true);
|
||||||
@@ -2385,7 +2390,80 @@ pub(crate) async fn should_skip_lifecycle_for_data_movement(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub struct HealLifecycleExpiryContext {
|
||||||
|
configs: LifecycleExpiryConfigs,
|
||||||
|
}
|
||||||
|
|
||||||
impl ECStore {
|
impl ECStore {
|
||||||
|
pub async fn load_heal_lifecycle_expiry_context(&self, bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
|
||||||
|
if bucket == RUSTFS_META_BUCKET {
|
||||||
|
return Ok(None);
|
||||||
|
}
|
||||||
|
|
||||||
|
let configs = get_expiry_configs(self, bucket).await?;
|
||||||
|
if configs.lifecycle.is_none() {
|
||||||
|
return Ok(None);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(Some(HealLifecycleExpiryContext { configs }))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn enqueue_heal_lifecycle_expiry(
|
||||||
|
self: &Arc<Self>,
|
||||||
|
context: &HealLifecycleExpiryContext,
|
||||||
|
bucket: &str,
|
||||||
|
object: &str,
|
||||||
|
version_id: Option<&str>,
|
||||||
|
object_info: Option<&crate::object_api::ObjectInfo>,
|
||||||
|
) -> Result<bool> {
|
||||||
|
let Some(lifecycle_config) = context.configs.lifecycle.as_ref() else {
|
||||||
|
return Ok(false);
|
||||||
|
};
|
||||||
|
|
||||||
|
let object_info = if let Some(object_info) = object_info {
|
||||||
|
if object_info.bucket != bucket || object_info.name != object {
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
let snapshot_version_id = object_info
|
||||||
|
.version_id
|
||||||
|
.filter(|version_id| !version_id.is_nil())
|
||||||
|
.map(|version_id| version_id.to_string());
|
||||||
|
if snapshot_version_id.as_deref() != version_id {
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
object_info.clone()
|
||||||
|
} else {
|
||||||
|
match self
|
||||||
|
.get_object_info(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&ObjectOptions {
|
||||||
|
version_id: version_id.map(str::to_string),
|
||||||
|
versioned: version_id.is_some(),
|
||||||
|
expected_bucket_incarnation_id: Some(context.configs.bucket_incarnation_id),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(object_info) => object_info,
|
||||||
|
Err(err) if is_err_object_not_found(&err) || is_err_version_not_found(&err) => return Ok(false),
|
||||||
|
Err(err) => return Err(err),
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
let event = eval_action_from_lifecycle(lifecycle_config, context.configs.object_lock.as_deref(), &object_info).await;
|
||||||
|
if !lifecycle_action_skips_heal_version(event.action) {
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
|
||||||
|
if lifecycle_delete_all_versions_blocked_by_replication(self.clone(), bucket, &object_info.name, event.action).await? {
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(apply_expiry_rule_in(self.clone(), &event, &LcEventSrc::Scanner, &object_info).await)
|
||||||
|
}
|
||||||
|
|
||||||
async fn save_current_pool_meta(&self) -> Result<()> {
|
async fn save_current_pool_meta(&self) -> Result<()> {
|
||||||
let _save_guard = self.pool_meta_save_gate.lock().await;
|
let _save_guard = self.pool_meta_save_gate.lock().await;
|
||||||
let snapshot = {
|
let snapshot = {
|
||||||
@@ -4287,6 +4365,19 @@ mod tests {
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn lifecycle_action_skips_heal_version_for_every_delete_action() {
|
||||||
|
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteAction));
|
||||||
|
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteVersionAction));
|
||||||
|
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteRestoredAction));
|
||||||
|
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteRestoredVersionAction));
|
||||||
|
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteAllVersionsAction));
|
||||||
|
assert!(lifecycle_action_skips_heal_version(IlmAction::DelMarkerDeleteAllVersionsAction));
|
||||||
|
assert!(!lifecycle_action_skips_heal_version(IlmAction::TransitionAction));
|
||||||
|
assert!(!lifecycle_action_skips_heal_version(IlmAction::TransitionVersionAction));
|
||||||
|
assert!(!lifecycle_action_skips_heal_version(IlmAction::NoneAction));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn resolve_data_movement_lifecycle_expiry_result_allows_dry_run_skip() {
|
fn resolve_data_movement_lifecycle_expiry_result_allows_dry_run_skip() {
|
||||||
let skip = resolve_data_movement_lifecycle_expiry_result(IlmAction::DeleteVersionAction, false, false)
|
let skip = resolve_data_movement_lifecycle_expiry_result(IlmAction::DeleteVersionAction, false, false)
|
||||||
@@ -4958,13 +5049,19 @@ fn is_disk_online_state(state: &str) -> bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[deprecated(since = "0.1.0", note = "Use fallback_total_capacity_dedup instead")]
|
#[deprecated(since = "0.1.0", note = "Use fallback_total_capacity_dedup instead")]
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "superseded by the replacement named in the comment at pools.rs:5071 (backlog#1823)"
|
||||||
|
)]
|
||||||
fn fallback_total_capacity(disks: &[rustfs_madmin::Disk]) -> usize {
|
fn fallback_total_capacity(disks: &[rustfs_madmin::Disk]) -> usize {
|
||||||
fallback_total_capacity_dedup(disks)
|
fallback_total_capacity_dedup(disks)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[deprecated(since = "0.1.0", note = "Use fallback_free_capacity_dedup instead")]
|
#[deprecated(since = "0.1.0", note = "Use fallback_free_capacity_dedup instead")]
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "superseded by the replacement named in the comment at pools.rs:5071 (backlog#1823)"
|
||||||
|
)]
|
||||||
fn fallback_free_capacity(disks: &[rustfs_madmin::Disk]) -> usize {
|
fn fallback_free_capacity(disks: &[rustfs_madmin::Disk]) -> usize {
|
||||||
fallback_free_capacity_dedup(disks)
|
fallback_free_capacity_dedup(disks)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1140,11 +1140,11 @@ impl crate::storage_api_contracts::heal::HealOperations for Sets {
|
|||||||
|
|
||||||
Err(Error::DiskNotFound)
|
Err(Error::DiskNotFound)
|
||||||
}
|
}
|
||||||
#[tracing::instrument(skip(self))]
|
#[tracing::instrument(level = "debug", skip(self, opts), fields(bucket = %bucket, object = %object, dry_run = opts.dry_run))]
|
||||||
async fn check_abandoned_parts(&self, _bucket: &str, _object: &str, _opts: &HealOpts) -> Result<()> {
|
async fn check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
|
||||||
// Multipart orphan reconciliation is intentionally retained above the pool/set layers
|
self.get_disks_for_heal_object(object, opts)?
|
||||||
// until there is a concrete caller and a stable lower-level contract to implement.
|
.check_abandoned_parts(bucket, object, opts)
|
||||||
Err(StorageError::NotImplemented)
|
.await
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1996,7 +1996,7 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn sets_check_abandoned_parts_returns_typed_not_implemented_error() {
|
async fn sets_check_abandoned_parts_rejects_invalid_set_scope() {
|
||||||
let format = FormatV3::new(1, 1);
|
let format = FormatV3::new(1, 1);
|
||||||
let sets = Sets {
|
let sets = Sets {
|
||||||
id: format.id,
|
id: format.id,
|
||||||
@@ -2021,10 +2021,21 @@ mod tests {
|
|||||||
};
|
};
|
||||||
|
|
||||||
let err = sets
|
let err = sets
|
||||||
.check_abandoned_parts("bucket", "object", &HealOpts::default())
|
.check_abandoned_parts(
|
||||||
|
"bucket",
|
||||||
|
"object",
|
||||||
|
&HealOpts {
|
||||||
|
set: Some(1),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.expect_err("abandoned-parts ownership should stay above the pool/set storage layers");
|
.expect_err("out-of-range abandoned-parts set scope must fail closed");
|
||||||
assert!(matches!(err, StorageError::NotImplemented));
|
assert!(
|
||||||
|
matches!(err, StorageError::InvalidArgument(_, ref field, ref reason)
|
||||||
|
if field == "set" && reason.contains("invalid heal set index 1")),
|
||||||
|
"unexpected invalid set error: {err:?}"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Builds a single-set `Sets` over `SET_DRIVE_COUNT` local temp-dir disks,
|
// Builds a single-set `Sets` over `SET_DRIVE_COUNT` local temp-dir disks,
|
||||||
|
|||||||
@@ -6562,7 +6562,7 @@ impl LocalDisk {
|
|||||||
Ok(f)
|
Ok(f)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn get_metrics(&self) -> DiskMetrics {
|
fn get_metrics(&self) -> DiskMetrics {
|
||||||
DiskMetrics::default()
|
DiskMetrics::default()
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -820,10 +820,263 @@ impl BitrotWriterWrapper {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// --- startup bitrot self-test (rustfs/backlog#1873, MinIO bitrotSelfTest parity) ---
|
||||||
|
//
|
||||||
|
// A broken hash implementation (bad SIMD feature combination, platform drift, a
|
||||||
|
// key-handling regression) fails silently: every shard reads back "corrupt",
|
||||||
|
// heal rewrites data that was fine, and cross-platform clusters disagree about
|
||||||
|
// which copy is healthy. The self-test below pins the algorithms the moment a
|
||||||
|
// process starts, so a drifted build announces itself instead of quietly
|
||||||
|
// rewriting objects. See docs/rustfs-heal-scanner-vs-minio-comprehensive-
|
||||||
|
// analysis-2026-08-16.md §6 HS-11.
|
||||||
|
|
||||||
|
/// Length of the deterministic self-test payload.
|
||||||
|
pub const BITROT_SELF_TEST_PAYLOAD_LEN: usize = 4096;
|
||||||
|
|
||||||
|
/// Known-answer digest of [`bitrot_self_test_payload`] under `HighwayHash256S`
|
||||||
|
/// (the production default). Pinned so any platform or build where the
|
||||||
|
/// implementation drifts fails startup instead of miss-hashing shards.
|
||||||
|
const BITROT_SELF_TEST_KAT_HIGHWAY_HASH256S: [u8; 32] = [
|
||||||
|
0xb9, 0x32, 0xa2, 0xaa, 0x4a, 0xb7, 0x33, 0x6a, 0xa3, 0xca, 0x7e, 0x61, 0x9d, 0x86, 0x52, 0x14, 0x6e, 0x7f, 0xd8, 0x9e, 0xea,
|
||||||
|
0x08, 0xd9, 0x8c, 0x33, 0x85, 0x87, 0x19, 0x30, 0xd6, 0xed, 0x06,
|
||||||
|
];
|
||||||
|
|
||||||
|
/// Known-answer digest of the same payload under `HighwayHash256SLegacy`.
|
||||||
|
const BITROT_SELF_TEST_KAT_HIGHWAY_HASH256S_LEGACY: [u8; 32] = [
|
||||||
|
0x98, 0x24, 0x71, 0x4f, 0x16, 0xbb, 0x48, 0x39, 0xed, 0x68, 0xfa, 0x63, 0x5e, 0xd9, 0x07, 0x61, 0xdf, 0x0a, 0xff, 0xcf, 0x7d,
|
||||||
|
0x8c, 0xa8, 0xc7, 0xc0, 0xb6, 0x6f, 0x05, 0xdb, 0xda, 0x5a, 0x22,
|
||||||
|
];
|
||||||
|
|
||||||
|
/// FIPS 180-2 test vector: SHA-256 of the ASCII string "abc". Unlike the
|
||||||
|
/// Highway digests above this one is externally verifiable, so it guards the
|
||||||
|
/// whole `HashAlgorithm` plumbing even for readers who distrust pinned
|
||||||
|
/// self-computed constants.
|
||||||
|
const BITROT_SELF_TEST_KAT_SHA256_ABC: [u8; 32] = [
|
||||||
|
0xba, 0x78, 0x16, 0xbf, 0x8f, 0x01, 0xcf, 0xea, 0x41, 0x41, 0x40, 0xde, 0x5d, 0xae, 0x22, 0x23, 0xb0, 0x03, 0x61, 0xa3, 0x96,
|
||||||
|
0x17, 0x7a, 0x9c, 0xb4, 0x10, 0xff, 0x61, 0xf2, 0x00, 0x15, 0xad,
|
||||||
|
];
|
||||||
|
|
||||||
|
/// Deterministic self-test payload: xorshift64* from a fixed seed, so every
|
||||||
|
/// platform and every run hashes the same 4096 bytes.
|
||||||
|
fn bitrot_self_test_payload() -> [u8; BITROT_SELF_TEST_PAYLOAD_LEN] {
|
||||||
|
let mut state = 0x9E37_79B9_7F4A_7C15u64;
|
||||||
|
let mut payload = [0u8; BITROT_SELF_TEST_PAYLOAD_LEN];
|
||||||
|
for byte in payload.iter_mut() {
|
||||||
|
state ^= state >> 12;
|
||||||
|
state ^= state << 25;
|
||||||
|
state ^= state >> 27;
|
||||||
|
*byte = state.wrapping_mul(0x2545_F491_4F6C_DD1D) as u8;
|
||||||
|
}
|
||||||
|
payload
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a bitrot self-test failed.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum BitrotSelfTestError {
|
||||||
|
/// A known-answer digest mismatched the pinned constant.
|
||||||
|
KnownAnswerMismatch {
|
||||||
|
algorithm: &'static str,
|
||||||
|
got: String,
|
||||||
|
want: String,
|
||||||
|
},
|
||||||
|
/// A freshly encoded shard failed `bitrot_verify`.
|
||||||
|
RoundtripVerify { algorithm: &'static str, detail: String },
|
||||||
|
/// A verified roundtrip read back different bytes than were written.
|
||||||
|
RoundtripReadback { algorithm: &'static str },
|
||||||
|
/// A deliberately tampered shard was not rejected by `bitrot_verify`.
|
||||||
|
TamperNotRejected {
|
||||||
|
algorithm: &'static str,
|
||||||
|
tampered: &'static str,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for BitrotSelfTestError {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
match self {
|
||||||
|
Self::KnownAnswerMismatch { algorithm, got, want } => {
|
||||||
|
write!(f, "known-answer mismatch for {algorithm}: got {got}, want {want}")
|
||||||
|
}
|
||||||
|
Self::RoundtripVerify { algorithm, detail } => write!(f, "{algorithm} roundtrip shard failed verification: {detail}"),
|
||||||
|
Self::RoundtripReadback { algorithm } => write!(f, "{algorithm} roundtrip read back different bytes"),
|
||||||
|
Self::TamperNotRejected { algorithm, tampered } => {
|
||||||
|
write!(f, "{algorithm} tampered shard ({tampered}) was not rejected")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::error::Error for BitrotSelfTestError {}
|
||||||
|
|
||||||
|
fn self_test_hex(bytes: &[u8]) -> String {
|
||||||
|
rustfs_utils::hex(bytes)
|
||||||
|
}
|
||||||
|
|
||||||
|
// (kept as a named one-liner so every KAT failure site reads the same; the
|
||||||
|
// underlying formatter is the shared `rustfs_utils::hex`)
|
||||||
|
|
||||||
|
/// Compare a digest against its pinned constant. Split out so a test can drive
|
||||||
|
/// it with a wrong constant and prove the mismatch path fires.
|
||||||
|
fn bitrot_kat_check(
|
||||||
|
algorithm: &'static str,
|
||||||
|
algo: &HashAlgorithm,
|
||||||
|
payload: &[u8],
|
||||||
|
expected: &[u8; 32],
|
||||||
|
) -> Result<(), BitrotSelfTestError> {
|
||||||
|
let digest = algo.hash_encode(payload);
|
||||||
|
let digest = digest.as_ref();
|
||||||
|
if digest.len() != expected.len() || digest != expected.as_slice() {
|
||||||
|
return Err(BitrotSelfTestError::KnownAnswerMismatch {
|
||||||
|
algorithm,
|
||||||
|
got: self_test_hex(digest),
|
||||||
|
want: self_test_hex(expected),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode `payload` with `shard_size` blocks, verify it end to end, and read
|
||||||
|
/// every block back through `BitrotReader` comparing bytes.
|
||||||
|
async fn bitrot_roundtrip_check(
|
||||||
|
algorithm: &'static str,
|
||||||
|
algo: HashAlgorithm,
|
||||||
|
payload: &[u8],
|
||||||
|
shard_size: usize,
|
||||||
|
) -> Result<(), BitrotSelfTestError> {
|
||||||
|
let mut writer = BitrotWriter::new(std::io::Cursor::new(Vec::<u8>::new()), shard_size, algo.clone());
|
||||||
|
for chunk in payload.chunks(shard_size) {
|
||||||
|
writer
|
||||||
|
.write(chunk)
|
||||||
|
.await
|
||||||
|
.map_err(|err| BitrotSelfTestError::RoundtripVerify {
|
||||||
|
algorithm,
|
||||||
|
detail: format!("encode failed: {err}"),
|
||||||
|
})?;
|
||||||
|
}
|
||||||
|
let encoded = writer.into_inner().into_inner();
|
||||||
|
|
||||||
|
let on_disk = bitrot_shard_file_size(payload.len(), shard_size, algo.clone());
|
||||||
|
if encoded.len() != on_disk {
|
||||||
|
return Err(BitrotSelfTestError::RoundtripVerify {
|
||||||
|
algorithm,
|
||||||
|
detail: format!("encoded {} bytes, size formula says {on_disk}", encoded.len()),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
bitrot_verify(std::io::Cursor::new(encoded.clone()), on_disk, payload.len(), algo.clone(), shard_size)
|
||||||
|
.await
|
||||||
|
.map_err(|err| BitrotSelfTestError::RoundtripVerify {
|
||||||
|
algorithm,
|
||||||
|
detail: err.to_string(),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
let mut reader = BitrotReader::new(std::io::Cursor::new(encoded), shard_size, algo, false);
|
||||||
|
let mut offset = 0usize;
|
||||||
|
while offset < payload.len() {
|
||||||
|
let want = shard_size.min(payload.len() - offset);
|
||||||
|
let mut buf = vec![0u8; want];
|
||||||
|
let read = reader
|
||||||
|
.read(&mut buf)
|
||||||
|
.await
|
||||||
|
.map_err(|err| BitrotSelfTestError::RoundtripVerify {
|
||||||
|
algorithm,
|
||||||
|
detail: format!("read back failed at offset {offset}: {err}"),
|
||||||
|
})?;
|
||||||
|
if read != want || buf[..read] != payload[offset..offset + read] {
|
||||||
|
return Err(BitrotSelfTestError::RoundtripReadback { algorithm });
|
||||||
|
}
|
||||||
|
offset += read;
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Flip one byte and require `bitrot_verify` to reject the result.
|
||||||
|
async fn bitrot_tamper_check(
|
||||||
|
algorithm: &'static str,
|
||||||
|
algo: HashAlgorithm,
|
||||||
|
payload: &[u8],
|
||||||
|
shard_size: usize,
|
||||||
|
tampered: &'static str,
|
||||||
|
flip_at: usize,
|
||||||
|
) -> Result<(), BitrotSelfTestError> {
|
||||||
|
let mut writer = BitrotWriter::new(std::io::Cursor::new(Vec::<u8>::new()), shard_size, algo.clone());
|
||||||
|
for chunk in payload.chunks(shard_size) {
|
||||||
|
writer.write(chunk).await.expect("self-test encode should not fail");
|
||||||
|
}
|
||||||
|
let mut corrupt = writer.into_inner().into_inner();
|
||||||
|
let flip_index = flip_at % corrupt.len();
|
||||||
|
corrupt[flip_index] ^= 0x80;
|
||||||
|
|
||||||
|
let on_disk = bitrot_shard_file_size(payload.len(), shard_size, algo.clone());
|
||||||
|
match bitrot_verify(std::io::Cursor::new(corrupt), on_disk, payload.len(), algo, shard_size).await {
|
||||||
|
// The flipped byte must be rejected as a hash mismatch specifically, not
|
||||||
|
// by any incidental read error: an in-memory cursor cannot fail reads,
|
||||||
|
// so accepting any other failure here would mask a verify path that
|
||||||
|
// errors out before it ever compares hashes.
|
||||||
|
Err(err) if err.to_string().contains("hash mismatch") => Ok(()),
|
||||||
|
Ok(()) => Err(BitrotSelfTestError::TamperNotRejected { algorithm, tampered }),
|
||||||
|
Err(err) => Err(BitrotSelfTestError::RoundtripVerify {
|
||||||
|
algorithm,
|
||||||
|
detail: format!("tampered shard rejected with an unexpected error: {err}"),
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Verify every bitrot algorithm this crate can write or verify in production:
|
||||||
|
/// both streaming Highway variants roundtrip end to end (encode → size formula
|
||||||
|
/// → `bitrot_verify` → read back) and reject a flipped byte in both the data
|
||||||
|
/// and the leading hash, while all three hashed algorithms reproduce their
|
||||||
|
/// pinned known-answer digests.
|
||||||
|
///
|
||||||
|
/// Runs in well under a millisecond on 4 KiB of data; callers may run it inline
|
||||||
|
/// at startup. Pure CPU, no allocation beyond a few KiB of scratch.
|
||||||
|
pub async fn bitrot_self_test() -> Result<(), BitrotSelfTestError> {
|
||||||
|
let payload = bitrot_self_test_payload();
|
||||||
|
|
||||||
|
// Externally verifiable vector first: it guards the HashAlgorithm plumbing
|
||||||
|
// itself, before any self-pinned constants are consulted.
|
||||||
|
let abc = HashAlgorithm::SHA256.hash_encode(b"abc");
|
||||||
|
if abc.as_ref() != BITROT_SELF_TEST_KAT_SHA256_ABC.as_slice() {
|
||||||
|
return Err(BitrotSelfTestError::KnownAnswerMismatch {
|
||||||
|
algorithm: "SHA256",
|
||||||
|
got: self_test_hex(abc.as_ref()),
|
||||||
|
want: self_test_hex(&BITROT_SELF_TEST_KAT_SHA256_ABC),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
bitrot_kat_check(
|
||||||
|
"HighwayHash256S",
|
||||||
|
&HashAlgorithm::HighwayHash256S,
|
||||||
|
&payload,
|
||||||
|
&BITROT_SELF_TEST_KAT_HIGHWAY_HASH256S,
|
||||||
|
)?;
|
||||||
|
bitrot_kat_check(
|
||||||
|
"HighwayHash256SLegacy",
|
||||||
|
&HashAlgorithm::HighwayHash256SLegacy,
|
||||||
|
&payload,
|
||||||
|
&BITROT_SELF_TEST_KAT_HIGHWAY_HASH256S_LEGACY,
|
||||||
|
)?;
|
||||||
|
|
||||||
|
for (algorithm, algo) in [
|
||||||
|
("HighwayHash256S", HashAlgorithm::HighwayHash256S),
|
||||||
|
("HighwayHash256SLegacy", HashAlgorithm::HighwayHash256SLegacy),
|
||||||
|
] {
|
||||||
|
// Full blocks plus a partial tail, exactly like a real part stripe.
|
||||||
|
let tail_len = 2 * 1024 + 333;
|
||||||
|
bitrot_roundtrip_check(algorithm, algo.clone(), &payload, 1024).await?;
|
||||||
|
bitrot_roundtrip_check(algorithm, algo.clone(), &payload[..tail_len], 1024).await?;
|
||||||
|
// One flipped byte in the final data block, one in the first leading
|
||||||
|
// hash: both must fail verification.
|
||||||
|
bitrot_tamper_check(algorithm, algo.clone(), &payload, 1024, "final data byte", payload.len() - 1).await?;
|
||||||
|
bitrot_tamper_check(algorithm, algo, &payload, 1024, "leading hash byte", 0).await?;
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::{
|
use super::{
|
||||||
BitrotReader, BitrotWriter, BitrotWriterWrapper, CustomWriter, bitrot_shard_file_size, bitrot_verify, write_all_vectored,
|
BitrotReader, BitrotWriter, BitrotWriterWrapper, CustomWriter, bitrot_kat_check, bitrot_self_test,
|
||||||
|
bitrot_self_test_payload, bitrot_shard_file_size, bitrot_verify, write_all_vectored,
|
||||||
};
|
};
|
||||||
use super::{MAX_RETAINED_CHUNKS_PER_BLOCK, ShardChunkRead, ShardSource};
|
use super::{MAX_RETAINED_CHUNKS_PER_BLOCK, ShardChunkRead, ShardSource};
|
||||||
use bytes::Bytes;
|
use bytes::Bytes;
|
||||||
@@ -1090,6 +1343,32 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bitrot_self_test_payload_is_deterministic() {
|
||||||
|
// Two independent builds of the payload must agree byte for byte, or
|
||||||
|
// the pinned known-answer digests below would be meaningless.
|
||||||
|
assert_eq!(bitrot_self_test_payload(), bitrot_self_test_payload());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bitrot_self_test_rejects_a_wrong_known_answer_digest() {
|
||||||
|
let payload = bitrot_self_test_payload();
|
||||||
|
let wrong = [0u8; 32];
|
||||||
|
let err = bitrot_kat_check("HighwayHash256S", &HashAlgorithm::HighwayHash256S, &payload, &wrong)
|
||||||
|
.expect_err("a zeroed digest must never match");
|
||||||
|
match err {
|
||||||
|
super::BitrotSelfTestError::KnownAnswerMismatch { algorithm, .. } => assert_eq!(algorithm, "HighwayHash256S"),
|
||||||
|
other => panic!("expected KnownAnswerMismatch, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn bitrot_self_test_passes() {
|
||||||
|
bitrot_self_test()
|
||||||
|
.await
|
||||||
|
.expect("the pinned digests and roundtrip checks must all pass on this platform");
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn vectored_test_writers_cover_fallback_flush_and_shutdown_paths() {
|
async fn vectored_test_writers_cover_fallback_flush_and_shutdown_paths() {
|
||||||
let mut counting = VectoredCountingWriter::default();
|
let mut counting = VectoredCountingWriter::default();
|
||||||
@@ -1189,7 +1468,7 @@ mod tests {
|
|||||||
let last = corrupt.len() - 1;
|
let last = corrupt.len() - 1;
|
||||||
corrupt[last] ^= 0x80;
|
corrupt[last] ^= 0x80;
|
||||||
let err = bitrot_verify(
|
let err = bitrot_verify(
|
||||||
Cursor::new(corrupt),
|
std::io::Cursor::new(corrupt),
|
||||||
super::bitrot_shard_file_size(data.len(), shard_size, algo.clone()),
|
super::bitrot_shard_file_size(data.len(), shard_size, algo.clone()),
|
||||||
data.len(),
|
data.len(),
|
||||||
algo,
|
algo,
|
||||||
@@ -1282,7 +1561,7 @@ mod tests {
|
|||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn bitrot_reader_rejects_output_buffers_larger_than_shard_size() {
|
async fn bitrot_reader_rejects_output_buffers_larger_than_shard_size() {
|
||||||
let mut reader = BitrotReader::new(Cursor::new(Vec::<u8>::new()), 4, HashAlgorithm::None, false);
|
let mut reader = BitrotReader::new(std::io::Cursor::new(Vec::<u8>::new()), 4, HashAlgorithm::None, false);
|
||||||
let mut out = [0u8; 5];
|
let mut out = [0u8; 5];
|
||||||
let err = reader
|
let err = reader
|
||||||
.read(&mut out)
|
.read(&mut out)
|
||||||
@@ -1407,7 +1686,7 @@ mod tests {
|
|||||||
(HashAlgorithm::HighwayHash256, true),
|
(HashAlgorithm::HighwayHash256, true),
|
||||||
] {
|
] {
|
||||||
let label = format!("{algo:?}");
|
let label = format!("{algo:?}");
|
||||||
let writer = Cursor::new(Vec::<u8>::new());
|
let writer = std::io::Cursor::new(Vec::<u8>::new());
|
||||||
let mut w = BitrotWriter::new(writer, shard_size, algo.clone());
|
let mut w = BitrotWriter::new(writer, shard_size, algo.clone());
|
||||||
w.write(&[7u8; 16]).await.unwrap();
|
w.write(&[7u8; 16]).await.unwrap();
|
||||||
let written = w.into_inner().into_inner();
|
let written = w.into_inner().into_inner();
|
||||||
@@ -1492,7 +1771,7 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn encode_one_block(payload: &[u8], shard_size: usize, algo: HashAlgorithm) -> Vec<u8> {
|
async fn encode_one_block(payload: &[u8], shard_size: usize, algo: HashAlgorithm) -> Vec<u8> {
|
||||||
let mut w = BitrotWriter::new(Cursor::new(Vec::<u8>::new()), shard_size, algo);
|
let mut w = BitrotWriter::new(std::io::Cursor::new(Vec::<u8>::new()), shard_size, algo);
|
||||||
w.write(payload).await.unwrap();
|
w.write(payload).await.unwrap();
|
||||||
w.into_inner().into_inner()
|
w.into_inner().into_inner()
|
||||||
}
|
}
|
||||||
@@ -1600,7 +1879,7 @@ mod tests {
|
|||||||
for algo in [HashAlgorithm::HighwayHash256S, HashAlgorithm::HighwayHash256SLegacy] {
|
for algo in [HashAlgorithm::HighwayHash256S, HashAlgorithm::HighwayHash256SLegacy] {
|
||||||
for &size in &[1usize, 16, 17, 32, 40, 48] {
|
for &size in &[1usize, 16, 17, 32, 40, 48] {
|
||||||
let payload: Vec<u8> = (0..size).map(|i| i as u8).collect();
|
let payload: Vec<u8> = (0..size).map(|i| i as u8).collect();
|
||||||
let mut w = BitrotWriter::new(Cursor::new(Vec::<u8>::new()), shard_size, algo.clone());
|
let mut w = BitrotWriter::new(std::io::Cursor::new(Vec::<u8>::new()), shard_size, algo.clone());
|
||||||
for chunk in payload.chunks(shard_size) {
|
for chunk in payload.chunks(shard_size) {
|
||||||
w.write(chunk).await.unwrap();
|
w.write(chunk).await.unwrap();
|
||||||
}
|
}
|
||||||
@@ -1674,14 +1953,14 @@ mod tests {
|
|||||||
w.write(&data).await.expect("write shard");
|
w.write(&data).await.expect("write shard");
|
||||||
|
|
||||||
let mut via_read = vec![0u8; SHARD];
|
let mut via_read = vec![0u8; SHARD];
|
||||||
let n1 = BitrotReader::new(Cursor::new(encoded.clone()), SHARD, algo.clone(), false)
|
let n1 = BitrotReader::new(std::io::Cursor::new(encoded.clone()), SHARD, algo.clone(), false)
|
||||||
.read(&mut via_read)
|
.read(&mut via_read)
|
||||||
.await
|
.await
|
||||||
.expect("read");
|
.expect("read");
|
||||||
|
|
||||||
// A buffer with only capacity — no initialized bytes at all.
|
// A buffer with only capacity — no initialized bytes at all.
|
||||||
let mut via_append: Vec<u8> = Vec::with_capacity(SHARD);
|
let mut via_append: Vec<u8> = Vec::with_capacity(SHARD);
|
||||||
let n2 = BitrotReader::new(Cursor::new(encoded), SHARD, algo.clone(), false)
|
let n2 = BitrotReader::new(std::io::Cursor::new(encoded), SHARD, algo.clone(), false)
|
||||||
.read_appending(&mut via_append, SHARD)
|
.read_appending(&mut via_append, SHARD)
|
||||||
.await
|
.await
|
||||||
.expect("read_appending");
|
.expect("read_appending");
|
||||||
@@ -1706,7 +1985,7 @@ mod tests {
|
|||||||
encoded.truncate(encoded.len() - 1);
|
encoded.truncate(encoded.len() - 1);
|
||||||
|
|
||||||
let mut out: Vec<u8> = Vec::with_capacity(SHARD);
|
let mut out: Vec<u8> = Vec::with_capacity(SHARD);
|
||||||
let err = BitrotReader::new(Cursor::new(encoded), SHARD, algo.clone(), false)
|
let err = BitrotReader::new(std::io::Cursor::new(encoded), SHARD, algo.clone(), false)
|
||||||
.read_appending(&mut out, SHARD)
|
.read_appending(&mut out, SHARD)
|
||||||
.await
|
.await
|
||||||
.expect_err("a truncated shard must not succeed");
|
.expect_err("a truncated shard must not succeed");
|
||||||
@@ -1732,7 +2011,7 @@ mod tests {
|
|||||||
encoded[last] ^= 0xff;
|
encoded[last] ^= 0xff;
|
||||||
|
|
||||||
let mut out: Vec<u8> = Vec::with_capacity(SHARD);
|
let mut out: Vec<u8> = Vec::with_capacity(SHARD);
|
||||||
let err = BitrotReader::new(Cursor::new(encoded), SHARD, algo, false)
|
let err = BitrotReader::new(std::io::Cursor::new(encoded), SHARD, algo, false)
|
||||||
.read_appending(&mut out, SHARD)
|
.read_appending(&mut out, SHARD)
|
||||||
.await
|
.await
|
||||||
.expect_err("a corrupt shard must not verify");
|
.expect_err("a corrupt shard must not verify");
|
||||||
@@ -1844,7 +2123,7 @@ mod tests {
|
|||||||
"Cursor<Bytes> must be able to hand out a block, otherwise the fast path is dead code"
|
"Cursor<Bytes> must be able to hand out a block, otherwise the fast path is dead code"
|
||||||
);
|
);
|
||||||
assert_eq!(mem.position(), 8, "taking a block must advance like a read of the same length");
|
assert_eq!(mem.position(), 8, "taking a block must advance like a read of the same length");
|
||||||
let mut streamed = Cursor::new(encoded.clone());
|
let mut streamed = std::io::Cursor::new(encoded.clone());
|
||||||
assert!(
|
assert!(
|
||||||
ShardSource::try_take_block(&mut streamed, 8).is_none(),
|
ShardSource::try_take_block(&mut streamed, 8).is_none(),
|
||||||
"a non-Bytes source must stay on the streaming path"
|
"a non-Bytes source must stay on the streaming path"
|
||||||
@@ -1872,7 +2151,7 @@ mod tests {
|
|||||||
);
|
);
|
||||||
|
|
||||||
let mut via_stream: Vec<u8> = Vec::with_capacity(SHARD);
|
let mut via_stream: Vec<u8> = Vec::with_capacity(SHARD);
|
||||||
BitrotReader::new(Cursor::new(encoded), SHARD, algo, false)
|
BitrotReader::new(std::io::Cursor::new(encoded), SHARD, algo, false)
|
||||||
.read_appending(&mut via_stream, SHARD)
|
.read_appending(&mut via_stream, SHARD)
|
||||||
.await
|
.await
|
||||||
.expect("streaming read");
|
.expect("streaming read");
|
||||||
|
|||||||
@@ -132,7 +132,6 @@ impl RebalanceStopPropagationRecord {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
#[derive(Debug, Clone, Default)]
|
#[derive(Debug, Clone, Default)]
|
||||||
pub struct DiskStat {
|
pub struct DiskStat {
|
||||||
pub total_space: u64,
|
pub total_space: u64,
|
||||||
|
|||||||
@@ -16,8 +16,16 @@ use serde::{Deserialize, Serialize};
|
|||||||
use std::{fmt::Display, io};
|
use std::{fmt::Display, io};
|
||||||
use tracing::info;
|
use tracing::info;
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "tier config wire version stamped by the parity constructors below (backlog#1823)"
|
||||||
|
)]
|
||||||
const C_TIER_CONFIG_VER: &str = "v1";
|
const C_TIER_CONFIG_VER: &str = "v1";
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "tier-name validation message reached only from the parity constructors below (backlog#1823)"
|
||||||
|
)]
|
||||||
const ERR_TIER_NAME_EMPTY: &str = "remote tier name empty";
|
const ERR_TIER_NAME_EMPTY: &str = "remote tier name empty";
|
||||||
const WASABI_US_EAST_ENDPOINT: &str = "https://s3.wasabisys.com";
|
const WASABI_US_EAST_ENDPOINT: &str = "https://s3.wasabisys.com";
|
||||||
const WASABI_ALTERNATIVE_ENDPOINTS: &[(&str, &str)] = &[
|
const WASABI_ALTERNATIVE_ENDPOINTS: &[(&str, &str)] = &[
|
||||||
@@ -264,7 +272,6 @@ impl Clone for TierConfig {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl TierConfig {
|
impl TierConfig {
|
||||||
pub(crate) fn clone_with_credentials(&self) -> Self {
|
pub(crate) fn clone_with_credentials(&self) -> Self {
|
||||||
Self {
|
Self {
|
||||||
@@ -284,6 +291,7 @@ impl TierConfig {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn endpoint(&self) -> String {
|
fn endpoint(&self) -> String {
|
||||||
match self.tier_type {
|
match self.tier_type {
|
||||||
TierType::S3 => self.s3.as_ref().map(|s| s.endpoint.clone()).unwrap_or_default(),
|
TierType::S3 => self.s3.as_ref().map(|s| s.endpoint.clone()).unwrap_or_default(),
|
||||||
@@ -303,6 +311,7 @@ impl TierConfig {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn bucket(&self) -> String {
|
fn bucket(&self) -> String {
|
||||||
match self.tier_type {
|
match self.tier_type {
|
||||||
TierType::S3 => self.s3.as_ref().map(|s| s.bucket.clone()).unwrap_or_default(),
|
TierType::S3 => self.s3.as_ref().map(|s| s.bucket.clone()).unwrap_or_default(),
|
||||||
@@ -322,6 +331,7 @@ impl TierConfig {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn prefix(&self) -> String {
|
fn prefix(&self) -> String {
|
||||||
match self.tier_type {
|
match self.tier_type {
|
||||||
TierType::S3 => self.s3.as_ref().map(|s| s.prefix.clone()).unwrap_or_default(),
|
TierType::S3 => self.s3.as_ref().map(|s| s.prefix.clone()).unwrap_or_default(),
|
||||||
@@ -341,6 +351,7 @@ impl TierConfig {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn region(&self) -> String {
|
fn region(&self) -> String {
|
||||||
match self.tier_type {
|
match self.tier_type {
|
||||||
TierType::S3 => self.s3.as_ref().map(|s| s.region.clone()).unwrap_or_default(),
|
TierType::S3 => self.s3.as_ref().map(|s| s.region.clone()).unwrap_or_default(),
|
||||||
@@ -457,7 +468,7 @@ impl TierWasabi {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl TierS3 {
|
impl TierS3 {
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn create<F>(
|
fn create<F>(
|
||||||
name: &str,
|
name: &str,
|
||||||
access_key: &str,
|
access_key: &str,
|
||||||
@@ -528,7 +539,7 @@ pub struct TierMinIO {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl TierMinIO {
|
impl TierMinIO {
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
fn create<F>(
|
fn create<F>(
|
||||||
name: &str,
|
name: &str,
|
||||||
endpoint: &str,
|
endpoint: &str,
|
||||||
|
|||||||
@@ -14,7 +14,6 @@
|
|||||||
|
|
||||||
use crate::services::tier::tier::TierConfigMgr;
|
use crate::services::tier::tier::TierConfigMgr;
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl TierConfigMgr {
|
impl TierConfigMgr {
|
||||||
pub fn msg_size(&self) -> usize {
|
pub fn msg_size(&self) -> usize {
|
||||||
100
|
100
|
||||||
|
|||||||
@@ -4860,6 +4860,14 @@ impl SetDisks {
|
|||||||
/// is best-effort maintenance: individual delete failures are logged and
|
/// is best-effort maintenance: individual delete failures are logged and
|
||||||
/// skipped rather than propagated.
|
/// skipped rather than propagated.
|
||||||
pub(crate) async fn reclaim_orphan_data_dirs(&self, bucket: &str, object: &str) -> disk::error::Result<usize> {
|
pub(crate) async fn reclaim_orphan_data_dirs(&self, bucket: &str, object: &str) -> disk::error::Result<usize> {
|
||||||
|
self.reclaim_orphan_data_dirs_inner(bucket, object, false).await
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn dry_run_reclaim_orphan_data_dirs(&self, bucket: &str, object: &str) -> disk::error::Result<usize> {
|
||||||
|
self.reclaim_orphan_data_dirs_inner(bucket, object, true).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn reclaim_orphan_data_dirs_inner(&self, bucket: &str, object: &str, dry_run: bool) -> disk::error::Result<usize> {
|
||||||
let disks = self.get_disks_internal().await;
|
let disks = self.get_disks_internal().await;
|
||||||
|
|
||||||
// Phase 1 (read-only): build the referenced-data-dir union and record the
|
// Phase 1 (read-only): build the referenced-data-dir union and record the
|
||||||
@@ -4967,6 +4975,20 @@ impl SetDisks {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let stray = format!("{object}/{dir}");
|
let stray = format!("{object}/{dir}");
|
||||||
|
if dry_run {
|
||||||
|
removed += 1;
|
||||||
|
debug!(
|
||||||
|
target: "rustfs_ecstore::set_disk",
|
||||||
|
event = "heal_abandoned_parts",
|
||||||
|
component = "ecstore",
|
||||||
|
subsystem = "heal",
|
||||||
|
state = "dry_run_matched",
|
||||||
|
result = "matched",
|
||||||
|
bucket, object, data_dir = %dir,
|
||||||
|
"Heal abandoned parts dry-run matched orphaned data directory"
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
match disk
|
match disk
|
||||||
.delete(
|
.delete(
|
||||||
bucket,
|
bucket,
|
||||||
|
|||||||
@@ -6998,6 +6998,100 @@ mod tests {
|
|||||||
assert!(object_dir.join(STORAGE_FORMAT_FILE).exists(), "metadata must be preserved");
|
assert!(object_dir.join(STORAGE_FORMAT_FILE).exists(), "metadata must be preserved");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn recv_abandoned_parts_trace(
|
||||||
|
trace: &mut rustfs_common::trace_bus::TraceSubscription,
|
||||||
|
bucket: &str,
|
||||||
|
object: &str,
|
||||||
|
state: &str,
|
||||||
|
) -> rustfs_common::trace_bus::TraceEvent {
|
||||||
|
for _ in 0..32 {
|
||||||
|
let event = tokio::time::timeout(std::time::Duration::from_secs(1), trace.recv())
|
||||||
|
.await
|
||||||
|
.expect("abandoned-parts trace event should arrive")
|
||||||
|
.expect("trace bus should stay open");
|
||||||
|
if event.kind == rustfs_common::trace_bus::TraceKind::Heal
|
||||||
|
&& event.func == rustfs_common::trace_bus::TraceFunc::HealCheckAbandonedParts
|
||||||
|
&& event.bucket.as_deref() == Some(bucket)
|
||||||
|
&& event.object.as_deref() == Some(object)
|
||||||
|
&& trace_attr_string(&event, "state").as_deref() == Some(state)
|
||||||
|
{
|
||||||
|
return (*event).clone();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
panic!("expected abandoned-parts trace state {state} for {bucket}/{object}");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn trace_attr_string(event: &rustfs_common::trace_bus::TraceEvent, key: &str) -> Option<String> {
|
||||||
|
event.attrs.iter().find_map(|attr| {
|
||||||
|
if attr.key != key {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(match &attr.value {
|
||||||
|
rustfs_common::trace_bus::TraceVal::Bool(value) => value.to_string(),
|
||||||
|
rustfs_common::trace_bus::TraceVal::U64(value) => value.to_string(),
|
||||||
|
rustfs_common::trace_bus::TraceVal::I64(value) => value.to_string(),
|
||||||
|
rustfs_common::trace_bus::TraceVal::Str(value) => value.to_string(),
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn check_abandoned_parts_dry_run_counts_without_deleting() {
|
||||||
|
let mut trace = rustfs_common::trace_bus::subscribe_trace_events();
|
||||||
|
let (dir, disk) = make_single_local_disk().await;
|
||||||
|
let live = Uuid::new_v4();
|
||||||
|
let orphan = Uuid::new_v4();
|
||||||
|
|
||||||
|
let object_dir = dir.path().join("bucket").join("obj");
|
||||||
|
write_object_meta_with_data_dirs(&object_dir, "bucket", "obj", &[live]).await;
|
||||||
|
fs::create_dir_all(object_dir.join(live.to_string()))
|
||||||
|
.await
|
||||||
|
.expect("live data dir should be created");
|
||||||
|
fs::create_dir_all(object_dir.join(orphan.to_string()))
|
||||||
|
.await
|
||||||
|
.expect("orphan data dir should be created");
|
||||||
|
|
||||||
|
let set = make_set_disks_with(vec![Some(disk)]).await;
|
||||||
|
set.check_abandoned_parts(
|
||||||
|
"bucket",
|
||||||
|
"obj",
|
||||||
|
&HealOpts {
|
||||||
|
dry_run: true,
|
||||||
|
no_lock: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("dry-run abandoned-parts check should succeed");
|
||||||
|
let dry_run_trace = recv_abandoned_parts_trace(&mut trace, "bucket", "obj", "dry_run_matched").await;
|
||||||
|
assert_eq!(trace_attr_string(&dry_run_trace, "dry_run").as_deref(), Some("true"));
|
||||||
|
assert_eq!(trace_attr_string(&dry_run_trace, "data_dirs").as_deref(), Some("1"));
|
||||||
|
|
||||||
|
assert!(object_dir.join(live.to_string()).exists(), "referenced data dir must be preserved");
|
||||||
|
assert!(object_dir.join(orphan.to_string()).exists(), "dry-run must not remove orphaned data dir");
|
||||||
|
|
||||||
|
set.check_abandoned_parts(
|
||||||
|
"bucket",
|
||||||
|
"obj",
|
||||||
|
&HealOpts {
|
||||||
|
no_lock: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("abandoned-parts check should reclaim stale data dir");
|
||||||
|
let reclaim_trace = recv_abandoned_parts_trace(&mut trace, "bucket", "obj", "reclaimed").await;
|
||||||
|
assert_eq!(trace_attr_string(&reclaim_trace, "dry_run").as_deref(), Some("false"));
|
||||||
|
assert_eq!(trace_attr_string(&reclaim_trace, "data_dirs").as_deref(), Some("1"));
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
object_dir.join(live.to_string()).exists(),
|
||||||
|
"referenced data dir must remain after reclaim"
|
||||||
|
);
|
||||||
|
assert!(!object_dir.join(orphan.to_string()).exists(), "orphaned data dir must be removed");
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn reclaim_orphan_data_dirs_recovers_deferred_cleanup_after_restart() {
|
async fn reclaim_orphan_data_dirs_recovers_deferred_cleanup_after_restart() {
|
||||||
let (dir, disk) = make_single_local_disk().await;
|
let (dir, disk) = make_single_local_disk().await;
|
||||||
@@ -12233,11 +12327,18 @@ mod tests {
|
|||||||
.expect_err("unsupported copy_object_part should return a typed error");
|
.expect_err("unsupported copy_object_part should return a typed error");
|
||||||
assert!(matches!(copy_part_err, StorageError::NotImplemented));
|
assert!(matches!(copy_part_err, StorageError::NotImplemented));
|
||||||
|
|
||||||
let abandoned_err = set_disks
|
set_disks
|
||||||
.check_abandoned_parts("bucket", "object", &HealOpts::default())
|
.check_abandoned_parts(
|
||||||
|
"bucket",
|
||||||
|
"object",
|
||||||
|
&HealOpts {
|
||||||
|
dry_run: true,
|
||||||
|
no_lock: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.expect_err("abandoned-parts check should stay in the upper reconciliation layer");
|
.expect("abandoned-parts check should be callable on empty disk sets");
|
||||||
assert!(matches!(abandoned_err, StorageError::NotImplemented));
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ use super::super::*;
|
|||||||
use crate::disk::disk_store::DiskStoreRenameDataExt;
|
use crate::disk::disk_store::DiskStoreRenameDataExt;
|
||||||
use crate::io_support::bitrot::object_mmap_read_enabled;
|
use crate::io_support::bitrot::object_mmap_read_enabled;
|
||||||
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
|
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
|
||||||
|
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, trace_emit};
|
||||||
use tracing::trace;
|
use tracing::trace;
|
||||||
|
|
||||||
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
||||||
@@ -2057,11 +2058,61 @@ impl crate::storage_api_contracts::heal::HealOperations for SetDisks {
|
|||||||
Err(Error::DiskNotFound)
|
Err(Error::DiskNotFound)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tracing::instrument(skip(self))]
|
#[tracing::instrument(level = "debug", skip(self, opts), fields(bucket = %bucket, object = %object, dry_run = opts.dry_run))]
|
||||||
async fn check_abandoned_parts(&self, _bucket: &str, _object: &str, _opts: &HealOpts) -> Result<()> {
|
async fn check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
|
||||||
// Multipart orphan reconciliation is intentionally retained above the set layer
|
let started_at = std::time::Instant::now();
|
||||||
// until there is a concrete caller and a stable lower-level contract to implement.
|
let _write_lock_guard = if !opts.no_lock {
|
||||||
Err(StorageError::NotImplemented)
|
let ns_lock = self.new_ns_lock(bucket, object).await?;
|
||||||
|
Some(
|
||||||
|
ns_lock
|
||||||
|
.get_write_lock(get_lock_acquire_timeout())
|
||||||
|
.await
|
||||||
|
.map_err(|e| self.map_namespace_lock_error(bucket, object, "write", e))?,
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
|
||||||
|
let removed = if opts.dry_run {
|
||||||
|
self.dry_run_reclaim_orphan_data_dirs(bucket, object).await?
|
||||||
|
} else {
|
||||||
|
self.reclaim_orphan_data_dirs(bucket, object).await?
|
||||||
|
};
|
||||||
|
let state = if opts.dry_run && removed > 0 {
|
||||||
|
"dry_run_matched"
|
||||||
|
} else if removed > 0 {
|
||||||
|
"reclaimed"
|
||||||
|
} else {
|
||||||
|
"checked"
|
||||||
|
};
|
||||||
|
let data_dirs = u64::try_from(removed).unwrap_or(u64::MAX);
|
||||||
|
|
||||||
|
trace_emit(|| {
|
||||||
|
TraceEvent::new(TraceKind::Heal, TraceFunc::HealCheckAbandonedParts)
|
||||||
|
.with_bucket(bucket)
|
||||||
|
.with_object(object)
|
||||||
|
.with_duration(started_at.elapsed())
|
||||||
|
.with_attr("state", state)
|
||||||
|
.with_attr("dry_run", opts.dry_run)
|
||||||
|
.with_attr("data_dirs", data_dirs)
|
||||||
|
});
|
||||||
|
|
||||||
|
if removed > 0 {
|
||||||
|
trace!(
|
||||||
|
event = "heal_abandoned_parts",
|
||||||
|
component = LOG_COMPONENT_ECSTORE,
|
||||||
|
subsystem = LOG_SUBSYSTEM_HEAL,
|
||||||
|
state = if opts.dry_run { "dry_run_matched" } else { "reclaimed" },
|
||||||
|
result = "ok",
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
dry_run = opts.dry_run,
|
||||||
|
data_dirs = removed,
|
||||||
|
"Heal abandoned parts checked object data directories"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -3246,4 +3297,223 @@ mod heal_result_report_tests {
|
|||||||
assert!(result.detail.contains("part 1"));
|
assert!(result.detail.contains("part 1"));
|
||||||
assert!(result.detail.contains("bitrot_failure=true"));
|
assert!(result.detail.contains("bitrot_failure=true"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// HS-12 (backlog#1874): a versioned DELETE racing an object heal must never
|
||||||
|
// resurrect the deleted version. The heal has real reconstruction work (a
|
||||||
|
// shard of the doomed version is removed), so both sides touch the same
|
||||||
|
// (bucket, object, data_dir); whichever order the ns write lock serializes
|
||||||
|
// them in, the committed delete must win.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial]
|
||||||
|
async fn heal_racing_version_delete_never_resurrects_the_deleted_version() {
|
||||||
|
let (temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
|
||||||
|
let bucket = "heal-race-delete-no-resurrect";
|
||||||
|
let object = "object.bin";
|
||||||
|
set.make_bucket(
|
||||||
|
bucket,
|
||||||
|
&MakeBucketOptions {
|
||||||
|
versioning_enabled: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("versioned bucket should be created");
|
||||||
|
|
||||||
|
let mut first_reader = PutObjReader::from_vec(vec![0x11; 1024 * 1024]);
|
||||||
|
let first_info = set
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut first_reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
versioned: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("first version should be written");
|
||||||
|
let first_version = first_info
|
||||||
|
.version_id
|
||||||
|
.expect("versioned put should return the first version id")
|
||||||
|
.to_string();
|
||||||
|
|
||||||
|
let mut second_reader = PutObjReader::from_vec(vec![0x22; 1024 * 1024]);
|
||||||
|
let second_info = set
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut second_reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
versioned: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("second version should be written");
|
||||||
|
let second_version = second_info
|
||||||
|
.version_id
|
||||||
|
.expect("versioned put should return the second version id")
|
||||||
|
.to_string();
|
||||||
|
|
||||||
|
// Damage one shard of the doomed version so the racing heal performs an
|
||||||
|
// actual reconstruction over its data dir instead of an early exit.
|
||||||
|
let doomed_source = disks[0]
|
||||||
|
.read_version("", bucket, object, &first_version, &ReadOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("doomed version metadata should be readable");
|
||||||
|
let doomed_data_dir = doomed_source
|
||||||
|
.data_dir
|
||||||
|
.expect("non-inline version should have a data directory");
|
||||||
|
tokio::fs::remove_file(
|
||||||
|
temp_dirs[1]
|
||||||
|
.path()
|
||||||
|
.join(bucket)
|
||||||
|
.join(object)
|
||||||
|
.join(doomed_data_dir.to_string())
|
||||||
|
.join("part.1"),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("shard damage should be injected before the race");
|
||||||
|
|
||||||
|
let delete_set = set.clone();
|
||||||
|
let (delete_res, heal_res) = tokio::join!(
|
||||||
|
async {
|
||||||
|
delete_set
|
||||||
|
.delete_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
ObjectOptions {
|
||||||
|
versioned: true,
|
||||||
|
version_id: Some(first_version.clone()),
|
||||||
|
object_lock_config_snapshot: Some(Arc::new(crate::set_disk::ObjectLockConfigSnapshot::new(
|
||||||
|
crate::bucket::metadata_sys::ObjectLockConfigState::ConfirmedAbsent,
|
||||||
|
))),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
},
|
||||||
|
async {
|
||||||
|
set.heal_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
"",
|
||||||
|
&HealOpts {
|
||||||
|
scan_mode: HealScanMode::Deep,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
},
|
||||||
|
);
|
||||||
|
delete_res.expect("version delete must succeed under lock serialization");
|
||||||
|
// The heal may legitimately report a transient failure when the version
|
||||||
|
// it was rebuilding disappears mid-flight; only the end state matters.
|
||||||
|
drop(heal_res);
|
||||||
|
|
||||||
|
let resurrected = set
|
||||||
|
.get_object_info(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&ObjectOptions {
|
||||||
|
versioned: true,
|
||||||
|
version_id: Some(first_version.clone()),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
assert!(
|
||||||
|
matches!(&resurrected, Err(Error::FileVersionNotFound) | Err(Error::ObjectNotFound(..))),
|
||||||
|
"a racing heal must not resurrect the deleted version: {resurrected:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
let survivor = set
|
||||||
|
.get_object_info(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&ObjectOptions {
|
||||||
|
versioned: true,
|
||||||
|
version_id: Some(second_version.clone()),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("surviving version must remain readable after the race");
|
||||||
|
assert_eq!(survivor.size, 1024 * 1024, "survivor size must be intact");
|
||||||
|
}
|
||||||
|
|
||||||
|
// HS-12 (backlog#1874): unversioned overwrite commits race a Deep heal on
|
||||||
|
// the same object. The overwrite's post-commit tail deletes the replaced
|
||||||
|
// data dir without the ns lock (object.rs commit tail), which is exactly
|
||||||
|
// the intersection the audit flagged: the heal must tolerate the tail race
|
||||||
|
// (retryable outcome) and every committed overwrite must survive — the
|
||||||
|
// final current version is exactly the last payload written.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial]
|
||||||
|
async fn heal_racing_unversioned_overwrites_preserves_the_last_commit() {
|
||||||
|
let (temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
|
||||||
|
let bucket = "heal-race-put-overwrite";
|
||||||
|
let object = "object.bin";
|
||||||
|
set.make_bucket(bucket, &MakeBucketOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("bucket should be created");
|
||||||
|
|
||||||
|
const ROUNDS: usize = 8;
|
||||||
|
const PAYLOAD_SIZE: usize = 256 * 1024;
|
||||||
|
let mut last_etag = String::new();
|
||||||
|
for round in 0..ROUNDS {
|
||||||
|
// Give the heal something to rebuild on alternating rounds: remove a
|
||||||
|
// shard of the current data dir right before the race.
|
||||||
|
if round % 2 == 1 {
|
||||||
|
let current = disks[2]
|
||||||
|
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("current metadata should be readable");
|
||||||
|
if let Some(data_dir) = current.data_dir {
|
||||||
|
let shard = temp_dirs[3]
|
||||||
|
.path()
|
||||||
|
.join(bucket)
|
||||||
|
.join(object)
|
||||||
|
.join(data_dir.to_string())
|
||||||
|
.join("part.1");
|
||||||
|
if shard.exists() {
|
||||||
|
tokio::fs::remove_file(&shard)
|
||||||
|
.await
|
||||||
|
.expect("shard damage should be injectable mid-race");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let payload = vec![round as u8; PAYLOAD_SIZE];
|
||||||
|
let mut put_reader = PutObjReader::from_vec(payload);
|
||||||
|
let put_opts = ObjectOptions::default();
|
||||||
|
let heal_opts = HealOpts {
|
||||||
|
scan_mode: HealScanMode::Deep,
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let (put_res, heal_res) = tokio::join!(
|
||||||
|
set.put_object(bucket, object, &mut put_reader, &put_opts),
|
||||||
|
set.heal_object(bucket, object, "", &heal_opts),
|
||||||
|
);
|
||||||
|
let put_info = put_res.expect("overwrite must succeed under lock serialization");
|
||||||
|
last_etag = put_info.etag.clone().unwrap_or_default();
|
||||||
|
// Heal outcome is unconstrained (may hit the tail race and report a
|
||||||
|
// retryable error); the invariant is checked on the end state.
|
||||||
|
drop(heal_res);
|
||||||
|
}
|
||||||
|
|
||||||
|
let final_info = set
|
||||||
|
.get_object_info(bucket, object, &ObjectOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("object must remain readable after the race loop");
|
||||||
|
assert_eq!(
|
||||||
|
final_info.size, PAYLOAD_SIZE as i64,
|
||||||
|
"final current version must be the last committed overwrite"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
final_info.etag.unwrap_or_default(),
|
||||||
|
last_etag,
|
||||||
|
"the racing heal loop must never leave a stale or resurrected current version"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -23,6 +23,7 @@
|
|||||||
//! per-version `SetDisks::heal_object`.
|
//! per-version `SetDisks::heal_object`.
|
||||||
|
|
||||||
use super::super::*;
|
use super::super::*;
|
||||||
|
use crate::object_api::ObjectInfo;
|
||||||
use std::collections::HashSet;
|
use std::collections::HashSet;
|
||||||
use std::sync::Mutex;
|
use std::sync::Mutex;
|
||||||
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
|
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
|
||||||
@@ -39,12 +40,16 @@ const BACKGROUND_WALKDIR_STALL_TIMEOUT: Duration = Duration::from_secs(60);
|
|||||||
/// it must not gate healing logic — the delete-marker vs data path is chosen
|
/// it must not gate healing logic — the delete-marker vs data path is chosen
|
||||||
/// inside `ops/heal.rs` from the resolved latest metadata. `version_id` is
|
/// inside `ops/heal.rs` from the resolved latest metadata. `version_id` is
|
||||||
/// normalized (nil/absent UUID => `None`).
|
/// normalized (nil/absent UUID => `None`).
|
||||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct HealWalkVersion {
|
pub struct HealWalkVersion {
|
||||||
/// object key
|
/// object key
|
||||||
pub name: String,
|
pub name: String,
|
||||||
/// normalized version id (`None` when the version is nil/absent)
|
/// normalized version id (`None` when the version is nil/absent)
|
||||||
pub version_id: Option<String>,
|
pub version_id: Option<String>,
|
||||||
|
/// version modification time as Unix nanoseconds
|
||||||
|
pub mod_time_unix_nanos: Option<i128>,
|
||||||
|
/// object snapshot for lifecycle evaluation
|
||||||
|
pub lifecycle_object_info: Option<ObjectInfo>,
|
||||||
/// whether this version is a delete marker (observability only)
|
/// whether this version is a delete marker (observability only)
|
||||||
pub is_delete_marker: bool,
|
pub is_delete_marker: bool,
|
||||||
}
|
}
|
||||||
@@ -63,6 +68,7 @@ struct HealWalkCollector {
|
|||||||
bucket: String,
|
bucket: String,
|
||||||
batch_objects: usize,
|
batch_objects: usize,
|
||||||
version_budget: usize,
|
version_budget: usize,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
objects: Mutex<Vec<HealWalkObject>>,
|
objects: Mutex<Vec<HealWalkObject>>,
|
||||||
decode_error: Mutex<Option<DiskError>>,
|
decode_error: Mutex<Option<DiskError>>,
|
||||||
version_total: AtomicUsize,
|
version_total: AtomicUsize,
|
||||||
@@ -116,10 +122,25 @@ impl HealWalkCollector {
|
|||||||
|
|
||||||
let mut versions = Vec::with_capacity(fiv.versions.len() + fiv.free_versions.len());
|
let mut versions = Vec::with_capacity(fiv.versions.len() + fiv.free_versions.len());
|
||||||
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
|
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
|
||||||
|
let version_uuid = fi.version_id.filter(|version_id| !version_id.is_nil());
|
||||||
|
let lifecycle_object_info = if self.include_lifecycle_object_info {
|
||||||
|
let mut lifecycle_fi = fi.clone();
|
||||||
|
lifecycle_fi.version_id = version_uuid;
|
||||||
|
Some(ObjectInfo::from_file_info(
|
||||||
|
&lifecycle_fi,
|
||||||
|
&self.bucket,
|
||||||
|
&entry.name,
|
||||||
|
version_uuid.is_some(),
|
||||||
|
))
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
versions.push(HealWalkVersion {
|
versions.push(HealWalkVersion {
|
||||||
name: entry.name.clone(),
|
name: entry.name.clone(),
|
||||||
// Normalize: nil/absent version id => None.
|
// Normalize: nil/absent version id => None.
|
||||||
version_id: fi.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string()),
|
version_id: version_uuid.map(|u| u.to_string()),
|
||||||
|
mod_time_unix_nanos: fi.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos()),
|
||||||
|
lifecycle_object_info,
|
||||||
is_delete_marker: fi.deleted,
|
is_delete_marker: fi.deleted,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -173,11 +194,26 @@ impl HealWalkCollector {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
|
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
|
||||||
let vid = fi.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string());
|
let version_uuid = fi.version_id.filter(|version_id| !version_id.is_nil());
|
||||||
|
let vid = version_uuid.map(|u| u.to_string());
|
||||||
if seen.insert(vid.clone()) {
|
if seen.insert(vid.clone()) {
|
||||||
|
let lifecycle_object_info = if self.include_lifecycle_object_info {
|
||||||
|
let mut lifecycle_fi = fi.clone();
|
||||||
|
lifecycle_fi.version_id = version_uuid;
|
||||||
|
Some(ObjectInfo::from_file_info(
|
||||||
|
&lifecycle_fi,
|
||||||
|
&self.bucket,
|
||||||
|
&entry.name,
|
||||||
|
version_uuid.is_some(),
|
||||||
|
))
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
versions.push(HealWalkVersion {
|
versions.push(HealWalkVersion {
|
||||||
name: entry.name.clone(),
|
name: entry.name.clone(),
|
||||||
version_id: vid,
|
version_id: vid,
|
||||||
|
mod_time_unix_nanos: fi.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos()),
|
||||||
|
lifecycle_object_info,
|
||||||
is_delete_marker: fi.deleted,
|
is_delete_marker: fi.deleted,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
@@ -255,6 +291,7 @@ impl SetDisks {
|
|||||||
forward_to: Option<&str>,
|
forward_to: Option<&str>,
|
||||||
batch_objects: usize,
|
batch_objects: usize,
|
||||||
version_budget: usize,
|
version_budget: usize,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
) -> disk::error::Result<(Vec<HealWalkVersion>, Option<String>, bool)> {
|
) -> disk::error::Result<(Vec<HealWalkVersion>, Option<String>, bool)> {
|
||||||
assert!(batch_objects >= 2, "heal_walk_versions_page requires batch_objects >= 2");
|
assert!(batch_objects >= 2, "heal_walk_versions_page requires batch_objects >= 2");
|
||||||
|
|
||||||
@@ -264,6 +301,7 @@ impl SetDisks {
|
|||||||
bucket: bucket.to_string(),
|
bucket: bucket.to_string(),
|
||||||
batch_objects,
|
batch_objects,
|
||||||
version_budget: version_budget.max(1),
|
version_budget: version_budget.max(1),
|
||||||
|
include_lifecycle_object_info,
|
||||||
objects: Mutex::new(Vec::new()),
|
objects: Mutex::new(Vec::new()),
|
||||||
decode_error: Mutex::new(None),
|
decode_error: Mutex::new(None),
|
||||||
version_total: AtomicUsize::new(0),
|
version_total: AtomicUsize::new(0),
|
||||||
@@ -347,6 +385,7 @@ mod tests {
|
|||||||
bucket: "bucket".to_string(),
|
bucket: "bucket".to_string(),
|
||||||
batch_objects: 2,
|
batch_objects: 2,
|
||||||
version_budget: 2,
|
version_budget: 2,
|
||||||
|
include_lifecycle_object_info: false,
|
||||||
objects: Mutex::new(Vec::new()),
|
objects: Mutex::new(Vec::new()),
|
||||||
decode_error: Mutex::new(None),
|
decode_error: Mutex::new(None),
|
||||||
version_total: AtomicUsize::new(0),
|
version_total: AtomicUsize::new(0),
|
||||||
@@ -388,6 +427,8 @@ mod tests {
|
|||||||
HealWalkVersion {
|
HealWalkVersion {
|
||||||
name: name.to_string(),
|
name: name.to_string(),
|
||||||
version_id: Some(id.to_string()),
|
version_id: Some(id.to_string()),
|
||||||
|
mod_time_unix_nanos: None,
|
||||||
|
lifecycle_object_info: None,
|
||||||
is_delete_marker: dm,
|
is_delete_marker: dm,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -491,6 +532,7 @@ mod tests {
|
|||||||
bucket: "bucket".to_string(),
|
bucket: "bucket".to_string(),
|
||||||
batch_objects: 1000,
|
batch_objects: 1000,
|
||||||
version_budget: 10_000,
|
version_budget: 10_000,
|
||||||
|
include_lifecycle_object_info: false,
|
||||||
objects: Mutex::new(Vec::new()),
|
objects: Mutex::new(Vec::new()),
|
||||||
version_total: AtomicUsize::new(0),
|
version_total: AtomicUsize::new(0),
|
||||||
decode_error: Mutex::new(None),
|
decode_error: Mutex::new(None),
|
||||||
@@ -567,7 +609,7 @@ mod tests {
|
|||||||
.expect("corrupt test metadata should be written");
|
.expect("corrupt test metadata should be written");
|
||||||
|
|
||||||
let error = set_disks
|
let error = set_disks
|
||||||
.heal_walk_versions_page(bucket, "", None, 2, 2)
|
.heal_walk_versions_page(bucket, "", None, 2, 2, false)
|
||||||
.await
|
.await
|
||||||
.expect_err("semantic metadata corruption must fail the heal disk walk");
|
.expect_err("semantic metadata corruption must fail the heal disk walk");
|
||||||
|
|
||||||
|
|||||||
@@ -5845,6 +5845,14 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
|
|||||||
|
|
||||||
#[tracing::instrument(skip(self))]
|
#[tracing::instrument(skip(self))]
|
||||||
async fn add_partial(&self, bucket: &str, object: &str, version_id: &str) -> Result<()> {
|
async fn add_partial(&self, bucket: &str, object: &str, version_id: &str) -> Result<()> {
|
||||||
|
// MRF journal intent: partial-write recovery must survive a restart
|
||||||
|
// (HS-01); the heal request below remains the in-memory fast path.
|
||||||
|
rustfs_common::mrf_channel::try_send_mrf_intent(
|
||||||
|
rustfs_common::mrf_channel::MrfKind::PartialWrite,
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
uuid::Uuid::try_parse(version_id).ok(),
|
||||||
|
);
|
||||||
let mut request = rustfs_common::heal_channel::create_heal_request_with_options(
|
let mut request = rustfs_common::heal_channel::create_heal_request_with_options(
|
||||||
bucket.to_string(),
|
bucket.to_string(),
|
||||||
Some(object.to_string()),
|
Some(object.to_string()),
|
||||||
|
|||||||
@@ -1077,6 +1077,15 @@ impl SetDisks {
|
|||||||
"Recoverable decode error triggered read repair"
|
"Recoverable decode error triggered read repair"
|
||||||
);
|
);
|
||||||
let version_id = fi.version_id.as_ref().map(ToString::to_string);
|
let version_id = fi.version_id.as_ref().map(ToString::to_string);
|
||||||
|
// MRF journal intent: keeps a durable Urgent ECDecode
|
||||||
|
// request alive across restarts even when the in-memory
|
||||||
|
// read-repair request is dropped or lost (HS-01).
|
||||||
|
rustfs_common::mrf_channel::try_send_mrf_intent(
|
||||||
|
rustfs_common::mrf_channel::MrfKind::DecodeFailure,
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
fi.version_id,
|
||||||
|
);
|
||||||
submit_read_repair_heal(
|
submit_read_repair_heal(
|
||||||
bucket,
|
bucket,
|
||||||
object,
|
object,
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ use tracing::trace;
|
|||||||
|
|
||||||
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
||||||
const LOG_SUBSYSTEM_HEAL: &str = "heal";
|
const LOG_SUBSYSTEM_HEAL: &str = "heal";
|
||||||
|
const EVENT_HEAL_ABANDONED_PARTS: &str = "heal_abandoned_parts";
|
||||||
const EVENT_HEAL_FORMAT_COMPLETED: &str = "heal_format_completed";
|
const EVENT_HEAL_FORMAT_COMPLETED: &str = "heal_format_completed";
|
||||||
const EVENT_HEAL_OBJECT_STARTED: &str = "heal_object_started";
|
const EVENT_HEAL_OBJECT_STARTED: &str = "heal_object_started";
|
||||||
|
|
||||||
@@ -256,13 +257,40 @@ impl ECStore {
|
|||||||
|
|
||||||
#[instrument(skip(self))]
|
#[instrument(skip(self))]
|
||||||
pub(super) async fn handle_check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
|
pub(super) async fn handle_check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
|
||||||
let _ = (bucket, object, opts);
|
let object = encode_dir_object(object);
|
||||||
// Stale multipart reconciliation is already owned by the lifecycle-driven
|
let pools = self.get_pools_for_heal_object(opts)?;
|
||||||
// background cleanup path in `bucket_lifecycle_ops.rs`. There is currently
|
|
||||||
// no stable object-heal contract that should fan this request out through
|
let mut futures = Vec::with_capacity(pools.len());
|
||||||
// pool/set storage layers, so keep the placeholder explicit at the ECStore
|
for pool in pools.iter() {
|
||||||
// boundary instead of dispatching into lower layers.
|
futures.push(pool.check_abandoned_parts(bucket, &object, opts));
|
||||||
Err(StorageError::NotImplemented)
|
}
|
||||||
|
|
||||||
|
let mut first_error = None;
|
||||||
|
for result in join_all(futures).await {
|
||||||
|
if let Err(err) = result
|
||||||
|
&& first_error.is_none()
|
||||||
|
{
|
||||||
|
first_error = Some(err);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(err) = first_error {
|
||||||
|
return Err(err);
|
||||||
|
}
|
||||||
|
|
||||||
|
trace!(
|
||||||
|
event = EVENT_HEAL_ABANDONED_PARTS,
|
||||||
|
component = LOG_COMPONENT_ECSTORE,
|
||||||
|
subsystem = LOG_SUBSYSTEM_HEAL,
|
||||||
|
state = "completed",
|
||||||
|
result = "ok",
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
dry_run = opts.dry_run,
|
||||||
|
"Heal abandoned parts completed"
|
||||||
|
);
|
||||||
|
|
||||||
|
Ok(())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -34,6 +34,7 @@ impl ECStore {
|
|||||||
forward_to: Option<&str>,
|
forward_to: Option<&str>,
|
||||||
batch_objects: usize,
|
batch_objects: usize,
|
||||||
version_budget: usize,
|
version_budget: usize,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealWalkVersion>, Option<String>, bool)> {
|
) -> Result<(Vec<HealWalkVersion>, Option<String>, bool)> {
|
||||||
if pool_idx >= self.pools.len() || set_idx >= self.pools[pool_idx].disk_set.len() {
|
if pool_idx >= self.pools.len() || set_idx >= self.pools[pool_idx].disk_set.len() {
|
||||||
return Err(Error::other(format!(
|
return Err(Error::other(format!(
|
||||||
@@ -43,7 +44,7 @@ impl ECStore {
|
|||||||
}
|
}
|
||||||
|
|
||||||
self.pools[pool_idx].disk_set[set_idx]
|
self.pools[pool_idx].disk_set[set_idx]
|
||||||
.heal_walk_versions_page(bucket, prefix, forward_to, batch_objects, version_budget)
|
.heal_walk_versions_page(bucket, prefix, forward_to, batch_objects, version_budget, include_lifecycle_object_info)
|
||||||
.await
|
.await
|
||||||
.map_err(Error::from)
|
.map_err(Error::from)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -216,6 +216,16 @@ impl std::fmt::Debug for ECStore {
|
|||||||
/// These delegate to the process-global statics. No local state — the globals
|
/// These delegate to the process-global statics. No local state — the globals
|
||||||
/// remain the single source of truth until the migration is complete.
|
/// remain the single source of truth until the migration is complete.
|
||||||
impl ECStore {
|
impl ECStore {
|
||||||
|
/// Every erasure set across all pools, pool-major order.
|
||||||
|
///
|
||||||
|
/// Read-only queries that must consult each set's own copy of a
|
||||||
|
/// per-bucket object (e.g. the scanner's `.usage-cache.bin`) iterate
|
||||||
|
/// this instead of the hash-routed store path, which would always land
|
||||||
|
/// on one set (rustfs/backlog#1872).
|
||||||
|
pub fn all_set_disks(&self) -> Vec<Arc<crate::set_disk::SetDisks>> {
|
||||||
|
self.pools.iter().flat_map(|pool| pool.disk_set.iter().cloned()).collect()
|
||||||
|
}
|
||||||
|
|
||||||
/// Get server configuration (delegates to global)
|
/// Get server configuration (delegates to global)
|
||||||
pub fn get_server_config(&self) -> Option<Config> {
|
pub fn get_server_config(&self) -> Option<Config> {
|
||||||
runtime_sources::server_config()
|
runtime_sources::server_config()
|
||||||
|
|||||||
@@ -89,6 +89,8 @@ async-trait = { workspace = true }
|
|||||||
futures = { workspace = true }
|
futures = { workspace = true }
|
||||||
metrics = { workspace = true }
|
metrics = { workspace = true }
|
||||||
base64 = { workspace = true }
|
base64 = { workspace = true }
|
||||||
|
bytes = { workspace = true }
|
||||||
|
crc-fast = { workspace = true }
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
serde_json = { workspace = true, features = ["raw_value"] }
|
serde_json = { workspace = true, features = ["raw_value"] }
|
||||||
|
|||||||
@@ -612,7 +612,8 @@ impl HealChannelProcessor {
|
|||||||
HealRequestSource::Admin
|
HealRequestSource::Admin
|
||||||
| HealRequestSource::AutoHeal
|
| HealRequestSource::AutoHeal
|
||||||
| HealRequestSource::Internal
|
| HealRequestSource::Internal
|
||||||
| HealRequestSource::ReadRepair => true,
|
| HealRequestSource::ReadRepair
|
||||||
|
| HealRequestSource::Mrf => true,
|
||||||
});
|
});
|
||||||
|
|
||||||
// Build HealOptions with all available fields
|
// Build HealOptions with all available fields
|
||||||
@@ -767,6 +768,7 @@ mod tests {
|
|||||||
_bucket: &str,
|
_bucket: &str,
|
||||||
_prefix: &str,
|
_prefix: &str,
|
||||||
_continuation_token: Option<&str>,
|
_continuation_token: Option<&str>,
|
||||||
|
_include_lifecycle_object_info: bool,
|
||||||
) -> crate::Result<(Vec<crate::heal::storage::HealListItem>, Option<String>, bool)> {
|
) -> crate::Result<(Vec<crate::heal::storage::HealListItem>, Option<String>, bool)> {
|
||||||
Ok((vec![], None, false))
|
Ok((vec![], None, false))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -23,13 +23,14 @@ use crate::heal::{
|
|||||||
};
|
};
|
||||||
use crate::{Error, Result};
|
use crate::{Error, Result};
|
||||||
use futures::{StreamExt, stream::FuturesUnordered};
|
use futures::{StreamExt, stream::FuturesUnordered};
|
||||||
use metrics::gauge;
|
use metrics::{counter, gauge};
|
||||||
use rustfs_common::heal_channel::{HealOpts, HealRequestSource, HealScanMode};
|
use rustfs_common::heal_channel::{HealOpts, HealRequestSource, HealScanMode};
|
||||||
use rustfs_madmin::heal_commands::HealResultItem;
|
use rustfs_madmin::heal_commands::HealResultItem;
|
||||||
use std::sync::{
|
use std::sync::{
|
||||||
Arc,
|
Arc,
|
||||||
atomic::{AtomicUsize, Ordering},
|
atomic::{AtomicUsize, Ordering},
|
||||||
};
|
};
|
||||||
|
use std::time::{Duration, UNIX_EPOCH};
|
||||||
use tokio::sync::{RwLock, Semaphore};
|
use tokio::sync::{RwLock, Semaphore};
|
||||||
use tracing::{debug, error, warn};
|
use tracing::{debug, error, warn};
|
||||||
|
|
||||||
@@ -47,6 +48,21 @@ enum HealObjectOutcome {
|
|||||||
Failed,
|
Failed,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn result_object_size_u64(result: &HealResultItem) -> u64 {
|
||||||
|
u64::try_from(result.object_size).unwrap_or(u64::MAX)
|
||||||
|
}
|
||||||
|
|
||||||
|
const NEW_VERSION_SKIP_GRACE_SECS: u64 = 60;
|
||||||
|
const NANOS_PER_SECOND: i128 = 1_000_000_000;
|
||||||
|
|
||||||
|
fn should_skip_new_version(mod_time_unix_nanos: Option<i128>, started_at_secs: u64) -> bool {
|
||||||
|
let Some(mod_time_unix_nanos) = mod_time_unix_nanos else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
let cutoff_secs = started_at_secs.saturating_add(NEW_VERSION_SKIP_GRACE_SECS);
|
||||||
|
mod_time_unix_nanos > i128::from(cutoff_secs).saturating_mul(NANOS_PER_SECOND)
|
||||||
|
}
|
||||||
|
|
||||||
struct PageConcurrencyGuard {
|
struct PageConcurrencyGuard {
|
||||||
in_flight: Arc<AtomicUsize>,
|
in_flight: Arc<AtomicUsize>,
|
||||||
set_label: String,
|
set_label: String,
|
||||||
@@ -492,6 +508,7 @@ impl ErasureSetHealer {
|
|||||||
&mut skipped_objects,
|
&mut skipped_objects,
|
||||||
resume_manager,
|
resume_manager,
|
||||||
checkpoint_manager,
|
checkpoint_manager,
|
||||||
|
state.start_time,
|
||||||
)
|
)
|
||||||
.await;
|
.await;
|
||||||
|
|
||||||
@@ -658,6 +675,7 @@ impl ErasureSetHealer {
|
|||||||
skipped_objects: &mut u64,
|
skipped_objects: &mut u64,
|
||||||
resume_manager: &ResumeManager,
|
resume_manager: &ResumeManager,
|
||||||
checkpoint_manager: &CheckpointManager,
|
checkpoint_manager: &CheckpointManager,
|
||||||
|
started_at_secs: u64,
|
||||||
) -> Result<()> {
|
) -> Result<()> {
|
||||||
debug!(
|
debug!(
|
||||||
target: "rustfs::heal::erasure_healer",
|
target: "rustfs::heal::erasure_healer",
|
||||||
@@ -710,6 +728,7 @@ impl ErasureSetHealer {
|
|||||||
// The end-of-pass summary reports the full failed/skipped counts.
|
// The end-of-pass summary reports the full failed/skipped counts.
|
||||||
let mut transient_skip_samples_logged = 0_u64;
|
let mut transient_skip_samples_logged = 0_u64;
|
||||||
let mut failure_samples_logged = 0_u64;
|
let mut failure_samples_logged = 0_u64;
|
||||||
|
let mut bytes_processed = self.progress.read().await.bytes_processed;
|
||||||
|
|
||||||
// backlog#920: select the per-erasure-set DISK-WALK union enumerator when
|
// backlog#920: select the per-erasure-set DISK-WALK union enumerator when
|
||||||
// the scan is Deep OR the request came from AutoHeal — these are the paths
|
// the scan is Deep OR the request came from AutoHeal — these are the paths
|
||||||
@@ -718,17 +737,25 @@ impl ErasureSetHealer {
|
|||||||
// which stays the default.
|
// which stays the default.
|
||||||
let use_disk_walk =
|
let use_disk_walk =
|
||||||
matches!(self.heal_opts.scan_mode, HealScanMode::Deep) || matches!(self.source, HealRequestSource::AutoHeal);
|
matches!(self.heal_opts.scan_mode, HealScanMode::Deep) || matches!(self.source, HealRequestSource::AutoHeal);
|
||||||
|
let lifecycle_expiry_context = self.storage.load_heal_lifecycle_expiry_context(bucket).await?;
|
||||||
|
let include_lifecycle_object_info = lifecycle_expiry_context.is_some();
|
||||||
|
|
||||||
loop {
|
loop {
|
||||||
self.verify_replacement_identity_fence("page scan").await?;
|
self.verify_replacement_identity_fence("page scan").await?;
|
||||||
// Get one page of object versions
|
// Get one page of object versions
|
||||||
let (objects, next_token, is_truncated) = if use_disk_walk {
|
let (objects, next_token, is_truncated) = if use_disk_walk {
|
||||||
self.storage
|
self.storage
|
||||||
.list_versions_for_heal_page_disk_walk(set_disk_id, bucket, "", continuation_token.as_deref())
|
.list_versions_for_heal_page_disk_walk(
|
||||||
|
set_disk_id,
|
||||||
|
bucket,
|
||||||
|
"",
|
||||||
|
continuation_token.as_deref(),
|
||||||
|
include_lifecycle_object_info,
|
||||||
|
)
|
||||||
.await?
|
.await?
|
||||||
} else {
|
} else {
|
||||||
self.storage
|
self.storage
|
||||||
.list_objects_for_heal_page(bucket, "", continuation_token.as_deref())
|
.list_objects_for_heal_page(bucket, "", continuation_token.as_deref(), include_lifecycle_object_info)
|
||||||
.await?
|
.await?
|
||||||
};
|
};
|
||||||
let page_is_empty = objects.is_empty();
|
let page_is_empty = objects.is_empty();
|
||||||
@@ -736,6 +763,7 @@ impl ErasureSetHealer {
|
|||||||
let page_resume_index = *current_object_index;
|
let page_resume_index = *current_object_index;
|
||||||
let semaphore = Arc::new(Semaphore::new(page_concurrency_limit));
|
let semaphore = Arc::new(Semaphore::new(page_concurrency_limit));
|
||||||
let mut page_tasks = FuturesUnordered::new();
|
let mut page_tasks = FuturesUnordered::new();
|
||||||
|
let mut completed_in_page = 0usize;
|
||||||
|
|
||||||
// Capture the last version identity of this page for the anti-loop guard.
|
// Capture the last version identity of this page for the anti-loop guard.
|
||||||
let page_last = objects.last().map(|item| (item.name.clone(), item.version_id.clone()));
|
let page_last = objects.last().map(|item| (item.name.clone(), item.version_id.clone()));
|
||||||
@@ -751,6 +779,75 @@ impl ErasureSetHealer {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if should_skip_new_version(item.mod_time_unix_nanos, started_at_secs) {
|
||||||
|
checkpoint_manager.add_processed_object(key).await?;
|
||||||
|
*processed_objects = processed_objects.saturating_add(1);
|
||||||
|
completed_in_page = completed_in_page.saturating_add(1);
|
||||||
|
counter!("rustfs_heal_skipped_new_versions_total").increment(1);
|
||||||
|
{
|
||||||
|
let mut progress = self.progress.write().await;
|
||||||
|
progress.record_skipped_new_version();
|
||||||
|
progress.set_current_object(Some(format!("skipped_new: {bucket}/{}", item.name)));
|
||||||
|
progress.update_progress(*processed_objects, *successful_objects, *failed_objects, bytes_processed);
|
||||||
|
}
|
||||||
|
debug!(
|
||||||
|
target: "rustfs::heal::erasure_healer",
|
||||||
|
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
||||||
|
component = LOG_COMPONENT_HEAL,
|
||||||
|
subsystem = LOG_SUBSYSTEM_ERASURE_HEALER,
|
||||||
|
set_disk_id,
|
||||||
|
bucket,
|
||||||
|
object = %item.name,
|
||||||
|
version_id = ?item.version_id,
|
||||||
|
state = "skipped_new_version",
|
||||||
|
"Erasure set object version skipped because it was written after heal started"
|
||||||
|
);
|
||||||
|
if completed_in_page.is_multiple_of(100) {
|
||||||
|
checkpoint_manager.update_position(bucket_index, page_resume_index).await?;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(context) = lifecycle_expiry_context.as_ref()
|
||||||
|
&& self
|
||||||
|
.storage
|
||||||
|
.enqueue_heal_lifecycle_expiry(
|
||||||
|
context,
|
||||||
|
bucket,
|
||||||
|
&item.name,
|
||||||
|
item.version_id.as_deref(),
|
||||||
|
item.lifecycle_object_info.as_ref(),
|
||||||
|
)
|
||||||
|
.await?
|
||||||
|
{
|
||||||
|
checkpoint_manager.add_processed_object(key).await?;
|
||||||
|
*processed_objects = processed_objects.saturating_add(1);
|
||||||
|
completed_in_page = completed_in_page.saturating_add(1);
|
||||||
|
counter!("rustfs_heal_skipped_ilm_expired_total").increment(1);
|
||||||
|
{
|
||||||
|
let mut progress = self.progress.write().await;
|
||||||
|
progress.record_skipped_ilm_expired();
|
||||||
|
progress.set_current_object(Some(format!("skipped_ilm: {bucket}/{}", item.name)));
|
||||||
|
progress.update_progress(*processed_objects, *successful_objects, *failed_objects, bytes_processed);
|
||||||
|
}
|
||||||
|
debug!(
|
||||||
|
target: "rustfs::heal::erasure_healer",
|
||||||
|
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
||||||
|
component = LOG_COMPONENT_HEAL,
|
||||||
|
subsystem = LOG_SUBSYSTEM_ERASURE_HEALER,
|
||||||
|
set_disk_id,
|
||||||
|
bucket,
|
||||||
|
object = %item.name,
|
||||||
|
version_id = ?item.version_id,
|
||||||
|
state = "skipped_ilm_expired",
|
||||||
|
"Erasure set object version skipped because lifecycle expiry was queued"
|
||||||
|
);
|
||||||
|
if completed_in_page.is_multiple_of(100) {
|
||||||
|
checkpoint_manager.update_position(bucket_index, page_resume_index).await?;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
resume_manager
|
resume_manager
|
||||||
.set_current_item(Some(bucket.to_string()), Some(item.name.clone()))
|
.set_current_item(Some(bucket.to_string()), Some(item.name.clone()))
|
||||||
.await?;
|
.await?;
|
||||||
@@ -777,7 +874,7 @@ impl ErasureSetHealer {
|
|||||||
|
|
||||||
let _permit = match permit {
|
let _permit = match permit {
|
||||||
Ok(permit) => permit,
|
Ok(permit) => permit,
|
||||||
Err(err) => return (dedup_key, object_name, version_id, Err(err)),
|
Err(err) => return (dedup_key, object_name, version_id, (0, Err(err))),
|
||||||
};
|
};
|
||||||
|
|
||||||
let _in_flight_guard = PageConcurrencyGuard::new(in_flight, set_label);
|
let _in_flight_guard = PageConcurrencyGuard::new(in_flight, set_label);
|
||||||
@@ -788,7 +885,7 @@ impl ErasureSetHealer {
|
|||||||
// recorded as skipped-ok rather than failed. The delete-marker
|
// recorded as skipped-ok rather than failed. The delete-marker
|
||||||
// vs data path is chosen internally in ops/heal.rs.
|
// vs data path is chosen internally in ops/heal.rs.
|
||||||
let result = if cancel_token.is_cancelled() {
|
let result = if cancel_token.is_cancelled() {
|
||||||
Err(Error::TaskCancelled)
|
(0, Err(Error::TaskCancelled))
|
||||||
} else {
|
} else {
|
||||||
match storage
|
match storage
|
||||||
.heal_object(&bucket_name, &object_name, version_id.as_deref(), &heal_opts)
|
.heal_object(&bucket_name, &object_name, version_id.as_deref(), &heal_opts)
|
||||||
@@ -797,8 +894,9 @@ impl ErasureSetHealer {
|
|||||||
Ok((result, None))
|
Ok((result, None))
|
||||||
if target_outcomes_complete(&result, &target_endpoints) =>
|
if target_outcomes_complete(&result, &target_endpoints) =>
|
||||||
{
|
{
|
||||||
|
let object_size = result_object_size_u64(&result);
|
||||||
if !replacement_commit_evidence_required {
|
if !replacement_commit_evidence_required {
|
||||||
Ok(true)
|
(object_size, Ok(true))
|
||||||
} else {
|
} else {
|
||||||
match storage
|
match storage
|
||||||
.replacement_targets_have_version(
|
.replacement_targets_have_version(
|
||||||
@@ -810,27 +908,42 @@ impl ErasureSetHealer {
|
|||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
Ok(true) => Ok(true),
|
Ok(true) => (object_size, Ok(true)),
|
||||||
Ok(false) => Err(Error::transient_skip(format!(
|
Ok(false) => (object_size, Err(Error::transient_skip(format!(
|
||||||
"Skipped heal for {bucket_name}/{object_name} because replacement target readback did not confirm the committed version"
|
"Skipped heal for {bucket_name}/{object_name} because replacement target readback did not confirm the committed version"
|
||||||
))),
|
)))),
|
||||||
Err(err) => Err(Error::transient_skip(format!(
|
Err(err) => (object_size, Err(Error::transient_skip(format!(
|
||||||
"Skipped heal for {bucket_name}/{object_name} because replacement target readback failed: {err}"
|
"Skipped heal for {bucket_name}/{object_name} because replacement target readback failed: {err}"
|
||||||
))),
|
)))),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
},
|
||||||
Ok((_result, None)) if !target_endpoints.is_empty() => Err(Error::transient_skip(format!(
|
Ok((result, None)) if !target_endpoints.is_empty() => (
|
||||||
"Skipped heal for {bucket_name}/{object_name} because a replacement target was not committed"
|
result_object_size_u64(&result),
|
||||||
))),
|
Err(Error::transient_skip(format!(
|
||||||
Ok((_result, None)) => Ok(true),
|
"Skipped heal for {bucket_name}/{object_name} because a replacement target was not committed"
|
||||||
Ok((_, Some(err))) if is_missing_object_dir_heal_result(&object_name, &err) => Ok(false),
|
|
||||||
Ok((_, Some(err))) | Err(err) => match Self::classify_heal_object_error(&err) {
|
|
||||||
HealObjectOutcome::Absent => Ok(false),
|
|
||||||
HealObjectOutcome::Transient => Err(Error::transient_skip(format!(
|
|
||||||
"Skipped heal for {bucket_name}/{object_name} due to transient error: {err}"
|
|
||||||
))),
|
))),
|
||||||
HealObjectOutcome::Failed => Err(err),
|
),
|
||||||
|
Ok((result, None)) => (result_object_size_u64(&result), Ok(true)),
|
||||||
|
Ok((result, Some(err))) if is_missing_object_dir_heal_result(&object_name, &err) => {
|
||||||
|
(result_object_size_u64(&result), Ok(false))
|
||||||
|
}
|
||||||
|
Ok((result, Some(err))) => {
|
||||||
|
let object_size = result_object_size_u64(&result);
|
||||||
|
match Self::classify_heal_object_error(&err) {
|
||||||
|
HealObjectOutcome::Absent => (object_size, Ok(false)),
|
||||||
|
HealObjectOutcome::Transient => (object_size, Err(Error::transient_skip(format!(
|
||||||
|
"Skipped heal for {bucket_name}/{object_name} due to transient error: {err}"
|
||||||
|
)))),
|
||||||
|
HealObjectOutcome::Failed => (object_size, Err(err)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(err) => match Self::classify_heal_object_error(&err) {
|
||||||
|
HealObjectOutcome::Absent => (0, Ok(false)),
|
||||||
|
HealObjectOutcome::Transient => (0, Err(Error::transient_skip(format!(
|
||||||
|
"Skipped heal for {bucket_name}/{object_name} due to transient error: {err}"
|
||||||
|
)))),
|
||||||
|
HealObjectOutcome::Failed => (0, Err(err)),
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -839,11 +952,12 @@ impl ErasureSetHealer {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut completed_in_page = 0usize;
|
|
||||||
while let Some((key, object, version_id, result)) = page_tasks.next().await {
|
while let Some((key, object, version_id, result)) = page_tasks.next().await {
|
||||||
|
let (object_size, result) = result;
|
||||||
match result {
|
match result {
|
||||||
Ok(true) => {
|
Ok(true) => {
|
||||||
*successful_objects += 1;
|
*successful_objects += 1;
|
||||||
|
bytes_processed = bytes_processed.saturating_add(object_size);
|
||||||
checkpoint_manager.add_processed_object(key).await?;
|
checkpoint_manager.add_processed_object(key).await?;
|
||||||
debug!(
|
debug!(
|
||||||
target: "rustfs::heal::erasure_healer",
|
target: "rustfs::heal::erasure_healer",
|
||||||
@@ -861,6 +975,7 @@ impl ErasureSetHealer {
|
|||||||
Ok(false) => {
|
Ok(false) => {
|
||||||
checkpoint_manager.add_processed_object(key).await?;
|
checkpoint_manager.add_processed_object(key).await?;
|
||||||
*successful_objects += 1;
|
*successful_objects += 1;
|
||||||
|
bytes_processed = bytes_processed.saturating_add(object_size);
|
||||||
debug!(
|
debug!(
|
||||||
target: "rustfs::heal::erasure_healer",
|
target: "rustfs::heal::erasure_healer",
|
||||||
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
||||||
@@ -877,6 +992,7 @@ impl ErasureSetHealer {
|
|||||||
Err(err @ Error::TaskCancelled) | Err(err @ Error::TaskTimeout) => return Err(err),
|
Err(err @ Error::TaskCancelled) | Err(err @ Error::TaskTimeout) => return Err(err),
|
||||||
Err(Error::TransientSkip { message }) => {
|
Err(Error::TransientSkip { message }) => {
|
||||||
*skipped_objects += 1;
|
*skipped_objects += 1;
|
||||||
|
bytes_processed = bytes_processed.saturating_add(object_size);
|
||||||
checkpoint_manager.add_skipped_object(key).await?;
|
checkpoint_manager.add_skipped_object(key).await?;
|
||||||
demote_to_debug_when!(!take_failure_log_sample(&mut transient_skip_samples_logged), warn, target: "rustfs::heal::erasure_healer", {
|
demote_to_debug_when!(!take_failure_log_sample(&mut transient_skip_samples_logged), warn, target: "rustfs::heal::erasure_healer", {
|
||||||
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
||||||
@@ -893,6 +1009,7 @@ impl ErasureSetHealer {
|
|||||||
}
|
}
|
||||||
Err(err) => {
|
Err(err) => {
|
||||||
*failed_objects += 1;
|
*failed_objects += 1;
|
||||||
|
bytes_processed = bytes_processed.saturating_add(object_size);
|
||||||
checkpoint_manager.add_failed_object(key).await?;
|
checkpoint_manager.add_failed_object(key).await?;
|
||||||
demote_to_debug_when!(!take_failure_log_sample(&mut failure_samples_logged), warn, target: "rustfs::heal::erasure_healer", {
|
demote_to_debug_when!(!take_failure_log_sample(&mut failure_samples_logged), warn, target: "rustfs::heal::erasure_healer", {
|
||||||
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
|
||||||
@@ -911,6 +1028,11 @@ impl ErasureSetHealer {
|
|||||||
|
|
||||||
*processed_objects += 1;
|
*processed_objects += 1;
|
||||||
completed_in_page += 1;
|
completed_in_page += 1;
|
||||||
|
{
|
||||||
|
let mut progress = self.progress.write().await;
|
||||||
|
progress.set_current_object(Some(format!("{bucket}/{object}")));
|
||||||
|
progress.update_progress(*processed_objects, *successful_objects, *failed_objects, bytes_processed);
|
||||||
|
}
|
||||||
|
|
||||||
if completed_in_page.is_multiple_of(100) {
|
if completed_in_page.is_multiple_of(100) {
|
||||||
checkpoint_manager.update_position(bucket_index, page_resume_index).await?;
|
checkpoint_manager.update_position(bucket_index, page_resume_index).await?;
|
||||||
@@ -964,7 +1086,9 @@ impl ErasureSetHealer {
|
|||||||
progress.objects_scanned = state.total_objects;
|
progress.objects_scanned = state.total_objects;
|
||||||
progress.objects_healed = state.successful_objects;
|
progress.objects_healed = state.successful_objects;
|
||||||
progress.objects_failed = state.failed_objects;
|
progress.objects_failed = state.failed_objects;
|
||||||
progress.bytes_processed = 0; // set to 0 for now, can be extended later
|
progress.bytes_processed = 0; // Resume state tracks object counts, not byte counters.
|
||||||
|
progress.start_time = UNIX_EPOCH.checked_add(Duration::from_secs(state.start_time));
|
||||||
|
progress.last_update_time = UNIX_EPOCH.checked_add(Duration::from_secs(state.last_update));
|
||||||
progress.set_current_object(state.current_object.clone());
|
progress.set_current_object(state.current_object.clone());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1135,13 +1259,15 @@ mod resume_loop_tests {
|
|||||||
//! that emits programmable multi-version pages. These exercise the real loop
|
//! that emits programmable multi-version pages. These exercise the real loop
|
||||||
//! logic (cursor seeding, per-version dedup, anti-loop guard, absence
|
//! logic (cursor seeding, per-version dedup, anti-loop guard, absence
|
||||||
//! handling) — not merely a mock's own output.
|
//! handling) — not merely a mock's own output.
|
||||||
use super::{ErasureSetHealer, target_outcomes_complete};
|
use super::{
|
||||||
|
ErasureSetHealer, NANOS_PER_SECOND, NEW_VERSION_SKIP_GRACE_SECS, should_skip_new_version, target_outcomes_complete,
|
||||||
|
};
|
||||||
use crate::heal::progress::HealProgress;
|
use crate::heal::progress::HealProgress;
|
||||||
use crate::heal::resume::{
|
use crate::heal::resume::{
|
||||||
CheckpointManager, RESUME_CHECKPOINT_FILE, ReplacementTargetIdentity, ResumeDeleteFailure, ResumeManager, ResumeUtils,
|
CheckpointManager, RESUME_CHECKPOINT_FILE, ReplacementTargetIdentity, ResumeDeleteFailure, ResumeManager, ResumeUtils,
|
||||||
compose_key,
|
compose_key,
|
||||||
};
|
};
|
||||||
use crate::heal::storage::{DiskStatus, HealListItem, HealObjectInfo, HealStorageAPI};
|
use crate::heal::storage::{DiskStatus, HealLifecycleExpiryContext, HealListItem, HealObjectInfo, HealStorageAPI};
|
||||||
use crate::heal::storage_api::status::BucketInfo;
|
use crate::heal::storage_api::status::BucketInfo;
|
||||||
use crate::heal::{
|
use crate::heal::{
|
||||||
BUCKET_META_PREFIX, DiskOption, DiskStore, EcstoreError, Endpoint, HealDiskExt as _, RUSTFS_META_BUCKET, new_disk,
|
BUCKET_META_PREFIX, DiskOption, DiskStore, EcstoreError, Endpoint, HealDiskExt as _, RUSTFS_META_BUCKET, new_disk,
|
||||||
@@ -1149,7 +1275,7 @@ mod resume_loop_tests {
|
|||||||
use crate::{Error, Result};
|
use crate::{Error, Result};
|
||||||
use rustfs_common::heal_channel::{HealOpts, HealRequestSource};
|
use rustfs_common::heal_channel::{HealOpts, HealRequestSource};
|
||||||
use rustfs_madmin::heal_commands::{HealDriveInfo, HealResultItem, Infos};
|
use rustfs_madmin::heal_commands::{HealDriveInfo, HealResultItem, Infos};
|
||||||
use std::collections::{HashMap, VecDeque};
|
use std::collections::{HashMap, HashSet, VecDeque};
|
||||||
use std::sync::atomic::{AtomicBool, Ordering};
|
use std::sync::atomic::{AtomicBool, Ordering};
|
||||||
use std::sync::{Arc, Mutex};
|
use std::sync::{Arc, Mutex};
|
||||||
use tempfile::TempDir;
|
use tempfile::TempDir;
|
||||||
@@ -1160,10 +1286,37 @@ mod resume_loop_tests {
|
|||||||
HealListItem {
|
HealListItem {
|
||||||
name: name.to_string(),
|
name: name.to_string(),
|
||||||
version_id: version.map(str::to_string),
|
version_id: version.map(str::to_string),
|
||||||
|
mod_time_unix_nanos: None,
|
||||||
|
lifecycle_object_info: None,
|
||||||
is_delete_marker: delete_marker,
|
is_delete_marker: delete_marker,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn item_with_mod_time(name: &str, version: Option<&str>, mod_time_secs: u64) -> HealListItem {
|
||||||
|
HealListItem {
|
||||||
|
name: name.to_string(),
|
||||||
|
version_id: version.map(str::to_string),
|
||||||
|
mod_time_unix_nanos: Some(i128::from(mod_time_secs).saturating_mul(NANOS_PER_SECOND)),
|
||||||
|
lifecycle_object_info: None,
|
||||||
|
is_delete_marker: false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_version_filter_respects_grace_boundary() {
|
||||||
|
let started_at = 1_700_000_000;
|
||||||
|
|
||||||
|
assert!(!should_skip_new_version(None, started_at));
|
||||||
|
assert!(!should_skip_new_version(
|
||||||
|
Some(i128::from(started_at + NEW_VERSION_SKIP_GRACE_SECS).saturating_mul(NANOS_PER_SECOND)),
|
||||||
|
started_at,
|
||||||
|
));
|
||||||
|
assert!(should_skip_new_version(
|
||||||
|
Some(i128::from(started_at + NEW_VERSION_SKIP_GRACE_SECS + 1).saturating_mul(NANOS_PER_SECOND)),
|
||||||
|
started_at,
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn target_outcomes_require_each_requested_endpoint_once_and_ok() {
|
fn target_outcomes_require_each_requested_endpoint_once_and_ok() {
|
||||||
let result = HealResultItem {
|
let result = HealResultItem {
|
||||||
@@ -1246,8 +1399,10 @@ mod resume_loop_tests {
|
|||||||
/// Target-specific physical readback evidence per `compose_key`; the
|
/// Target-specific physical readback evidence per `compose_key`; the
|
||||||
/// fake models a healthy backend unless a test explicitly revokes it.
|
/// fake models a healthy backend unless a test explicitly revokes it.
|
||||||
replacement_commit_evidence: Mutex<HashMap<String, ReplacementCommitEvidence>>,
|
replacement_commit_evidence: Mutex<HashMap<String, ReplacementCommitEvidence>>,
|
||||||
|
lifecycle_expired: Mutex<HashSet<String>>,
|
||||||
/// every heal_object call recorded as (name, version_id)
|
/// every heal_object call recorded as (name, version_id)
|
||||||
heal_calls: Mutex<Vec<(String, Option<String>)>>,
|
heal_calls: Mutex<Vec<(String, Option<String>)>>,
|
||||||
|
list_include_lifecycle_object_info: Mutex<Vec<bool>>,
|
||||||
replacement_target_identity_sequences: Mutex<VecDeque<Vec<ReplacementTargetIdentity>>>,
|
replacement_target_identity_sequences: Mutex<VecDeque<Vec<ReplacementTargetIdentity>>>,
|
||||||
fail_listing: AtomicBool,
|
fail_listing: AtomicBool,
|
||||||
}
|
}
|
||||||
@@ -1274,9 +1429,15 @@ mod resume_loop_tests {
|
|||||||
.unwrap()
|
.unwrap()
|
||||||
.insert(compose_key(name, version), ReplacementCommitEvidence::Error(message.to_string()));
|
.insert(compose_key(name, version), ReplacementCommitEvidence::Error(message.to_string()));
|
||||||
}
|
}
|
||||||
|
fn set_lifecycle_expired(&self, name: &str, version: Option<&str>) {
|
||||||
|
self.lifecycle_expired.lock().unwrap().insert(compose_key(name, version));
|
||||||
|
}
|
||||||
fn calls(&self) -> Vec<(String, Option<String>)> {
|
fn calls(&self) -> Vec<(String, Option<String>)> {
|
||||||
self.heal_calls.lock().unwrap().clone()
|
self.heal_calls.lock().unwrap().clone()
|
||||||
}
|
}
|
||||||
|
fn list_include_lifecycle_object_info_calls(&self) -> Vec<bool> {
|
||||||
|
self.list_include_lifecycle_object_info.lock().unwrap().clone()
|
||||||
|
}
|
||||||
fn fail_listing(&self) {
|
fn fail_listing(&self) {
|
||||||
self.fail_listing.store(true, Ordering::SeqCst);
|
self.fail_listing.store(true, Ordering::SeqCst);
|
||||||
}
|
}
|
||||||
@@ -1330,6 +1491,23 @@ mod resume_loop_tests {
|
|||||||
async fn get_object_checksum(&self, _b: &str, _o: &str) -> Result<Option<String>> {
|
async fn get_object_checksum(&self, _b: &str, _o: &str) -> Result<Option<String>> {
|
||||||
Ok(None)
|
Ok(None)
|
||||||
}
|
}
|
||||||
|
async fn load_heal_lifecycle_expiry_context(&self, _bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
|
||||||
|
Ok((!self.lifecycle_expired.lock().unwrap().is_empty()).then(HealLifecycleExpiryContext::test))
|
||||||
|
}
|
||||||
|
async fn enqueue_heal_lifecycle_expiry(
|
||||||
|
&self,
|
||||||
|
_context: &HealLifecycleExpiryContext,
|
||||||
|
_bucket: &str,
|
||||||
|
object: &str,
|
||||||
|
version_id: Option<&str>,
|
||||||
|
_object_info: Option<&HealObjectInfo>,
|
||||||
|
) -> Result<bool> {
|
||||||
|
Ok(self
|
||||||
|
.lifecycle_expired
|
||||||
|
.lock()
|
||||||
|
.unwrap()
|
||||||
|
.contains(&compose_key(object, version_id)))
|
||||||
|
}
|
||||||
async fn heal_object(
|
async fn heal_object(
|
||||||
&self,
|
&self,
|
||||||
_bucket: &str,
|
_bucket: &str,
|
||||||
@@ -1386,7 +1564,12 @@ mod resume_loop_tests {
|
|||||||
_bucket: &str,
|
_bucket: &str,
|
||||||
_prefix: &str,
|
_prefix: &str,
|
||||||
continuation_token: Option<&str>,
|
continuation_token: Option<&str>,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
||||||
|
self.list_include_lifecycle_object_info
|
||||||
|
.lock()
|
||||||
|
.unwrap()
|
||||||
|
.push(include_lifecycle_object_info);
|
||||||
if self.fail_listing.load(Ordering::SeqCst) {
|
if self.fail_listing.load(Ordering::SeqCst) {
|
||||||
return Err(Error::other("injected listing failure"));
|
return Err(Error::other("injected listing failure"));
|
||||||
}
|
}
|
||||||
@@ -1476,6 +1659,7 @@ mod resume_loop_tests {
|
|||||||
|
|
||||||
/// Drive one bucket heal pass; returns (processed, successful, failed, skipped, result).
|
/// Drive one bucket heal pass; returns (processed, successful, failed, skipped, result).
|
||||||
async fn run(env: &Env) -> (u64, u64, u64, u64, Result<()>) {
|
async fn run(env: &Env) -> (u64, u64, u64, u64, Result<()>) {
|
||||||
|
let state = env.resume.get_state().await;
|
||||||
let mut current_object_index = 0usize;
|
let mut current_object_index = 0usize;
|
||||||
let mut processed = 0u64;
|
let mut processed = 0u64;
|
||||||
let mut successful = 0u64;
|
let mut successful = 0u64;
|
||||||
@@ -1494,6 +1678,7 @@ mod resume_loop_tests {
|
|||||||
&mut skipped,
|
&mut skipped,
|
||||||
&env.resume,
|
&env.resume,
|
||||||
&env.checkpoint,
|
&env.checkpoint,
|
||||||
|
state.start_time,
|
||||||
)
|
)
|
||||||
.await;
|
.await;
|
||||||
(processed, successful, failed, skipped, result)
|
(processed, successful, failed, skipped, result)
|
||||||
@@ -1559,6 +1744,7 @@ mod resume_loop_tests {
|
|||||||
let mut successful = 0;
|
let mut successful = 0;
|
||||||
let mut failed = 0;
|
let mut failed = 0;
|
||||||
let mut skipped = 0;
|
let mut skipped = 0;
|
||||||
|
let started_at = env.resume.get_state().await.start_time;
|
||||||
|
|
||||||
let error = healer
|
let error = healer
|
||||||
.heal_bucket_with_resume(
|
.heal_bucket_with_resume(
|
||||||
@@ -1572,6 +1758,7 @@ mod resume_loop_tests {
|
|||||||
&mut skipped,
|
&mut skipped,
|
||||||
&env.resume,
|
&env.resume,
|
||||||
&env.checkpoint,
|
&env.checkpoint,
|
||||||
|
started_at,
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
.expect_err("a remounted target must not begin a new page scan");
|
.expect_err("a remounted target must not begin a new page scan");
|
||||||
@@ -1641,6 +1828,109 @@ mod resume_loop_tests {
|
|||||||
assert_eq!(skipped, 0);
|
assert_eq!(skipped, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn erasure_set_progress_accumulates_healed_object_bytes() {
|
||||||
|
let env = make_env().await;
|
||||||
|
env.storage.set_page(
|
||||||
|
None,
|
||||||
|
Page {
|
||||||
|
items: vec![item("first", Some("v1"), false), item("second", Some("v2"), false)],
|
||||||
|
next: None,
|
||||||
|
truncated: false,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
env.storage.set_result(
|
||||||
|
"first",
|
||||||
|
Some("v1"),
|
||||||
|
HealResultItem {
|
||||||
|
object_size: 1024,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
);
|
||||||
|
env.storage.set_result(
|
||||||
|
"second",
|
||||||
|
Some("v2"),
|
||||||
|
HealResultItem {
|
||||||
|
object_size: 2048,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
let (processed, successful, failed, skipped, result) = run(&env).await;
|
||||||
|
|
||||||
|
result.expect("page heal should succeed");
|
||||||
|
assert_eq!(processed, 2);
|
||||||
|
assert_eq!(successful, 2);
|
||||||
|
assert_eq!(failed, 0);
|
||||||
|
assert_eq!(skipped, 0);
|
||||||
|
let progress = env.healer.progress.read().await;
|
||||||
|
assert_eq!(progress.objects_scanned, 2);
|
||||||
|
assert_eq!(progress.objects_healed, 2);
|
||||||
|
assert_eq!(progress.objects_failed, 0);
|
||||||
|
assert_eq!(progress.bytes_processed, 3072);
|
||||||
|
assert!(matches!(progress.current_object.as_deref(), Some("b/first" | "b/second")));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn erasure_set_skips_versions_written_after_heal_started() {
|
||||||
|
let env = make_env().await;
|
||||||
|
let started_at = env.resume.get_state().await.start_time;
|
||||||
|
env.storage.set_page(
|
||||||
|
None,
|
||||||
|
Page {
|
||||||
|
items: vec![
|
||||||
|
item_with_mod_time("old", Some("v1"), started_at + NEW_VERSION_SKIP_GRACE_SECS),
|
||||||
|
item_with_mod_time("new", Some("v2"), started_at + NEW_VERSION_SKIP_GRACE_SECS + 1),
|
||||||
|
],
|
||||||
|
next: None,
|
||||||
|
truncated: false,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
let (processed, successful, failed, skipped, result) = run(&env).await;
|
||||||
|
|
||||||
|
result.expect("page heal should succeed");
|
||||||
|
assert_eq!(processed, 2);
|
||||||
|
assert_eq!(successful, 1);
|
||||||
|
assert_eq!(failed, 0);
|
||||||
|
assert_eq!(skipped, 0);
|
||||||
|
assert_eq!(env.storage.calls(), vec![("old".to_string(), Some("v1".to_string()))]);
|
||||||
|
let progress = env.healer.progress.read().await;
|
||||||
|
assert_eq!(progress.skipped_new_versions, 1);
|
||||||
|
assert_eq!(progress.objects_scanned, 2);
|
||||||
|
assert_eq!(progress.objects_healed, 1);
|
||||||
|
assert_eq!(progress.objects_failed, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn erasure_set_skips_versions_queued_for_lifecycle_expiry() {
|
||||||
|
let env = make_env().await;
|
||||||
|
env.storage.set_page(
|
||||||
|
None,
|
||||||
|
Page {
|
||||||
|
items: vec![item("expired", Some("v1"), false), item("kept", Some("v2"), false)],
|
||||||
|
next: None,
|
||||||
|
truncated: false,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
env.storage.set_lifecycle_expired("expired", Some("v1"));
|
||||||
|
|
||||||
|
let (processed, successful, failed, skipped, result) = run(&env).await;
|
||||||
|
|
||||||
|
result.expect("page heal should succeed");
|
||||||
|
assert_eq!(processed, 2);
|
||||||
|
assert_eq!(successful, 1);
|
||||||
|
assert_eq!(failed, 0);
|
||||||
|
assert_eq!(skipped, 0);
|
||||||
|
assert_eq!(env.storage.calls(), vec![("kept".to_string(), Some("v2".to_string()))]);
|
||||||
|
assert_eq!(env.storage.list_include_lifecycle_object_info_calls(), vec![true]);
|
||||||
|
let progress = env.healer.progress.read().await;
|
||||||
|
assert_eq!(progress.skipped_ilm_expired, 1);
|
||||||
|
assert_eq!(progress.objects_scanned, 2);
|
||||||
|
assert_eq!(progress.objects_healed, 1);
|
||||||
|
assert_eq!(progress.objects_failed, 0);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn bucket_listing_failure_does_not_mark_set_completed() {
|
async fn bucket_listing_failure_does_not_mark_set_completed() {
|
||||||
let env = make_env().await;
|
let env = make_env().await;
|
||||||
|
|||||||
@@ -270,6 +270,8 @@ pub struct HealSourceCounts {
|
|||||||
pub auto_heal: u64,
|
pub auto_heal: u64,
|
||||||
pub internal: u64,
|
pub internal: u64,
|
||||||
pub read_repair: u64,
|
pub read_repair: u64,
|
||||||
|
#[serde(default)]
|
||||||
|
pub mrf: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl HealSourceCounts {
|
impl HealSourceCounts {
|
||||||
@@ -280,6 +282,7 @@ impl HealSourceCounts {
|
|||||||
HealRequestSource::AutoHeal => self.auto_heal += 1,
|
HealRequestSource::AutoHeal => self.auto_heal += 1,
|
||||||
HealRequestSource::Internal => self.internal += 1,
|
HealRequestSource::Internal => self.internal += 1,
|
||||||
HealRequestSource::ReadRepair => self.read_repair += 1,
|
HealRequestSource::ReadRepair => self.read_repair += 1,
|
||||||
|
HealRequestSource::Mrf => self.mrf += 1,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -2385,8 +2388,27 @@ impl HealManager {
|
|||||||
snapshot.objects_scanned = snapshot.objects_scanned.saturating_add(progress.objects_scanned);
|
snapshot.objects_scanned = snapshot.objects_scanned.saturating_add(progress.objects_scanned);
|
||||||
snapshot.objects_healed = snapshot.objects_healed.saturating_add(progress.objects_healed);
|
snapshot.objects_healed = snapshot.objects_healed.saturating_add(progress.objects_healed);
|
||||||
snapshot.objects_failed = snapshot.objects_failed.saturating_add(progress.objects_failed);
|
snapshot.objects_failed = snapshot.objects_failed.saturating_add(progress.objects_failed);
|
||||||
|
snapshot.skipped_new_versions = snapshot.skipped_new_versions.saturating_add(progress.skipped_new_versions);
|
||||||
|
snapshot.skipped_ilm_expired = snapshot.skipped_ilm_expired.saturating_add(progress.skipped_ilm_expired);
|
||||||
|
snapshot.objects_total_count = snapshot.objects_total_count.saturating_add(progress.objects_total_count);
|
||||||
|
snapshot.objects_total_size = snapshot.objects_total_size.saturating_add(progress.objects_total_size);
|
||||||
snapshot.bytes_processed = snapshot.bytes_processed.saturating_add(progress.bytes_processed);
|
snapshot.bytes_processed = snapshot.bytes_processed.saturating_add(progress.bytes_processed);
|
||||||
|
snapshot.start_time = match (snapshot.start_time, progress.start_time) {
|
||||||
|
(Some(current), Some(next)) => Some(current.min(next)),
|
||||||
|
(None, next) => next,
|
||||||
|
(current, None) => current,
|
||||||
|
};
|
||||||
|
snapshot.last_update_time = match (snapshot.last_update_time, progress.last_update_time) {
|
||||||
|
(Some(current), Some(next)) => Some(current.max(next)),
|
||||||
|
(None, next) => next,
|
||||||
|
(current, None) => current,
|
||||||
|
};
|
||||||
|
if progress.current_object.is_some() {
|
||||||
|
snapshot.current_object = progress.current_object;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
snapshot.refresh_progress_percentage();
|
||||||
|
snapshot.refresh_estimated_completion_time();
|
||||||
Some(snapshot)
|
Some(snapshot)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -3208,6 +3230,7 @@ impl HealManager {
|
|||||||
} else {
|
} else {
|
||||||
completed_task.get_status().await
|
completed_task.get_status().await
|
||||||
};
|
};
|
||||||
|
let completed_progress = completed_task.get_progress().await;
|
||||||
let completed_status_entry = CompletedHealStatus {
|
let completed_status_entry = CompletedHealStatus {
|
||||||
heal_type: completed_task.heal_type.clone(),
|
heal_type: completed_task.heal_type.clone(),
|
||||||
status: completed_status.clone(),
|
status: completed_status.clone(),
|
||||||
@@ -3223,6 +3246,7 @@ impl HealManager {
|
|||||||
match completed_status {
|
match completed_status {
|
||||||
HealTaskStatus::Completed => {
|
HealTaskStatus::Completed => {
|
||||||
stats.update_task_completion(true);
|
stats.update_task_completion(true);
|
||||||
|
stats.add_healed_objects(completed_progress.objects_healed, completed_progress.bytes_processed);
|
||||||
}
|
}
|
||||||
HealTaskStatus::Retrying { .. } => {}
|
HealTaskStatus::Retrying { .. } => {}
|
||||||
_ => {
|
_ => {
|
||||||
@@ -3749,6 +3773,7 @@ mod tests {
|
|||||||
_bucket: &str,
|
_bucket: &str,
|
||||||
_prefix: &str,
|
_prefix: &str,
|
||||||
_continuation_token: Option<&str>,
|
_continuation_token: Option<&str>,
|
||||||
|
_include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<crate::heal::storage::HealListItem>, Option<String>, bool)> {
|
) -> Result<(Vec<crate::heal::storage::HealListItem>, Option<String>, bool)> {
|
||||||
Ok((Vec::new(), None, false))
|
Ok((Vec::new(), None, false))
|
||||||
}
|
}
|
||||||
@@ -5396,6 +5421,8 @@ mod tests {
|
|||||||
));
|
));
|
||||||
{
|
{
|
||||||
let mut progress = first.progress.write().await;
|
let mut progress = first.progress.write().await;
|
||||||
|
progress.start_time = Some(SystemTime::now() - Duration::from_secs(20));
|
||||||
|
progress.set_total_baseline(12, 8192);
|
||||||
progress.update_progress(7, 3, 1, 4096);
|
progress.update_progress(7, 3, 1, 4096);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -5405,6 +5432,8 @@ mod tests {
|
|||||||
));
|
));
|
||||||
{
|
{
|
||||||
let mut progress = second.progress.write().await;
|
let mut progress = second.progress.write().await;
|
||||||
|
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
|
||||||
|
progress.set_total_baseline(8, 4096);
|
||||||
progress.update_progress(11, 5, 2, 2048);
|
progress.update_progress(11, 5, 2, 2048);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -5419,7 +5448,11 @@ mod tests {
|
|||||||
assert_eq!(progress.objects_scanned, 18);
|
assert_eq!(progress.objects_scanned, 18);
|
||||||
assert_eq!(progress.objects_healed, 8);
|
assert_eq!(progress.objects_healed, 8);
|
||||||
assert_eq!(progress.objects_failed, 3);
|
assert_eq!(progress.objects_failed, 3);
|
||||||
|
assert_eq!(progress.objects_total_count, 20);
|
||||||
|
assert_eq!(progress.objects_total_size, 12288);
|
||||||
assert_eq!(progress.bytes_processed, 6144);
|
assert_eq!(progress.bytes_processed, 6144);
|
||||||
|
assert!((progress.progress_percentage - 50.0).abs() < 0.001);
|
||||||
|
assert!(progress.estimated_completion_time.is_some());
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ pub mod channel;
|
|||||||
pub mod erasure_healer;
|
pub mod erasure_healer;
|
||||||
pub mod event;
|
pub mod event;
|
||||||
pub mod manager;
|
pub mod manager;
|
||||||
|
pub mod mrf_queue;
|
||||||
pub mod progress;
|
pub mod progress;
|
||||||
pub(crate) mod replacement_readiness;
|
pub(crate) mod replacement_readiness;
|
||||||
pub mod resume;
|
pub mod resume;
|
||||||
|
|||||||
@@ -0,0 +1,682 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! Mission Repair Feed (MRF) queue, journal, and consumer.
|
||||||
|
//!
|
||||||
|
//! Intents arriving on the global channel (see `rustfs_common::mrf_channel`)
|
||||||
|
//! are buffered in a bounded in-memory queue, translated into prioritized
|
||||||
|
//! heal requests, and — while they are not yet accepted by the heal manager —
|
||||||
|
//! mirrored into a durable journal so a crash or restart can replay them.
|
||||||
|
//! This is the RustFS counterpart of MinIO's `.heal/mrf/list.bin` replay,
|
||||||
|
//! layered on top of (not replacing) read-repair and scanner heal.
|
||||||
|
//!
|
||||||
|
//! Durability model: the journal is a snapshot of the *unaccepted* pending
|
||||||
|
//! set, rewritten on a group-commit cadence (every flush interval or flush
|
||||||
|
//! threshold new intents). A rewrite is atomic at the record level only — a
|
||||||
|
//! torn tail simply truncates during replay because every record carries its
|
||||||
|
//! own CRC32. Losing the last flush window (≤500 ms) is acceptable: replayed
|
||||||
|
//! duplicates are merged by the manager's dedup key, and read-repair remains
|
||||||
|
//! the safety net.
|
||||||
|
|
||||||
|
use super::{DiskStore, HealDiskExt as _, local_disk_map_read};
|
||||||
|
use crate::heal::manager::HealManager;
|
||||||
|
use metrics::{counter, gauge};
|
||||||
|
use rustfs_common::heal_channel::{HealAdmissionDropReason, HealAdmissionResult};
|
||||||
|
use rustfs_common::mrf_channel::{MRF_MAX_ATTEMPTS, MrfIntent};
|
||||||
|
use std::collections::VecDeque;
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::Duration;
|
||||||
|
use tokio::sync::mpsc;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
use crate::heal::task::{HealOptions, HealPriority, HealRequest, HealType};
|
||||||
|
|
||||||
|
/// Journal location inside the metadata bucket, following the resume-state
|
||||||
|
/// layout.
|
||||||
|
pub(crate) const MRF_JOURNAL_PATH: &str = "buckets/.heal/mrf/journal.bin";
|
||||||
|
|
||||||
|
/// Record format tag.
|
||||||
|
const MRF_JOURNAL_FORMAT: u8 = 1;
|
||||||
|
/// Record layout version.
|
||||||
|
const MRF_JOURNAL_VERSION: u8 = 1;
|
||||||
|
|
||||||
|
/// Fixed header size: format, version, kind, attempts, enqueued_at_ms,
|
||||||
|
/// has_version flag.
|
||||||
|
const MRF_RECORD_FIXED_HEAD: usize = 1 + 1 + 1 + 1 + 8 + 1;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub(crate) struct MrfConsumerConfig {
|
||||||
|
/// In-memory queue capacity in intents.
|
||||||
|
pub queue_capacity: usize,
|
||||||
|
/// Journal byte budget; a pending snapshot above this bound is rejected
|
||||||
|
/// oldest-first so the journal can never grow unbounded.
|
||||||
|
pub journal_max_bytes: usize,
|
||||||
|
/// How many journal intents to re-arm per replay round.
|
||||||
|
pub replay_batch: usize,
|
||||||
|
/// Group-commit cadence for the journal snapshot.
|
||||||
|
pub flush_interval: Duration,
|
||||||
|
/// New intents between flushes that force an early snapshot.
|
||||||
|
pub flush_threshold: usize,
|
||||||
|
/// Backoff after the heal manager reports a full admission.
|
||||||
|
pub admission_backoff: Duration,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Default for MrfConsumerConfig {
|
||||||
|
fn default() -> Self {
|
||||||
|
Self {
|
||||||
|
queue_capacity: rustfs_utils::get_env_usize(
|
||||||
|
rustfs_config::ENV_HEAL_MRF_QUEUE_SIZE,
|
||||||
|
rustfs_config::DEFAULT_HEAL_MRF_QUEUE_SIZE,
|
||||||
|
),
|
||||||
|
journal_max_bytes: rustfs_utils::get_env_usize(
|
||||||
|
rustfs_config::ENV_HEAL_MRF_JOURNAL_MAX_BYTES,
|
||||||
|
rustfs_config::DEFAULT_HEAL_MRF_JOURNAL_MAX_BYTES,
|
||||||
|
),
|
||||||
|
replay_batch: rustfs_utils::get_env_usize(
|
||||||
|
rustfs_config::ENV_HEAL_MRF_REPLAY_BATCH,
|
||||||
|
rustfs_config::DEFAULT_HEAL_MRF_REPLAY_BATCH,
|
||||||
|
),
|
||||||
|
flush_interval: Duration::from_millis(500),
|
||||||
|
flush_threshold: 1000,
|
||||||
|
admission_backoff: Duration::from_secs(5),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bounded pending set with count and byte ceilings. Overflow drops the
|
||||||
|
/// incoming intent (never a resident one) and counts the loss.
|
||||||
|
pub(crate) struct MrfQueue {
|
||||||
|
pending: VecDeque<MrfIntent>,
|
||||||
|
bytes: usize,
|
||||||
|
capacity: usize,
|
||||||
|
byte_budget: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl MrfQueue {
|
||||||
|
pub(crate) fn new(capacity: usize, byte_budget: usize) -> Self {
|
||||||
|
Self {
|
||||||
|
pending: VecDeque::new(),
|
||||||
|
bytes: 0,
|
||||||
|
capacity,
|
||||||
|
byte_budget,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `false` (after counting) when either ceiling would be crossed.
|
||||||
|
pub(crate) fn try_push(&mut self, intent: MrfIntent) -> bool {
|
||||||
|
let cost = intent.estimated_bytes();
|
||||||
|
if self.pending.len() >= self.capacity || self.bytes + cost > self.byte_budget {
|
||||||
|
counter!("rustfs_heal_mrf_dropped_total", "reason" => "queue_overflow").increment(1);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
self.bytes += cost;
|
||||||
|
self.pending.push_back(intent);
|
||||||
|
true
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn pop_front(&mut self) -> Option<MrfIntent> {
|
||||||
|
let intent = self.pending.pop_front()?;
|
||||||
|
self.bytes = self.bytes.saturating_sub(intent.estimated_bytes());
|
||||||
|
Some(intent)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn push_back(&mut self, intent: MrfIntent) {
|
||||||
|
self.bytes += intent.estimated_bytes();
|
||||||
|
self.pending.push_back(intent);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn depth(&self) -> usize {
|
||||||
|
self.pending.len()
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn bytes(&self) -> usize {
|
||||||
|
self.bytes
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn intents(&self) -> impl Iterator<Item = &MrfIntent> {
|
||||||
|
self.pending.iter()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Journal record codec
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Append one encoded record to `out`.
|
||||||
|
pub(crate) fn encode_intent(intent: &MrfIntent, out: &mut Vec<u8>) {
|
||||||
|
let start = out.len();
|
||||||
|
out.push(MRF_JOURNAL_FORMAT);
|
||||||
|
out.push(MRF_JOURNAL_VERSION);
|
||||||
|
out.push(match intent.kind {
|
||||||
|
rustfs_common::mrf_channel::MrfKind::DecodeFailure => 1,
|
||||||
|
rustfs_common::mrf_channel::MrfKind::MetadataCorruption => 2,
|
||||||
|
rustfs_common::mrf_channel::MrfKind::PartialWrite => 3,
|
||||||
|
});
|
||||||
|
out.push(intent.attempts);
|
||||||
|
out.extend_from_slice(&intent.enqueued_at_ms.to_le_bytes());
|
||||||
|
match intent.version_id {
|
||||||
|
Some(bytes) => {
|
||||||
|
out.push(1);
|
||||||
|
out.extend_from_slice(&bytes);
|
||||||
|
}
|
||||||
|
None => out.push(0),
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&(intent.bucket.len() as u32).to_le_bytes());
|
||||||
|
out.extend_from_slice(&(intent.object.len() as u32).to_le_bytes());
|
||||||
|
out.extend_from_slice(intent.bucket.as_bytes());
|
||||||
|
out.extend_from_slice(intent.object.as_bytes());
|
||||||
|
let mut hasher = crc_fast::Digest::new(crc_fast::CrcAlgorithm::Crc32IsoHdlc);
|
||||||
|
hasher.update(&out[start..]);
|
||||||
|
out.extend_from_slice(&(hasher.finalize() as u32).to_le_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
fn decode_one(data: &[u8]) -> Option<(MrfIntent, usize)> {
|
||||||
|
if data.len() < MRF_RECORD_FIXED_HEAD + 8 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
if data[0] != MRF_JOURNAL_FORMAT || data[1] != MRF_JOURNAL_VERSION {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let kind = match data[2] {
|
||||||
|
1 => rustfs_common::mrf_channel::MrfKind::DecodeFailure,
|
||||||
|
2 => rustfs_common::mrf_channel::MrfKind::MetadataCorruption,
|
||||||
|
3 => rustfs_common::mrf_channel::MrfKind::PartialWrite,
|
||||||
|
_ => return None,
|
||||||
|
};
|
||||||
|
let attempts = data[3];
|
||||||
|
let enqueued_at_ms = u64::from_le_bytes(data[4..12].try_into().expect("slice length checked"));
|
||||||
|
let has_version = data[12] != 0;
|
||||||
|
let mut cursor = MRF_RECORD_FIXED_HEAD;
|
||||||
|
let version_id = if has_version {
|
||||||
|
if data.len() < cursor + 16 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let bytes: [u8; 16] = data[cursor..cursor + 16].try_into().expect("slice length checked");
|
||||||
|
cursor += 16;
|
||||||
|
Some(bytes)
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
if data.len() < cursor + 8 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let bucket_len = u32::from_le_bytes(data[cursor..cursor + 4].try_into().expect("slice length checked")) as usize;
|
||||||
|
let object_len = u32::from_le_bytes(data[cursor + 4..cursor + 8].try_into().expect("slice length checked")) as usize;
|
||||||
|
cursor += 8;
|
||||||
|
let body_end = cursor.checked_add(bucket_len)?.checked_add(object_len)?;
|
||||||
|
let record_end = body_end.checked_add(4)?;
|
||||||
|
if data.len() < record_end {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let mut hasher = crc_fast::Digest::new(crc_fast::CrcAlgorithm::Crc32IsoHdlc);
|
||||||
|
hasher.update(&data[..body_end]);
|
||||||
|
if (hasher.finalize() as u32) != u32::from_le_bytes(data[body_end..record_end].try_into().expect("slice length checked")) {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let bucket = std::sync::Arc::from(std::str::from_utf8(&data[cursor..cursor + bucket_len]).ok()?);
|
||||||
|
let object = std::sync::Arc::from(std::str::from_utf8(&data[cursor + bucket_len..body_end]).ok()?);
|
||||||
|
Some((
|
||||||
|
MrfIntent {
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id,
|
||||||
|
kind,
|
||||||
|
enqueued_at_ms,
|
||||||
|
attempts,
|
||||||
|
},
|
||||||
|
record_end,
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a whole journal, stopping at the first torn or corrupt record.
|
||||||
|
/// Returns the decoded intents and the number of trailing bytes discarded.
|
||||||
|
pub(crate) fn decode_journal(data: &[u8]) -> (Vec<MrfIntent>, usize) {
|
||||||
|
let mut intents = Vec::new();
|
||||||
|
let mut cursor = 0usize;
|
||||||
|
while cursor < data.len() {
|
||||||
|
match decode_one(&data[cursor..]) {
|
||||||
|
Some((intent, consumed)) => {
|
||||||
|
intents.push(intent);
|
||||||
|
cursor += consumed;
|
||||||
|
}
|
||||||
|
None => break,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let truncated = data.len() - cursor;
|
||||||
|
(intents, truncated)
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Journal disk IO (all local disks, first successful read wins)
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
async fn journal_disks() -> Vec<DiskStore> {
|
||||||
|
let map = local_disk_map_read().await;
|
||||||
|
map.values().flatten().cloned().collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn read_journal() -> Option<Vec<u8>> {
|
||||||
|
for disk in journal_disks().await {
|
||||||
|
match disk.read_all(super::RUSTFS_META_BUCKET, MRF_JOURNAL_PATH).await {
|
||||||
|
Ok(bytes) => return Some(bytes.to_vec()),
|
||||||
|
Err(_) => continue,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn write_journal(data: &[u8]) {
|
||||||
|
let payload = bytes::Bytes::copy_from_slice(data);
|
||||||
|
for disk in journal_disks().await {
|
||||||
|
if let Err(err) = disk
|
||||||
|
.write_all(super::RUSTFS_META_BUCKET, MRF_JOURNAL_PATH, payload.clone())
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
warn_mrf_journal_write(&err);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if !data.is_empty() {
|
||||||
|
counter!("rustfs_heal_mrf_journal_fsync_total").increment(1);
|
||||||
|
}
|
||||||
|
gauge!("rustfs_heal_mrf_journal_bytes").set(data.len() as f64);
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn delete_journal() {
|
||||||
|
for disk in journal_disks().await {
|
||||||
|
let _ = disk
|
||||||
|
.delete(
|
||||||
|
super::RUSTFS_META_BUCKET,
|
||||||
|
MRF_JOURNAL_PATH,
|
||||||
|
crate::heal::storage_api::owner::EcstoreDeleteOptions::default(),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn warn_mrf_journal_write(err: &super::DiskError) {
|
||||||
|
tracing::warn!(
|
||||||
|
target: "rustfs::heal::mrf",
|
||||||
|
error = %err,
|
||||||
|
"MRF journal write failed; unconsumed intents may be lost on restart"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Consumer
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Translate an intent into the prioritized heal request the issue specifies:
|
||||||
|
/// decode failures go Urgent ECDecode, metadata corruption goes High
|
||||||
|
/// Metadata, partial writes go Normal object heal.
|
||||||
|
pub(crate) fn build_heal_request(intent: &MrfIntent) -> HealRequest {
|
||||||
|
let bucket = intent.bucket.to_string();
|
||||||
|
let object = intent.object.to_string();
|
||||||
|
let version_id = intent.version_id.map(|bytes| Uuid::from_bytes(bytes).to_string());
|
||||||
|
let (heal_type, priority) = match intent.kind {
|
||||||
|
rustfs_common::mrf_channel::MrfKind::DecodeFailure => (
|
||||||
|
HealType::ECDecode {
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id,
|
||||||
|
},
|
||||||
|
HealPriority::Urgent,
|
||||||
|
),
|
||||||
|
rustfs_common::mrf_channel::MrfKind::MetadataCorruption => (HealType::Metadata { bucket, object }, HealPriority::High),
|
||||||
|
rustfs_common::mrf_channel::MrfKind::PartialWrite => (
|
||||||
|
HealType::Object {
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id,
|
||||||
|
},
|
||||||
|
HealPriority::Normal,
|
||||||
|
),
|
||||||
|
};
|
||||||
|
let mut request = HealRequest::new(heal_type, HealOptions::default(), priority);
|
||||||
|
request.source = rustfs_common::heal_channel::HealRequestSource::Mrf;
|
||||||
|
request
|
||||||
|
}
|
||||||
|
|
||||||
|
struct MrfRuntime {
|
||||||
|
queue: MrfQueue,
|
||||||
|
config: MrfConsumerConfig,
|
||||||
|
new_since_flush: usize,
|
||||||
|
/// True while a journal snapshot exists on disk that no longer reflects
|
||||||
|
/// an all-consumed pending set; the next idle tick removes it (MinIO
|
||||||
|
/// deletes its `list.bin` after replay for the same reason).
|
||||||
|
journal_on_disk: bool,
|
||||||
|
/// Earliest instant a full-admission retry may proceed.
|
||||||
|
backoff_until: Option<tokio::time::Instant>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl MrfRuntime {
|
||||||
|
fn record_accept(&mut self) {
|
||||||
|
// Accepted intents leave the pending set; the next flush persists the
|
||||||
|
// smaller snapshot, which is the journal's compaction.
|
||||||
|
}
|
||||||
|
|
||||||
|
fn snapshot(&self) -> Vec<u8> {
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
for intent in self.queue.intents() {
|
||||||
|
encode_intent(intent, &mut buf);
|
||||||
|
}
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn flush(&mut self) {
|
||||||
|
write_journal(&self.snapshot()).await;
|
||||||
|
self.new_since_flush = 0;
|
||||||
|
self.journal_on_disk = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drain pending intents into the heal manager until it is full, the
|
||||||
|
/// queue empties, or attempts are exhausted.
|
||||||
|
async fn dispatch(&mut self, manager: &HealManager) {
|
||||||
|
if let Some(until) = self.backoff_until {
|
||||||
|
if tokio::time::Instant::now() < until {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
self.backoff_until = None;
|
||||||
|
}
|
||||||
|
while let Some(mut intent) = self.queue.pop_front() {
|
||||||
|
let request = build_heal_request(&intent);
|
||||||
|
match manager.submit_heal_request(request).await {
|
||||||
|
Ok(HealAdmissionResult::Accepted) | Ok(HealAdmissionResult::Merged) => self.record_accept(),
|
||||||
|
Ok(HealAdmissionResult::Full) | Ok(HealAdmissionResult::Dropped(HealAdmissionDropReason::QueueFull)) => {
|
||||||
|
intent.attempts = intent.attempts.saturating_add(1);
|
||||||
|
if intent.attempts >= MRF_MAX_ATTEMPTS {
|
||||||
|
counter!("rustfs_heal_mrf_dropped_total", "reason" => "attempts_exhausted").increment(1);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
self.queue.push_back(intent);
|
||||||
|
self.backoff_until = Some(tokio::time::Instant::now() + self.config.admission_backoff);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
Ok(HealAdmissionResult::Dropped(_)) => {
|
||||||
|
counter!("rustfs_heal_mrf_dropped_total", "reason" => "admission_policy").increment(1);
|
||||||
|
}
|
||||||
|
Err(_) => {
|
||||||
|
intent.attempts = intent.attempts.saturating_add(1);
|
||||||
|
if intent.attempts >= MRF_MAX_ATTEMPTS {
|
||||||
|
counter!("rustfs_heal_mrf_dropped_total", "reason" => "attempts_exhausted").increment(1);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
self.queue.push_back(intent);
|
||||||
|
self.backoff_until = Some(tokio::time::Instant::now() + self.config.admission_backoff);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
gauge!("rustfs_heal_mrf_queue_depth").set(self.queue.depth() as f64);
|
||||||
|
gauge!("rustfs_heal_mrf_queue_bytes").set(self.queue.bytes() as f64);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Initialize the global MRF channel (honoring `RUSTFS_HEAL_MRF_ENABLE`) and
|
||||||
|
/// spawn the consumer task. Called once from the heal runtime bootstrap right
|
||||||
|
/// after the manager started; a disabled feature or a double call is a no-op.
|
||||||
|
/// Public for integration tests that drive the real consumer loop.
|
||||||
|
pub fn spawn_mrf_consumer(manager: Arc<HealManager>) {
|
||||||
|
let enabled = rustfs_utils::get_env_bool(rustfs_config::ENV_HEAL_MRF_ENABLE, rustfs_config::DEFAULT_HEAL_MRF_ENABLE);
|
||||||
|
rustfs_common::mrf_channel::set_mrf_delivery_enabled(enabled);
|
||||||
|
if !enabled {
|
||||||
|
tracing::info!(
|
||||||
|
target: "rustfs::heal::mrf",
|
||||||
|
"MRF intent pipeline disabled by configuration; producers will not deliver"
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let receiver = match rustfs_common::mrf_channel::init_mrf_channel() {
|
||||||
|
Ok(receiver) => receiver,
|
||||||
|
Err(err) => {
|
||||||
|
tracing::warn!(
|
||||||
|
target: "rustfs::heal::mrf",
|
||||||
|
error = err,
|
||||||
|
"MRF channel initialization failed; intents will be dropped at producers"
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
tokio::spawn(async move {
|
||||||
|
run_mrf_consumer(manager, receiver).await;
|
||||||
|
});
|
||||||
|
tracing::info!(target: "rustfs::heal::mrf", "MRF intent consumer started");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Replay the durable journal into a fresh pending queue and submit whatever
|
||||||
|
/// it armed. Returns the number of intact intents replayed. Duplicates are
|
||||||
|
/// merged by the manager's dedup key; the journal file is removed once read
|
||||||
|
/// (torn tails truncate via the per-record CRC). Public for integration tests;
|
||||||
|
/// the live consumer invokes this through [`replay_into`] at startup.
|
||||||
|
pub async fn replay_journal_once(manager: &Arc<HealManager>) -> usize {
|
||||||
|
let config = MrfConsumerConfig::default();
|
||||||
|
let mut queue = MrfQueue::new(config.queue_capacity, config.journal_max_bytes);
|
||||||
|
let mut backoff_until: Option<tokio::time::Instant> = None;
|
||||||
|
replay_into(manager, &mut queue, &mut backoff_until).await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Shared replay core: read + decode + re-arm + delete, then drain what fits.
|
||||||
|
async fn replay_into(
|
||||||
|
manager: &Arc<HealManager>,
|
||||||
|
queue: &mut MrfQueue,
|
||||||
|
backoff_until: &mut Option<tokio::time::Instant>,
|
||||||
|
) -> usize {
|
||||||
|
let Some(data) = read_journal().await else {
|
||||||
|
return 0;
|
||||||
|
};
|
||||||
|
let (intents, truncated) = decode_journal(&data);
|
||||||
|
if truncated > 0 {
|
||||||
|
tracing::warn!(
|
||||||
|
target: "rustfs::heal::mrf",
|
||||||
|
truncated_bytes = truncated,
|
||||||
|
"MRF journal had a torn tail; truncated records were discarded"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
counter!("rustfs_heal_mrf_replayed_total").increment(intents.len() as u64);
|
||||||
|
let replayed = intents.len();
|
||||||
|
for intent in intents {
|
||||||
|
queue.try_push(intent);
|
||||||
|
}
|
||||||
|
delete_journal().await;
|
||||||
|
|
||||||
|
// Drain the replayed intents immediately; whatever the manager refuses
|
||||||
|
// stays armed in `queue` for the consumer's retry loop.
|
||||||
|
if backoff_until.is_none() {
|
||||||
|
while let Some(mut intent) = queue.pop_front() {
|
||||||
|
let request = build_heal_request(&intent);
|
||||||
|
match manager.submit_heal_request(request).await {
|
||||||
|
Ok(HealAdmissionResult::Accepted) | Ok(HealAdmissionResult::Merged) => {}
|
||||||
|
Ok(HealAdmissionResult::Full) | Ok(HealAdmissionResult::Dropped(HealAdmissionDropReason::QueueFull)) => {
|
||||||
|
intent.attempts = intent.attempts.saturating_add(1);
|
||||||
|
if intent.attempts < MRF_MAX_ATTEMPTS {
|
||||||
|
queue.push_back(intent);
|
||||||
|
*backoff_until = Some(tokio::time::Instant::now());
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
Ok(HealAdmissionResult::Dropped(_)) | Err(_) => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
replayed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Replay the journal, then keep draining the channel into the heal manager
|
||||||
|
/// while persisting the pending snapshot.
|
||||||
|
async fn run_mrf_consumer(manager: Arc<HealManager>, mut receiver: mpsc::Receiver<MrfIntent>) {
|
||||||
|
let config = MrfConsumerConfig::default();
|
||||||
|
let mut runtime = MrfRuntime {
|
||||||
|
queue: MrfQueue::new(config.queue_capacity, config.journal_max_bytes),
|
||||||
|
config: config.clone(),
|
||||||
|
new_since_flush: 0,
|
||||||
|
journal_on_disk: false,
|
||||||
|
backoff_until: None,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Replay: read the journal, re-arm intents (duplicates are merged by the
|
||||||
|
// manager's dedup key), then drop the file so the next flush starts clean.
|
||||||
|
replay_into(&manager, &mut runtime.queue, &mut runtime.backoff_until).await;
|
||||||
|
|
||||||
|
let mut flush_tick = tokio::time::interval(runtime.config.flush_interval);
|
||||||
|
flush_tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay);
|
||||||
|
let mut batch: Vec<MrfIntent> = Vec::with_capacity(runtime.config.replay_batch);
|
||||||
|
|
||||||
|
loop {
|
||||||
|
tokio::select! {
|
||||||
|
received = receiver.recv_many(&mut batch, runtime.config.replay_batch) => {
|
||||||
|
if received == 0 {
|
||||||
|
// Channel closed: flush once more and stop.
|
||||||
|
runtime.flush().await;
|
||||||
|
tracing::info!(
|
||||||
|
target: "rustfs::heal::mrf",
|
||||||
|
"MRF channel closed; consumer stopped after final flush"
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
for intent in batch.drain(..) {
|
||||||
|
runtime.queue.try_push(intent);
|
||||||
|
runtime.new_since_flush += 1;
|
||||||
|
}
|
||||||
|
runtime.dispatch(manager.as_ref()).await;
|
||||||
|
if runtime.new_since_flush >= runtime.config.flush_threshold {
|
||||||
|
runtime.flush().await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ = flush_tick.tick() => {
|
||||||
|
if runtime.new_since_flush > 0 || runtime.queue.depth() > 0 {
|
||||||
|
runtime.flush().await;
|
||||||
|
runtime.dispatch(manager.as_ref()).await;
|
||||||
|
} else if runtime.journal_on_disk {
|
||||||
|
// All intents consumed: remove the journal so a restart
|
||||||
|
// replays nothing (mirrors MinIO's post-replay unlink).
|
||||||
|
delete_journal().await;
|
||||||
|
runtime.journal_on_disk = false;
|
||||||
|
gauge!("rustfs_heal_mrf_journal_bytes").set(0.0);
|
||||||
|
}
|
||||||
|
gauge!("rustfs_heal_mrf_queue_depth").set(runtime.queue.depth() as f64);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use rustfs_common::mrf_channel::{MrfIntent, MrfKind};
|
||||||
|
use std::sync::Arc as StdArc;
|
||||||
|
|
||||||
|
fn intent(bucket: &str, object: &str, attempts: u8) -> MrfIntent {
|
||||||
|
MrfIntent {
|
||||||
|
bucket: StdArc::from(bucket),
|
||||||
|
object: StdArc::from(object),
|
||||||
|
version_id: Some([7u8; 16]),
|
||||||
|
kind: MrfKind::DecodeFailure,
|
||||||
|
enqueued_at_ms: 1_700_000_000_000,
|
||||||
|
attempts,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn queue_enforces_count_and_byte_ceilings() {
|
||||||
|
let mut queue = MrfQueue::new(2, usize::MAX);
|
||||||
|
assert!(queue.try_push(intent("b", "o", 0)));
|
||||||
|
assert!(queue.try_push(intent("b", "o", 0)));
|
||||||
|
assert!(!queue.try_push(intent("b", "o", 0)), "count ceiling must drop");
|
||||||
|
|
||||||
|
let mut tiny = MrfQueue::new(usize::MAX, intent("bucket", "object", 0).estimated_bytes());
|
||||||
|
assert!(tiny.try_push(intent("bucket", "object", 0)));
|
||||||
|
assert!(
|
||||||
|
!tiny.try_push(intent("bucket", "object", 0)),
|
||||||
|
"byte budget must drop before the second intent fits"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn journal_roundtrip_preserves_intents() {
|
||||||
|
let intents = vec![
|
||||||
|
intent("bucket-a", "object/a", 0),
|
||||||
|
intent("bucket-b", "object/b", 2),
|
||||||
|
MrfIntent {
|
||||||
|
bucket: StdArc::from("bucket-c"),
|
||||||
|
object: StdArc::from("object/c"),
|
||||||
|
version_id: None,
|
||||||
|
kind: MrfKind::MetadataCorruption,
|
||||||
|
enqueued_at_ms: 5,
|
||||||
|
attempts: 1,
|
||||||
|
},
|
||||||
|
];
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
for intent in &intents {
|
||||||
|
encode_intent(intent, &mut buf);
|
||||||
|
}
|
||||||
|
let (decoded, truncated) = decode_journal(&buf);
|
||||||
|
assert_eq!(truncated, 0);
|
||||||
|
assert_eq!(decoded.len(), intents.len());
|
||||||
|
for (left, right) in decoded.iter().zip(intents.iter()) {
|
||||||
|
assert_eq!(left.bucket, right.bucket);
|
||||||
|
assert_eq!(left.object, right.object);
|
||||||
|
assert_eq!(left.version_id, right.version_id);
|
||||||
|
assert_eq!(left.kind, right.kind);
|
||||||
|
assert_eq!(left.attempts, right.attempts);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn journal_torn_tail_is_truncated() {
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
encode_intent(&intent("b", "o", 0), &mut buf);
|
||||||
|
let mut torn = buf.clone();
|
||||||
|
torn.extend_from_slice(&buf[..buf.len() / 2]);
|
||||||
|
|
||||||
|
let (decoded, truncated) = decode_journal(&torn);
|
||||||
|
assert_eq!(decoded.len(), 1, "the intact record must survive");
|
||||||
|
assert!(truncated > 0, "the partial tail must be discarded");
|
||||||
|
|
||||||
|
// A corrupted body (CRC mismatch) also truncates from that record on.
|
||||||
|
let mut corrupt = buf.clone();
|
||||||
|
let mid = MRF_RECORD_FIXED_HEAD + 4;
|
||||||
|
corrupt[mid] ^= 0xff;
|
||||||
|
let (decoded, truncated) = decode_journal(&corrupt);
|
||||||
|
assert!(decoded.is_empty());
|
||||||
|
assert_eq!(truncated, corrupt.len());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heal_request_mapping_follows_priority_matrix() {
|
||||||
|
let decode = build_heal_request(&intent("b", "o", 0));
|
||||||
|
assert!(matches!(decode.heal_type, HealType::ECDecode { .. }));
|
||||||
|
assert_eq!(decode.priority, HealPriority::Urgent);
|
||||||
|
|
||||||
|
let metadata = build_heal_request(&MrfIntent {
|
||||||
|
bucket: StdArc::from("b"),
|
||||||
|
object: StdArc::from("o"),
|
||||||
|
version_id: None,
|
||||||
|
kind: MrfKind::MetadataCorruption,
|
||||||
|
enqueued_at_ms: 0,
|
||||||
|
attempts: 0,
|
||||||
|
});
|
||||||
|
assert!(matches!(metadata.heal_type, HealType::Metadata { .. }));
|
||||||
|
assert_eq!(metadata.priority, HealPriority::High);
|
||||||
|
|
||||||
|
let partial = build_heal_request(&MrfIntent {
|
||||||
|
bucket: StdArc::from("b"),
|
||||||
|
object: StdArc::from("o"),
|
||||||
|
version_id: None,
|
||||||
|
kind: MrfKind::PartialWrite,
|
||||||
|
enqueued_at_ms: 0,
|
||||||
|
attempts: 0,
|
||||||
|
});
|
||||||
|
assert!(matches!(partial.heal_type, HealType::Object { .. }));
|
||||||
|
assert_eq!(partial.priority, HealPriority::Normal);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -13,7 +13,7 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
use std::time::SystemTime;
|
use std::time::{Duration, SystemTime};
|
||||||
|
|
||||||
#[derive(Debug, Default, Clone, Serialize, Deserialize)]
|
#[derive(Debug, Default, Clone, Serialize, Deserialize)]
|
||||||
#[serde(rename_all = "camelCase")]
|
#[serde(rename_all = "camelCase")]
|
||||||
@@ -24,6 +24,14 @@ pub struct HealProgress {
|
|||||||
pub objects_healed: u64,
|
pub objects_healed: u64,
|
||||||
/// Objects failed
|
/// Objects failed
|
||||||
pub objects_failed: u64,
|
pub objects_failed: u64,
|
||||||
|
/// Versions skipped because they were written after this heal started
|
||||||
|
pub skipped_new_versions: u64,
|
||||||
|
/// Versions skipped because lifecycle already selected them for expiry
|
||||||
|
pub skipped_ilm_expired: u64,
|
||||||
|
/// Baseline object count from the latest complete usage snapshot
|
||||||
|
pub objects_total_count: u64,
|
||||||
|
/// Baseline object bytes from the latest complete usage snapshot
|
||||||
|
pub objects_total_size: u64,
|
||||||
/// Bytes processed
|
/// Bytes processed
|
||||||
pub bytes_processed: u64,
|
pub bytes_processed: u64,
|
||||||
/// Current object
|
/// Current object
|
||||||
@@ -54,10 +62,56 @@ impl HealProgress {
|
|||||||
self.bytes_processed = bytes;
|
self.bytes_processed = bytes;
|
||||||
self.last_update_time = Some(SystemTime::now());
|
self.last_update_time = Some(SystemTime::now());
|
||||||
|
|
||||||
// calculate progress percentage
|
self.refresh_progress_percentage();
|
||||||
let total = scanned + healed + failed;
|
self.refresh_estimated_completion_time();
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn set_total_baseline(&mut self, objects_total_count: u64, objects_total_size: u64) {
|
||||||
|
self.objects_total_count = objects_total_count;
|
||||||
|
self.objects_total_size = objects_total_size;
|
||||||
|
self.last_update_time = Some(SystemTime::now());
|
||||||
|
self.refresh_progress_percentage();
|
||||||
|
self.refresh_estimated_completion_time();
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn record_skipped_new_version(&mut self) {
|
||||||
|
self.skipped_new_versions = self.skipped_new_versions.saturating_add(1);
|
||||||
|
self.last_update_time = Some(SystemTime::now());
|
||||||
|
self.refresh_progress_percentage();
|
||||||
|
self.refresh_estimated_completion_time();
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn record_skipped_ilm_expired(&mut self) {
|
||||||
|
self.skipped_ilm_expired = self.skipped_ilm_expired.saturating_add(1);
|
||||||
|
self.last_update_time = Some(SystemTime::now());
|
||||||
|
self.refresh_progress_percentage();
|
||||||
|
self.refresh_estimated_completion_time();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn completed_for_baseline(&self) -> u64 {
|
||||||
|
self.objects_healed
|
||||||
|
.saturating_add(self.objects_failed)
|
||||||
|
.saturating_add(self.skipped_new_versions)
|
||||||
|
.saturating_add(self.skipped_ilm_expired)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn refresh_progress_percentage(&mut self) {
|
||||||
|
if self.objects_total_size > 0 {
|
||||||
|
self.progress_percentage = ((self.bytes_processed as f64 / self.objects_total_size as f64) * 100.0).min(100.0);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if self.objects_total_count > 0 {
|
||||||
|
let completed = self.completed_for_baseline();
|
||||||
|
self.progress_percentage = ((completed as f64 / self.objects_total_count as f64) * 100.0).min(100.0);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
let total = self
|
||||||
|
.objects_scanned
|
||||||
|
.saturating_add(self.objects_healed)
|
||||||
|
.saturating_add(self.objects_failed);
|
||||||
if total > 0 {
|
if total > 0 {
|
||||||
self.progress_percentage = (healed as f64 / total as f64) * 100.0;
|
self.progress_percentage = (self.objects_healed as f64 / total as f64) * 100.0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -66,9 +120,36 @@ impl HealProgress {
|
|||||||
self.last_update_time = Some(SystemTime::now());
|
self.last_update_time = Some(SystemTime::now());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn refresh_estimated_completion_time(&mut self) {
|
||||||
|
let Some(start_time) = self.start_time else {
|
||||||
|
self.estimated_completion_time = None;
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if self.is_completed() || !(0.0..100.0).contains(&self.progress_percentage) || self.bytes_processed == 0 {
|
||||||
|
self.estimated_completion_time = None;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
let elapsed = match SystemTime::now().duration_since(start_time) {
|
||||||
|
Ok(elapsed) if !elapsed.is_zero() => elapsed,
|
||||||
|
_ => {
|
||||||
|
self.estimated_completion_time = None;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let estimated_total_secs = elapsed.as_secs_f64() * 100.0 / self.progress_percentage;
|
||||||
|
self.estimated_completion_time = start_time.checked_add(Duration::from_secs_f64(estimated_total_secs));
|
||||||
|
}
|
||||||
|
|
||||||
pub fn is_completed(&self) -> bool {
|
pub fn is_completed(&self) -> bool {
|
||||||
self.progress_percentage >= 100.0
|
if self.progress_percentage >= 100.0 {
|
||||||
|| self.objects_scanned > 0 && self.objects_healed + self.objects_failed >= self.objects_scanned
|
return true;
|
||||||
|
}
|
||||||
|
if self.objects_total_count > 0 || self.objects_total_size > 0 {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
self.objects_scanned > 0 && self.objects_healed.saturating_add(self.objects_failed) >= self.objects_scanned
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn get_success_rate(&self) -> f64 {
|
pub fn get_success_rate(&self) -> f64 {
|
||||||
@@ -158,6 +239,10 @@ mod tests {
|
|||||||
assert_eq!(progress.objects_scanned, 0);
|
assert_eq!(progress.objects_scanned, 0);
|
||||||
assert_eq!(progress.objects_healed, 0);
|
assert_eq!(progress.objects_healed, 0);
|
||||||
assert_eq!(progress.objects_failed, 0);
|
assert_eq!(progress.objects_failed, 0);
|
||||||
|
assert_eq!(progress.skipped_new_versions, 0);
|
||||||
|
assert_eq!(progress.skipped_ilm_expired, 0);
|
||||||
|
assert_eq!(progress.objects_total_count, 0);
|
||||||
|
assert_eq!(progress.objects_total_size, 0);
|
||||||
assert_eq!(progress.bytes_processed, 0);
|
assert_eq!(progress.bytes_processed, 0);
|
||||||
assert_eq!(progress.progress_percentage, 0.0);
|
assert_eq!(progress.progress_percentage, 0.0);
|
||||||
assert!(progress.start_time.is_some());
|
assert!(progress.start_time.is_some());
|
||||||
@@ -181,6 +266,73 @@ mod tests {
|
|||||||
assert!(progress.last_update_time.is_some());
|
assert!(progress.last_update_time.is_some());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_heal_progress_estimates_completion_time_from_progress() {
|
||||||
|
let mut progress = HealProgress::new();
|
||||||
|
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
|
||||||
|
|
||||||
|
progress.update_progress(100, 25, 0, 4096);
|
||||||
|
|
||||||
|
let eta = progress
|
||||||
|
.estimated_completion_time
|
||||||
|
.expect("partial byte progress should estimate completion");
|
||||||
|
assert!(eta > SystemTime::now());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_heal_progress_uses_byte_baseline_for_percentage() {
|
||||||
|
let mut progress = HealProgress::new();
|
||||||
|
progress.set_total_baseline(10, 8192);
|
||||||
|
|
||||||
|
progress.update_progress(100, 25, 0, 4096);
|
||||||
|
|
||||||
|
assert!((progress.progress_percentage - 50.0).abs() < 0.001);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_heal_progress_uses_object_baseline_when_bytes_unknown() {
|
||||||
|
let mut progress = HealProgress::new();
|
||||||
|
progress.set_total_baseline(10, 0);
|
||||||
|
|
||||||
|
progress.update_progress(100, 3, 2, 0);
|
||||||
|
|
||||||
|
assert!((progress.progress_percentage - 50.0).abs() < 0.001);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_heal_progress_counts_skipped_versions_for_object_baseline() {
|
||||||
|
let mut progress = HealProgress::new();
|
||||||
|
progress.set_total_baseline(10, 0);
|
||||||
|
|
||||||
|
progress.update_progress(100, 3, 2, 0);
|
||||||
|
progress.record_skipped_new_version();
|
||||||
|
|
||||||
|
assert_eq!(progress.skipped_new_versions, 1);
|
||||||
|
assert!((progress.progress_percentage - 60.0).abs() < 0.001);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_heal_progress_does_not_estimate_completion_without_bytes() {
|
||||||
|
let mut progress = HealProgress::new();
|
||||||
|
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
|
||||||
|
|
||||||
|
progress.update_progress(100, 25, 0, 0);
|
||||||
|
|
||||||
|
assert!(progress.estimated_completion_time.is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_heal_progress_with_baseline_is_not_completed_by_processed_count() {
|
||||||
|
let mut progress = HealProgress::new();
|
||||||
|
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
|
||||||
|
progress.set_total_baseline(10, 8192);
|
||||||
|
|
||||||
|
progress.update_progress(1, 1, 0, 1024);
|
||||||
|
|
||||||
|
assert!(!progress.is_completed());
|
||||||
|
assert!(progress.estimated_completion_time.is_some());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_heal_progress_update_progress_zero_total() {
|
fn test_heal_progress_update_progress_zero_total() {
|
||||||
let mut progress = HealProgress::new();
|
let mut progress = HealProgress::new();
|
||||||
@@ -251,6 +403,8 @@ mod tests {
|
|||||||
assert_eq!(json["objectsScanned"], 10);
|
assert_eq!(json["objectsScanned"], 10);
|
||||||
assert_eq!(json["objectsHealed"], 8);
|
assert_eq!(json["objectsHealed"], 8);
|
||||||
assert_eq!(json["objectsFailed"], 2);
|
assert_eq!(json["objectsFailed"], 2);
|
||||||
|
assert_eq!(json["skippedNewVersions"], 0);
|
||||||
|
assert_eq!(json["skippedIlmExpired"], 0);
|
||||||
assert_eq!(json["bytesProcessed"], 1024);
|
assert_eq!(json["bytesProcessed"], 1024);
|
||||||
assert_eq!(json["currentObject"], "test-bucket/test-object");
|
assert_eq!(json["currentObject"], "test-bucket/test-object");
|
||||||
assert!(json["progressPercentage"].is_number());
|
assert!(json["progressPercentage"].is_number());
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ use serde::{Deserialize, Serialize};
|
|||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use tracing::{debug, error, warn};
|
use tracing::{debug, error, warn};
|
||||||
|
|
||||||
|
use super::storage_api::owner::{EcstoreHealLifecycleExpiryContext, ecstore_load_admin_data_usage_from_backend_cached};
|
||||||
use super::storage_api::storage::{
|
use super::storage_api::storage::{
|
||||||
BucketInfo, BucketOperations, DiskSetSelector, HealOperations as _, ListOperations as _, ObjectIO as _,
|
BucketInfo, BucketOperations, DiskSetSelector, HealOperations as _, ListOperations as _, ObjectIO as _,
|
||||||
ObjectOperations as _, StorageAdminApi,
|
ObjectOperations as _, StorageAdminApi,
|
||||||
@@ -29,6 +30,37 @@ use super::storage_api::storage::{
|
|||||||
use super::{DiskStore, ECStore, Endpoint, HealDiskExt as _, StorageError, resume::ReplacementTargetIdentity};
|
use super::{DiskStore, ECStore, Endpoint, HealDiskExt as _, StorageError, resume::ReplacementTargetIdentity};
|
||||||
pub use super::{HealObjectInfo, HealObjectOptions, HealPutObjReader};
|
pub use super::{HealObjectInfo, HealObjectOptions, HealPutObjReader};
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||||
|
pub struct HealBucketUsageBaseline {
|
||||||
|
pub objects_count: u64,
|
||||||
|
pub bytes: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct HealLifecycleExpiryContext {
|
||||||
|
inner: HealLifecycleExpiryContextInner,
|
||||||
|
}
|
||||||
|
|
||||||
|
enum HealLifecycleExpiryContextInner {
|
||||||
|
Ecstore(EcstoreHealLifecycleExpiryContext),
|
||||||
|
#[allow(dead_code)]
|
||||||
|
Test,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl HealLifecycleExpiryContext {
|
||||||
|
fn ecstore(inner: EcstoreHealLifecycleExpiryContext) -> Self {
|
||||||
|
Self {
|
||||||
|
inner: HealLifecycleExpiryContextInner::Ecstore(inner),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) fn test() -> Self {
|
||||||
|
Self {
|
||||||
|
inner: HealLifecycleExpiryContextInner::Test,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
const LOG_COMPONENT_HEAL: &str = "heal";
|
const LOG_COMPONENT_HEAL: &str = "heal";
|
||||||
const LOG_SUBSYSTEM_STORAGE: &str = "storage";
|
const LOG_SUBSYSTEM_STORAGE: &str = "storage";
|
||||||
const EVENT_HEAL_STORAGE_OBJECT_IO: &str = "heal_storage_object_io";
|
const EVENT_HEAL_STORAGE_OBJECT_IO: &str = "heal_storage_object_io";
|
||||||
@@ -272,6 +304,10 @@ pub struct HealListItem {
|
|||||||
pub name: String,
|
pub name: String,
|
||||||
/// normalized version id (`None` when the version is nil/absent)
|
/// normalized version id (`None` when the version is nil/absent)
|
||||||
pub version_id: Option<String>,
|
pub version_id: Option<String>,
|
||||||
|
/// version modification time as Unix nanoseconds
|
||||||
|
pub mod_time_unix_nanos: Option<i128>,
|
||||||
|
/// object snapshot for lifecycle evaluation
|
||||||
|
pub lifecycle_object_info: Option<HealObjectInfo>,
|
||||||
/// whether this version is a delete marker (observability only)
|
/// whether this version is a delete marker (observability only)
|
||||||
pub is_delete_marker: bool,
|
pub is_delete_marker: bool,
|
||||||
}
|
}
|
||||||
@@ -329,6 +365,28 @@ pub trait HealStorageAPI: Send + Sync {
|
|||||||
/// Get bucket info
|
/// Get bucket info
|
||||||
async fn get_bucket_info(&self, bucket: &str) -> Result<Option<BucketInfo>>;
|
async fn get_bucket_info(&self, bucket: &str) -> Result<Option<BucketInfo>>;
|
||||||
|
|
||||||
|
/// Aggregate usage-cache baselines for the requested buckets.
|
||||||
|
async fn erasure_set_usage_baseline(&self, _buckets: &[String]) -> Result<Option<HealBucketUsageBaseline>> {
|
||||||
|
Ok(None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Load per-bucket lifecycle expiry context for heal skips.
|
||||||
|
async fn load_heal_lifecycle_expiry_context(&self, _bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
|
||||||
|
Ok(None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Queue lifecycle expiry for a version that heal can skip.
|
||||||
|
async fn enqueue_heal_lifecycle_expiry(
|
||||||
|
&self,
|
||||||
|
_context: &HealLifecycleExpiryContext,
|
||||||
|
_bucket: &str,
|
||||||
|
_object: &str,
|
||||||
|
_version_id: Option<&str>,
|
||||||
|
_object_info: Option<&HealObjectInfo>,
|
||||||
|
) -> Result<bool> {
|
||||||
|
Ok(false)
|
||||||
|
}
|
||||||
|
|
||||||
/// Fix bucket metadata
|
/// Fix bucket metadata
|
||||||
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()>;
|
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()>;
|
||||||
|
|
||||||
@@ -409,6 +467,7 @@ pub trait HealStorageAPI: Send + Sync {
|
|||||||
bucket: &str,
|
bucket: &str,
|
||||||
prefix: &str,
|
prefix: &str,
|
||||||
continuation_token: Option<&str>,
|
continuation_token: Option<&str>,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealListItem>, Option<String>, bool)>;
|
) -> Result<(Vec<HealListItem>, Option<String>, bool)>;
|
||||||
|
|
||||||
/// List versions for healing via a per-erasure-set DISK-WALK union enumerator
|
/// List versions for healing via a per-erasure-set DISK-WALK union enumerator
|
||||||
@@ -427,8 +486,10 @@ pub trait HealStorageAPI: Send + Sync {
|
|||||||
bucket: &str,
|
bucket: &str,
|
||||||
prefix: &str,
|
prefix: &str,
|
||||||
continuation_token: Option<&str>,
|
continuation_token: Option<&str>,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
||||||
self.list_objects_for_heal_page(bucket, prefix, continuation_token).await
|
self.list_objects_for_heal_page(bucket, prefix, continuation_token, include_lifecycle_object_info)
|
||||||
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get disk for resume functionality.
|
/// Get disk for resume functionality.
|
||||||
@@ -1021,6 +1082,85 @@ impl HealStorageAPI for ECStoreHealStorage {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn erasure_set_usage_baseline(&self, buckets: &[String]) -> Result<Option<HealBucketUsageBaseline>> {
|
||||||
|
if buckets.is_empty() {
|
||||||
|
return Ok(None);
|
||||||
|
}
|
||||||
|
|
||||||
|
let info = match ecstore_load_admin_data_usage_from_backend_cached(self.ecstore.clone()).await {
|
||||||
|
Ok(info) if info.is_complete_bucket_usage_snapshot() => info,
|
||||||
|
Ok(_) | Err(_) => return Ok(None),
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut baseline = HealBucketUsageBaseline::default();
|
||||||
|
for bucket in buckets {
|
||||||
|
if let Some(usage) = info.buckets_usage.get(bucket) {
|
||||||
|
baseline.objects_count = baseline.objects_count.saturating_add(usage.objects_count);
|
||||||
|
baseline.bytes = baseline.bytes.saturating_add(usage.size);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(Some(baseline))
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn load_heal_lifecycle_expiry_context(&self, bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
|
||||||
|
match self.ecstore.load_heal_lifecycle_expiry_context(bucket).await {
|
||||||
|
Ok(Some(context)) => Ok(Some(HealLifecycleExpiryContext::ecstore(context))),
|
||||||
|
Ok(None) => Ok(None),
|
||||||
|
Err(err) => {
|
||||||
|
debug!(
|
||||||
|
target: "rustfs::heal::storage",
|
||||||
|
event = EVENT_HEAL_STORAGE_ADMIN_OP,
|
||||||
|
component = LOG_COMPONENT_HEAL,
|
||||||
|
subsystem = LOG_SUBSYSTEM_STORAGE,
|
||||||
|
operation = "load_heal_lifecycle_expiry_context",
|
||||||
|
bucket,
|
||||||
|
result = "failed",
|
||||||
|
error = %err,
|
||||||
|
"Heal storage lifecycle expiry context load failed"
|
||||||
|
);
|
||||||
|
Ok(None)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn enqueue_heal_lifecycle_expiry(
|
||||||
|
&self,
|
||||||
|
context: &HealLifecycleExpiryContext,
|
||||||
|
bucket: &str,
|
||||||
|
object: &str,
|
||||||
|
version_id: Option<&str>,
|
||||||
|
object_info: Option<&HealObjectInfo>,
|
||||||
|
) -> Result<bool> {
|
||||||
|
let context = match &context.inner {
|
||||||
|
HealLifecycleExpiryContextInner::Ecstore(context) => context,
|
||||||
|
HealLifecycleExpiryContextInner::Test => return Ok(false),
|
||||||
|
};
|
||||||
|
match self
|
||||||
|
.ecstore
|
||||||
|
.enqueue_heal_lifecycle_expiry(context, bucket, object, version_id, object_info)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(queued) => Ok(queued),
|
||||||
|
Err(err) => {
|
||||||
|
debug!(
|
||||||
|
target: "rustfs::heal::storage",
|
||||||
|
event = EVENT_HEAL_STORAGE_ADMIN_OP,
|
||||||
|
component = LOG_COMPONENT_HEAL,
|
||||||
|
subsystem = LOG_SUBSYSTEM_STORAGE,
|
||||||
|
operation = "enqueue_heal_lifecycle_expiry",
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id = ?version_id,
|
||||||
|
result = "failed",
|
||||||
|
error = %err,
|
||||||
|
"Heal storage lifecycle expiry check failed"
|
||||||
|
);
|
||||||
|
Ok(false)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()> {
|
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()> {
|
||||||
debug!(
|
debug!(
|
||||||
target: "rustfs::heal::storage",
|
target: "rustfs::heal::storage",
|
||||||
@@ -1436,7 +1576,7 @@ impl HealStorageAPI for ECStoreHealStorage {
|
|||||||
|
|
||||||
loop {
|
loop {
|
||||||
let (page_objects, next_token, is_truncated) = self
|
let (page_objects, next_token, is_truncated) = self
|
||||||
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref())
|
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref(), false)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
all_objects.extend(page_objects);
|
all_objects.extend(page_objects);
|
||||||
@@ -1471,6 +1611,7 @@ impl HealStorageAPI for ECStoreHealStorage {
|
|||||||
bucket: &str,
|
bucket: &str,
|
||||||
prefix: &str,
|
prefix: &str,
|
||||||
continuation_token: Option<&str>,
|
continuation_token: Option<&str>,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
||||||
debug!(
|
debug!(
|
||||||
target: "rustfs::heal::storage",
|
target: "rustfs::heal::storage",
|
||||||
@@ -1522,10 +1663,19 @@ impl HealStorageAPI for ECStoreHealStorage {
|
|||||||
let page_objects: Vec<HealListItem> = list_info
|
let page_objects: Vec<HealListItem> = list_info
|
||||||
.objects
|
.objects
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.map(|obj| HealListItem {
|
.map(|mut obj| {
|
||||||
name: obj.name,
|
obj.version_id = obj.version_id.filter(|u| !u.is_nil());
|
||||||
version_id: obj.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string()),
|
let version_id = obj.version_id.map(|u| u.to_string());
|
||||||
is_delete_marker: obj.delete_marker,
|
let mod_time_unix_nanos = obj.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos());
|
||||||
|
let is_delete_marker = obj.delete_marker;
|
||||||
|
let lifecycle_object_info = include_lifecycle_object_info.then(|| obj.clone());
|
||||||
|
HealListItem {
|
||||||
|
name: obj.name,
|
||||||
|
version_id,
|
||||||
|
mod_time_unix_nanos,
|
||||||
|
lifecycle_object_info,
|
||||||
|
is_delete_marker,
|
||||||
|
}
|
||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
let page_count = page_objects.len();
|
let page_count = page_objects.len();
|
||||||
@@ -1562,6 +1712,7 @@ impl HealStorageAPI for ECStoreHealStorage {
|
|||||||
bucket: &str,
|
bucket: &str,
|
||||||
prefix: &str,
|
prefix: &str,
|
||||||
continuation_token: Option<&str>,
|
continuation_token: Option<&str>,
|
||||||
|
include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
||||||
// Per-page bounds for the disk-walk union enumerator. Objects are atomic
|
// Per-page bounds for the disk-walk union enumerator. Objects are atomic
|
||||||
// (never split across pages), so version_budget only bounds how many
|
// (never split across pages), so version_budget only bounds how many
|
||||||
@@ -1590,7 +1741,16 @@ impl HealStorageAPI for ECStoreHealStorage {
|
|||||||
|
|
||||||
let (versions, next_forward, is_truncated) = self
|
let (versions, next_forward, is_truncated) = self
|
||||||
.ecstore
|
.ecstore
|
||||||
.heal_walk_versions_page(pool_idx, set_idx, bucket, prefix, forward_to.as_deref(), BATCH_OBJECTS, VERSION_BUDGET)
|
.heal_walk_versions_page(
|
||||||
|
pool_idx,
|
||||||
|
set_idx,
|
||||||
|
bucket,
|
||||||
|
prefix,
|
||||||
|
forward_to.as_deref(),
|
||||||
|
BATCH_OBJECTS,
|
||||||
|
VERSION_BUDGET,
|
||||||
|
include_lifecycle_object_info,
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| {
|
.map_err(|e| {
|
||||||
error!(
|
error!(
|
||||||
@@ -1614,6 +1774,8 @@ impl HealStorageAPI for ECStoreHealStorage {
|
|||||||
.map(|v| HealListItem {
|
.map(|v| HealListItem {
|
||||||
name: v.name,
|
name: v.name,
|
||||||
version_id: v.version_id,
|
version_id: v.version_id,
|
||||||
|
mod_time_unix_nanos: v.mod_time_unix_nanos,
|
||||||
|
lifecycle_object_info: v.lifecycle_object_info,
|
||||||
is_delete_marker: v.is_delete_marker,
|
is_delete_marker: v.is_delete_marker,
|
||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
|
|||||||
@@ -12,7 +12,10 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
pub(crate) use rustfs_ecstore::api::data_usage::DATA_USAGE_CACHE_NAME as ECSTORE_DATA_USAGE_CACHE_NAME;
|
pub(crate) use rustfs_ecstore::api::data_usage::{
|
||||||
|
DATA_USAGE_CACHE_NAME as ECSTORE_DATA_USAGE_CACHE_NAME,
|
||||||
|
load_admin_data_usage_from_backend_cached as ecstore_load_admin_data_usage_from_backend_cached,
|
||||||
|
};
|
||||||
pub(crate) use rustfs_ecstore::api::disk::endpoint::Endpoint as EcstoreEndpoint;
|
pub(crate) use rustfs_ecstore::api::disk::endpoint::Endpoint as EcstoreEndpoint;
|
||||||
pub(crate) use rustfs_ecstore::api::disk::error::{DiskError as EcstoreDiskError, Result as EcstoreDiskResult};
|
pub(crate) use rustfs_ecstore::api::disk::error::{DiskError as EcstoreDiskError, Result as EcstoreDiskResult};
|
||||||
pub(crate) use rustfs_ecstore::api::disk::{
|
pub(crate) use rustfs_ecstore::api::disk::{
|
||||||
@@ -25,7 +28,9 @@ pub(crate) use rustfs_ecstore::api::disk::{
|
|||||||
pub(crate) use rustfs_ecstore::api::disk::{DiskOption as EcstoreDiskOption, new_disk as ecstore_new_disk};
|
pub(crate) use rustfs_ecstore::api::disk::{DiskOption as EcstoreDiskOption, new_disk as ecstore_new_disk};
|
||||||
pub(crate) use rustfs_ecstore::api::error::{Error as EcstoreErrorType, StorageError as EcstoreStorageError};
|
pub(crate) use rustfs_ecstore::api::error::{Error as EcstoreErrorType, StorageError as EcstoreStorageError};
|
||||||
pub(crate) use rustfs_ecstore::api::runtime::local_disk_map_read as ecstore_local_disk_map_read;
|
pub(crate) use rustfs_ecstore::api::runtime::local_disk_map_read as ecstore_local_disk_map_read;
|
||||||
pub(crate) use rustfs_ecstore::api::storage::ECStore as EcstoreStore;
|
pub(crate) use rustfs_ecstore::api::storage::{
|
||||||
|
ECStore as EcstoreStore, HealLifecycleExpiryContext as EcstoreHealLifecycleExpiryContext,
|
||||||
|
};
|
||||||
use rustfs_storage_api as storage_contracts;
|
use rustfs_storage_api as storage_contracts;
|
||||||
|
|
||||||
pub(crate) mod owner {
|
pub(crate) mod owner {
|
||||||
@@ -34,8 +39,8 @@ pub(crate) mod owner {
|
|||||||
pub(crate) use super::{
|
pub(crate) use super::{
|
||||||
ECSTORE_BUCKET_META_PREFIX, ECSTORE_DATA_USAGE_CACHE_NAME, ECSTORE_HEALING_MARKER_PATH, ECSTORE_RUSTFS_META_BUCKET,
|
ECSTORE_BUCKET_META_PREFIX, ECSTORE_DATA_USAGE_CACHE_NAME, ECSTORE_HEALING_MARKER_PATH, ECSTORE_RUSTFS_META_BUCKET,
|
||||||
EcstoreConditionalFileUpdate, EcstoreDeleteOptions, EcstoreDiskAPI, EcstoreDiskBytes, EcstoreDiskError,
|
EcstoreConditionalFileUpdate, EcstoreDeleteOptions, EcstoreDiskAPI, EcstoreDiskBytes, EcstoreDiskError,
|
||||||
EcstoreDiskResult, EcstoreDiskStore, EcstoreEndpoint, EcstoreErrorType, EcstoreStorageError, EcstoreStore,
|
EcstoreDiskResult, EcstoreDiskStore, EcstoreEndpoint, EcstoreErrorType, EcstoreHealLifecycleExpiryContext,
|
||||||
ecstore_local_disk_map_read,
|
EcstoreStorageError, EcstoreStore, ecstore_load_admin_data_usage_from_backend_cached, ecstore_local_disk_map_read,
|
||||||
};
|
};
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
|
|||||||
@@ -19,11 +19,12 @@ use crate::heal::{
|
|||||||
resume::{
|
resume::{
|
||||||
CheckpointManager, ReplacementPhase, ReplacementTargetIdentity, ResumeManager, replacement_target_identities_match,
|
CheckpointManager, ReplacementPhase, ReplacementTargetIdentity, ResumeManager, replacement_target_identities_match,
|
||||||
},
|
},
|
||||||
storage::{HealStorageAPI, next_heal_listing_token},
|
storage::{HealBucketUsageBaseline, HealStorageAPI, next_heal_listing_token},
|
||||||
};
|
};
|
||||||
use crate::{Error, Result};
|
use crate::{Error, Result};
|
||||||
use metrics::{counter, histogram};
|
use metrics::{counter, histogram};
|
||||||
use rustfs_common::heal_channel::{HealOpts, HealRequestSource, HealScanMode};
|
use rustfs_common::heal_channel::{HealOpts, HealRequestSource, HealScanMode};
|
||||||
|
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, trace_emit};
|
||||||
use rustfs_madmin::heal_commands::HealResultItem;
|
use rustfs_madmin::heal_commands::HealResultItem;
|
||||||
use rustfs_utils::path::SLASH_SEPARATOR;
|
use rustfs_utils::path::SLASH_SEPARATOR;
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
@@ -178,6 +179,17 @@ pub enum HealPriority {
|
|||||||
Urgent = 3,
|
Urgent = 3,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl HealPriority {
|
||||||
|
fn as_str(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
Self::Low => "low",
|
||||||
|
Self::Normal => "normal",
|
||||||
|
Self::High => "high",
|
||||||
|
Self::Urgent => "urgent",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Heal options
|
/// Heal options
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||||
pub struct HealOptions {
|
pub struct HealOptions {
|
||||||
@@ -498,6 +510,61 @@ impl HealTask {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn emit_trace_task_state(&self, state: &'static str, duration: Duration, error: Option<&Error>) {
|
||||||
|
trace_emit(|| {
|
||||||
|
let mut event = TraceEvent::new(TraceKind::Heal, TraceFunc::HealTask)
|
||||||
|
.with_duration(duration)
|
||||||
|
.with_attr("task_id", self.id.as_str())
|
||||||
|
.with_attr("heal_type", self.heal_type.log_kind())
|
||||||
|
.with_attr("state", state)
|
||||||
|
.with_attr("source", self.source.as_str())
|
||||||
|
.with_attr("priority", self.priority.as_str())
|
||||||
|
.with_attr("retry_attempts", u64::from(self.retry_attempts))
|
||||||
|
.with_attr("dry_run", self.options.dry_run);
|
||||||
|
|
||||||
|
event = match &self.heal_type {
|
||||||
|
HealType::Cluster => event,
|
||||||
|
HealType::Object {
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id,
|
||||||
|
} => {
|
||||||
|
let event = event.with_bucket(bucket.as_str()).with_object(object.as_str());
|
||||||
|
match version_id {
|
||||||
|
Some(version_id) => event.with_attr("version_id", version_id.as_str()),
|
||||||
|
None => event,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
HealType::Bucket { bucket } => event.with_bucket(bucket.as_str()),
|
||||||
|
HealType::Prefix { bucket, prefix } => event.with_bucket(bucket.as_str()).with_object(prefix.as_str()),
|
||||||
|
HealType::ErasureSet { buckets, set_disk_id } => {
|
||||||
|
let bucket_count = u64::try_from(buckets.len()).unwrap_or(u64::MAX);
|
||||||
|
event
|
||||||
|
.with_attr("set_disk_id", set_disk_id.as_str())
|
||||||
|
.with_attr("bucket_count", bucket_count)
|
||||||
|
}
|
||||||
|
HealType::Metadata { bucket, object } => event.with_bucket(bucket.as_str()).with_object(object.as_str()),
|
||||||
|
HealType::ECDecode {
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id,
|
||||||
|
} => {
|
||||||
|
let event = event.with_bucket(bucket.as_str()).with_object(object.as_str());
|
||||||
|
match version_id {
|
||||||
|
Some(version_id) => event.with_attr("version_id", version_id.as_str()),
|
||||||
|
None => event,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
HealType::MRF { meta_path } => event.with_object(meta_path.as_str()),
|
||||||
|
};
|
||||||
|
|
||||||
|
match error {
|
||||||
|
Some(error) => event.with_attr("error", error.to_string()),
|
||||||
|
None => event,
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
async fn remaining_timeout(&self) -> Result<Option<Duration>> {
|
async fn remaining_timeout(&self) -> Result<Option<Duration>> {
|
||||||
if let Some(total) = self.options.timeout {
|
if let Some(total) = self.options.timeout {
|
||||||
let start_instant = { *self.task_start_instant.read().await };
|
let start_instant = { *self.task_start_instant.read().await };
|
||||||
@@ -717,6 +784,7 @@ impl HealTask {
|
|||||||
queue_delay = ?queue_delay,
|
queue_delay = ?queue_delay,
|
||||||
"Heal task started"
|
"Heal task started"
|
||||||
});
|
});
|
||||||
|
self.emit_trace_task_state("started", Duration::ZERO, None);
|
||||||
|
|
||||||
let result = match &self.heal_type {
|
let result = match &self.heal_type {
|
||||||
HealType::Cluster => self.heal_cluster().await,
|
HealType::Cluster => self.heal_cluster().await,
|
||||||
@@ -805,6 +873,14 @@ impl HealTask {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
let terminal_state = match &result {
|
||||||
|
Ok(_) => "completed",
|
||||||
|
Err(Error::TaskCancelled) => "cancelled",
|
||||||
|
Err(Error::TaskTimeout) => "timed_out",
|
||||||
|
Err(_) => "failed",
|
||||||
|
};
|
||||||
|
self.emit_trace_task_state(terminal_state, start_instant.elapsed(), result.as_ref().err());
|
||||||
|
|
||||||
result
|
result
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1535,7 +1611,7 @@ impl HealTask {
|
|||||||
let (objects, next_token, is_truncated) = self
|
let (objects, next_token, is_truncated) = self
|
||||||
.await_with_control(
|
.await_with_control(
|
||||||
self.storage
|
self.storage
|
||||||
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref()),
|
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref(), false),
|
||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
@@ -1697,6 +1773,23 @@ impl HealTask {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn apply_erasure_set_usage_baseline(&self, buckets: &[String]) -> Result<()> {
|
||||||
|
let baseline = match self
|
||||||
|
.await_with_control(self.storage.erasure_set_usage_baseline(buckets))
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(Some(baseline)) => baseline,
|
||||||
|
Ok(None) => return Ok(()),
|
||||||
|
Err(err @ Error::TaskCancelled) | Err(err @ Error::TaskTimeout) => return Err(err),
|
||||||
|
Err(_) => return Ok(()),
|
||||||
|
};
|
||||||
|
|
||||||
|
let HealBucketUsageBaseline { objects_count, bytes } = baseline;
|
||||||
|
let mut progress = self.progress.write().await;
|
||||||
|
progress.set_total_baseline(objects_count, bytes);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
async fn heal_metadata(&self, bucket: &str, object: &str) -> Result<()> {
|
async fn heal_metadata(&self, bucket: &str, object: &str) -> Result<()> {
|
||||||
debug!(
|
debug!(
|
||||||
target: "rustfs::heal::task",
|
target: "rustfs::heal::task",
|
||||||
@@ -2298,6 +2391,8 @@ impl HealTask {
|
|||||||
None
|
None
|
||||||
};
|
};
|
||||||
|
|
||||||
|
self.apply_erasure_set_usage_baseline(&buckets).await?;
|
||||||
|
|
||||||
let healing_marker = format!("{set_disk_id}:{}", self.id);
|
let healing_marker = format!("{set_disk_id}:{}", self.id);
|
||||||
if let Some((disk, resume_manager, _)) = replacement_resume.as_ref() {
|
if let Some((disk, resume_manager, _)) = replacement_resume.as_ref() {
|
||||||
let state = resume_manager.get_state().await;
|
let state = resume_manager.get_state().await;
|
||||||
@@ -2602,7 +2697,8 @@ impl HealTask {
|
|||||||
|
|
||||||
{
|
{
|
||||||
let mut progress = self.progress.write().await;
|
let mut progress = self.progress.write().await;
|
||||||
progress.update_progress(4, 4, 0, 0);
|
let bytes_processed = progress.bytes_processed;
|
||||||
|
progress.update_progress(4, 4, 0, bytes_processed);
|
||||||
}
|
}
|
||||||
|
|
||||||
match result {
|
match result {
|
||||||
@@ -2658,6 +2754,7 @@ mod tests {
|
|||||||
use super::super::{DiskOption, DiskStore, Endpoint, HealDiskExt as _, new_disk};
|
use super::super::{DiskOption, DiskStore, Endpoint, HealDiskExt as _, new_disk};
|
||||||
use super::*;
|
use super::*;
|
||||||
use crate::heal::storage::{DiskStatus, HealListItem, HealObjectInfo};
|
use crate::heal::storage::{DiskStatus, HealListItem, HealObjectInfo};
|
||||||
|
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, TraceSubscription, TraceVal, subscribe_trace_events};
|
||||||
use rustfs_madmin::heal_commands::{HealDriveInfo, HealResultItem, Infos};
|
use rustfs_madmin::heal_commands::{HealDriveInfo, HealResultItem, Infos};
|
||||||
use std::collections::{HashMap, VecDeque};
|
use std::collections::{HashMap, VecDeque};
|
||||||
use std::sync::Mutex;
|
use std::sync::Mutex;
|
||||||
@@ -3203,6 +3300,8 @@ mod tests {
|
|||||||
block_heal_object: Mutex<bool>,
|
block_heal_object: Mutex<bool>,
|
||||||
resume_disk: Mutex<Option<DiskStore>>,
|
resume_disk: Mutex<Option<DiskStore>>,
|
||||||
replacement_resume_disk: Mutex<Option<DiskStore>>,
|
replacement_resume_disk: Mutex<Option<DiskStore>>,
|
||||||
|
usage_baseline: Mutex<Option<HealBucketUsageBaseline>>,
|
||||||
|
usage_baseline_error: Mutex<bool>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -3265,11 +3364,69 @@ mod tests {
|
|||||||
assert_eq!(samples_logged, MAX_BUCKET_FAILURE_LOG_SAMPLES);
|
assert_eq!(samples_logged, MAX_BUCKET_FAILURE_LOG_SAMPLES);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn execute_emits_heal_trace_task_state() {
|
||||||
|
let mut trace = subscribe_trace_events();
|
||||||
|
let storage = Arc::new(MockStorage::default());
|
||||||
|
let task = HealTask::from_request(
|
||||||
|
HealRequest::object("bucket-a".to_string(), "object-a".to_string(), Some("version-a".to_string())),
|
||||||
|
storage,
|
||||||
|
);
|
||||||
|
|
||||||
|
task.execute().await.expect("mock object heal should complete");
|
||||||
|
|
||||||
|
let started = recv_trace_task_state(&mut trace, &task.id, "started").await;
|
||||||
|
assert_eq!(started.kind, TraceKind::Heal);
|
||||||
|
assert_eq!(started.func, TraceFunc::HealTask);
|
||||||
|
assert_eq!(started.bucket.as_deref(), Some("bucket-a"));
|
||||||
|
assert_eq!(started.object.as_deref(), Some("object-a"));
|
||||||
|
assert_eq!(trace_attr_string(&started, "heal_type").as_deref(), Some("object"));
|
||||||
|
assert_eq!(trace_attr_string(&started, "source").as_deref(), Some("internal"));
|
||||||
|
assert_eq!(trace_attr_string(&started, "version_id").as_deref(), Some("version-a"));
|
||||||
|
|
||||||
|
let completed = recv_trace_task_state(&mut trace, &task.id, "completed").await;
|
||||||
|
assert_eq!(completed.kind, TraceKind::Heal);
|
||||||
|
assert_eq!(completed.func, TraceFunc::HealTask);
|
||||||
|
assert_eq!(trace_attr_string(&completed, "state").as_deref(), Some("completed"));
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn recv_trace_task_state(trace: &mut TraceSubscription, task_id: &str, state: &str) -> TraceEvent {
|
||||||
|
for _ in 0..32 {
|
||||||
|
let event = tokio::time::timeout(Duration::from_secs(1), trace.recv())
|
||||||
|
.await
|
||||||
|
.expect("trace event should arrive")
|
||||||
|
.expect("trace bus should stay open");
|
||||||
|
if trace_attr_string(&event, "task_id").as_deref() == Some(task_id)
|
||||||
|
&& trace_attr_string(&event, "state").as_deref() == Some(state)
|
||||||
|
{
|
||||||
|
return (*event).clone();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
panic!("expected trace state {state} for task {task_id}");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn trace_attr_string(event: &TraceEvent, key: &str) -> Option<String> {
|
||||||
|
event.attrs.iter().find_map(|attr| {
|
||||||
|
if attr.key != key {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(match &attr.value {
|
||||||
|
TraceVal::Bool(value) => value.to_string(),
|
||||||
|
TraceVal::U64(value) => value.to_string(),
|
||||||
|
TraceVal::I64(value) => value.to_string(),
|
||||||
|
TraceVal::Str(value) => value.to_string(),
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
/// Build a latest, non-delete-marker heal list item with no version id.
|
/// Build a latest, non-delete-marker heal list item with no version id.
|
||||||
fn heal_item(name: &str) -> HealListItem {
|
fn heal_item(name: &str) -> HealListItem {
|
||||||
HealListItem {
|
HealListItem {
|
||||||
name: name.to_string(),
|
name: name.to_string(),
|
||||||
version_id: None,
|
version_id: None,
|
||||||
|
mod_time_unix_nanos: None,
|
||||||
|
lifecycle_object_info: None,
|
||||||
is_delete_marker: false,
|
is_delete_marker: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -3357,6 +3514,13 @@ mod tests {
|
|||||||
}))
|
}))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn erasure_set_usage_baseline(&self, _buckets: &[String]) -> Result<Option<HealBucketUsageBaseline>> {
|
||||||
|
if *self.usage_baseline_error.lock().unwrap() {
|
||||||
|
return Err(Error::Other("usage baseline unavailable".to_string()));
|
||||||
|
}
|
||||||
|
Ok(*self.usage_baseline.lock().unwrap())
|
||||||
|
}
|
||||||
|
|
||||||
async fn heal_bucket_metadata(&self, _bucket: &str) -> Result<()> {
|
async fn heal_bucket_metadata(&self, _bucket: &str) -> Result<()> {
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
@@ -3540,6 +3704,7 @@ mod tests {
|
|||||||
bucket: &str,
|
bucket: &str,
|
||||||
prefix: &str,
|
prefix: &str,
|
||||||
continuation_token: Option<&str>,
|
continuation_token: Option<&str>,
|
||||||
|
_include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
|
||||||
self.listed_prefixes.lock().unwrap().push(prefix.to_string());
|
self.listed_prefixes.lock().unwrap().push(prefix.to_string());
|
||||||
if *self.truncate_without_token.lock().unwrap() {
|
if *self.truncate_without_token.lock().unwrap() {
|
||||||
@@ -4654,6 +4819,73 @@ mod tests {
|
|||||||
assert!(storage.object_heal_opts.lock().unwrap().is_empty());
|
assert!(storage.object_heal_opts.lock().unwrap().is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn erasure_set_heal_applies_usage_baseline_to_progress() {
|
||||||
|
let temp = TempDir::new().expect("temporary directory should be created");
|
||||||
|
let disk = make_resume_disk(&temp).await;
|
||||||
|
let storage = Arc::new(MockStorage {
|
||||||
|
resume_disk: Mutex::new(Some(disk)),
|
||||||
|
usage_baseline: Mutex::new(Some(HealBucketUsageBaseline {
|
||||||
|
objects_count: 10,
|
||||||
|
bytes: 8,
|
||||||
|
})),
|
||||||
|
..Default::default()
|
||||||
|
});
|
||||||
|
let request = HealRequest::new(
|
||||||
|
HealType::ErasureSet {
|
||||||
|
buckets: vec!["bucket-a".to_string()],
|
||||||
|
set_disk_id: "pool_0_set_0".to_string(),
|
||||||
|
},
|
||||||
|
HealOptions {
|
||||||
|
timeout: None,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
HealPriority::Normal,
|
||||||
|
);
|
||||||
|
let task = HealTask::from_request(request, storage);
|
||||||
|
|
||||||
|
task.heal_erasure_set(vec!["bucket-a".to_string()], "pool_0_set_0".to_string())
|
||||||
|
.await
|
||||||
|
.expect("erasure set heal should complete");
|
||||||
|
|
||||||
|
let progress = task.get_progress().await;
|
||||||
|
assert_eq!(progress.objects_total_count, 10);
|
||||||
|
assert_eq!(progress.objects_total_size, 8);
|
||||||
|
assert_eq!(progress.bytes_processed, 2);
|
||||||
|
assert!((progress.progress_percentage - 25.0).abs() < 0.001);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn erasure_set_heal_ignores_usage_baseline_errors() {
|
||||||
|
let temp = TempDir::new().expect("temporary directory should be created");
|
||||||
|
let disk = make_resume_disk(&temp).await;
|
||||||
|
let storage = Arc::new(MockStorage {
|
||||||
|
resume_disk: Mutex::new(Some(disk)),
|
||||||
|
usage_baseline_error: Mutex::new(true),
|
||||||
|
..Default::default()
|
||||||
|
});
|
||||||
|
let request = HealRequest::new(
|
||||||
|
HealType::ErasureSet {
|
||||||
|
buckets: vec!["bucket-a".to_string()],
|
||||||
|
set_disk_id: "pool_0_set_0".to_string(),
|
||||||
|
},
|
||||||
|
HealOptions {
|
||||||
|
timeout: None,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
HealPriority::Normal,
|
||||||
|
);
|
||||||
|
let task = HealTask::from_request(request, storage);
|
||||||
|
|
||||||
|
task.heal_erasure_set(vec!["bucket-a".to_string()], "pool_0_set_0".to_string())
|
||||||
|
.await
|
||||||
|
.expect("usage baseline failures should not fail erasure set heal");
|
||||||
|
|
||||||
|
let progress = task.get_progress().await;
|
||||||
|
assert_eq!(progress.objects_total_count, 0);
|
||||||
|
assert_eq!(progress.objects_total_size, 0);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn resumable_erasure_set_execution_is_cancelled_while_object_heal_is_pending() {
|
async fn resumable_erasure_set_execution_is_cancelled_while_object_heal_is_pending() {
|
||||||
let temp = TempDir::new().expect("temporary directory should be created");
|
let temp = TempDir::new().expect("temporary directory should be created");
|
||||||
|
|||||||
@@ -158,6 +158,10 @@ pub async fn init_heal_manager_with_workload_provider(
|
|||||||
return Err(err);
|
return Err(err);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Start the MRF intent consumer (error-path repair intents + durable
|
||||||
|
// journal replay) now that the manager can accept submissions.
|
||||||
|
heal::mrf_queue::spawn_mrf_consumer(heal_manager.clone());
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
test_hook_after_manager_start().await;
|
test_hook_after_manager_start().await;
|
||||||
|
|
||||||
@@ -445,6 +449,7 @@ mod tests {
|
|||||||
_bucket: &str,
|
_bucket: &str,
|
||||||
_prefix: &str,
|
_prefix: &str,
|
||||||
_continuation_token: Option<&str>,
|
_continuation_token: Option<&str>,
|
||||||
|
_include_lifecycle_object_info: bool,
|
||||||
) -> Result<(Vec<HealListItem>, Option<String>, bool), Error> {
|
) -> Result<(Vec<HealListItem>, Option<String>, bool), Error> {
|
||||||
Ok((Vec::new(), None, false))
|
Ok((Vec::new(), None, false))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -176,7 +176,7 @@ async fn enumerate_all_versions(heal_storage: &Arc<ECStoreHealStorage>, bucket:
|
|||||||
let mut token: Option<String> = None;
|
let mut token: Option<String> = None;
|
||||||
loop {
|
loop {
|
||||||
let (page, next, truncated) = heal_storage
|
let (page, next, truncated) = heal_storage
|
||||||
.list_objects_for_heal_page(bucket, "", token.as_deref())
|
.list_objects_for_heal_page(bucket, "", token.as_deref(), false)
|
||||||
.await
|
.await
|
||||||
.expect("list_objects_for_heal_page failed");
|
.expect("list_objects_for_heal_page failed");
|
||||||
items.extend(page);
|
items.extend(page);
|
||||||
|
|||||||
@@ -166,7 +166,7 @@ async fn enumerate_b5(heal_storage: &Arc<ECStoreHealStorage>, bucket: &str) -> V
|
|||||||
let mut token: Option<String> = None;
|
let mut token: Option<String> = None;
|
||||||
loop {
|
loop {
|
||||||
let (page, next, truncated) = heal_storage
|
let (page, next, truncated) = heal_storage
|
||||||
.list_objects_for_heal_page(bucket, "", token.as_deref())
|
.list_objects_for_heal_page(bucket, "", token.as_deref(), false)
|
||||||
.await
|
.await
|
||||||
.expect("b5 list page failed");
|
.expect("b5 list page failed");
|
||||||
items.extend(page);
|
items.extend(page);
|
||||||
@@ -187,7 +187,7 @@ async fn enumerate_disk_walk(heal_storage: &Arc<ECStoreHealStorage>, bucket: &st
|
|||||||
let mut token: Option<String> = None;
|
let mut token: Option<String> = None;
|
||||||
loop {
|
loop {
|
||||||
let (page, next, truncated) = heal_storage
|
let (page, next, truncated) = heal_storage
|
||||||
.list_versions_for_heal_page_disk_walk(SET_DISK_ID, bucket, "", token.as_deref())
|
.list_versions_for_heal_page_disk_walk(SET_DISK_ID, bucket, "", token.as_deref(), false)
|
||||||
.await
|
.await
|
||||||
.expect("disk-walk list page failed");
|
.expect("disk-walk list page failed");
|
||||||
items.extend(page);
|
items.extend(page);
|
||||||
@@ -418,7 +418,7 @@ mod serial_tests {
|
|||||||
let mut pages = 0usize;
|
let mut pages = 0usize;
|
||||||
loop {
|
loop {
|
||||||
let (versions, next_forward, truncated) = ecstore
|
let (versions, next_forward, truncated) = ecstore
|
||||||
.heal_walk_versions_page(0, 0, bucket, "", forward.as_deref(), 2, 100_000)
|
.heal_walk_versions_page(0, 0, bucket, "", forward.as_deref(), 2, 100_000, false)
|
||||||
.await
|
.await
|
||||||
.expect("heal_walk_versions_page failed");
|
.expect("heal_walk_versions_page failed");
|
||||||
pages += 1;
|
pages += 1;
|
||||||
|
|||||||
@@ -242,6 +242,7 @@ fn test_heal_task_status_atomic_update() {
|
|||||||
_bucket: &str,
|
_bucket: &str,
|
||||||
_prefix: &str,
|
_prefix: &str,
|
||||||
_continuation_token: Option<&str>,
|
_continuation_token: Option<&str>,
|
||||||
|
_include_lifecycle_object_info: bool,
|
||||||
) -> rustfs_heal::Result<(Vec<HealListItem>, Option<String>, bool)> {
|
) -> rustfs_heal::Result<(Vec<HealListItem>, Option<String>, bool)> {
|
||||||
Ok((vec![], None, false))
|
Ok((vec![], None, false))
|
||||||
}
|
}
|
||||||
@@ -385,6 +386,7 @@ async fn test_heal_task_transient_object_exists_skip_avoids_recreate() {
|
|||||||
_bucket: &str,
|
_bucket: &str,
|
||||||
_prefix: &str,
|
_prefix: &str,
|
||||||
_continuation_token: Option<&str>,
|
_continuation_token: Option<&str>,
|
||||||
|
_include_lifecycle_object_info: bool,
|
||||||
) -> rustfs_heal::Result<(Vec<HealListItem>, Option<String>, bool)> {
|
) -> rustfs_heal::Result<(Vec<HealListItem>, Option<String>, bool)> {
|
||||||
Ok((Vec::new(), None, false))
|
Ok((Vec::new(), None, false))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,189 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! HS-01 (rustfs/backlog#1865): MRF intent pipeline integration tests.
|
||||||
|
//!
|
||||||
|
//! Drives the real consumer loop (`spawn_mrf_consumer`) against a real
|
||||||
|
//! 4-disk `ECStore` heal storage and a `HealManager` that has not started its
|
||||||
|
//! scheduler, so submitted intents stay observable in the admission queue.
|
||||||
|
//! Under `cargo nextest` each test runs in its own process, which keeps the
|
||||||
|
//! process-global MRF channel singleton safe.
|
||||||
|
|
||||||
|
use rustfs_common::mrf_channel::{self, MrfKind};
|
||||||
|
use rustfs_heal::heal::{
|
||||||
|
manager::{HealConfig, HealManager},
|
||||||
|
mrf_queue,
|
||||||
|
storage::{ECStoreHealStorage, HealStorageAPI},
|
||||||
|
};
|
||||||
|
use serial_test::serial;
|
||||||
|
use std::{path::Path, sync::Arc, time::Duration};
|
||||||
|
|
||||||
|
mod storage_api;
|
||||||
|
|
||||||
|
use storage_api::endpoint_index::{Endpoint, EndpointServerPools, Endpoints, PoolEndpoints, init_local_disks};
|
||||||
|
|
||||||
|
const META_BUCKET: &str = ".rustfs.sys";
|
||||||
|
const JOURNAL_REL: &str = "buckets/.heal/mrf/journal.bin";
|
||||||
|
|
||||||
|
async fn heal_env() -> (Vec<std::path::PathBuf>, Arc<dyn HealStorageAPI>) {
|
||||||
|
let env = rustfs_test_utils::TestECStoreEnv::builder()
|
||||||
|
.prefix("rustfs_heal_mrf_test")
|
||||||
|
.build()
|
||||||
|
.await;
|
||||||
|
let heal_storage: Arc<dyn HealStorageAPI> = Arc::new(ECStoreHealStorage::new(env.ecstore.clone()));
|
||||||
|
(env.disk_paths, heal_storage)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn make_manager(storage: Arc<dyn HealStorageAPI>) -> Arc<HealManager> {
|
||||||
|
Arc::new(HealManager::new(
|
||||||
|
storage,
|
||||||
|
Some(HealConfig {
|
||||||
|
// Keep the scheduler from draining the queue before assertions.
|
||||||
|
heal_interval: Duration::from_secs(3600),
|
||||||
|
enable_auto_heal: false,
|
||||||
|
..Default::default()
|
||||||
|
}),
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode one journal record independently of the implementation, so a format
|
||||||
|
/// drift between writer and this fixture fails loudly here.
|
||||||
|
fn journal_record(kind: u8, bucket: &str, object: &str, version: Option<[u8; 16]>, attempts: u8) -> Vec<u8> {
|
||||||
|
let mut body = vec![1u8, 1, kind, attempts];
|
||||||
|
body.extend_from_slice(&1_700_000_000_000u64.to_le_bytes());
|
||||||
|
match version {
|
||||||
|
Some(bytes) => {
|
||||||
|
body.push(1);
|
||||||
|
body.extend_from_slice(&bytes);
|
||||||
|
}
|
||||||
|
None => body.push(0),
|
||||||
|
}
|
||||||
|
body.extend_from_slice(&(bucket.len() as u32).to_le_bytes());
|
||||||
|
body.extend_from_slice(&(object.len() as u32).to_le_bytes());
|
||||||
|
body.extend_from_slice(bucket.as_bytes());
|
||||||
|
body.extend_from_slice(object.as_bytes());
|
||||||
|
let mut hasher = crc_fast::Digest::new(crc_fast::CrcAlgorithm::Crc32IsoHdlc);
|
||||||
|
hasher.update(&body);
|
||||||
|
body.extend_from_slice(&(hasher.finalize() as u32).to_le_bytes());
|
||||||
|
body
|
||||||
|
}
|
||||||
|
|
||||||
|
fn write_journal_to_disks(disk_paths: &[std::path::PathBuf], data: &[u8]) {
|
||||||
|
for path in disk_paths {
|
||||||
|
let journal = path.join(META_BUCKET).join(JOURNAL_REL);
|
||||||
|
std::fs::create_dir_all(journal.parent().expect("journal parent")).expect("create journal dir");
|
||||||
|
std::fs::write(&journal, data).expect("write journal fixture");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn wait_until<F, Fut>(deadline: Duration, mut probe: F) -> bool
|
||||||
|
where
|
||||||
|
F: FnMut() -> Fut,
|
||||||
|
Fut: std::future::Future<Output = bool>,
|
||||||
|
{
|
||||||
|
let start = std::time::Instant::now();
|
||||||
|
while start.elapsed() < deadline {
|
||||||
|
if probe().await {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
tokio::time::sleep(Duration::from_millis(50)).await;
|
||||||
|
}
|
||||||
|
false
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A decode-failure intent delivered on the global channel must surface in the
|
||||||
|
/// heal manager as an Urgent request attributed to the MRF source.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn decode_failure_intent_maps_to_urgent_mrf_heal_request() {
|
||||||
|
let (_disk_paths, storage) = heal_env().await;
|
||||||
|
let manager = make_manager(storage);
|
||||||
|
|
||||||
|
mrf_queue::spawn_mrf_consumer(manager.clone());
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
mrf_channel::try_send_mrf_intent(MrfKind::DecodeFailure, "mrf-bucket", "mrf-object", None),
|
||||||
|
"intent should be accepted while the consumer holds the channel"
|
||||||
|
);
|
||||||
|
|
||||||
|
let appeared = wait_until(Duration::from_secs(10), || async {
|
||||||
|
let snapshot = manager.operations_snapshot().await;
|
||||||
|
snapshot.queued_by_source.mrf >= 1 && snapshot.queued_by_priority.urgent >= 1
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
assert!(
|
||||||
|
appeared,
|
||||||
|
"MRF intent must reach the manager queue as an Urgent request (snapshot: {:?})",
|
||||||
|
manager.operations_snapshot().await
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A journal left behind by a previous process must be replayed into the
|
||||||
|
/// manager queue and then removed, and a torn tail must not block replay of
|
||||||
|
/// the intact records.
|
||||||
|
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||||
|
#[serial]
|
||||||
|
async fn journal_replay_arms_intents_and_deletes_the_file() {
|
||||||
|
let (disk_paths, storage) = heal_env().await;
|
||||||
|
|
||||||
|
// The journal reader resolves disks through the process-local disk map;
|
||||||
|
// register the environment's disks the same way server startup does.
|
||||||
|
let mut endpoints: Vec<Endpoint> = disk_paths
|
||||||
|
.iter()
|
||||||
|
.map(|p| Endpoint::try_from(p.to_string_lossy().as_ref()).expect("endpoint from disk path"))
|
||||||
|
.collect();
|
||||||
|
for (i, endpoint) in endpoints.iter_mut().enumerate() {
|
||||||
|
endpoint.set_pool_index(0);
|
||||||
|
endpoint.set_set_index(0);
|
||||||
|
endpoint.set_disk_index(i);
|
||||||
|
}
|
||||||
|
let pool = PoolEndpoints {
|
||||||
|
legacy: false,
|
||||||
|
set_count: 1,
|
||||||
|
drives_per_set: endpoints.len(),
|
||||||
|
endpoints: Endpoints::from(endpoints),
|
||||||
|
cmd_line: "mrf-test".to_string(),
|
||||||
|
platform: String::new(),
|
||||||
|
};
|
||||||
|
init_local_disks(EndpointServerPools::from(vec![pool]))
|
||||||
|
.await
|
||||||
|
.expect("local disks should register");
|
||||||
|
|
||||||
|
let mut journal = journal_record(1, "replay-bucket", "replay-object", Some([9u8; 16]), 0);
|
||||||
|
journal.extend(journal_record(3, "replay-bucket", "partial-object", None, 1));
|
||||||
|
// Torn tail: a third record truncated mid-way must not block the two
|
||||||
|
// intact records above.
|
||||||
|
journal.extend_from_slice(&journal_record(2, "replay-bucket", "metadata-object", None, 0)[..8]);
|
||||||
|
write_journal_to_disks(&disk_paths, &journal);
|
||||||
|
|
||||||
|
let manager = make_manager(storage);
|
||||||
|
// Replay directly (not via the process-global channel consumer, which the
|
||||||
|
// sibling test already claimed in this process under plain `cargo test`).
|
||||||
|
let replayed = mrf_queue::replay_journal_once(&manager).await;
|
||||||
|
assert_eq!(replayed, 2, "the two intact records must be replayed");
|
||||||
|
|
||||||
|
let snapshot = manager.operations_snapshot().await;
|
||||||
|
assert_eq!(snapshot.queued_by_source.mrf, 2, "replayed intents must be attributed to the MRF source");
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
disk_paths
|
||||||
|
.iter()
|
||||||
|
.all(|path| !Path::new(path).join(META_BUCKET).join(JOURNAL_REL).exists()),
|
||||||
|
"the journal file must be removed after a successful replay"
|
||||||
|
);
|
||||||
|
|
||||||
|
let snapshot = manager.operations_snapshot().await;
|
||||||
|
assert_eq!(snapshot.queued_by_priority.urgent, 1, "the decode-failure record must replay as Urgent");
|
||||||
|
assert!(snapshot.queued_by_priority.normal >= 1, "the partial-write record must replay as Normal");
|
||||||
|
}
|
||||||
@@ -26,7 +26,6 @@ pub enum StorageMedia {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl StorageMedia {
|
impl StorageMedia {
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn as_str(&self) -> &'static str {
|
pub fn as_str(&self) -> &'static str {
|
||||||
match self {
|
match self {
|
||||||
Self::Nvme => "nvme",
|
Self::Nvme => "nvme",
|
||||||
@@ -60,7 +59,6 @@ pub enum AccessPattern {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl AccessPattern {
|
impl AccessPattern {
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn as_str(&self) -> &'static str {
|
pub fn as_str(&self) -> &'static str {
|
||||||
match self {
|
match self {
|
||||||
Self::Sequential => "sequential",
|
Self::Sequential => "sequential",
|
||||||
@@ -71,25 +69,21 @@ impl AccessPattern {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Check if this is a sequential access pattern.
|
/// Check if this is a sequential access pattern.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn is_sequential(&self) -> bool {
|
pub fn is_sequential(&self) -> bool {
|
||||||
matches!(self, Self::Sequential)
|
matches!(self, Self::Sequential)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Check if this is a random access pattern.
|
/// Check if this is a random access pattern.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn is_random(&self) -> bool {
|
pub fn is_random(&self) -> bool {
|
||||||
matches!(self, Self::Random)
|
matches!(self, Self::Random)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Check if this is a mixed access pattern.
|
/// Check if this is a mixed access pattern.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn is_mixed(&self) -> bool {
|
pub fn is_mixed(&self) -> bool {
|
||||||
matches!(self, Self::Mixed)
|
matches!(self, Self::Mixed)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Check if this pattern is unknown.
|
/// Check if this pattern is unknown.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn is_unknown(&self) -> bool {
|
pub fn is_unknown(&self) -> bool {
|
||||||
matches!(self, Self::Unknown)
|
matches!(self, Self::Unknown)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -31,7 +31,10 @@ pub struct KeystoneClient {
|
|||||||
admin_password: Option<String>,
|
admin_password: Option<String>,
|
||||||
admin_project: Option<String>,
|
admin_project: Option<String>,
|
||||||
admin_domain: String,
|
admin_domain: String,
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "TLS verification flag parsed from config; the reqwest client is built before it is consulted, so nothing reads it back (backlog#1823)"
|
||||||
|
)]
|
||||||
verify_ssl: bool,
|
verify_ssl: bool,
|
||||||
/// Request timeout applied to the underlying HTTP client.
|
/// Request timeout applied to the underlying HTTP client.
|
||||||
timeout: std::time::Duration,
|
timeout: std::time::Duration,
|
||||||
|
|||||||
@@ -20,7 +20,10 @@ use tracing::{debug, info};
|
|||||||
|
|
||||||
/// Maps Keystone identities to RustFS concepts
|
/// Maps Keystone identities to RustFS concepts
|
||||||
pub struct KeystoneIdentityMapper {
|
pub struct KeystoneIdentityMapper {
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "keeps the Keystone client alive for the mapper's lifetime; the mapping paths do not call through it yet (backlog#1823)"
|
||||||
|
)]
|
||||||
client: Arc<KeystoneClient>,
|
client: Arc<KeystoneClient>,
|
||||||
role_policy_map: HashMap<String, String>,
|
role_policy_map: HashMap<String, String>,
|
||||||
enable_tenant_prefix: bool,
|
enable_tenant_prefix: bool,
|
||||||
|
|||||||
@@ -293,6 +293,15 @@ enum StrictVaultAuthMethod {
|
|||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
refresh_safety_window_secs: Option<u64>,
|
refresh_safety_window_secs: Option<u64>,
|
||||||
},
|
},
|
||||||
|
Kubernetes {
|
||||||
|
role: String,
|
||||||
|
#[serde(default)]
|
||||||
|
mount: Option<String>,
|
||||||
|
#[serde(default)]
|
||||||
|
jwt_path: Option<std::path::PathBuf>,
|
||||||
|
#[serde(default)]
|
||||||
|
refresh_safety_window_secs: Option<u64>,
|
||||||
|
},
|
||||||
TokenFile {
|
TokenFile {
|
||||||
path: std::path::PathBuf,
|
path: std::path::PathBuf,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
@@ -319,6 +328,17 @@ impl From<StrictVaultAuthMethod> for VaultAuthMethod {
|
|||||||
mount: mount.unwrap_or_else(|| crate::config::DEFAULT_VAULT_APPROLE_MOUNT.to_string()),
|
mount: mount.unwrap_or_else(|| crate::config::DEFAULT_VAULT_APPROLE_MOUNT.to_string()),
|
||||||
refresh_safety_window_secs,
|
refresh_safety_window_secs,
|
||||||
},
|
},
|
||||||
|
StrictVaultAuthMethod::Kubernetes {
|
||||||
|
role,
|
||||||
|
mount,
|
||||||
|
jwt_path,
|
||||||
|
refresh_safety_window_secs,
|
||||||
|
} => Self::Kubernetes {
|
||||||
|
role,
|
||||||
|
mount: mount.unwrap_or_else(|| crate::config::DEFAULT_VAULT_KUBERNETES_MOUNT.to_string()),
|
||||||
|
jwt_path: jwt_path.unwrap_or_else(|| std::path::PathBuf::from(crate::config::DEFAULT_VAULT_KUBERNETES_JWT_PATH)),
|
||||||
|
refresh_safety_window_secs,
|
||||||
|
},
|
||||||
StrictVaultAuthMethod::TokenFile {
|
StrictVaultAuthMethod::TokenFile {
|
||||||
path,
|
path,
|
||||||
poll_interval_secs,
|
poll_interval_secs,
|
||||||
@@ -499,6 +519,7 @@ impl From<&KmsConfig> for KmsConfigSummary {
|
|||||||
auth_method_type: match &vault_config.auth_method {
|
auth_method_type: match &vault_config.auth_method {
|
||||||
VaultAuthMethod::Token { .. } => "token".to_string(),
|
VaultAuthMethod::Token { .. } => "token".to_string(),
|
||||||
VaultAuthMethod::AppRole { .. } => "approle".to_string(),
|
VaultAuthMethod::AppRole { .. } => "approle".to_string(),
|
||||||
|
VaultAuthMethod::Kubernetes { .. } => "kubernetes".to_string(),
|
||||||
VaultAuthMethod::TokenFile { .. } => "token_file".to_string(),
|
VaultAuthMethod::TokenFile { .. } => "token_file".to_string(),
|
||||||
},
|
},
|
||||||
has_stored_credentials: true,
|
has_stored_credentials: true,
|
||||||
@@ -513,6 +534,7 @@ impl From<&KmsConfig> for KmsConfigSummary {
|
|||||||
auth_method_type: match &vault_config.auth_method {
|
auth_method_type: match &vault_config.auth_method {
|
||||||
VaultAuthMethod::Token { .. } => "token".to_string(),
|
VaultAuthMethod::Token { .. } => "token".to_string(),
|
||||||
VaultAuthMethod::AppRole { .. } => "approle".to_string(),
|
VaultAuthMethod::AppRole { .. } => "approle".to_string(),
|
||||||
|
VaultAuthMethod::Kubernetes { .. } => "kubernetes".to_string(),
|
||||||
VaultAuthMethod::TokenFile { .. } => "token_file".to_string(),
|
VaultAuthMethod::TokenFile { .. } => "token_file".to_string(),
|
||||||
},
|
},
|
||||||
has_stored_credentials: true,
|
has_stored_credentials: true,
|
||||||
@@ -901,6 +923,42 @@ mod tests {
|
|||||||
assert!(request.to_kms_config().validate().is_ok());
|
assert!(request.to_kms_config().validate().is_ok());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The admin API reaches Kubernetes auth with the role alone; the mount and
|
||||||
|
/// the projected token path fall back to the cluster defaults, so a Tenant
|
||||||
|
/// manifest carries no credential and no cluster-specific paths.
|
||||||
|
#[test]
|
||||||
|
fn test_deserialize_vault_configure_request_accepts_kubernetes_auth() {
|
||||||
|
let raw = serde_json::json!({
|
||||||
|
"backend_type": "vault-transit",
|
||||||
|
"address": "https://vault.example.com:8200",
|
||||||
|
"mount_path": "rustfs",
|
||||||
|
"auth_method": { "Kubernetes": { "role": "rustfs" } }
|
||||||
|
});
|
||||||
|
|
||||||
|
let request: ConfigureKmsRequest = serde_json::from_value(raw).expect("kubernetes auth should deserialize");
|
||||||
|
let config = request.to_kms_config();
|
||||||
|
config.validate().expect("kubernetes auth must validate");
|
||||||
|
|
||||||
|
let vault = config.vault_transit_config().expect("vault transit backend config");
|
||||||
|
let VaultAuthMethod::Kubernetes {
|
||||||
|
role, mount, jwt_path, ..
|
||||||
|
} = &vault.auth_method
|
||||||
|
else {
|
||||||
|
panic!("expected Kubernetes auth, got {:?}", vault.auth_method);
|
||||||
|
};
|
||||||
|
assert_eq!(role, "rustfs");
|
||||||
|
assert_eq!(mount, crate::config::DEFAULT_VAULT_KUBERNETES_MOUNT);
|
||||||
|
assert_eq!(jwt_path, std::path::Path::new(crate::config::DEFAULT_VAULT_KUBERNETES_JWT_PATH));
|
||||||
|
|
||||||
|
let unknown_field = serde_json::json!({
|
||||||
|
"backend_type": "vault-transit",
|
||||||
|
"address": "https://vault.example.com:8200",
|
||||||
|
"auth_method": { "Kubernetes": { "role": "rustfs", "service_account": "rustfs" } }
|
||||||
|
});
|
||||||
|
serde_json::from_value::<ConfigureKmsRequest>(unknown_field)
|
||||||
|
.expect_err("an unknown auth field must be rejected rather than silently dropped");
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_deserialize_aws_configure_request_accepts_type_aliases() {
|
fn test_deserialize_aws_configure_request_accepts_type_aliases() {
|
||||||
for backend_type in ["AWS", "AwsKms", "aws", "aws-kms", "aws_kms"] {
|
for backend_type in ["AWS", "AwsKms", "aws", "aws-kms", "aws_kms"] {
|
||||||
|
|||||||
@@ -550,6 +550,7 @@ impl VaultKmsClient {
|
|||||||
address: config.address.clone(),
|
address: config.address.clone(),
|
||||||
namespace: config.namespace.clone(),
|
namespace: config.namespace.clone(),
|
||||||
attempt_timeout: kms_config.effective_timeout(),
|
attempt_timeout: kms_config.effective_timeout(),
|
||||||
|
skip_tls_verify: config.tls.as_ref().is_some_and(|tls| tls.skip_verify),
|
||||||
};
|
};
|
||||||
let source = token_source_for(&config.auth_method, &settings)?;
|
let source = token_source_for(&config.auth_method, &settings)?;
|
||||||
let policy = VaultCredentialPolicy::from_kms_config(
|
let policy = VaultCredentialPolicy::from_kms_config(
|
||||||
|
|||||||
@@ -326,6 +326,97 @@ impl fmt::Debug for AppRoleLogin {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Token source for [`VaultAuthMethod::Kubernetes`]: exchanges the pod's
|
||||||
|
/// projected ServiceAccount token for a lease-bound Vault token.
|
||||||
|
///
|
||||||
|
/// The JWT is re-read on every login because the kubelet rotates a projected
|
||||||
|
/// token well inside the pod's lifetime; caching it would strand the source on
|
||||||
|
/// an expired assertion once the current Vault token can no longer be renewed.
|
||||||
|
///
|
||||||
|
/// Unlike [`TokenFileSource`], the file mode is not checked: the kubelet owns
|
||||||
|
/// the projected token and mounts it world-readable by default, so rejecting
|
||||||
|
/// group/other bits would refuse every standard pod rather than catch a
|
||||||
|
/// deployment error.
|
||||||
|
pub(crate) struct KubernetesLogin {
|
||||||
|
/// Unauthenticated client used only for the login exchange.
|
||||||
|
login_client: VaultClient,
|
||||||
|
mount: String,
|
||||||
|
role: String,
|
||||||
|
jwt_path: PathBuf,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl KubernetesLogin {
|
||||||
|
pub(crate) fn new(settings: &VaultConnectionSettings, mount: String, role: String, jwt_path: PathBuf) -> Result<Self> {
|
||||||
|
Ok(Self {
|
||||||
|
login_client: settings.build_login_client()?,
|
||||||
|
mount,
|
||||||
|
role,
|
||||||
|
jwt_path,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read the ServiceAccount token for one login attempt.
|
||||||
|
///
|
||||||
|
/// Mirrors [`AppRoleLogin::resolve_secret_id`]: a read failure is fatal for
|
||||||
|
/// the attempt but the refresh loop keeps retrying, so a token the kubelet
|
||||||
|
/// has not projected yet heals the source without a restart.
|
||||||
|
async fn resolve_jwt(&self) -> AttemptResult<SecretString> {
|
||||||
|
let mut raw = tokio::fs::read_to_string(&self.jwt_path)
|
||||||
|
.await
|
||||||
|
.map_err(|error| AttemptError {
|
||||||
|
class: ErrorClass::Fatal,
|
||||||
|
error: KmsError::configuration_error(format!(
|
||||||
|
"Failed to read Kubernetes ServiceAccount token {}: {error}",
|
||||||
|
self.jwt_path.display()
|
||||||
|
)),
|
||||||
|
})?;
|
||||||
|
let trimmed = raw.trim();
|
||||||
|
if trimmed.is_empty() {
|
||||||
|
raw.zeroize();
|
||||||
|
return Err(AttemptError {
|
||||||
|
class: ErrorClass::Fatal,
|
||||||
|
error: KmsError::configuration_error(format!(
|
||||||
|
"Kubernetes ServiceAccount token {} is empty",
|
||||||
|
self.jwt_path.display()
|
||||||
|
)),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
let jwt = SecretString::new(trimmed.to_string());
|
||||||
|
raw.zeroize();
|
||||||
|
Ok(jwt)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[async_trait]
|
||||||
|
impl TokenSource for KubernetesLogin {
|
||||||
|
async fn acquire(&self) -> AttemptResult<TokenLease> {
|
||||||
|
let jwt = self.resolve_jwt().await?;
|
||||||
|
let auth = vaultrs::auth::kubernetes::login(&self.login_client, &self.mount, &self.role, jwt.expose())
|
||||||
|
.await
|
||||||
|
.map_err(|error| attempt_error("Kubernetes login", error))?;
|
||||||
|
Ok(TokenLease::from_auth(auth))
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn renew(&self, client: &VaultClient) -> AttemptResult<TokenLease> {
|
||||||
|
let auth = vaultrs::token::renew_self(client, None)
|
||||||
|
.await
|
||||||
|
.map_err(|error| attempt_error("token renewal", error))?;
|
||||||
|
Ok(TokenLease::from_auth(auth))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl fmt::Debug for KubernetesLogin {
|
||||||
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||||
|
// The login client embeds Vault client settings and must stay out of
|
||||||
|
// Debug output; the role name is not a secret, and the JWT is never held.
|
||||||
|
f.debug_struct("KubernetesLogin")
|
||||||
|
.field("mount", &self.mount)
|
||||||
|
.field("role", &self.role)
|
||||||
|
.field("jwt_path", &self.jwt_path)
|
||||||
|
.finish_non_exhaustive()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Token source for [`VaultAuthMethod::TokenFile`]: reads an agent-managed
|
/// Token source for [`VaultAuthMethod::TokenFile`]: reads an agent-managed
|
||||||
/// token file (for example a Vault Agent auto-auth sink).
|
/// token file (for example a Vault Agent auto-auth sink).
|
||||||
///
|
///
|
||||||
@@ -464,6 +555,9 @@ pub(crate) fn token_source_for(
|
|||||||
secret_id.clone(),
|
secret_id.clone(),
|
||||||
secret_id_file.clone(),
|
secret_id_file.clone(),
|
||||||
)?)),
|
)?)),
|
||||||
|
VaultAuthMethod::Kubernetes {
|
||||||
|
role, mount, jwt_path, ..
|
||||||
|
} => Ok(Box::new(KubernetesLogin::new(settings, mount.clone(), role.clone(), jwt_path.clone())?)),
|
||||||
VaultAuthMethod::TokenFile {
|
VaultAuthMethod::TokenFile {
|
||||||
path,
|
path,
|
||||||
poll_interval_secs,
|
poll_interval_secs,
|
||||||
@@ -486,6 +580,9 @@ pub(crate) struct VaultConnectionSettings {
|
|||||||
pub(crate) namespace: Option<String>,
|
pub(crate) namespace: Option<String>,
|
||||||
/// Per-attempt HTTP timeout applied to the underlying reqwest client.
|
/// Per-attempt HTTP timeout applied to the underlying reqwest client.
|
||||||
pub(crate) attempt_timeout: Duration,
|
pub(crate) attempt_timeout: Duration,
|
||||||
|
/// Whether to accept an unverified Vault server certificate. Gated on
|
||||||
|
/// `allow_insecure_dev_defaults` by `KmsConfig::validate`.
|
||||||
|
pub(crate) skip_tls_verify: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl VaultConnectionSettings {
|
impl VaultConnectionSettings {
|
||||||
@@ -499,6 +596,11 @@ impl VaultConnectionSettings {
|
|||||||
// operation-level retry policy.
|
// operation-level retry policy.
|
||||||
settings_builder.timeout(Some(self.attempt_timeout));
|
settings_builder.timeout(Some(self.attempt_timeout));
|
||||||
settings_builder.token(token);
|
settings_builder.token(token);
|
||||||
|
// Always set explicitly: left unset, vaultrs derives this from its own
|
||||||
|
// VAULT_SKIP_VERIFY variable, so a stray value in the environment would
|
||||||
|
// disable certificate verification behind the KMS configuration and its
|
||||||
|
// insecure-defaults gate.
|
||||||
|
settings_builder.verify(!self.skip_tls_verify);
|
||||||
|
|
||||||
if let Some(namespace) = &self.namespace {
|
if let Some(namespace) = &self.namespace {
|
||||||
settings_builder.namespace(Some(namespace.clone()));
|
settings_builder.namespace(Some(namespace.clone()));
|
||||||
@@ -551,6 +653,10 @@ impl VaultCredentialPolicy {
|
|||||||
refresh_safety_window_secs: Some(secs),
|
refresh_safety_window_secs: Some(secs),
|
||||||
..
|
..
|
||||||
}
|
}
|
||||||
|
| VaultAuthMethod::Kubernetes {
|
||||||
|
refresh_safety_window_secs: Some(secs),
|
||||||
|
..
|
||||||
|
}
|
||||||
| VaultAuthMethod::TokenFile {
|
| VaultAuthMethod::TokenFile {
|
||||||
refresh_safety_window_secs: Some(secs),
|
refresh_safety_window_secs: Some(secs),
|
||||||
..
|
..
|
||||||
@@ -584,15 +690,25 @@ pub(crate) struct VaultClientHandle {
|
|||||||
|
|
||||||
impl VaultClientHandle {
|
impl VaultClientHandle {
|
||||||
/// Absolute expiry of this generation's token.
|
/// Absolute expiry of this generation's token.
|
||||||
|
///
|
||||||
|
/// `lease.ttl` is built from the `lease_duration` the Vault server sent, so
|
||||||
|
/// a value too large to add to `issued_at` would panic on the bare `+`. A
|
||||||
|
/// TTL that cannot be represented is indistinguishable from no expiry, so it
|
||||||
|
/// collapses to `None` — the same answer already given for the zero-lease
|
||||||
|
/// tokens Vault issues, which keeps the token in use and still fully
|
||||||
|
/// validated by Vault on every call.
|
||||||
fn expires_at(&self) -> Option<Instant> {
|
fn expires_at(&self) -> Option<Instant> {
|
||||||
self.lease.map(|lease| self.issued_at + lease.ttl)
|
self.lease.and_then(|lease| self.issued_at.checked_add(lease.ttl))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// When the renewal task should refresh this generation: half the TTL,
|
/// When the renewal task should refresh this generation: half the TTL,
|
||||||
/// leaving the second half as budget for retries before the fail-closed
|
/// leaving the second half as budget for retries before the fail-closed
|
||||||
/// window is reached.
|
/// window is reached.
|
||||||
|
///
|
||||||
|
/// Unrepresentable TTLs collapse to `None` as in [`Self::expires_at`],
|
||||||
|
/// leaving a token that never expires with nothing to renew.
|
||||||
fn renew_at(&self) -> Option<Instant> {
|
fn renew_at(&self) -> Option<Instant> {
|
||||||
self.lease.map(|lease| self.issued_at + lease.ttl / 2)
|
self.lease.and_then(|lease| self.issued_at.checked_add(lease.ttl / 2))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -662,7 +778,7 @@ impl VaultCredentialProvider {
|
|||||||
let handle = self.current.load_full();
|
let handle = self.current.load_full();
|
||||||
if let Some(expires_at) = handle.expires_at() {
|
if let Some(expires_at) = handle.expires_at() {
|
||||||
let now = Instant::now();
|
let now = Instant::now();
|
||||||
if now + self.policy.safety_window >= expires_at {
|
if self.inside_safety_window(now, expires_at) {
|
||||||
return Err(KmsError::credentials_unavailable(format!(
|
return Err(KmsError::credentials_unavailable(format!(
|
||||||
"Vault token (generation {}) is within {:?} of expiry and has not been refreshed; refusing to use it",
|
"Vault token (generation {}) is within {:?} of expiry and has not been refreshed; refusing to use it",
|
||||||
handle.generation, self.policy.safety_window
|
handle.generation, self.policy.safety_window
|
||||||
@@ -672,6 +788,18 @@ impl VaultCredentialProvider {
|
|||||||
Ok(handle)
|
Ok(handle)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Whether the token expiring at `expires_at` is close enough to refuse.
|
||||||
|
///
|
||||||
|
/// `safety_window` reaches here from persisted configuration, so it is not
|
||||||
|
/// guaranteed to have passed this version's validation: a window too large
|
||||||
|
/// to add to the current instant would panic on the bare `+`. Such a window
|
||||||
|
/// means every token is always inside it, so saturating to "refuse" is both
|
||||||
|
/// the fail-closed answer and the one the arithmetic was reaching for.
|
||||||
|
fn inside_safety_window(&self, now: Instant, expires_at: Instant) -> bool {
|
||||||
|
now.checked_add(self.policy.safety_window)
|
||||||
|
.is_none_or(|deadline| deadline >= expires_at)
|
||||||
|
}
|
||||||
|
|
||||||
/// Publish the credential gauges for the generation currently installed.
|
/// Publish the credential gauges for the generation currently installed.
|
||||||
///
|
///
|
||||||
/// The fail-closed gauge re-evaluates the very gate
|
/// The fail-closed gauge re-evaluates the very gate
|
||||||
@@ -683,7 +811,7 @@ impl VaultCredentialProvider {
|
|||||||
let fail_closed = match handle.expires_at() {
|
let fail_closed = match handle.expires_at() {
|
||||||
Some(expires_at) => {
|
Some(expires_at) => {
|
||||||
metrics::gauge!(METRIC_TOKEN_TTL_SECONDS).set(expires_at.saturating_duration_since(now).as_secs_f64());
|
metrics::gauge!(METRIC_TOKEN_TTL_SECONDS).set(expires_at.saturating_duration_since(now).as_secs_f64());
|
||||||
now + self.policy.safety_window >= expires_at
|
self.inside_safety_window(now, expires_at)
|
||||||
}
|
}
|
||||||
// A generation without an expiry has no remaining TTL to report
|
// A generation without an expiry has no remaining TTL to report
|
||||||
// and can never lapse, so it can never fail closed either.
|
// and can never lapse, so it can never fail closed either.
|
||||||
@@ -860,7 +988,7 @@ impl Drop for CredentialTaskHandle {
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
use crate::config::REDACTED_SECRET;
|
use crate::config::{DEFAULT_VAULT_KUBERNETES_MOUNT, REDACTED_SECRET};
|
||||||
use std::sync::atomic::{AtomicBool, AtomicU32, Ordering};
|
use std::sync::atomic::{AtomicBool, AtomicU32, Ordering};
|
||||||
|
|
||||||
const TEST_TOKEN: &str = "vault-token-debug-leak-canary";
|
const TEST_TOKEN: &str = "vault-token-debug-leak-canary";
|
||||||
@@ -871,6 +999,7 @@ mod tests {
|
|||||||
address: "http://127.0.0.1:8200".to_string(),
|
address: "http://127.0.0.1:8200".to_string(),
|
||||||
namespace: Some("team-namespace".to_string()),
|
namespace: Some("team-namespace".to_string()),
|
||||||
attempt_timeout: Duration::from_secs(30),
|
attempt_timeout: Duration::from_secs(30),
|
||||||
|
skip_tls_verify: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1057,6 +1186,143 @@ mod tests {
|
|||||||
assert!(format!("{source:?}").contains("AppRoleLogin"));
|
assert!(format!("{source:?}").contains("AppRoleLogin"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_kubernetes_auth_method_maps_to_login_source() {
|
||||||
|
let settings = test_settings();
|
||||||
|
let source = token_source_for(&VaultAuthMethod::kubernetes("rustfs".to_string()), &settings)
|
||||||
|
.expect("kubernetes auth must map to a login source");
|
||||||
|
|
||||||
|
assert!(format!("{source:?}").contains("KubernetesLogin"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `refresh_safety_window_secs` is operator-supplied and reaches the request
|
||||||
|
/// path from persisted configuration, so the fail-closed comparison must
|
||||||
|
/// survive a window too large to add to the current instant. Before the
|
||||||
|
/// checked arithmetic this panicked with "overflow when adding duration to
|
||||||
|
/// instant" on the first request after a lease-bearing login.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_current_refuses_rather_than_panics_on_an_unrepresentable_safety_window() {
|
||||||
|
let (provider, _state) = scripted_provider(
|
||||||
|
Duration::from_secs(60),
|
||||||
|
true,
|
||||||
|
test_policy(Duration::from_secs(u64::MAX), Duration::from_secs(5)),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let error = provider
|
||||||
|
.current()
|
||||||
|
.expect_err("a window wider than any lease must refuse the token");
|
||||||
|
assert!(
|
||||||
|
matches!(error, KmsError::CredentialsUnavailable { .. }),
|
||||||
|
"expected CredentialsUnavailable, got {error:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `lease_duration` is a bare u64 straight off the Vault response and forms
|
||||||
|
/// the other side of the same comparison, so an absurd one must not panic
|
||||||
|
/// either. It is indistinguishable from a non-expiring token, which is how
|
||||||
|
/// the zero-lease case already behaves.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_an_unrepresentable_lease_is_treated_as_non_expiring() {
|
||||||
|
let (provider, _state) = scripted_provider(
|
||||||
|
Duration::from_secs(u64::MAX),
|
||||||
|
true,
|
||||||
|
test_policy(Duration::from_secs(30), Duration::from_secs(5)),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
provider
|
||||||
|
.current()
|
||||||
|
.expect("a token whose expiry cannot be represented must stay usable");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The configured flag has to reach the HTTP client, not just the config
|
||||||
|
/// struct: every generation (authenticated and login) builds its own client,
|
||||||
|
/// and a Vault with a self-signed certificate fails the handshake unless
|
||||||
|
/// each one carries the setting.
|
||||||
|
#[test]
|
||||||
|
fn test_skip_tls_verify_reaches_every_vault_client_generation() {
|
||||||
|
for skip_tls_verify in [false, true] {
|
||||||
|
let settings = VaultConnectionSettings {
|
||||||
|
address: "https://vault.example.com:8200".to_string(),
|
||||||
|
namespace: None,
|
||||||
|
attempt_timeout: Duration::from_secs(30),
|
||||||
|
skip_tls_verify,
|
||||||
|
};
|
||||||
|
|
||||||
|
let authenticated = settings.build_client(TEST_TOKEN).expect("authenticated client must build");
|
||||||
|
assert_eq!(authenticated.settings.verify, !skip_tls_verify);
|
||||||
|
|
||||||
|
let login = settings.build_login_client().expect("login client must build");
|
||||||
|
assert_eq!(login.settings.verify, !skip_tls_verify);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// vaultrs derives `verify` from its own VAULT_SKIP_VERIFY variable when the
|
||||||
|
/// builder leaves it unset, which would disable certificate verification
|
||||||
|
/// without passing the KMS insecure-defaults gate.
|
||||||
|
#[test]
|
||||||
|
fn test_vaultrs_skip_verify_env_cannot_override_the_configured_setting() {
|
||||||
|
temp_env::with_var("VAULT_SKIP_VERIFY", Some("true"), || {
|
||||||
|
let client = test_settings().build_client(TEST_TOKEN).expect("client must build");
|
||||||
|
assert!(
|
||||||
|
client.settings.verify,
|
||||||
|
"a stray VAULT_SKIP_VERIFY must not disable verification behind the KMS configuration"
|
||||||
|
);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The projected token is read fresh per login attempt and trimmed, so a
|
||||||
|
/// kubelet rotation is picked up without a restart and a trailing newline
|
||||||
|
/// does not corrupt the assertion sent to Vault.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_kubernetes_login_rereads_and_trims_the_service_account_token() {
|
||||||
|
let dir = tempfile::tempdir().expect("temp dir");
|
||||||
|
let path = dir.path().join("token");
|
||||||
|
tokio::fs::write(&path, " first-jwt\n").await.expect("write token");
|
||||||
|
|
||||||
|
let login = KubernetesLogin::new(
|
||||||
|
&test_settings(),
|
||||||
|
DEFAULT_VAULT_KUBERNETES_MOUNT.to_string(),
|
||||||
|
"rustfs".to_string(),
|
||||||
|
path.clone(),
|
||||||
|
)
|
||||||
|
.expect("login source must build");
|
||||||
|
|
||||||
|
assert_eq!(login.resolve_jwt().await.expect("first read").expose(), "first-jwt");
|
||||||
|
|
||||||
|
tokio::fs::write(&path, "rotated-jwt").await.expect("rotate token");
|
||||||
|
assert_eq!(
|
||||||
|
login.resolve_jwt().await.expect("second read").expose(),
|
||||||
|
"rotated-jwt",
|
||||||
|
"a rotated projected token must be picked up without a restart"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The ServiceAccount token is re-read per attempt, so an unreadable or
|
||||||
|
/// empty one fails that attempt without reaching Vault; the refresh loop
|
||||||
|
/// keeps retrying, which is what lets a late projection heal the source.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_kubernetes_login_rejects_an_unusable_service_account_token() {
|
||||||
|
let dir = tempfile::tempdir().expect("temp dir");
|
||||||
|
let missing = dir.path().join("absent-token");
|
||||||
|
let empty = dir.path().join("empty-token");
|
||||||
|
tokio::fs::write(&empty, " \n").await.expect("write empty token");
|
||||||
|
|
||||||
|
for (path, expected) in [(missing, "Failed to read"), (empty, "is empty")] {
|
||||||
|
let login =
|
||||||
|
KubernetesLogin::new(&test_settings(), DEFAULT_VAULT_KUBERNETES_MOUNT.to_string(), "rustfs".to_string(), path)
|
||||||
|
.expect("login source must build");
|
||||||
|
|
||||||
|
let error = login
|
||||||
|
.acquire()
|
||||||
|
.await
|
||||||
|
.expect_err("an unusable ServiceAccount token must fail the attempt");
|
||||||
|
assert!(matches!(error.class, ErrorClass::Fatal));
|
||||||
|
assert!(error.error.to_string().contains(expected), "got {}", error.error);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test(start_paused = true)]
|
#[tokio::test(start_paused = true)]
|
||||||
async fn test_renewal_task_renews_at_half_ttl() {
|
async fn test_renewal_task_renews_at_half_ttl() {
|
||||||
let (provider, state) = scripted_provider(
|
let (provider, state) = scripted_provider(
|
||||||
|
|||||||
@@ -415,6 +415,7 @@ impl VaultTransitKmsClient {
|
|||||||
address: config.address.clone(),
|
address: config.address.clone(),
|
||||||
namespace: config.namespace.clone(),
|
namespace: config.namespace.clone(),
|
||||||
attempt_timeout: kms_config.effective_timeout(),
|
attempt_timeout: kms_config.effective_timeout(),
|
||||||
|
skip_tls_verify: config.tls.as_ref().is_some_and(|tls| tls.skip_verify),
|
||||||
};
|
};
|
||||||
let source = token_source_for(&config.auth_method, &settings)?;
|
let source = token_source_for(&config.auth_method, &settings)?;
|
||||||
let policy = VaultCredentialPolicy::from_kms_config(
|
let policy = VaultCredentialPolicy::from_kms_config(
|
||||||
|
|||||||
@@ -450,6 +450,10 @@ impl VaultRestoreClient {
|
|||||||
address: target.address.clone(),
|
address: target.address.clone(),
|
||||||
namespace: target.namespace.clone(),
|
namespace: target.namespace.clone(),
|
||||||
attempt_timeout: kms_config.effective_timeout(),
|
attempt_timeout: kms_config.effective_timeout(),
|
||||||
|
// A restore target carries no TLS settings, so certificates are
|
||||||
|
// always verified: recovery is the last path that should accept an
|
||||||
|
// unauthenticated Vault.
|
||||||
|
skip_tls_verify: false,
|
||||||
};
|
};
|
||||||
let source = token_source_for(&target.auth_method, &settings)?;
|
let source = token_source_for(&target.auth_method, &settings)?;
|
||||||
let policy = VaultCredentialPolicy::from_kms_config(
|
let policy = VaultCredentialPolicy::from_kms_config(
|
||||||
|
|||||||
+295
-54
@@ -25,6 +25,10 @@ use url::Url;
|
|||||||
|
|
||||||
pub const ENV_KMS_ALLOW_INSECURE_DEV_DEFAULTS: &str = "RUSTFS_KMS_ALLOW_INSECURE_DEV_DEFAULTS";
|
pub const ENV_KMS_ALLOW_INSECURE_DEV_DEFAULTS: &str = "RUSTFS_KMS_ALLOW_INSECURE_DEV_DEFAULTS";
|
||||||
pub const ENV_KMS_ALLOW_IMMEDIATE_DELETION: &str = "RUSTFS_KMS_ALLOW_IMMEDIATE_DELETION";
|
pub const ENV_KMS_ALLOW_IMMEDIATE_DELETION: &str = "RUSTFS_KMS_ALLOW_IMMEDIATE_DELETION";
|
||||||
|
pub const ENV_KMS_VAULT_ADDRESS: &str = "RUSTFS_KMS_VAULT_ADDRESS";
|
||||||
|
pub const ENV_KMS_VAULT_TOKEN: &str = "RUSTFS_KMS_VAULT_TOKEN";
|
||||||
|
pub const ENV_KMS_VAULT_NAMESPACE: &str = "RUSTFS_KMS_VAULT_NAMESPACE";
|
||||||
|
pub const ENV_KMS_VAULT_MOUNT_PATH: &str = "RUSTFS_KMS_VAULT_MOUNT_PATH";
|
||||||
pub const ENV_KMS_VAULT_SKIP_TLS_VERIFY: &str = "RUSTFS_KMS_VAULT_SKIP_TLS_VERIFY";
|
pub const ENV_KMS_VAULT_SKIP_TLS_VERIFY: &str = "RUSTFS_KMS_VAULT_SKIP_TLS_VERIFY";
|
||||||
pub const ENV_KMS_VAULT_TRANSIT_METADATA_KV_MOUNT: &str = "RUSTFS_KMS_VAULT_TRANSIT_METADATA_KV_MOUNT";
|
pub const ENV_KMS_VAULT_TRANSIT_METADATA_KV_MOUNT: &str = "RUSTFS_KMS_VAULT_TRANSIT_METADATA_KV_MOUNT";
|
||||||
pub const ENV_KMS_VAULT_TRANSIT_METADATA_PREFIX: &str = "RUSTFS_KMS_VAULT_TRANSIT_METADATA_PREFIX";
|
pub const ENV_KMS_VAULT_TRANSIT_METADATA_PREFIX: &str = "RUSTFS_KMS_VAULT_TRANSIT_METADATA_PREFIX";
|
||||||
@@ -35,6 +39,9 @@ pub const ENV_KMS_VAULT_APPROLE_SECRET_ID: &str = "RUSTFS_KMS_VAULT_APPROLE_SECR
|
|||||||
pub const ENV_KMS_VAULT_APPROLE_SECRET_ID_FILE: &str = "RUSTFS_KMS_VAULT_APPROLE_SECRET_ID_FILE";
|
pub const ENV_KMS_VAULT_APPROLE_SECRET_ID_FILE: &str = "RUSTFS_KMS_VAULT_APPROLE_SECRET_ID_FILE";
|
||||||
pub const ENV_KMS_VAULT_APPROLE_MOUNT: &str = "RUSTFS_KMS_VAULT_APPROLE_MOUNT";
|
pub const ENV_KMS_VAULT_APPROLE_MOUNT: &str = "RUSTFS_KMS_VAULT_APPROLE_MOUNT";
|
||||||
pub const ENV_KMS_VAULT_TOKEN_FILE: &str = "RUSTFS_KMS_VAULT_TOKEN_FILE";
|
pub const ENV_KMS_VAULT_TOKEN_FILE: &str = "RUSTFS_KMS_VAULT_TOKEN_FILE";
|
||||||
|
pub const ENV_KMS_VAULT_KUBERNETES_ROLE: &str = "RUSTFS_KMS_VAULT_KUBERNETES_ROLE";
|
||||||
|
pub const ENV_KMS_VAULT_KUBERNETES_MOUNT: &str = "RUSTFS_KMS_VAULT_KUBERNETES_MOUNT";
|
||||||
|
pub const ENV_KMS_VAULT_KUBERNETES_JWT_PATH: &str = "RUSTFS_KMS_VAULT_KUBERNETES_JWT_PATH";
|
||||||
pub const ENV_KMS_AWS_REGION: &str = "RUSTFS_KMS_AWS_REGION";
|
pub const ENV_KMS_AWS_REGION: &str = "RUSTFS_KMS_AWS_REGION";
|
||||||
pub const ENV_KMS_AWS_ENDPOINT_URL: &str = "RUSTFS_KMS_AWS_ENDPOINT_URL";
|
pub const ENV_KMS_AWS_ENDPOINT_URL: &str = "RUSTFS_KMS_AWS_ENDPOINT_URL";
|
||||||
/// Age in whole seconds beyond which a key is reported as due for rotation;
|
/// Age in whole seconds beyond which a key is reported as due for rotation;
|
||||||
@@ -45,6 +52,9 @@ pub const ENV_KMS_ROTATION_MAX_WRAPS: &str = "RUSTFS_KMS_ROTATION_MAX_WRAPS";
|
|||||||
pub const DEFAULT_VAULT_TRANSIT_METADATA_KV_MOUNT: &str = "secret";
|
pub const DEFAULT_VAULT_TRANSIT_METADATA_KV_MOUNT: &str = "secret";
|
||||||
pub const DEFAULT_VAULT_TRANSIT_METADATA_KEY_PREFIX: &str = "rustfs/kms/transit-metadata";
|
pub const DEFAULT_VAULT_TRANSIT_METADATA_KEY_PREFIX: &str = "rustfs/kms/transit-metadata";
|
||||||
pub const DEFAULT_VAULT_APPROLE_MOUNT: &str = "approle";
|
pub const DEFAULT_VAULT_APPROLE_MOUNT: &str = "approle";
|
||||||
|
pub const DEFAULT_VAULT_KUBERNETES_MOUNT: &str = "kubernetes";
|
||||||
|
/// Where the kubelet projects a pod's ServiceAccount token by default.
|
||||||
|
pub const DEFAULT_VAULT_KUBERNETES_JWT_PATH: &str = "/var/run/secrets/kubernetes.io/serviceaccount/token";
|
||||||
|
|
||||||
/// Upper bound applied to `KmsConfig::timeout` when deriving backend behavior.
|
/// Upper bound applied to `KmsConfig::timeout` when deriving backend behavior.
|
||||||
///
|
///
|
||||||
@@ -84,6 +94,14 @@ fn default_vault_approle_mount() -> String {
|
|||||||
DEFAULT_VAULT_APPROLE_MOUNT.to_string()
|
DEFAULT_VAULT_APPROLE_MOUNT.to_string()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn default_vault_kubernetes_mount() -> String {
|
||||||
|
DEFAULT_VAULT_KUBERNETES_MOUNT.to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn default_vault_kubernetes_jwt_path() -> PathBuf {
|
||||||
|
PathBuf::from(DEFAULT_VAULT_KUBERNETES_JWT_PATH)
|
||||||
|
}
|
||||||
|
|
||||||
pub const KMS_CONFIG_REDACTION_RULES: &[RedactionRule] = &[
|
pub const KMS_CONFIG_REDACTION_RULES: &[RedactionRule] = &[
|
||||||
RedactionRule::new("kms.local.master_key", RedactionLevel::Secret, "local backend key encryption material"),
|
RedactionRule::new("kms.local.master_key", RedactionLevel::Secret, "local backend key encryption material"),
|
||||||
RedactionRule::new("kms.vault.token", RedactionLevel::Secret, "vault authentication token"),
|
RedactionRule::new("kms.vault.token", RedactionLevel::Secret, "vault authentication token"),
|
||||||
@@ -490,6 +508,23 @@ pub enum VaultAuthMethod {
|
|||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
refresh_safety_window_secs: Option<u64>,
|
refresh_safety_window_secs: Option<u64>,
|
||||||
},
|
},
|
||||||
|
/// Kubernetes authentication: the pod's ServiceAccount token is exchanged
|
||||||
|
/// for a lease-bound Vault token that is renewed in the background.
|
||||||
|
Kubernetes {
|
||||||
|
/// Vault role bound to this ServiceAccount.
|
||||||
|
role: String,
|
||||||
|
/// Kubernetes auth engine mount path.
|
||||||
|
#[serde(default = "default_vault_kubernetes_mount")]
|
||||||
|
mount: String,
|
||||||
|
/// Projected ServiceAccount token to present. Re-read on every login so
|
||||||
|
/// a token the kubelet rotates is picked up without a restart.
|
||||||
|
#[serde(default = "default_vault_kubernetes_jwt_path")]
|
||||||
|
jwt_path: PathBuf,
|
||||||
|
/// Fail-closed margin in seconds, as on `AppRole`. Defaults to the
|
||||||
|
/// per-attempt timeout.
|
||||||
|
#[serde(default)]
|
||||||
|
refresh_safety_window_secs: Option<u64>,
|
||||||
|
},
|
||||||
/// Agent-managed token file (for example a Vault Agent auto-auth sink):
|
/// Agent-managed token file (for example a Vault Agent auto-auth sink):
|
||||||
/// the token is read from `path` and re-read periodically so a token
|
/// the token is read from `path` and re-read periodically so a token
|
||||||
/// rotated by the agent is picked up without a restart.
|
/// rotated by the agent is picked up without a restart.
|
||||||
@@ -520,6 +555,16 @@ impl VaultAuthMethod {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Kubernetes authentication with the default mount and projected token path.
|
||||||
|
pub fn kubernetes(role: String) -> Self {
|
||||||
|
Self::Kubernetes {
|
||||||
|
role,
|
||||||
|
mount: default_vault_kubernetes_mount(),
|
||||||
|
jwt_path: default_vault_kubernetes_jwt_path(),
|
||||||
|
refresh_safety_window_secs: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Agent-managed token file with the default poll interval.
|
/// Agent-managed token file with the default poll interval.
|
||||||
pub fn token_file(path: PathBuf) -> Self {
|
pub fn token_file(path: PathBuf) -> Self {
|
||||||
Self::TokenFile {
|
Self::TokenFile {
|
||||||
@@ -548,6 +593,20 @@ impl fmt::Debug for VaultAuthMethod {
|
|||||||
.field("mount", mount)
|
.field("mount", mount)
|
||||||
.field("refresh_safety_window_secs", refresh_safety_window_secs)
|
.field("refresh_safety_window_secs", refresh_safety_window_secs)
|
||||||
.finish(),
|
.finish(),
|
||||||
|
// No redaction: the role and mount name a Vault binding, and the
|
||||||
|
// ServiceAccount token itself is never held on this type.
|
||||||
|
Self::Kubernetes {
|
||||||
|
role,
|
||||||
|
mount,
|
||||||
|
jwt_path,
|
||||||
|
refresh_safety_window_secs,
|
||||||
|
} => f
|
||||||
|
.debug_struct("Kubernetes")
|
||||||
|
.field("role", role)
|
||||||
|
.field("mount", mount)
|
||||||
|
.field("jwt_path", jwt_path)
|
||||||
|
.field("refresh_safety_window_secs", refresh_safety_window_secs)
|
||||||
|
.finish(),
|
||||||
Self::TokenFile {
|
Self::TokenFile {
|
||||||
path,
|
path,
|
||||||
poll_interval_secs,
|
poll_interval_secs,
|
||||||
@@ -1028,50 +1087,12 @@ impl KmsConfig {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
KmsBackend::VaultKv2 => {
|
KmsBackend::VaultKv2 => {
|
||||||
let address = get_env_str("RUSTFS_KMS_VAULT_ADDRESS", "http://localhost:8200");
|
config.backend_config =
|
||||||
let auth_method = vault_auth_method_from_env()?;
|
BackendConfig::VaultKv2(Box::new(vault_kv2_config_from_env(VaultCliOverrides::default())?));
|
||||||
let skip_tls_verify = get_env_bool(ENV_KMS_VAULT_SKIP_TLS_VERIFY, false);
|
|
||||||
|
|
||||||
let mount_path = match get_env_opt_str("RUSTFS_KMS_VAULT_MOUNT_PATH") {
|
|
||||||
Some(path) => {
|
|
||||||
tracing::warn!(
|
|
||||||
"RUSTFS_KMS_VAULT_MOUNT_PATH is deprecated for the Vault KV2 backend: it never calls the Transit engine and the value is stored but unused"
|
|
||||||
);
|
|
||||||
path
|
|
||||||
}
|
|
||||||
None => default_vault_kv2_mount_path(),
|
|
||||||
};
|
|
||||||
|
|
||||||
config.backend_config = BackendConfig::VaultKv2(Box::new(VaultConfig {
|
|
||||||
address,
|
|
||||||
auth_method,
|
|
||||||
namespace: get_env_opt_str("RUSTFS_KMS_VAULT_NAMESPACE"),
|
|
||||||
mount_path,
|
|
||||||
kv_mount: get_env_str("RUSTFS_KMS_VAULT_KV_MOUNT", "secret"),
|
|
||||||
key_path_prefix: get_env_str("RUSTFS_KMS_VAULT_KEY_PREFIX", "rustfs/kms/keys"),
|
|
||||||
tls: vault_tls_config(skip_tls_verify),
|
|
||||||
}));
|
|
||||||
}
|
}
|
||||||
KmsBackend::VaultTransit => {
|
KmsBackend::VaultTransit => {
|
||||||
let address = get_env_str("RUSTFS_KMS_VAULT_ADDRESS", "http://localhost:8200");
|
config.backend_config =
|
||||||
let auth_method = vault_auth_method_from_env()?;
|
BackendConfig::VaultTransit(Box::new(vault_transit_config_from_env(VaultCliOverrides::default())?));
|
||||||
let skip_tls_verify = get_env_bool(ENV_KMS_VAULT_SKIP_TLS_VERIFY, false);
|
|
||||||
|
|
||||||
config.backend_config = BackendConfig::VaultTransit(Box::new(VaultTransitConfig {
|
|
||||||
address,
|
|
||||||
auth_method,
|
|
||||||
namespace: get_env_opt_str("RUSTFS_KMS_VAULT_NAMESPACE"),
|
|
||||||
mount_path: get_env_str("RUSTFS_KMS_VAULT_MOUNT_PATH", "transit"),
|
|
||||||
metadata_kv_mount: get_env_str(
|
|
||||||
ENV_KMS_VAULT_TRANSIT_METADATA_KV_MOUNT,
|
|
||||||
DEFAULT_VAULT_TRANSIT_METADATA_KV_MOUNT,
|
|
||||||
),
|
|
||||||
metadata_key_prefix: get_env_str(
|
|
||||||
ENV_KMS_VAULT_TRANSIT_METADATA_PREFIX,
|
|
||||||
DEFAULT_VAULT_TRANSIT_METADATA_KEY_PREFIX,
|
|
||||||
),
|
|
||||||
tls: vault_tls_config(skip_tls_verify),
|
|
||||||
}));
|
|
||||||
}
|
}
|
||||||
KmsBackend::Static => {
|
KmsBackend::Static => {
|
||||||
// Read from file first, then fall back to direct env var
|
// Read from file first, then fall back to direct env var
|
||||||
@@ -1202,6 +1223,78 @@ fn is_under_temp_dir(path: &Path) -> bool {
|
|||||||
path.starts_with(std::env::temp_dir())
|
path.starts_with(std::env::temp_dir())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Command-line values that take precedence over the matching environment
|
||||||
|
/// variables when assembling a Vault backend configuration.
|
||||||
|
///
|
||||||
|
/// Every field has a `RUSTFS_KMS_VAULT_*` equivalent that the CLI layer already
|
||||||
|
/// reads, so these are only set when the operator passed an explicit flag.
|
||||||
|
///
|
||||||
|
/// Deliberately not `Debug`: `token` holds the raw Vault token, and the
|
||||||
|
/// redacting `Debug` impls elsewhere in this module exist because a derived one
|
||||||
|
/// would print it. Denying the derive makes a future `{overrides:?}` a compile
|
||||||
|
/// error instead of a leak.
|
||||||
|
#[derive(Default, Clone, Copy)]
|
||||||
|
pub struct VaultCliOverrides<'a> {
|
||||||
|
pub address: Option<&'a str>,
|
||||||
|
pub token: Option<&'a str>,
|
||||||
|
pub mount_path: Option<&'a str>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assemble the Vault KV2 backend configuration from the environment.
|
||||||
|
///
|
||||||
|
/// Shared by [`KmsConfig::from_env`] and the server's command-line startup path
|
||||||
|
/// so both resolve the same auth method, namespace, TLS and mount settings.
|
||||||
|
pub fn vault_kv2_config_from_env(overrides: VaultCliOverrides<'_>) -> Result<VaultConfig> {
|
||||||
|
let mount_path = match overrides
|
||||||
|
.mount_path
|
||||||
|
.map(str::to_string)
|
||||||
|
.or_else(|| get_env_opt_str(ENV_KMS_VAULT_MOUNT_PATH))
|
||||||
|
{
|
||||||
|
Some(path) => {
|
||||||
|
tracing::warn!(
|
||||||
|
"RUSTFS_KMS_VAULT_MOUNT_PATH is deprecated for the Vault KV2 backend: it never calls the Transit engine and the value is stored but unused"
|
||||||
|
);
|
||||||
|
path
|
||||||
|
}
|
||||||
|
None => default_vault_kv2_mount_path(),
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(VaultConfig {
|
||||||
|
address: vault_address_from_env(overrides.address),
|
||||||
|
auth_method: vault_auth_method_from_env(overrides.token)?,
|
||||||
|
namespace: get_env_opt_str(ENV_KMS_VAULT_NAMESPACE),
|
||||||
|
mount_path,
|
||||||
|
kv_mount: get_env_str("RUSTFS_KMS_VAULT_KV_MOUNT", "secret"),
|
||||||
|
key_path_prefix: get_env_str("RUSTFS_KMS_VAULT_KEY_PREFIX", "rustfs/kms/keys"),
|
||||||
|
tls: vault_tls_config(get_env_bool(ENV_KMS_VAULT_SKIP_TLS_VERIFY, false)),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assemble the Vault Transit backend configuration from the environment.
|
||||||
|
///
|
||||||
|
/// Companion to [`vault_kv2_config_from_env`]; see there for why both entry
|
||||||
|
/// points share it.
|
||||||
|
pub fn vault_transit_config_from_env(overrides: VaultCliOverrides<'_>) -> Result<VaultTransitConfig> {
|
||||||
|
Ok(VaultTransitConfig {
|
||||||
|
address: vault_address_from_env(overrides.address),
|
||||||
|
auth_method: vault_auth_method_from_env(overrides.token)?,
|
||||||
|
namespace: get_env_opt_str(ENV_KMS_VAULT_NAMESPACE),
|
||||||
|
mount_path: overrides
|
||||||
|
.mount_path
|
||||||
|
.map(str::to_string)
|
||||||
|
.unwrap_or_else(|| get_env_str(ENV_KMS_VAULT_MOUNT_PATH, "transit")),
|
||||||
|
metadata_kv_mount: get_env_str(ENV_KMS_VAULT_TRANSIT_METADATA_KV_MOUNT, DEFAULT_VAULT_TRANSIT_METADATA_KV_MOUNT),
|
||||||
|
metadata_key_prefix: get_env_str(ENV_KMS_VAULT_TRANSIT_METADATA_PREFIX, DEFAULT_VAULT_TRANSIT_METADATA_KEY_PREFIX),
|
||||||
|
tls: vault_tls_config(get_env_bool(ENV_KMS_VAULT_SKIP_TLS_VERIFY, false)),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn vault_address_from_env(override_value: Option<&str>) -> String {
|
||||||
|
override_value
|
||||||
|
.map(str::to_string)
|
||||||
|
.unwrap_or_else(|| get_env_str(ENV_KMS_VAULT_ADDRESS, "http://localhost:8200"))
|
||||||
|
}
|
||||||
|
|
||||||
/// Resolve the Vault auth method from environment variables.
|
/// Resolve the Vault auth method from environment variables.
|
||||||
///
|
///
|
||||||
/// Setting `RUSTFS_KMS_VAULT_APPROLE_ROLE_ID` selects AppRole authentication;
|
/// Setting `RUSTFS_KMS_VAULT_APPROLE_ROLE_ID` selects AppRole authentication;
|
||||||
@@ -1209,27 +1302,59 @@ fn is_under_temp_dir(path: &Path) -> bool {
|
|||||||
/// (re-read on every login, mirroring the `RUSTFS_KMS_STATIC_SECRET_KEY_FILE`
|
/// (re-read on every login, mirroring the `RUSTFS_KMS_STATIC_SECRET_KEY_FILE`
|
||||||
/// precedent) or inline from `RUSTFS_KMS_VAULT_APPROLE_SECRET_ID`, with the
|
/// precedent) or inline from `RUSTFS_KMS_VAULT_APPROLE_SECRET_ID`, with the
|
||||||
/// file taking precedence. Without a role id the legacy token flow applies.
|
/// file taking precedence. Without a role id the legacy token flow applies.
|
||||||
fn vault_auth_method_from_env() -> Result<VaultAuthMethod> {
|
///
|
||||||
|
/// `RUSTFS_KMS_VAULT_KUBERNETES_ROLE` selects Kubernetes authentication, which
|
||||||
|
/// presents the pod's projected ServiceAccount token.
|
||||||
|
///
|
||||||
|
/// `token_override` carries a token supplied on the command line; it stands in
|
||||||
|
/// for `RUSTFS_KMS_VAULT_TOKEN` everywhere below, including the conflict checks,
|
||||||
|
/// so a flag and the variable it mirrors select the same method.
|
||||||
|
fn vault_auth_method_from_env(token_override: Option<&str>) -> Result<VaultAuthMethod> {
|
||||||
|
let token = token_override
|
||||||
|
.map(str::to_string)
|
||||||
|
.or_else(|| get_env_opt_str(ENV_KMS_VAULT_TOKEN));
|
||||||
|
let role_id = get_env_opt_str(ENV_KMS_VAULT_APPROLE_ROLE_ID);
|
||||||
|
let kubernetes_role = get_env_opt_str(ENV_KMS_VAULT_KUBERNETES_ROLE);
|
||||||
|
|
||||||
if let Some(token_file) = get_env_opt_str(ENV_KMS_VAULT_TOKEN_FILE) {
|
if let Some(token_file) = get_env_opt_str(ENV_KMS_VAULT_TOKEN_FILE) {
|
||||||
// A token file names one authoritative credential source; combining it
|
// A token file names one authoritative credential source; combining it
|
||||||
// with another one would leave the effective identity ambiguous, so
|
// with another one would leave the effective identity ambiguous, so
|
||||||
// that is a configuration error rather than a precedence rule.
|
// that is a configuration error rather than a precedence rule.
|
||||||
if get_env_opt_str(ENV_KMS_VAULT_APPROLE_ROLE_ID).is_some() {
|
for (name, configured) in [
|
||||||
return Err(KmsError::configuration_error(format!(
|
(ENV_KMS_VAULT_APPROLE_ROLE_ID, role_id.is_some()),
|
||||||
"{ENV_KMS_VAULT_TOKEN_FILE} cannot be combined with {ENV_KMS_VAULT_APPROLE_ROLE_ID}; configure exactly one Vault auth method"
|
(ENV_KMS_VAULT_KUBERNETES_ROLE, kubernetes_role.is_some()),
|
||||||
)));
|
(ENV_KMS_VAULT_TOKEN, token.is_some()),
|
||||||
}
|
] {
|
||||||
if get_env_opt_str("RUSTFS_KMS_VAULT_TOKEN").is_some() {
|
if configured {
|
||||||
return Err(KmsError::configuration_error(format!(
|
return Err(KmsError::configuration_error(format!(
|
||||||
"{ENV_KMS_VAULT_TOKEN_FILE} cannot be combined with RUSTFS_KMS_VAULT_TOKEN; configure exactly one Vault auth method"
|
"{ENV_KMS_VAULT_TOKEN_FILE} cannot be combined with {name}; configure exactly one Vault auth method"
|
||||||
)));
|
)));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
return Ok(VaultAuthMethod::token_file(PathBuf::from(token_file)));
|
return Ok(VaultAuthMethod::token_file(PathBuf::from(token_file)));
|
||||||
}
|
}
|
||||||
|
|
||||||
let Some(role_id) = get_env_opt_str(ENV_KMS_VAULT_APPROLE_ROLE_ID) else {
|
if let Some(role) = kubernetes_role {
|
||||||
|
// Unlike a leftover static token, a second login method is never a
|
||||||
|
// stale remnant: both were configured deliberately and neither can be
|
||||||
|
// ranked over the other.
|
||||||
|
if role_id.is_some() {
|
||||||
|
return Err(KmsError::configuration_error(format!(
|
||||||
|
"{ENV_KMS_VAULT_KUBERNETES_ROLE} cannot be combined with {ENV_KMS_VAULT_APPROLE_ROLE_ID}; configure exactly one Vault auth method"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
return Ok(VaultAuthMethod::Kubernetes {
|
||||||
|
role,
|
||||||
|
mount: get_env_str(ENV_KMS_VAULT_KUBERNETES_MOUNT, DEFAULT_VAULT_KUBERNETES_MOUNT),
|
||||||
|
jwt_path: get_env_opt_str(ENV_KMS_VAULT_KUBERNETES_JWT_PATH)
|
||||||
|
.map_or_else(default_vault_kubernetes_jwt_path, PathBuf::from),
|
||||||
|
refresh_safety_window_secs: None,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
let Some(role_id) = role_id else {
|
||||||
return Ok(VaultAuthMethod::Token {
|
return Ok(VaultAuthMethod::Token {
|
||||||
token: get_env_str("RUSTFS_KMS_VAULT_TOKEN", "dev-token"),
|
token: token.unwrap_or_else(|| "dev-token".to_string()),
|
||||||
});
|
});
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -1273,6 +1398,22 @@ fn validate_vault_auth_method(backend_name: &str, auth_method: &VaultAuthMethod)
|
|||||||
}
|
}
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
VaultAuthMethod::Kubernetes {
|
||||||
|
role, mount, jwt_path, ..
|
||||||
|
} => {
|
||||||
|
if role.is_empty() {
|
||||||
|
return Err(KmsError::configuration_error(format!("{backend_name} Kubernetes role cannot be empty")));
|
||||||
|
}
|
||||||
|
if mount.is_empty() {
|
||||||
|
return Err(KmsError::configuration_error(format!("{backend_name} Kubernetes mount cannot be empty")));
|
||||||
|
}
|
||||||
|
if jwt_path.as_os_str().is_empty() {
|
||||||
|
return Err(KmsError::configuration_error(format!(
|
||||||
|
"{backend_name} Kubernetes ServiceAccount token path cannot be empty"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
VaultAuthMethod::TokenFile {
|
VaultAuthMethod::TokenFile {
|
||||||
path,
|
path,
|
||||||
poll_interval_secs,
|
poll_interval_secs,
|
||||||
@@ -1976,6 +2117,106 @@ mod tests {
|
|||||||
.expect("well-formed token file auth must validate");
|
.expect("well-formed token file auth must validate");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A Kubernetes role alone configures the method: the credential is the
|
||||||
|
/// pod's projected ServiceAccount token, so nothing secret is in the
|
||||||
|
/// environment and the mount and token path fall back to the cluster
|
||||||
|
/// defaults.
|
||||||
|
#[test]
|
||||||
|
fn test_from_env_selects_kubernetes() {
|
||||||
|
with_vars(
|
||||||
|
vec![
|
||||||
|
("RUSTFS_KMS_BACKEND", Some("vault-transit")),
|
||||||
|
(ENV_KMS_VAULT_ADDRESS, Some("https://vault.example.com")),
|
||||||
|
(ENV_KMS_VAULT_KUBERNETES_ROLE, Some("rustfs")),
|
||||||
|
(ENV_KMS_VAULT_KUBERNETES_MOUNT, None),
|
||||||
|
(ENV_KMS_VAULT_KUBERNETES_JWT_PATH, None),
|
||||||
|
(ENV_KMS_VAULT_TOKEN, None),
|
||||||
|
(ENV_KMS_VAULT_TOKEN_FILE, None),
|
||||||
|
(ENV_KMS_VAULT_APPROLE_ROLE_ID, None),
|
||||||
|
],
|
||||||
|
|| {
|
||||||
|
let config = KmsConfig::from_env().expect("kms config should load from env");
|
||||||
|
let vault = config.vault_transit_config().expect("vault transit backend config");
|
||||||
|
let VaultAuthMethod::Kubernetes {
|
||||||
|
role,
|
||||||
|
mount,
|
||||||
|
jwt_path,
|
||||||
|
refresh_safety_window_secs,
|
||||||
|
} = &vault.auth_method
|
||||||
|
else {
|
||||||
|
panic!(
|
||||||
|
"a kubernetes role in the environment must select Kubernetes auth, got {:?}",
|
||||||
|
vault.auth_method
|
||||||
|
);
|
||||||
|
};
|
||||||
|
assert_eq!(role, "rustfs");
|
||||||
|
assert_eq!(mount, DEFAULT_VAULT_KUBERNETES_MOUNT);
|
||||||
|
assert_eq!(jwt_path, Path::new(DEFAULT_VAULT_KUBERNETES_JWT_PATH));
|
||||||
|
assert_eq!(refresh_safety_window_secs, &None);
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_from_env_kubernetes_is_mutually_exclusive_with_other_auth() {
|
||||||
|
with_vars(
|
||||||
|
vec![
|
||||||
|
("RUSTFS_KMS_BACKEND", Some("vault-transit")),
|
||||||
|
(ENV_KMS_VAULT_KUBERNETES_ROLE, Some("rustfs")),
|
||||||
|
(ENV_KMS_VAULT_APPROLE_ROLE_ID, Some("env-role-id")),
|
||||||
|
(ENV_KMS_VAULT_TOKEN, None),
|
||||||
|
(ENV_KMS_VAULT_TOKEN_FILE, None),
|
||||||
|
],
|
||||||
|
|| {
|
||||||
|
let error = KmsConfig::from_env().expect_err("kubernetes combined with approle must be rejected");
|
||||||
|
assert!(error.to_string().contains(ENV_KMS_VAULT_KUBERNETES_ROLE));
|
||||||
|
assert!(error.to_string().contains(ENV_KMS_VAULT_APPROLE_ROLE_ID));
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_validate_rejects_bad_kubernetes_settings() {
|
||||||
|
let vault_config = |auth_method: VaultAuthMethod| KmsConfig {
|
||||||
|
backend: KmsBackend::VaultTransit,
|
||||||
|
backend_config: BackendConfig::VaultTransit(Box::new(VaultTransitConfig {
|
||||||
|
address: "https://vault.example.com:8200".to_string(),
|
||||||
|
auth_method,
|
||||||
|
..Default::default()
|
||||||
|
})),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let error = vault_config(VaultAuthMethod::kubernetes(String::new()))
|
||||||
|
.validate()
|
||||||
|
.expect_err("an empty kubernetes role must be rejected");
|
||||||
|
assert!(error.to_string().contains("role"), "got {error}");
|
||||||
|
|
||||||
|
let error = vault_config(VaultAuthMethod::Kubernetes {
|
||||||
|
role: "rustfs".to_string(),
|
||||||
|
mount: String::new(),
|
||||||
|
jwt_path: PathBuf::from(DEFAULT_VAULT_KUBERNETES_JWT_PATH),
|
||||||
|
refresh_safety_window_secs: None,
|
||||||
|
})
|
||||||
|
.validate()
|
||||||
|
.expect_err("an empty kubernetes mount must be rejected");
|
||||||
|
assert!(error.to_string().contains("mount"), "got {error}");
|
||||||
|
|
||||||
|
let error = vault_config(VaultAuthMethod::Kubernetes {
|
||||||
|
role: "rustfs".to_string(),
|
||||||
|
mount: DEFAULT_VAULT_KUBERNETES_MOUNT.to_string(),
|
||||||
|
jwt_path: PathBuf::new(),
|
||||||
|
refresh_safety_window_secs: None,
|
||||||
|
})
|
||||||
|
.validate()
|
||||||
|
.expect_err("an empty ServiceAccount token path must be rejected");
|
||||||
|
assert!(error.to_string().contains("token path"), "got {error}");
|
||||||
|
|
||||||
|
vault_config(VaultAuthMethod::kubernetes("rustfs".to_string()))
|
||||||
|
.validate()
|
||||||
|
.expect("well-formed kubernetes auth must validate");
|
||||||
|
}
|
||||||
|
|
||||||
/// Every KV2 read, write and listing is routed through `kv_mount`, so an
|
/// Every KV2 read, write and listing is routed through `kv_mount`, so an
|
||||||
/// empty one names a path no Vault engine answers. The Transit backend
|
/// empty one names a path no Vault engine answers. The Transit backend
|
||||||
/// already rejects its own empty mounts; this closes the same gap on the
|
/// already rejects its own empty mounts; this closes the same gap on the
|
||||||
|
|||||||
@@ -37,7 +37,11 @@ hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"]
|
|||||||
[dependencies]
|
[dependencies]
|
||||||
hotpath.workspace = true
|
hotpath.workspace = true
|
||||||
humantime.workspace = true
|
humantime.workspace = true
|
||||||
|
http.workspace = true
|
||||||
hyper = { workspace = true, features = ["http2", "http1", "server"] }
|
hyper = { workspace = true, features = ["http2", "http1", "server"] }
|
||||||
|
reqwest = { workspace = true, features = ["json"] }
|
||||||
|
rustfs-signer.workspace = true
|
||||||
|
s3s.workspace = true
|
||||||
jiff = { workspace = true, features = ["serde"] }
|
jiff = { workspace = true, features = ["serde"] }
|
||||||
serde = { workspace = true, features = ["derive"] }
|
serde = { workspace = true, features = ["derive"] }
|
||||||
serde_json = { workspace = true, features = ["raw_value"] }
|
serde_json = { workspace = true, features = ["raw_value"] }
|
||||||
@@ -49,3 +53,4 @@ doctest = false
|
|||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
rmp-serde.workspace = true
|
rmp-serde.workspace = true
|
||||||
|
tokio = { workspace = true, features = ["macros", "rt-multi-thread", "net"] }
|
||||||
|
|||||||
@@ -0,0 +1,851 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! Admin API HTTP client for heal and scanner management (rustfs/backlog#1869).
|
||||||
|
//!
|
||||||
|
//! [`AdminClient`] speaks the `/rustfs/admin/v3` surface with S3 SigV4
|
||||||
|
//! request signing (the same scheme the server's admin router authenticates),
|
||||||
|
//! so `mc`-style tooling and automation can drive heal start/query/cancel and
|
||||||
|
//! read background-heal / scanner status without hand-rolling HTTP.
|
||||||
|
//!
|
||||||
|
//! Wire structs in this module mirror the server-side shapes
|
||||||
|
//! (`rustfs/src/admin/handlers/heal.rs`, `handlers/scanner.rs`,
|
||||||
|
//! `rustfs-common/src/heal_channel.rs`), following the madmin-go model where
|
||||||
|
//! the SDK owns its own copies and round-trip tests pin the encoding. Deeply
|
||||||
|
//! nested status payloads that the server composes from runtime types are
|
||||||
|
//! carried through as `serde_json::Value` and flattened maps rather than
|
||||||
|
//! duplicated field-for-field, so the client cannot silently drift on fields
|
||||||
|
//! it never interprets.
|
||||||
|
|
||||||
|
use crate::heal_commands::HealResultItem;
|
||||||
|
use http::Method;
|
||||||
|
use serde::{Deserialize, Serialize, de};
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
/// Default admin API path prefix on a RustFS endpoint.
|
||||||
|
pub const DEFAULT_ADMIN_API_PREFIX: &str = "/rustfs/admin";
|
||||||
|
/// Default SigV4 region when the server has no explicit region configured.
|
||||||
|
pub const DEFAULT_REGION: &str = "us-east-1";
|
||||||
|
|
||||||
|
/// Scan mode for a heal request, mirroring the server's numeric-or-name wire
|
||||||
|
/// encoding (`0` unknown/default, `1` normal, `2` deep).
|
||||||
|
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||||
|
pub enum HealScanMode {
|
||||||
|
/// Server default; behaves as [`HealScanMode::Normal`].
|
||||||
|
#[default]
|
||||||
|
Unknown,
|
||||||
|
/// Metadata-level checks only.
|
||||||
|
Normal,
|
||||||
|
/// Full bitrot verification while healing.
|
||||||
|
Deep,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl HealScanMode {
|
||||||
|
fn wire_number(self) -> u8 {
|
||||||
|
match self {
|
||||||
|
Self::Unknown => 0,
|
||||||
|
Self::Normal => 1,
|
||||||
|
Self::Deep => 2,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn from_wire_number(value: u8) -> Option<Self> {
|
||||||
|
match value {
|
||||||
|
0 => Some(Self::Unknown),
|
||||||
|
1 => Some(Self::Normal),
|
||||||
|
2 => Some(Self::Deep),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn from_wire_name(value: &str) -> Option<Self> {
|
||||||
|
match value {
|
||||||
|
"unknown" => Some(Self::Unknown),
|
||||||
|
"normal" => Some(Self::Normal),
|
||||||
|
"deep" => Some(Self::Deep),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Serialize for HealScanMode {
|
||||||
|
fn serialize<S: serde::Serializer>(&self, serializer: S) -> Result<S::Ok, S::Error> {
|
||||||
|
serializer.serialize_u8(self.wire_number())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'de> Deserialize<'de> for HealScanMode {
|
||||||
|
fn deserialize<D: serde::Deserializer<'de>>(deserializer: D) -> Result<Self, D::Error> {
|
||||||
|
struct HealScanModeVisitor;
|
||||||
|
|
||||||
|
impl de::Visitor<'_> for HealScanModeVisitor {
|
||||||
|
type Value = HealScanMode;
|
||||||
|
|
||||||
|
fn expecting(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
formatter.write_str("a heal scan mode number or name")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn visit_u64<E: de::Error>(self, value: u64) -> Result<Self::Value, E> {
|
||||||
|
u8::try_from(value)
|
||||||
|
.ok()
|
||||||
|
.and_then(HealScanMode::from_wire_number)
|
||||||
|
.ok_or_else(|| E::custom(format!("unknown heal scan mode number: {value}")))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn visit_str<E: de::Error>(self, value: &str) -> Result<Self::Value, E> {
|
||||||
|
HealScanMode::from_wire_name(value).ok_or_else(|| E::custom(format!("unknown heal scan mode name: {value}")))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
deserializer.deserialize_any(HealScanModeVisitor)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Heal options for an admin heal request (mirror of the server body type).
|
||||||
|
/// Fields default on decode: a client should tolerate a server response whose
|
||||||
|
/// settings object omits fields it never set.
|
||||||
|
#[derive(Debug, Clone, Default, Serialize, Deserialize)]
|
||||||
|
pub struct HealOpts {
|
||||||
|
#[serde(default)]
|
||||||
|
pub recursive: bool,
|
||||||
|
#[serde(rename = "dryRun", default)]
|
||||||
|
pub dry_run: bool,
|
||||||
|
#[serde(default)]
|
||||||
|
pub remove: bool,
|
||||||
|
#[serde(default)]
|
||||||
|
pub recreate: bool,
|
||||||
|
#[serde(rename = "scanMode", default)]
|
||||||
|
pub scan_mode: HealScanMode,
|
||||||
|
#[serde(rename = "updateParity", default)]
|
||||||
|
pub update_parity: bool,
|
||||||
|
#[serde(rename = "nolock", default)]
|
||||||
|
pub no_lock: bool,
|
||||||
|
#[serde(rename = "pool", default)]
|
||||||
|
pub pool: Option<usize>,
|
||||||
|
#[serde(rename = "set", default)]
|
||||||
|
pub set: Option<usize>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Successful heal start / path-scoped cancel response.
|
||||||
|
#[derive(Debug, Clone, Deserialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct HealStartSuccess {
|
||||||
|
pub client_token: String,
|
||||||
|
pub client_address: String,
|
||||||
|
#[serde(default)]
|
||||||
|
pub start_time: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Heal task status response (query, cancel-with-token, start-then-poll).
|
||||||
|
#[derive(Debug, Clone, Deserialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct HealTaskStatus {
|
||||||
|
/// `running` | `finished` | `stopped` | `notFound`.
|
||||||
|
pub summary: String,
|
||||||
|
/// Failure detail for stopped tasks; empty otherwise.
|
||||||
|
#[serde(rename = "detail", default)]
|
||||||
|
pub failure_detail: String,
|
||||||
|
#[serde(default)]
|
||||||
|
pub start_time: String,
|
||||||
|
#[serde(default)]
|
||||||
|
pub settings: HealOpts,
|
||||||
|
#[serde(default)]
|
||||||
|
pub items: Vec<HealResultItem>,
|
||||||
|
#[serde(default)]
|
||||||
|
pub truncated: bool,
|
||||||
|
/// Live progress snapshot; the exact shape is owned by the heal runtime.
|
||||||
|
#[serde(default)]
|
||||||
|
pub progress: Option<serde_json::Value>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `POST /v3/background-heal/status` response. Known top-level fields are
|
||||||
|
/// typed; the flattened heal info and operations matrix pass through verbatim.
|
||||||
|
#[derive(Debug, Clone, Deserialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct BackgroundHealStatus {
|
||||||
|
/// `disabled` | `uninitialized` | `idle` | `active` | `degraded`.
|
||||||
|
pub state: String,
|
||||||
|
#[serde(default)]
|
||||||
|
pub heal_queue_length: u64,
|
||||||
|
#[serde(default)]
|
||||||
|
pub heal_active_tasks: u64,
|
||||||
|
#[serde(default)]
|
||||||
|
pub cluster_status_complete: bool,
|
||||||
|
#[serde(default)]
|
||||||
|
pub progress: Option<serde_json::Value>,
|
||||||
|
/// Remaining wire fields (flattened `BackgroundHealInfo` plus the
|
||||||
|
/// priority-by-source operations matrix), carried verbatim.
|
||||||
|
#[serde(flatten)]
|
||||||
|
pub extra: serde_json::Map<String, serde_json::Value>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /v3/scanner/status` response, typed at the fields operators branch
|
||||||
|
/// on; everything else passes through verbatim.
|
||||||
|
#[derive(Debug, Clone, Deserialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct ScannerStatus {
|
||||||
|
pub enabled: bool,
|
||||||
|
/// `fresh` | `stale` | `unknown`; absent when the scanner never completed
|
||||||
|
/// a cycle.
|
||||||
|
#[serde(default)]
|
||||||
|
pub freshness: Option<ScannerFreshness>,
|
||||||
|
#[serde(flatten)]
|
||||||
|
pub extra: serde_json::Map<String, serde_json::Value>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Freshness block of the scanner status response.
|
||||||
|
#[derive(Debug, Clone, Deserialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct ScannerFreshness {
|
||||||
|
/// `fresh` | `stale` | `unknown`.
|
||||||
|
pub state: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ScannerStatus {
|
||||||
|
/// Convenience accessor for the freshness state string.
|
||||||
|
pub fn freshness(&self) -> &str {
|
||||||
|
self.freshness
|
||||||
|
.as_ref()
|
||||||
|
.map(|freshness| freshness.state.as_str())
|
||||||
|
.unwrap_or("unknown")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Everything that can go wrong in an admin client call.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub enum AdminClientError {
|
||||||
|
/// The endpoint URL could not be parsed.
|
||||||
|
InvalidEndpoint(String),
|
||||||
|
/// Request build/send failed (DNS, connect, timeout, body read).
|
||||||
|
Transport(reqwest::Error),
|
||||||
|
/// The server answered a non-2xx status.
|
||||||
|
HttpStatus { status: u16, body: String },
|
||||||
|
/// The response body did not decode into the expected shape.
|
||||||
|
Decode { message: String },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for AdminClientError {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
match self {
|
||||||
|
Self::InvalidEndpoint(message) => write!(f, "invalid admin endpoint: {message}"),
|
||||||
|
Self::Transport(err) => write!(f, "admin request transport failure: {err}"),
|
||||||
|
Self::HttpStatus { status, body } => write!(f, "admin request failed with HTTP {status}: {body}"),
|
||||||
|
Self::Decode { message } => write!(f, "admin response decode failure: {message}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::error::Error for AdminClientError {}
|
||||||
|
|
||||||
|
impl From<reqwest::Error> for AdminClientError {
|
||||||
|
fn from(err: reqwest::Error) -> Self {
|
||||||
|
Self::Transport(err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A signed client for a RustFS admin API.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct AdminClient {
|
||||||
|
endpoint: reqwest::Url,
|
||||||
|
access_key: String,
|
||||||
|
secret_key: String,
|
||||||
|
session_token: String,
|
||||||
|
region: String,
|
||||||
|
api_prefix: String,
|
||||||
|
http: reqwest::Client,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl AdminClient {
|
||||||
|
/// Build a client for `endpoint` (e.g. `http://127.0.0.1:9000`) using root
|
||||||
|
/// or admin credentials. Requests are SigV4-signed with the same scheme
|
||||||
|
/// the server's admin router authenticates.
|
||||||
|
pub fn new(endpoint: &str, access_key: &str, secret_key: &str) -> Result<Self, AdminClientError> {
|
||||||
|
let url = reqwest::Url::parse(endpoint).map_err(|err| AdminClientError::InvalidEndpoint(err.to_string()))?;
|
||||||
|
if url.host_str().is_none() {
|
||||||
|
return Err(AdminClientError::InvalidEndpoint("endpoint has no host".to_string()));
|
||||||
|
}
|
||||||
|
let http = reqwest::Client::builder()
|
||||||
|
.connect_timeout(Duration::from_secs(10))
|
||||||
|
.timeout(Duration::from_secs(30))
|
||||||
|
.build()
|
||||||
|
.map_err(AdminClientError::Transport)?;
|
||||||
|
Ok(Self {
|
||||||
|
endpoint: url,
|
||||||
|
access_key: access_key.to_string(),
|
||||||
|
secret_key: secret_key.to_string(),
|
||||||
|
session_token: String::new(),
|
||||||
|
region: DEFAULT_REGION.to_string(),
|
||||||
|
api_prefix: DEFAULT_ADMIN_API_PREFIX.to_string(),
|
||||||
|
http,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Attach an STS session token (signed as `x-amz-security-token`).
|
||||||
|
pub fn with_session_token(mut self, session_token: impl Into<String>) -> Self {
|
||||||
|
self.session_token = session_token.into();
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Override the SigV4 region (defaults to `us-east-1`, matching a
|
||||||
|
/// region-less RustFS deployment).
|
||||||
|
pub fn with_region(mut self, region: impl Into<String>) -> Self {
|
||||||
|
self.region = region.into();
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Override the admin API path prefix (defaults to `/rustfs/admin`).
|
||||||
|
pub fn with_api_prefix(mut self, prefix: impl Into<String>) -> Self {
|
||||||
|
self.api_prefix = prefix.into();
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start a heal. `bucket` empty and `prefix` empty heals the whole
|
||||||
|
/// deployment (requires `recursive` or a `pool`/`set` pair in `opts`,
|
||||||
|
/// enforced server-side); a bucket alone heals the bucket (the server
|
||||||
|
/// forces `recursive` for bucket heals).
|
||||||
|
pub async fn heal_start(
|
||||||
|
&self,
|
||||||
|
bucket: Option<&str>,
|
||||||
|
prefix: Option<&str>,
|
||||||
|
opts: &HealOpts,
|
||||||
|
force_start: bool,
|
||||||
|
) -> Result<HealStartSuccess, AdminClientError> {
|
||||||
|
let body = serde_json::to_vec(opts).map_err(|err| AdminClientError::Decode {
|
||||||
|
message: err.to_string(),
|
||||||
|
})?;
|
||||||
|
let mut query = Vec::new();
|
||||||
|
if force_start {
|
||||||
|
query.push(("forceStart", "true".to_string()));
|
||||||
|
}
|
||||||
|
self.post_json(&heal_path(bucket, prefix), &query, body).await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Query the status of the heal identified by `client_token` (the token
|
||||||
|
/// returned by [`Self::heal_start`]) at the path it was started on.
|
||||||
|
pub async fn heal_status(
|
||||||
|
&self,
|
||||||
|
bucket: Option<&str>,
|
||||||
|
prefix: Option<&str>,
|
||||||
|
client_token: &str,
|
||||||
|
) -> Result<HealTaskStatus, AdminClientError> {
|
||||||
|
self.post_json(&heal_path(bucket, prefix), &[("clientToken", client_token.to_string())], Vec::new())
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stop a heal: with a `client_token` only that task is cancelled and its
|
||||||
|
/// final status returned; without one, every heal task at the path is
|
||||||
|
/// cancelled (the server answers with a start-success-shaped receipt).
|
||||||
|
pub async fn heal_stop(
|
||||||
|
&self,
|
||||||
|
bucket: Option<&str>,
|
||||||
|
prefix: Option<&str>,
|
||||||
|
client_token: Option<&str>,
|
||||||
|
) -> Result<HealStopOutcome, AdminClientError> {
|
||||||
|
let mut query = vec![("forceStop", "true".to_string())];
|
||||||
|
if let Some(token) = client_token {
|
||||||
|
query.push(("clientToken", token.to_string()));
|
||||||
|
}
|
||||||
|
match client_token {
|
||||||
|
Some(_) => {
|
||||||
|
let status: HealTaskStatus = self.post_json(&heal_path(bucket, prefix), &query, Vec::new()).await?;
|
||||||
|
Ok(HealStopOutcome::Stopped(status))
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
let success: HealStartSuccess = self.post_json(&heal_path(bucket, prefix), &query, Vec::new()).await?;
|
||||||
|
Ok(HealStopOutcome::PathStopped(success))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cluster-aggregated background heal status.
|
||||||
|
pub async fn background_heal_status(&self) -> Result<BackgroundHealStatus, AdminClientError> {
|
||||||
|
self.get_json("/v3/background-heal/status").await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Data scanner status (enabled state, freshness, runtime config).
|
||||||
|
pub async fn scanner_status(&self) -> Result<ScannerStatus, AdminClientError> {
|
||||||
|
self.get_json("/v3/scanner/status").await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ILM expiry worker status. The payload is owned by the expiry
|
||||||
|
/// subsystem and still evolving; returned verbatim.
|
||||||
|
pub async fn ilm_expiry_status(&self) -> Result<serde_json::Value, AdminClientError> {
|
||||||
|
self.get_json("/v3/ilm/expiry/status").await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Durable replacement-recovery status (admin v4). The payload is owned
|
||||||
|
/// by the heal runtime; returned verbatim.
|
||||||
|
pub async fn replacement_recovery_status(&self) -> Result<serde_json::Value, AdminClientError> {
|
||||||
|
self.get_json("/v4/heal/replacement-recovery").await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Signed GET returning a decoded JSON body; escape hatch for endpoints
|
||||||
|
/// this client does not wrap yet.
|
||||||
|
pub async fn get_json<T: for<'de> Deserialize<'de>>(&self, path: &str) -> Result<T, AdminClientError> {
|
||||||
|
let url = self.url_for(path, &[])?;
|
||||||
|
let request = self.sign_and_build(Method::GET, url, Vec::new(), None).await?;
|
||||||
|
self.execute(request).await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Signed POST returning a decoded JSON body.
|
||||||
|
async fn post_json<T: for<'de> Deserialize<'de>>(
|
||||||
|
&self,
|
||||||
|
path: &str,
|
||||||
|
query: &[(&str, String)],
|
||||||
|
body: Vec<u8>,
|
||||||
|
) -> Result<T, AdminClientError> {
|
||||||
|
let content_type = if body.is_empty() { None } else { Some("application/json") };
|
||||||
|
let url = self.url_for(path, query)?;
|
||||||
|
let request = self.sign_and_build(Method::POST, url, body, content_type).await?;
|
||||||
|
self.execute(request).await
|
||||||
|
}
|
||||||
|
|
||||||
|
fn url_for(&self, path: &str, query: &[(&str, String)]) -> Result<reqwest::Url, AdminClientError> {
|
||||||
|
let mut url = self
|
||||||
|
.endpoint
|
||||||
|
.join(&format!("{}{}", self.api_prefix.trim_end_matches('/'), path))
|
||||||
|
.map_err(|err| AdminClientError::InvalidEndpoint(err.to_string()))?;
|
||||||
|
if !query.is_empty() {
|
||||||
|
let mut pairs = url.query_pairs_mut();
|
||||||
|
for (key, value) in query {
|
||||||
|
pairs.append_pair(key, value);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(url)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build a SigV4-signed request via the same signer the server trusts,
|
||||||
|
/// then hand the signed headers to the HTTP client. The signature covers
|
||||||
|
/// method, path, query, and an unsigned-payload marker — the same shape
|
||||||
|
/// RustFS itself sends for peer admin calls.
|
||||||
|
async fn sign_and_build(
|
||||||
|
&self,
|
||||||
|
method: Method,
|
||||||
|
url: reqwest::Url,
|
||||||
|
body: Vec<u8>,
|
||||||
|
content_type: Option<&str>,
|
||||||
|
) -> Result<reqwest::Request, AdminClientError> {
|
||||||
|
let authority = match (url.host_str(), url.port_or_known_default()) {
|
||||||
|
(Some(host), Some(port)) => format!("{host}:{port}"),
|
||||||
|
_ => return Err(AdminClientError::InvalidEndpoint("endpoint has no authority".to_string())),
|
||||||
|
};
|
||||||
|
let mut builder = http::Request::builder()
|
||||||
|
.method(method.clone())
|
||||||
|
.uri(url.as_str())
|
||||||
|
.header(http::header::HOST, &authority)
|
||||||
|
.header("x-amz-content-sha256", rustfs_signer::constants::UNSIGNED_PAYLOAD);
|
||||||
|
if let Some(content_type) = content_type {
|
||||||
|
builder = builder.header(http::header::CONTENT_TYPE, content_type);
|
||||||
|
}
|
||||||
|
let unsigned = builder
|
||||||
|
.body(s3s::Body::empty())
|
||||||
|
.map_err(|err| AdminClientError::InvalidEndpoint(format!("build request failed: {err}")))?;
|
||||||
|
let signed = rustfs_signer::sign_v4(
|
||||||
|
unsigned,
|
||||||
|
body.len() as i64,
|
||||||
|
&self.access_key,
|
||||||
|
&self.secret_key,
|
||||||
|
&self.session_token,
|
||||||
|
&self.region,
|
||||||
|
);
|
||||||
|
|
||||||
|
let mut request = self
|
||||||
|
.http
|
||||||
|
.request(method, url)
|
||||||
|
.body(body)
|
||||||
|
.build()
|
||||||
|
.map_err(AdminClientError::Transport)?;
|
||||||
|
let headers = request.headers_mut();
|
||||||
|
for (name, value) in signed.headers().iter() {
|
||||||
|
// HOST is owned by the HTTP client; the signed value above was
|
||||||
|
// built from the same URL authority, so they always agree.
|
||||||
|
if name == http::header::HOST {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
headers.insert(name, value.clone());
|
||||||
|
}
|
||||||
|
Ok(request)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn execute<T: for<'de> Deserialize<'de>>(&self, request: reqwest::Request) -> Result<T, AdminClientError> {
|
||||||
|
let response = self.http.execute(request).await?;
|
||||||
|
let status = response.status();
|
||||||
|
let bytes = response.bytes().await?;
|
||||||
|
if !status.is_success() {
|
||||||
|
return Err(AdminClientError::HttpStatus {
|
||||||
|
status: status.as_u16(),
|
||||||
|
body: String::from_utf8_lossy(&bytes).into_owned(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
serde_json::from_slice(&bytes).map_err(|err| AdminClientError::Decode {
|
||||||
|
message: err.to_string(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Response of [`AdminClient::heal_stop`]: cancelling a single tokened task
|
||||||
|
/// answers with that task's status, cancelling a whole path answers with a
|
||||||
|
/// start-success-shaped receipt.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub enum HealStopOutcome {
|
||||||
|
Stopped(HealTaskStatus),
|
||||||
|
PathStopped(HealStartSuccess),
|
||||||
|
}
|
||||||
|
|
||||||
|
fn heal_path(bucket: Option<&str>, prefix: Option<&str>) -> String {
|
||||||
|
match (bucket, prefix) {
|
||||||
|
(Some(bucket), Some(prefix)) if !bucket.is_empty() && !prefix.is_empty() => {
|
||||||
|
format!("/v3/heal/{}/{}", percent_encode_path_segment(bucket), percent_encode_path_segment(prefix))
|
||||||
|
}
|
||||||
|
(Some(bucket), Some(_)) | (Some(bucket), None) if !bucket.is_empty() => {
|
||||||
|
format!("/v3/heal/{}", percent_encode_path_segment(bucket))
|
||||||
|
}
|
||||||
|
_ => "/v3/heal/".to_string(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a single path segment (slashes are content, not separators, inside
|
||||||
|
/// bucket/prefix path params).
|
||||||
|
fn percent_encode_path_segment(segment: &str) -> String {
|
||||||
|
let mut out = String::with_capacity(segment.len());
|
||||||
|
for byte in segment.bytes() {
|
||||||
|
match byte {
|
||||||
|
b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => out.push(byte as char),
|
||||||
|
_ => out.push_str(&format!("%{byte:02X}")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::{
|
||||||
|
AdminClient, AdminClientError, BackgroundHealStatus, HealOpts, HealScanMode, HealStartSuccess, HealTaskStatus,
|
||||||
|
ScannerStatus, heal_path, percent_encode_path_segment,
|
||||||
|
};
|
||||||
|
use serde_json::json;
|
||||||
|
use std::sync::{Arc, Mutex};
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heal_paths_cover_root_bucket_and_prefix() {
|
||||||
|
assert_eq!(heal_path(None, None), "/v3/heal/");
|
||||||
|
assert_eq!(heal_path(Some(""), Some("")), "/v3/heal/");
|
||||||
|
assert_eq!(heal_path(Some("bucket"), None), "/v3/heal/bucket");
|
||||||
|
assert_eq!(heal_path(Some("bucket"), Some("pre/fix")), "/v3/heal/bucket/pre%2Ffix");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn path_segments_percent_encode_reserved_characters() {
|
||||||
|
assert_eq!(percent_encode_path_segment("a b"), "a%20b");
|
||||||
|
assert_eq!(percent_encode_path_segment("a/b"), "a%2Fb");
|
||||||
|
assert_eq!(percent_encode_path_segment("ü"), "%C3%BC");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heal_opts_round_trip_through_the_server_wire_shape() {
|
||||||
|
let opts = HealOpts {
|
||||||
|
recursive: true,
|
||||||
|
dry_run: false,
|
||||||
|
remove: true,
|
||||||
|
recreate: false,
|
||||||
|
scan_mode: HealScanMode::Deep,
|
||||||
|
update_parity: true,
|
||||||
|
no_lock: false,
|
||||||
|
pool: Some(1),
|
||||||
|
set: Some(2),
|
||||||
|
};
|
||||||
|
let wire = serde_json::to_value(&opts).unwrap();
|
||||||
|
assert_eq!(wire["scanMode"], json!(2), "the server body decodes scanMode as a number");
|
||||||
|
let back: HealOpts = serde_json::from_value(wire).unwrap();
|
||||||
|
assert_eq!(back.scan_mode, HealScanMode::Deep);
|
||||||
|
assert_eq!(back.pool, Some(1));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heal_scan_mode_accepts_both_wire_encodings() {
|
||||||
|
assert_eq!(serde_json::from_value::<HealScanMode>(json!(1)).unwrap(), HealScanMode::Normal);
|
||||||
|
assert_eq!(serde_json::from_value::<HealScanMode>(json!("deep")).unwrap(), HealScanMode::Deep);
|
||||||
|
assert!(serde_json::from_value::<HealScanMode>(json!(9)).is_err());
|
||||||
|
assert!(serde_json::from_value::<HealScanMode>(json!("sideways")).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn heal_task_status_decodes_the_server_response_shape() {
|
||||||
|
let raw = json!({
|
||||||
|
"summary": "finished",
|
||||||
|
"detail": "",
|
||||||
|
"startTime": "2026-08-17T00:00:00Z",
|
||||||
|
"settings": {"recursive": false, "scanMode": 1},
|
||||||
|
"items": [{
|
||||||
|
"resultId": 1, "type": "object", "bucket": "b", "object": "o", "versionId": "", "detail": "",
|
||||||
|
"parityBlocks": 2, "dataBlocks": 2, "diskCount": 4, "setCount": 1,
|
||||||
|
"before": {"drives": []}, "after": {"drives": []}, "objectSize": 128
|
||||||
|
}],
|
||||||
|
"truncated": false
|
||||||
|
});
|
||||||
|
let status: HealTaskStatus = serde_json::from_value(raw).unwrap();
|
||||||
|
assert_eq!(status.summary, "finished");
|
||||||
|
assert_eq!(status.items.len(), 1);
|
||||||
|
assert_eq!(status.settings.scan_mode, HealScanMode::Normal);
|
||||||
|
assert!(status.progress.is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn background_heal_status_types_known_fields_and_passes_the_rest_through() {
|
||||||
|
let raw = json!({
|
||||||
|
"state": "active",
|
||||||
|
"bitrotStartTime": "t",
|
||||||
|
"healQueueLength": 3,
|
||||||
|
"healActiveTasks": 1,
|
||||||
|
"healOperations": {"queueLength": 3},
|
||||||
|
"clusterStatusComplete": true
|
||||||
|
});
|
||||||
|
let status: BackgroundHealStatus = serde_json::from_value(raw).unwrap();
|
||||||
|
assert_eq!(status.state, "active");
|
||||||
|
assert_eq!(status.heal_queue_length, 3);
|
||||||
|
assert!(status.cluster_status_complete);
|
||||||
|
assert!(status.extra.contains_key("healOperations"), "unknown nested payloads must pass through");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn scanner_status_defaults_freshness_to_unknown() {
|
||||||
|
let raw = json!({"enabled": true, "freshness": {"state": "stale"}, "metrics": {}});
|
||||||
|
let status: ScannerStatus = serde_json::from_value(raw).unwrap();
|
||||||
|
assert_eq!(status.freshness(), "stale");
|
||||||
|
let bare: ScannerStatus = serde_json::from_value(json!({"enabled": false})).unwrap();
|
||||||
|
assert_eq!(bare.freshness(), "unknown");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn invalid_endpoint_is_rejected_without_io() {
|
||||||
|
let err = AdminClient::new("not a url", "ak", "sk").unwrap_err();
|
||||||
|
assert!(matches!(err, AdminClientError::InvalidEndpoint(_)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn signed_requests_carry_sigv4_authorization_and_correct_target() {
|
||||||
|
let server = TestServer::spawn(r#"{"clientToken":"token-1","clientAddress":"127.0.0.1:9","startTime":"t"}"#, 200).await;
|
||||||
|
let client = AdminClient::new(&format!("http://{}", server.addr), "minioadmin", "minioadmin")
|
||||||
|
.expect("client builds against the test server");
|
||||||
|
|
||||||
|
let start: HealStartSuccess = client
|
||||||
|
.heal_start(
|
||||||
|
Some("bucket"),
|
||||||
|
None,
|
||||||
|
&HealOpts {
|
||||||
|
recursive: true,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
false,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("signed heal start decodes");
|
||||||
|
|
||||||
|
assert_eq!(start.client_token, "token-1");
|
||||||
|
let request = server.recorded();
|
||||||
|
assert_eq!(request.method, "POST");
|
||||||
|
assert_eq!(request.path, "/rustfs/admin/v3/heal/bucket");
|
||||||
|
assert!(!request.query.contains("forceStart"), "absent flags must not be sent");
|
||||||
|
let auth = request.header("authorization").expect("request must be signed");
|
||||||
|
assert!(auth.starts_with("AWS4-HMAC-SHA256"), "SigV4 scheme, got: {auth}");
|
||||||
|
assert!(auth.contains("Credential=minioadmin/"), "credentials must be in the Authorization header");
|
||||||
|
assert_eq!(
|
||||||
|
request.header("x-amz-content-sha256").as_deref(),
|
||||||
|
Some("UNSIGNED-PAYLOAD"),
|
||||||
|
"the client signs the same payload marker RustFS peer calls use"
|
||||||
|
);
|
||||||
|
assert_eq!(request.header("content-type").as_deref(), Some("application/json"));
|
||||||
|
assert!(request.body.contains("\"recursive\":true"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn query_sends_client_token_on_the_same_path() {
|
||||||
|
let body = r#"{"summary":"running","detail":"","settings":{"recursive":false},"items":[],"truncated":false}"#;
|
||||||
|
let server = TestServer::spawn(body, 200).await;
|
||||||
|
let client = AdminClient::new(&format!("http://{}", server.addr), "ak", "sk").unwrap();
|
||||||
|
|
||||||
|
let status = client
|
||||||
|
.heal_status(Some("bucket"), None, "token-1")
|
||||||
|
.await
|
||||||
|
.expect("status decodes");
|
||||||
|
assert_eq!(status.summary, "running");
|
||||||
|
let request = server.recorded();
|
||||||
|
assert_eq!(request.path, "/rustfs/admin/v3/heal/bucket");
|
||||||
|
assert!(request.query.contains("clientToken=token-1"));
|
||||||
|
assert!(!request.query.contains("forceStop"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn stop_without_token_takes_the_path_cancel_branch() {
|
||||||
|
let server = TestServer::spawn(r#"{"clientToken":"path","clientAddress":"c","startTime":"t"}"#, 200).await;
|
||||||
|
let client = AdminClient::new(&format!("http://{}", server.addr), "ak", "sk").unwrap();
|
||||||
|
|
||||||
|
let outcome = client.heal_stop(Some("bucket"), None, None).await.expect("path stop decodes");
|
||||||
|
assert!(matches!(outcome, super::HealStopOutcome::PathStopped(_)));
|
||||||
|
let request = server.recorded();
|
||||||
|
assert!(request.query.contains("forceStop=true"));
|
||||||
|
assert!(!request.query.contains("clientToken"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn http_error_status_maps_to_a_typed_error_with_body() {
|
||||||
|
let server = TestServer::spawn(r#"{"code":"AccessDenied","message":"denied"}"#, 403).await;
|
||||||
|
let client = AdminClient::new(&format!("http://{}", server.addr), "ak", "sk").unwrap();
|
||||||
|
let err = client.scanner_status().await.unwrap_err();
|
||||||
|
match err {
|
||||||
|
AdminClientError::HttpStatus { status, body } => {
|
||||||
|
assert_eq!(status, 403);
|
||||||
|
assert!(body.contains("AccessDenied"));
|
||||||
|
}
|
||||||
|
other => panic!("expected HttpStatus, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn malformed_success_body_maps_to_a_decode_error() {
|
||||||
|
let server = TestServer::spawn("not json", 200).await;
|
||||||
|
let client = AdminClient::new(&format!("http://{}", server.addr), "ak", "sk").unwrap();
|
||||||
|
assert!(matches!(client.scanner_status().await.unwrap_err(), AdminClientError::Decode { .. }));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One recorded request, parsed off the wire with the minimum needed for
|
||||||
|
/// assertions: method, path, query, headers, body.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
struct RecordedRequest {
|
||||||
|
method: String,
|
||||||
|
path: String,
|
||||||
|
query: String,
|
||||||
|
headers: Vec<(String, String)>,
|
||||||
|
body: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RecordedRequest {
|
||||||
|
fn header(&self, name: &str) -> Option<String> {
|
||||||
|
self.headers
|
||||||
|
.iter()
|
||||||
|
.find(|(key, _)| key.eq_ignore_ascii_case(name))
|
||||||
|
.map(|(_, value)| value.clone())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Minimal HTTP/1.1 server: one canned response per connection, every
|
||||||
|
/// request recorded behind an `Arc<Mutex>`. Deliberately dependency-free —
|
||||||
|
/// the assertions only need the raw request bytes.
|
||||||
|
struct TestServer {
|
||||||
|
addr: std::net::SocketAddr,
|
||||||
|
requests: Arc<Mutex<Vec<RecordedRequest>>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TestServer {
|
||||||
|
async fn spawn(response_body: &'static str, status: u16) -> Self {
|
||||||
|
use tokio::io::{AsyncReadExt, AsyncWriteExt};
|
||||||
|
|
||||||
|
let listener = tokio::net::TcpListener::bind("127.0.0.1:0")
|
||||||
|
.await
|
||||||
|
.expect("bind ephemeral port");
|
||||||
|
let addr = listener.local_addr().expect("local addr");
|
||||||
|
let requests: Arc<Mutex<Vec<RecordedRequest>>> = Arc::new(Mutex::new(Vec::new()));
|
||||||
|
|
||||||
|
let recorded = requests.clone();
|
||||||
|
tokio::spawn(async move {
|
||||||
|
let reason = if status == 200 { "OK" } else { "Forbidden" };
|
||||||
|
let response = format!(
|
||||||
|
"HTTP/1.1 {status} {reason}\r\ncontent-type: application/json\r\ncontent-length: {}\r\nconnection: close\r\n\r\n{response_body}",
|
||||||
|
response_body.len()
|
||||||
|
);
|
||||||
|
// Each request is a fresh connection (connection: close); a
|
||||||
|
// bounded loop serves every call a test makes while letting
|
||||||
|
// the task exit instead of lingering for the whole process.
|
||||||
|
for _ in 0..16 {
|
||||||
|
let Ok((mut stream, _)) = listener.accept().await else {
|
||||||
|
break;
|
||||||
|
};
|
||||||
|
let mut buffer = Vec::with_capacity(2048);
|
||||||
|
let mut chunk = [0u8; 2048];
|
||||||
|
// Read headers plus content-length body, or stop on close.
|
||||||
|
loop {
|
||||||
|
if let Some(end) = find_header_end(&buffer) {
|
||||||
|
let content_length = extract_content_length(&buffer[..end]);
|
||||||
|
if buffer.len() >= end + content_length {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let n = match stream.read(&mut chunk).await {
|
||||||
|
Ok(0) | Err(_) => break,
|
||||||
|
Ok(n) => n,
|
||||||
|
};
|
||||||
|
buffer.extend_from_slice(&chunk[..n]);
|
||||||
|
if buffer.len() > 64 * 1024 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if let Some(request) = parse_request(&buffer) {
|
||||||
|
recorded.lock().expect("recorded lock").push(request);
|
||||||
|
}
|
||||||
|
let _ = stream.write_all(response.as_bytes()).await;
|
||||||
|
let _ = stream.shutdown().await;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
Self { addr, requests }
|
||||||
|
}
|
||||||
|
|
||||||
|
fn recorded(&self) -> RecordedRequest {
|
||||||
|
self.requests
|
||||||
|
.lock()
|
||||||
|
.expect("recorded lock")
|
||||||
|
.last()
|
||||||
|
.cloned()
|
||||||
|
.expect("the client call must have produced one recorded request")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn find_header_end(buffer: &[u8]) -> Option<usize> {
|
||||||
|
buffer.windows(4).position(|window| window == b"\r\n\r\n").map(|pos| pos + 4)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn extract_content_length(headers: &[u8]) -> usize {
|
||||||
|
let text = String::from_utf8_lossy(headers).to_ascii_lowercase();
|
||||||
|
text.lines()
|
||||||
|
.find_map(|line| line.strip_prefix("content-length:"))
|
||||||
|
.and_then(|value| value.trim().parse().ok())
|
||||||
|
.unwrap_or(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_request(raw: &[u8]) -> Option<RecordedRequest> {
|
||||||
|
let end = find_header_end(raw)?;
|
||||||
|
let head = String::from_utf8_lossy(&raw[..end]);
|
||||||
|
let body = String::from_utf8_lossy(&raw[end..]).into_owned();
|
||||||
|
let mut lines = head.lines();
|
||||||
|
let request_line = lines.next()?;
|
||||||
|
let mut parts = request_line.split_whitespace();
|
||||||
|
let method = parts.next()?.to_string();
|
||||||
|
let target = parts.next()?.to_string();
|
||||||
|
let (path, query) = match target.split_once('?') {
|
||||||
|
Some((path, query)) => (path.to_string(), query.to_string()),
|
||||||
|
None => (target, String::new()),
|
||||||
|
};
|
||||||
|
let headers = lines
|
||||||
|
.filter_map(|line| line.split_once(':'))
|
||||||
|
.map(|(name, value)| (name.trim().to_string(), value.trim().to_string()))
|
||||||
|
.collect();
|
||||||
|
Some(RecordedRequest {
|
||||||
|
method,
|
||||||
|
path,
|
||||||
|
query,
|
||||||
|
headers,
|
||||||
|
body,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -12,6 +12,7 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
|
pub mod client;
|
||||||
pub mod group;
|
pub mod group;
|
||||||
pub mod heal_commands;
|
pub mod heal_commands;
|
||||||
pub mod health;
|
pub mod health;
|
||||||
@@ -25,6 +26,7 @@ pub mod trace;
|
|||||||
pub mod user;
|
pub mod user;
|
||||||
pub mod utils;
|
pub mod utils;
|
||||||
|
|
||||||
|
pub use client::*;
|
||||||
pub use group::*;
|
pub use group::*;
|
||||||
pub use info_commands::*;
|
pub use info_commands::*;
|
||||||
pub use policy::*;
|
pub use policy::*;
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ pub struct ServiceTraceOpts {
|
|||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(dead_code)]
|
||||||
impl ServiceTraceOpts {
|
impl ServiceTraceOpts {
|
||||||
fn trace_types(&self) -> TraceType {
|
pub fn trace_types(&self) -> TraceType {
|
||||||
let mut tt = TraceType::default();
|
let mut tt = TraceType::default();
|
||||||
tt.set_if(self.s3, &TraceType::S3);
|
tt.set_if(self.s3, &TraceType::S3);
|
||||||
tt.set_if(self.internal, &TraceType::INTERNAL);
|
tt.set_if(self.internal, &TraceType::INTERNAL);
|
||||||
@@ -72,6 +72,14 @@ impl ServiceTraceOpts {
|
|||||||
tt
|
tt
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn only_errors(&self) -> bool {
|
||||||
|
self.only_errors
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn threshold(&self) -> Duration {
|
||||||
|
self.threshold
|
||||||
|
}
|
||||||
|
|
||||||
pub fn parse_params(&mut self, uri: &Uri) -> Result<(), String> {
|
pub fn parse_params(&mut self, uri: &Uri) -> Result<(), String> {
|
||||||
let query_pairs: HashMap<_, _> = uri
|
let query_pairs: HashMap<_, _> = uri
|
||||||
.query()
|
.query()
|
||||||
|
|||||||
@@ -258,7 +258,7 @@ pub struct SRLDAPUser {
|
|||||||
pub api_version: Option<String>,
|
pub api_version: Option<String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, Default)]
|
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||||
pub struct SRIAMUser {
|
pub struct SRIAMUser {
|
||||||
#[serde(rename = "accessKey", default)]
|
#[serde(rename = "accessKey", default)]
|
||||||
pub access_key: String,
|
pub access_key: String,
|
||||||
@@ -270,7 +270,7 @@ pub struct SRIAMUser {
|
|||||||
pub api_version: Option<String>,
|
pub api_version: Option<String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, Default)]
|
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||||
pub struct SRGroupInfo {
|
pub struct SRGroupInfo {
|
||||||
#[serde(rename = "updateReq", default)]
|
#[serde(rename = "updateReq", default)]
|
||||||
pub update_req: GroupAddRemove,
|
pub update_req: GroupAddRemove,
|
||||||
@@ -346,7 +346,7 @@ pub struct SRCredInfo {
|
|||||||
pub api_version: Option<String>,
|
pub api_version: Option<String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Serialize, Deserialize, Default)]
|
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||||
pub struct SRIAMItem {
|
pub struct SRIAMItem {
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub r#type: String,
|
pub r#type: String,
|
||||||
|
|||||||
@@ -40,7 +40,10 @@ impl RuleEvents for RuleView {
|
|||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
struct CompiledRules {
|
struct CompiledRules {
|
||||||
// Keep RulesMap (can be used later if you want to make more complex judgments during the snapshot reading phase)
|
// Keep RulesMap (can be used later if you want to make more complex judgments during the snapshot reading phase)
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "speculative retention: the comment above keeps it for richer snapshot-time judgements that no code performs yet (backlog#1823)"
|
||||||
|
)]
|
||||||
rules_map: RulesMap,
|
rules_map: RulesMap,
|
||||||
// for RulesContainer::iter_rules
|
// for RulesContainer::iter_rules
|
||||||
rule_views: Vec<RuleView>,
|
rule_views: Vec<RuleView>,
|
||||||
|
|||||||
@@ -187,7 +187,6 @@ impl RulesMap {
|
|||||||
/// # Parameters
|
/// # Parameters
|
||||||
/// * `event_name` - The EventName from which to remove the rule.
|
/// * `event_name` - The EventName from which to remove the rule.
|
||||||
/// * `pattern` - The pattern of the rule to be removed.
|
/// * `pattern` - The pattern of the rule to be removed.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn remove_rule(&mut self, event_name: &EventName, pattern: &str) {
|
pub fn remove_rule(&mut self, event_name: &EventName, pattern: &str) {
|
||||||
let mut remove_event = false;
|
let mut remove_event = false;
|
||||||
|
|
||||||
@@ -209,7 +208,6 @@ impl RulesMap {
|
|||||||
///
|
///
|
||||||
/// # Parameters
|
/// # Parameters
|
||||||
/// * `event_names` - A slice of EventNames to be removed.
|
/// * `event_names` - A slice of EventNames to be removed.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn remove_rules(&mut self, event_names: &[EventName]) {
|
pub fn remove_rules(&mut self, event_names: &[EventName]) {
|
||||||
for event_name in event_names {
|
for event_name in event_names {
|
||||||
self.map.remove(event_name);
|
self.map.remove(event_name);
|
||||||
@@ -223,7 +221,6 @@ impl RulesMap {
|
|||||||
/// * `event_name` - The EventName to update.
|
/// * `event_name` - The EventName to update.
|
||||||
/// * `pattern` - The pattern of the rule to be updated.
|
/// * `pattern` - The pattern of the rule to be updated.
|
||||||
/// * `target_id` - The TargetID to be added.
|
/// * `target_id` - The TargetID to be added.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn update_rule(&mut self, event_name: EventName, pattern: String, target_id: TargetID) {
|
pub fn update_rule(&mut self, event_name: EventName, pattern: String, target_id: TargetID) {
|
||||||
self.map.entry(event_name).or_default().add(pattern, target_id);
|
self.map.entry(event_name).or_default().add(pattern, target_id);
|
||||||
self.total_events_mask |= event_name.mask(); // Update only the relevant bitmask
|
self.total_events_mask |= event_name.mask(); // Update only the relevant bitmask
|
||||||
|
|||||||
@@ -18,12 +18,6 @@ use rustfs_targets::arn::TargetID;
|
|||||||
/// TargetIDSet - A collection representation of TargetID.
|
/// TargetIDSet - A collection representation of TargetID.
|
||||||
pub type TargetIdSet = HashSet<TargetID>;
|
pub type TargetIdSet = HashSet<TargetID>;
|
||||||
|
|
||||||
/// Provides a Go-like method for TargetIdSet (can be implemented as trait if needed)
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub(crate) fn new_target_id_set(target_ids: Vec<TargetID>) -> TargetIdSet {
|
|
||||||
target_ids.into_iter().collect()
|
|
||||||
}
|
|
||||||
|
|
||||||
// HashSet has built-in clone, union, difference and other operations.
|
// HashSet has built-in clone, union, difference and other operations.
|
||||||
// But the Go version of the method returns a new Set, and the HashSet method is usually iterator or modify itself.
|
// But the Go version of the method returns a new Set, and the HashSet method is usually iterator or modify itself.
|
||||||
// If you need to exactly match Go's API style, you can add wrapper functions.
|
// If you need to exactly match Go's API style, you can add wrapper functions.
|
||||||
|
|||||||
@@ -427,7 +427,6 @@ pub enum DataSource {
|
|||||||
/// Write triggered
|
/// Write triggered
|
||||||
WriteTriggered,
|
WriteTriggered,
|
||||||
/// Fallback value
|
/// Fallback value
|
||||||
#[allow(dead_code)]
|
|
||||||
Fallback,
|
Fallback,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -603,7 +602,6 @@ impl WriteRecord {
|
|||||||
|
|
||||||
/// Hybrid strategy configuration
|
/// Hybrid strategy configuration
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct HybridStrategyConfig {
|
pub struct HybridStrategyConfig {
|
||||||
/// Scheduled update interval
|
/// Scheduled update interval
|
||||||
pub scheduled_update_interval: Duration,
|
pub scheduled_update_interval: Duration,
|
||||||
@@ -998,14 +996,12 @@ impl HybridCapacityManager {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Get cache age
|
/// Get cache age
|
||||||
#[allow(dead_code)]
|
|
||||||
pub async fn get_cache_age(&self) -> Option<Duration> {
|
pub async fn get_cache_age(&self) -> Option<Duration> {
|
||||||
let cache = self.cache.read().await;
|
let cache = self.cache.read().await;
|
||||||
cache.as_ref().map(|c| c.last_update.elapsed())
|
cache.as_ref().map(|c| c.last_update.elapsed())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get write frequency (writes/minute)
|
/// Get write frequency (writes/minute)
|
||||||
#[allow(dead_code)]
|
|
||||||
pub async fn get_write_frequency(&self) -> usize {
|
pub async fn get_write_frequency(&self) -> usize {
|
||||||
let record = &self.write_record;
|
let record = &self.write_record;
|
||||||
record.recent_write_count(record.monotonic_second())
|
record.recent_write_count(record.monotonic_second())
|
||||||
@@ -1300,7 +1296,6 @@ pub fn get_capacity_manager() -> Arc<HybridCapacityManager> {
|
|||||||
/// .update_capacity(CapacityUpdate::exact(1000, 0), DataSource::RealTime)
|
/// .update_capacity(CapacityUpdate::exact(1000, 0), DataSource::RealTime)
|
||||||
/// .await;
|
/// .await;
|
||||||
/// ```
|
/// ```
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn create_isolated_manager(config: HybridStrategyConfig) -> Arc<HybridCapacityManager> {
|
pub fn create_isolated_manager(config: HybridStrategyConfig) -> Arc<HybridCapacityManager> {
|
||||||
Arc::new(HybridCapacityManager::new(config))
|
Arc::new(HybridCapacityManager::new(config))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -17,7 +17,6 @@ use std::time::Duration;
|
|||||||
/// Environment variable key for the global default metrics interval (seconds).
|
/// Environment variable key for the global default metrics interval (seconds).
|
||||||
pub const ENV_DEFAULT_METRICS_INTERVAL: &str = "RUSTFS_METRICS_DEFAULT_INTERVAL_SEC";
|
pub const ENV_DEFAULT_METRICS_INTERVAL: &str = "RUSTFS_METRICS_DEFAULT_INTERVAL_SEC";
|
||||||
/// Default interval for metrics collection if not specified otherwise.
|
/// Default interval for metrics collection if not specified otherwise.
|
||||||
#[allow(dead_code)]
|
|
||||||
pub const DEFAULT_METRICS_INTERVAL: Duration = Duration::from_secs(60);
|
pub const DEFAULT_METRICS_INTERVAL: Duration = Duration::from_secs(60);
|
||||||
|
|
||||||
/// Environment variable key for cluster metrics interval (seconds).
|
/// Environment variable key for cluster metrics interval (seconds).
|
||||||
|
|||||||
@@ -145,21 +145,18 @@ impl PrometheusMetric {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[inline]
|
#[inline]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn with_label(mut self, key: &'static str, value: impl Into<Cow<'static, str>>) -> Self {
|
pub fn with_label(mut self, key: &'static str, value: impl Into<Cow<'static, str>>) -> Self {
|
||||||
self.labels.push((key, value.into()));
|
self.labels.push((key, value.into()));
|
||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
#[inline]
|
#[inline]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn with_label_owned(mut self, key: &'static str, value: String) -> Self {
|
pub fn with_label_owned(mut self, key: &'static str, value: String) -> Self {
|
||||||
self.labels.push((key, Cow::Owned(value)));
|
self.labels.push((key, Cow::Owned(value)));
|
||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
#[inline]
|
#[inline]
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn with_labels(mut self, labels: Vec<(&'static str, Cow<'static, str>)>) -> Self {
|
pub fn with_labels(mut self, labels: Vec<(&'static str, Cow<'static, str>)>) -> Self {
|
||||||
self.labels = labels;
|
self.labels = labels;
|
||||||
self
|
self
|
||||||
|
|||||||
@@ -16,7 +16,6 @@ use crate::{MetricName, MetricNamespace, MetricSubsystem, MetricType};
|
|||||||
use std::collections::HashSet;
|
use std::collections::HashSet;
|
||||||
|
|
||||||
/// MetricDescriptor - Metric descriptors
|
/// MetricDescriptor - Metric descriptors
|
||||||
#[allow(dead_code)]
|
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct MetricDescriptor {
|
pub struct MetricDescriptor {
|
||||||
pub name: MetricName,
|
pub name: MetricName,
|
||||||
@@ -52,7 +51,6 @@ impl MetricDescriptor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Get the full metric name in Prometheus style: <namespace>_<subsystem>_<name>
|
/// Get the full metric name in Prometheus style: <namespace>_<subsystem>_<name>
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn get_full_metric_name(&self) -> String {
|
pub fn get_full_metric_name(&self) -> String {
|
||||||
let namespace = self.namespace.as_str();
|
let namespace = self.namespace.as_str();
|
||||||
let formatted_subsystem = self.subsystem.as_str();
|
let formatted_subsystem = self.subsystem.as_str();
|
||||||
@@ -61,7 +59,6 @@ impl MetricDescriptor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// check whether the label is in the label set
|
/// check whether the label is in the label set
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn has_label(&mut self, label: &str) -> bool {
|
pub fn has_label(&mut self, label: &str) -> bool {
|
||||||
self.get_label_set().contains(label)
|
self.get_label_set().contains(label)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -13,7 +13,6 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
/// The metric name is the individual name of the metric
|
/// The metric name is the individual name of the metric
|
||||||
#[allow(dead_code)]
|
|
||||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
pub enum MetricName {
|
pub enum MetricName {
|
||||||
// The generic metric name
|
// The generic metric name
|
||||||
@@ -443,7 +442,6 @@ pub enum MetricName {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl MetricName {
|
impl MetricName {
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn as_str(&self) -> String {
|
pub fn as_str(&self) -> String {
|
||||||
match self {
|
match self {
|
||||||
Self::AuthTotal => "auth_total".to_string(),
|
Self::AuthTotal => "auth_total".to_string(),
|
||||||
|
|||||||
@@ -13,7 +13,6 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
/// MetricType - Indicates the type of indicator
|
/// MetricType - Indicates the type of indicator
|
||||||
#[allow(dead_code)]
|
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
pub enum MetricType {
|
pub enum MetricType {
|
||||||
Counter,
|
Counter,
|
||||||
@@ -23,7 +22,6 @@ pub enum MetricType {
|
|||||||
|
|
||||||
impl MetricType {
|
impl MetricType {
|
||||||
/// convert the metric type to a string representation
|
/// convert the metric type to a string representation
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn as_str(&self) -> &'static str {
|
pub fn as_str(&self) -> &'static str {
|
||||||
match self {
|
match self {
|
||||||
Self::Counter => "counter",
|
Self::Counter => "counter",
|
||||||
@@ -34,7 +32,6 @@ impl MetricType {
|
|||||||
|
|
||||||
/// Convert the metric type to the Prometheus value type
|
/// Convert the metric type to the Prometheus value type
|
||||||
/// In a Rust implementation, this might return the corresponding Prometheus Rust client type
|
/// In a Rust implementation, this might return the corresponding Prometheus Rust client type
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn as_prom(&self) -> &'static str {
|
pub fn as_prom(&self) -> &'static str {
|
||||||
match self {
|
match self {
|
||||||
Self::Counter => "counter.",
|
Self::Counter => "counter.",
|
||||||
|
|||||||
@@ -56,7 +56,6 @@ pub fn new_gauge_md(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// create a new histogram indicator descriptor
|
/// create a new histogram indicator descriptor
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn new_histogram_md(
|
pub fn new_histogram_md(
|
||||||
name: impl Into<MetricName>,
|
name: impl Into<MetricName>,
|
||||||
help: impl Into<String>,
|
help: impl Into<String>,
|
||||||
|
|||||||
@@ -19,7 +19,6 @@ pub enum MetricNamespace {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl MetricNamespace {
|
impl MetricNamespace {
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn as_str(&self) -> &'static str {
|
pub fn as_str(&self) -> &'static str {
|
||||||
match self {
|
match self {
|
||||||
Self::RustFS => "rustfs",
|
Self::RustFS => "rustfs",
|
||||||
|
|||||||
@@ -14,7 +14,6 @@
|
|||||||
|
|
||||||
/// Format the path to the metric name format
|
/// Format the path to the metric name format
|
||||||
/// Replace '/' and '-' with '_'
|
/// Replace '/' and '-' with '_'
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn format_path_to_metric_name(path: &str) -> String {
|
pub fn format_path_to_metric_name(path: &str) -> String {
|
||||||
path.trim_start_matches('/').replace(['/', '-'], "_")
|
path.trim_start_matches('/').replace(['/', '-'], "_")
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -102,7 +102,6 @@ impl MetricSubsystem {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Get the formatted metric name format string
|
/// Get the formatted metric name format string
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn as_str(&self) -> String {
|
pub fn as_str(&self) -> String {
|
||||||
format_path_to_metric_name(self.path())
|
format_path_to_metric_name(self.path())
|
||||||
}
|
}
|
||||||
@@ -151,7 +150,6 @@ impl MetricSubsystem {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// A convenient way to create custom subsystems directly
|
/// A convenient way to create custom subsystems directly
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn new(path: impl Into<String>) -> Self {
|
pub fn new(path: impl Into<String>) -> Self {
|
||||||
Self::Custom(path.into())
|
Self::Custom(path.into())
|
||||||
}
|
}
|
||||||
@@ -176,7 +174,6 @@ impl std::fmt::Display for MetricSubsystem {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub mod subsystems {
|
pub mod subsystems {
|
||||||
use super::MetricSubsystem;
|
use super::MetricSubsystem;
|
||||||
|
|
||||||
|
|||||||
@@ -38,7 +38,10 @@ pub enum Rotation {
|
|||||||
Minutely,
|
Minutely,
|
||||||
Hourly,
|
Hourly,
|
||||||
Daily,
|
Daily,
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "constructed only by this file's rolling-appender tests; the lib target cannot see them (backlog#1823)"
|
||||||
|
)]
|
||||||
Never,
|
Never,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -219,10 +219,6 @@ impl PartialEq for Functions {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Serialize, Deserialize)]
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub struct Value;
|
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use crate::policy::Functions;
|
use crate::policy::Functions;
|
||||||
|
|||||||
@@ -12,7 +12,6 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn is_simple_match<P, N>(pattern: P, name: N) -> bool
|
pub fn is_simple_match<P, N>(pattern: P, name: N) -> bool
|
||||||
where
|
where
|
||||||
P: AsRef<str>,
|
P: AsRef<str>,
|
||||||
@@ -29,7 +28,10 @@ where
|
|||||||
inner_match(pattern, name, false)
|
inner_match(pattern, name, false)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "prefix-matcher asserted by this file's tests; no production caller yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub fn is_match_as_pattern_prefix<P, N>(pattern: P, text: N) -> bool
|
pub fn is_match_as_pattern_prefix<P, N>(pattern: P, text: N) -> bool
|
||||||
where
|
where
|
||||||
P: AsRef<str>,
|
P: AsRef<str>,
|
||||||
|
|||||||
@@ -49,7 +49,6 @@ pub struct IndexInfo {
|
|||||||
pub uncompressed_offset: i64,
|
pub uncompressed_offset: i64,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
impl Index {
|
impl Index {
|
||||||
pub fn new() -> Self {
|
pub fn new() -> Self {
|
||||||
Self {
|
Self {
|
||||||
@@ -60,14 +59,6 @@ impl Index {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
fn reset(&mut self, max_block: usize) {
|
|
||||||
self.est_block_uncomp = max_block as i64;
|
|
||||||
self.total_compressed = -1;
|
|
||||||
self.total_uncompressed = -1;
|
|
||||||
self.info.clear();
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn len(&self) -> usize {
|
pub fn len(&self) -> usize {
|
||||||
self.info.len()
|
self.info.len()
|
||||||
}
|
}
|
||||||
@@ -511,47 +502,6 @@ fn read_varint(buf: &[u8]) -> io::Result<(i64, usize)> {
|
|||||||
Err(io::Error::new(io::ErrorKind::UnexpectedEof, "unexpected EOF"))
|
Err(io::Error::new(io::ErrorKind::UnexpectedEof, "unexpected EOF"))
|
||||||
}
|
}
|
||||||
|
|
||||||
// Helper functions for index header manipulation
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn remove_index_headers(b: &[u8]) -> Option<&[u8]> {
|
|
||||||
if b.len() < 4 + S2_INDEX_TRAILER.len() {
|
|
||||||
return None;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Skip size
|
|
||||||
let b = &b[4..];
|
|
||||||
|
|
||||||
// Check trailer
|
|
||||||
if !b.starts_with(S2_INDEX_TRAILER) {
|
|
||||||
return None;
|
|
||||||
}
|
|
||||||
|
|
||||||
Some(&b[S2_INDEX_TRAILER.len()..])
|
|
||||||
}
|
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
pub fn restore_index_headers(in_data: &[u8]) -> Vec<u8> {
|
|
||||||
if in_data.is_empty() {
|
|
||||||
return Vec::new();
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut b = Vec::with_capacity(4 + S2_INDEX_HEADER.len() + in_data.len() + S2_INDEX_TRAILER.len() + 4);
|
|
||||||
b.extend_from_slice(&[0x50, 0x2A, 0x4D, 0x18]);
|
|
||||||
b.extend_from_slice(S2_INDEX_HEADER);
|
|
||||||
b.extend_from_slice(in_data);
|
|
||||||
|
|
||||||
let total_size = (b.len() + 4 + S2_INDEX_TRAILER.len()) as u32;
|
|
||||||
b.extend_from_slice(&total_size.to_le_bytes());
|
|
||||||
b.extend_from_slice(S2_INDEX_TRAILER);
|
|
||||||
|
|
||||||
let chunk_len = b.len() - 4;
|
|
||||||
b[1] = chunk_len as u8;
|
|
||||||
b[2] = (chunk_len >> 8) as u8;
|
|
||||||
b[3] = (chunk_len >> 16) as u8;
|
|
||||||
|
|
||||||
b
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|||||||
@@ -108,6 +108,9 @@ temp-env = { workspace = true }
|
|||||||
tempfile = { workspace = true }
|
tempfile = { workspace = true }
|
||||||
uuid = { workspace = true, features = ["v4", "serde", "fast-rng", "macro-diagnostics"] }
|
uuid = { workspace = true, features = ["v4", "serde", "fast-rng", "macro-diagnostics"] }
|
||||||
tokio = { workspace = true, features = ["test-util", "fs", "rt-multi-thread"] }
|
tokio = { workspace = true, features = ["test-util", "fs", "rt-multi-thread"] }
|
||||||
|
# Test-only: pins the emitted scanner alert wire names against the canonical
|
||||||
|
# EventName string forms subscribers configure (rustfs/backlog#1868).
|
||||||
|
rustfs-s3-types.workspace = true
|
||||||
# Enables the shared MockWarmBackend / xl.meta assertion helpers exposed via
|
# Enables the shared MockWarmBackend / xl.meta assertion helpers exposed via
|
||||||
# the ecstore `api::tier::test_util` facade module (rustfs/backlog#1148 ilm-6).
|
# the ecstore `api::tier::test_util` facade module (rustfs/backlog#1148 ilm-6).
|
||||||
rustfs-ecstore = { workspace = true, features = ["test-util"] }
|
rustfs-ecstore = { workspace = true, features = ["test-util"] }
|
||||||
|
|||||||
@@ -28,7 +28,8 @@ use rustfs_common::heal_channel::HealScanMode;
|
|||||||
use rustfs_config::ENV_SCANNER_CACHE_SAVE_TIMEOUT_SECS;
|
use rustfs_config::ENV_SCANNER_CACHE_SAVE_TIMEOUT_SECS;
|
||||||
pub use rustfs_data_usage::{
|
pub use rustfs_data_usage::{
|
||||||
AllTierStats, BucketTargetUsageInfo, BucketUsageInfo, DATA_USAGE_OBJECT_NAME, DATA_USAGE_OBSERVED_OBJECT_NAME,
|
AllTierStats, BucketTargetUsageInfo, BucketUsageInfo, DATA_USAGE_OBJECT_NAME, DATA_USAGE_OBSERVED_OBJECT_NAME,
|
||||||
DataUsageEntry, DataUsageHash, DataUsageHashMap, DataUsageInfo, LEGACY_DATA_USAGE_OBJECT_NAME, TierStats, hash_path,
|
DataUsageEntry, DataUsageHash, DataUsageHashMap, DataUsageInfo, LEGACY_DATA_USAGE_OBJECT_NAME, PrefixUsageEntry,
|
||||||
|
PrefixUsageQuery, PrefixUsageSummary, TierStats, hash_path, prefix_usage_in_cache,
|
||||||
};
|
};
|
||||||
use rustfs_utils::path::{SLASH_SEPARATOR, path_join_buf};
|
use rustfs_utils::path::{SLASH_SEPARATOR, path_join_buf};
|
||||||
use tokio::time::{Duration, Instant, sleep, timeout};
|
use tokio::time::{Duration, Instant, sleep, timeout};
|
||||||
@@ -430,6 +431,13 @@ pub(crate) enum DataUsageCachePrepareOutcome {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl DataUsageCache {
|
impl DataUsageCache {
|
||||||
|
/// Prefix-level usage query over this (writer-side) cache; see
|
||||||
|
/// [`prefix_usage_in_cache`] for the semantics
|
||||||
|
/// (rustfs/backlog#1872).
|
||||||
|
pub fn prefix_usage(&self, bucket: &str, prefix: &str, max_entries: usize) -> Option<PrefixUsageQuery> {
|
||||||
|
prefix_usage_in_cache(&self.cache, bucket, prefix, max_entries)
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) fn prepare_for_scan(
|
pub(crate) fn prepare_for_scan(
|
||||||
&mut self,
|
&mut self,
|
||||||
name: &str,
|
name: &str,
|
||||||
|
|||||||
@@ -53,6 +53,7 @@ use tokio_util::sync::CancellationToken;
|
|||||||
|
|
||||||
pub mod data_usage_define;
|
pub mod data_usage_define;
|
||||||
pub mod error;
|
pub mod error;
|
||||||
|
pub mod prefix_usage;
|
||||||
mod remote_scanner;
|
mod remote_scanner;
|
||||||
pub mod runtime_config;
|
pub mod runtime_config;
|
||||||
pub mod scanner;
|
pub mod scanner;
|
||||||
@@ -64,6 +65,7 @@ pub(crate) mod storage_api;
|
|||||||
|
|
||||||
pub use data_usage_define::*;
|
pub use data_usage_define::*;
|
||||||
pub use error::ScannerError;
|
pub use error::ScannerError;
|
||||||
|
pub use prefix_usage::{BucketPrefixUsageResponse, bucket_prefix_usage, invalidate_prefix_usage_cache};
|
||||||
pub use remote_scanner::{
|
pub use remote_scanner::{
|
||||||
NS_SCANNER_MAX_REQUEST_BODY_SIZE, RemoteScannerAdmission, RemoteScannerRequest, admit_remote_scanner_request,
|
NS_SCANNER_MAX_REQUEST_BODY_SIZE, RemoteScannerAdmission, RemoteScannerRequest, admit_remote_scanner_request,
|
||||||
claim_remote_scanner_request, decode_remote_scanner_request, preflight_remote_scanner_request,
|
claim_remote_scanner_request, decode_remote_scanner_request, preflight_remote_scanner_request,
|
||||||
|
|||||||
@@ -0,0 +1,349 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! Prefix-level bucket usage for admin/console consumers (rustfs/backlog#1872,
|
||||||
|
//! MinIO `loadPrefixUsageFromBackend` parity).
|
||||||
|
//!
|
||||||
|
//! The per-bucket, per-set `.usage-cache.bin` objects already hold a
|
||||||
|
//! path-keyed prefix tree; this module reads every set's copy through that
|
||||||
|
//! set's own object layer (the hash-routed store path would always land on
|
||||||
|
//! one set), aggregates the overlapping trees, and serves the result from a
|
||||||
|
//! bounded 30-second cache. Bucket writes poke the cache through the
|
||||||
|
//! dirty-usage hook so a fresh scan is visible immediately.
|
||||||
|
|
||||||
|
use crate::data_usage_define::{DATA_USAGE_CACHE_NAME, DataUsageCache};
|
||||||
|
use crate::error::ScannerError;
|
||||||
|
use crate::storage_api::owner::{
|
||||||
|
EcstoreSetDisks, EcstoreStore, ecstore_is_reserved_or_invalid_bucket, ecstore_resolve_object_store_handle,
|
||||||
|
};
|
||||||
|
use futures::future::join_all;
|
||||||
|
use rustfs_data_usage::{PrefixUsageEntry, PrefixUsageSummary};
|
||||||
|
use std::collections::HashMap;
|
||||||
|
use std::sync::{Arc, Mutex};
|
||||||
|
use std::time::{Duration, SystemTime};
|
||||||
|
use tracing::{debug, warn};
|
||||||
|
|
||||||
|
const LOG_COMPONENT_SCANNER: &str = "scanner";
|
||||||
|
const LOG_SUBSYSTEM_PREFIX_USAGE: &str = "prefix_usage";
|
||||||
|
const EVENT_PREFIX_USAGE_CACHE_STATE: &str = "prefix_usage_cache_state";
|
||||||
|
|
||||||
|
/// How long a computed breakdown stays fresh. MinIO uses the same 30s for
|
||||||
|
/// its prefix-usage cache; bucket writes additionally invalidate on the spot.
|
||||||
|
const CACHE_TTL: Duration = Duration::from_secs(30);
|
||||||
|
/// Hard entry cap for the result cache; exceeded, expired entries go first
|
||||||
|
/// and the map clears rather than growing past the bound.
|
||||||
|
const CACHE_MAX_ENTRIES: usize = 128;
|
||||||
|
/// Per-set cache read budget. The underlying loader retries for up to a
|
||||||
|
/// minute per attempt on backend errors — far too long for an admin GET, so
|
||||||
|
/// a slow set degrades to "not reporting" instead of stalling the caller.
|
||||||
|
const PER_SET_LOAD_TIMEOUT: Duration = Duration::from_secs(5);
|
||||||
|
|
||||||
|
/// Aggregated prefix-usage answer across every erasure set.
|
||||||
|
#[derive(Clone, Debug, PartialEq, serde::Serialize)]
|
||||||
|
#[serde(rename_all = "camelCase")]
|
||||||
|
pub struct BucketPrefixUsageResponse {
|
||||||
|
pub bucket: String,
|
||||||
|
pub prefix: String,
|
||||||
|
pub usage: PrefixUsageSummary,
|
||||||
|
/// Every reporting set's prefix entry was compacted: the aggregate is
|
||||||
|
/// valid, the sub-prefix breakdown is empty on disk.
|
||||||
|
pub compacted: bool,
|
||||||
|
/// The sub-prefix breakdown is incomplete: at least one reporting set
|
||||||
|
/// had the prefix compacted (or absent while others found it), so its
|
||||||
|
/// objects cannot be attributed to a sub-prefix.
|
||||||
|
pub sub_prefixes_partial: bool,
|
||||||
|
/// The breakdown exceeded the caller's entry limit; largest remain.
|
||||||
|
pub truncated: bool,
|
||||||
|
pub sub_prefixes: Vec<PrefixUsageEntry>,
|
||||||
|
/// Sets whose cache held this bucket and prefix.
|
||||||
|
pub sets_reporting: usize,
|
||||||
|
pub sets_total: usize,
|
||||||
|
/// Newest `last_update` across reporting sets, unix seconds.
|
||||||
|
pub last_update_unix_secs: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
struct CachedResponse {
|
||||||
|
computed_at: std::time::Instant,
|
||||||
|
response: Arc<BucketPrefixUsageResponse>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cache key: (lowercased bucket, normalized prefix, max entries).
|
||||||
|
type PrefixUsageCacheKey = (String, String, usize);
|
||||||
|
type PrefixUsageCacheMap = Option<HashMap<PrefixUsageCacheKey, CachedResponse>>;
|
||||||
|
|
||||||
|
static PREFIX_USAGE_CACHE: Mutex<PrefixUsageCacheMap> = Mutex::new(None);
|
||||||
|
|
||||||
|
/// Drop cached results for `bucket` (empty string clears everything). Wired
|
||||||
|
/// into the dirty-usage recording path so a write makes the next prefix
|
||||||
|
/// query recompute instead of serving up to `CACHE_TTL` seconds of stale
|
||||||
|
/// numbers.
|
||||||
|
pub fn invalidate_prefix_usage_cache(bucket: &str) {
|
||||||
|
let mut guard = PREFIX_USAGE_CACHE.lock().unwrap_or_else(|poison| poison.into_inner());
|
||||||
|
let Some(map) = guard.as_mut() else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
if bucket.is_empty() {
|
||||||
|
map.clear();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
map.retain(|(cached_bucket, ..), _| !cached_bucket.eq_ignore_ascii_case(bucket));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Query prefix usage for `bucket` (arbitrary `prefix`, empty = whole
|
||||||
|
/// bucket), merging every erasure set's own cache copy. `max_entries` bounds
|
||||||
|
/// the sub-prefix rows (largest first).
|
||||||
|
pub async fn bucket_prefix_usage(
|
||||||
|
bucket: &str,
|
||||||
|
prefix: &str,
|
||||||
|
max_entries: usize,
|
||||||
|
) -> Result<BucketPrefixUsageResponse, ScannerError> {
|
||||||
|
if ecstore_is_reserved_or_invalid_bucket(bucket, true) {
|
||||||
|
return Err(ScannerError::Other(format!("invalid bucket name: {bucket}")));
|
||||||
|
}
|
||||||
|
let normalized_prefix = prefix.trim_matches('/').to_string();
|
||||||
|
let cache_key = (bucket.to_ascii_lowercase(), normalized_prefix.clone(), max_entries);
|
||||||
|
if let Some(response) = lookup_cached(&cache_key) {
|
||||||
|
return Ok((*response).clone());
|
||||||
|
}
|
||||||
|
|
||||||
|
let store = ecstore_resolve_object_store_handle()
|
||||||
|
.ok_or_else(|| ScannerError::Other("object store is not initialized".to_string()))?;
|
||||||
|
let response = Arc::new(compute_prefix_usage(store, bucket, &normalized_prefix, max_entries).await);
|
||||||
|
store_cached(cache_key, response.clone());
|
||||||
|
Ok((*response).clone())
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn compute_prefix_usage(
|
||||||
|
store: Arc<EcstoreStore>,
|
||||||
|
bucket: &str,
|
||||||
|
prefix: &str,
|
||||||
|
max_entries: usize,
|
||||||
|
) -> BucketPrefixUsageResponse {
|
||||||
|
let sets: Vec<Arc<EcstoreSetDisks>> = store.all_set_disks();
|
||||||
|
let sets_total = sets.len();
|
||||||
|
let cache_name = format!("{bucket}/{DATA_USAGE_CACHE_NAME}");
|
||||||
|
|
||||||
|
let per_set = join_all(sets.into_iter().map(|set| {
|
||||||
|
let cache_name = cache_name.clone();
|
||||||
|
async move {
|
||||||
|
let mut cache = DataUsageCache::default();
|
||||||
|
// A set that has never scanned this bucket (or cannot be read
|
||||||
|
// within the budget) reports nothing — the remaining sets still
|
||||||
|
// produce a usable, flagged answer.
|
||||||
|
let loaded = match tokio::time::timeout(PER_SET_LOAD_TIMEOUT, cache.load(set, &cache_name)).await {
|
||||||
|
Ok(Ok(())) => cache,
|
||||||
|
Ok(Err(err)) => {
|
||||||
|
debug!(
|
||||||
|
target: "rustfs::scanner::prefix_usage",
|
||||||
|
event = EVENT_PREFIX_USAGE_CACHE_STATE,
|
||||||
|
component = LOG_COMPONENT_SCANNER,
|
||||||
|
subsystem = LOG_SUBSYSTEM_PREFIX_USAGE,
|
||||||
|
bucket = %bucket,
|
||||||
|
state = "set_load_failed",
|
||||||
|
error = %err,
|
||||||
|
"Prefix usage set cache load failed"
|
||||||
|
);
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Err(_) => {
|
||||||
|
warn!(
|
||||||
|
target: "rustfs::scanner::prefix_usage",
|
||||||
|
event = EVENT_PREFIX_USAGE_CACHE_STATE,
|
||||||
|
component = LOG_COMPONENT_SCANNER,
|
||||||
|
subsystem = LOG_SUBSYSTEM_PREFIX_USAGE,
|
||||||
|
bucket = %bucket,
|
||||||
|
state = "set_load_timeout",
|
||||||
|
"Prefix usage set cache load timed out"
|
||||||
|
);
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
if loaded.info.name != bucket {
|
||||||
|
// Empty or stale-scoped cache: this set has no data for the bucket.
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let last_update = loaded.info.last_update;
|
||||||
|
let query = loaded.prefix_usage(bucket, prefix, max_entries);
|
||||||
|
Some((query, last_update))
|
||||||
|
}
|
||||||
|
}))
|
||||||
|
.await;
|
||||||
|
|
||||||
|
let mut usage = PrefixUsageSummary::default();
|
||||||
|
let mut sub_prefix_map: HashMap<String, PrefixUsageSummary> = HashMap::new();
|
||||||
|
let mut sets_reporting = 0usize;
|
||||||
|
let mut reporting_but_absent = 0usize;
|
||||||
|
let mut any_compacted = false;
|
||||||
|
let mut all_compacted = true;
|
||||||
|
let mut truncated = false;
|
||||||
|
let mut last_update: Option<SystemTime> = None;
|
||||||
|
|
||||||
|
for (query, set_last_update) in per_set.into_iter().flatten() {
|
||||||
|
// last_update counts every set that has scanned the bucket, even
|
||||||
|
// when the prefix itself is absent on that set.
|
||||||
|
if let Some(set_last_update) = set_last_update
|
||||||
|
&& last_update.map(|current| set_last_update > current).unwrap_or(true)
|
||||||
|
{
|
||||||
|
last_update = Some(set_last_update);
|
||||||
|
}
|
||||||
|
let Some(query) = query else {
|
||||||
|
// The set knows the bucket but not this prefix: legitimate when
|
||||||
|
// the prefix's objects all hash to other sets, but it means the
|
||||||
|
// breakdown below cannot attribute that set's (zero) objects.
|
||||||
|
reporting_but_absent += 1;
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
sets_reporting += 1;
|
||||||
|
usage.merge(&query.usage);
|
||||||
|
if query.compacted {
|
||||||
|
any_compacted = true;
|
||||||
|
} else {
|
||||||
|
all_compacted = false;
|
||||||
|
}
|
||||||
|
truncated |= query.truncated;
|
||||||
|
for entry in query.sub_prefixes {
|
||||||
|
sub_prefix_map.entry(entry.prefix).or_default().merge(&entry.usage);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut sub_prefixes: Vec<PrefixUsageEntry> = sub_prefix_map
|
||||||
|
.into_iter()
|
||||||
|
.map(|(prefix, usage)| PrefixUsageEntry { prefix, usage })
|
||||||
|
.collect();
|
||||||
|
sub_prefixes.sort_by(|left, right| {
|
||||||
|
right
|
||||||
|
.usage
|
||||||
|
.size
|
||||||
|
.cmp(&left.usage.size)
|
||||||
|
.then_with(|| left.prefix.cmp(&right.prefix))
|
||||||
|
});
|
||||||
|
// Merged rows can exceed max_entries only when per-set truncation
|
||||||
|
// already flagged; enforce the caller bound on the merged view too.
|
||||||
|
if sub_prefixes.len() > max_entries {
|
||||||
|
truncated = true;
|
||||||
|
sub_prefixes.truncate(max_entries);
|
||||||
|
}
|
||||||
|
|
||||||
|
let found = sets_reporting > 0;
|
||||||
|
BucketPrefixUsageResponse {
|
||||||
|
bucket: bucket.to_string(),
|
||||||
|
prefix: prefix.to_string(),
|
||||||
|
usage,
|
||||||
|
compacted: found && all_compacted,
|
||||||
|
sub_prefixes_partial: any_compacted || reporting_but_absent > 0,
|
||||||
|
truncated,
|
||||||
|
sub_prefixes,
|
||||||
|
sets_reporting,
|
||||||
|
sets_total,
|
||||||
|
last_update_unix_secs: last_update
|
||||||
|
.and_then(|time| time.duration_since(SystemTime::UNIX_EPOCH).ok())
|
||||||
|
.map(|dur| dur.as_secs()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn lookup_cached(key: &(String, String, usize)) -> Option<Arc<BucketPrefixUsageResponse>> {
|
||||||
|
let mut guard = PREFIX_USAGE_CACHE.lock().unwrap_or_else(|poison| poison.into_inner());
|
||||||
|
let map = guard.as_mut()?;
|
||||||
|
let cached = map.get(key)?;
|
||||||
|
if cached.computed_at.elapsed() > CACHE_TTL {
|
||||||
|
map.remove(key);
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(cached.response.clone())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn store_cached(key: (String, String, usize), response: Arc<BucketPrefixUsageResponse>) {
|
||||||
|
let mut guard = PREFIX_USAGE_CACHE.lock().unwrap_or_else(|poison| poison.into_inner());
|
||||||
|
let map = guard.get_or_insert_with(HashMap::new);
|
||||||
|
// Bound the cache: drop expired entries first, and if the cap is still
|
||||||
|
// exceeded clear wholesale — the next queries recompute in milliseconds.
|
||||||
|
if map.len() >= CACHE_MAX_ENTRIES {
|
||||||
|
map.retain(|_, cached| cached.computed_at.elapsed() <= CACHE_TTL);
|
||||||
|
if map.len() >= CACHE_MAX_ENTRIES {
|
||||||
|
map.clear();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
map.insert(
|
||||||
|
key,
|
||||||
|
CachedResponse {
|
||||||
|
computed_at: std::time::Instant::now(),
|
||||||
|
response,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::{CACHE_MAX_ENTRIES, PREFIX_USAGE_CACHE, invalidate_prefix_usage_cache, store_cached};
|
||||||
|
use rustfs_data_usage::PrefixUsageSummary;
|
||||||
|
|
||||||
|
fn response(bucket: &str) -> super::BucketPrefixUsageResponse {
|
||||||
|
super::BucketPrefixUsageResponse {
|
||||||
|
bucket: bucket.to_string(),
|
||||||
|
prefix: String::new(),
|
||||||
|
usage: PrefixUsageSummary::default(),
|
||||||
|
compacted: false,
|
||||||
|
sub_prefixes_partial: false,
|
||||||
|
truncated: false,
|
||||||
|
sub_prefixes: Vec::new(),
|
||||||
|
sets_reporting: 1,
|
||||||
|
sets_total: 1,
|
||||||
|
last_update_unix_secs: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn seed(bucket: &str, prefix: &str) {
|
||||||
|
store_cached(
|
||||||
|
(bucket.to_ascii_lowercase(), prefix.to_string(), 10),
|
||||||
|
std::sync::Arc::new(response(bucket)),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn contains(bucket: &str, prefix: &str) -> bool {
|
||||||
|
PREFIX_USAGE_CACHE
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(|poison| poison.into_inner())
|
||||||
|
.as_ref()
|
||||||
|
.is_some_and(|map| map.contains_key(&(bucket.to_ascii_lowercase(), prefix.to_string(), 10)))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// All cache tests run inside one test to keep the process-global map
|
||||||
|
/// free of cross-test ordering (the flake class this module avoids).
|
||||||
|
#[test]
|
||||||
|
fn invalidation_scopes_to_bucket_and_cache_stays_bounded() {
|
||||||
|
invalidate_prefix_usage_cache("");
|
||||||
|
seed("alpha", "x");
|
||||||
|
seed("beta", "y");
|
||||||
|
|
||||||
|
// Case-insensitive bucket scoping.
|
||||||
|
invalidate_prefix_usage_cache("ALPHA");
|
||||||
|
assert!(!contains("alpha", "x"));
|
||||||
|
assert!(contains("beta", "y"));
|
||||||
|
|
||||||
|
// Wholesale clear.
|
||||||
|
invalidate_prefix_usage_cache("");
|
||||||
|
assert!(!contains("beta", "y"));
|
||||||
|
|
||||||
|
// Hard cap: overflow clears rather than grows.
|
||||||
|
for index in 0..=(CACHE_MAX_ENTRIES / 2) {
|
||||||
|
let bucket = format!("cap-bucket-{index}");
|
||||||
|
seed(&bucket, "a");
|
||||||
|
seed(&bucket, "b");
|
||||||
|
}
|
||||||
|
let guard = PREFIX_USAGE_CACHE.lock().unwrap_or_else(|poison| poison.into_inner());
|
||||||
|
let map = guard.as_ref().expect("seeded");
|
||||||
|
assert!(map.len() <= CACHE_MAX_ENTRIES, "cache must stay bounded, got {}", map.len());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -12,10 +12,10 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
use std::collections::HashSet;
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::fs::FileType;
|
use std::fs::FileType;
|
||||||
use std::io::ErrorKind;
|
use std::io::ErrorKind;
|
||||||
use std::sync::{Arc, Once};
|
use std::sync::{Arc, Mutex, Once};
|
||||||
use std::time::{Duration, Instant, SystemTime};
|
use std::time::{Duration, Instant, SystemTime};
|
||||||
|
|
||||||
use crate::ReplTargetSizeSummary;
|
use crate::ReplTargetSizeSummary;
|
||||||
@@ -32,6 +32,7 @@ use crate::scanner_io::{
|
|||||||
SCANNER_SKIP_FILE_ERROR, ScannerIODisk as _, is_scanner_metadata_corrupt_error, is_scanner_metadata_transient_error,
|
SCANNER_SKIP_FILE_ERROR, ScannerIODisk as _, is_scanner_metadata_corrupt_error, is_scanner_metadata_transient_error,
|
||||||
};
|
};
|
||||||
use crate::sleeper::DynamicSleeper;
|
use crate::sleeper::DynamicSleeper;
|
||||||
|
use crate::storage_api::owner::{EcstoreEventArgs, ecstore_send_event};
|
||||||
use metrics::{counter, describe_counter};
|
use metrics::{counter, describe_counter};
|
||||||
use rustfs_common::heal_channel::{
|
use rustfs_common::heal_channel::{
|
||||||
HEAL_DELETE_DANGLING, HealAdmissionDropReason, HealAdmissionResult, HealChannelPriority, HealChannelRequest,
|
HEAL_DELETE_DANGLING, HealAdmissionDropReason, HealAdmissionResult, HealChannelPriority, HealChannelRequest,
|
||||||
@@ -41,6 +42,7 @@ use rustfs_common::metrics::{
|
|||||||
CloseDiskGuard, IlmAction, Metric, Metrics, ScannerReplicationRepairKind, ScannerSourceWorkUpdate, ScannerWorkSource,
|
CloseDiskGuard, IlmAction, Metric, Metrics, ScannerReplicationRepairKind, ScannerSourceWorkUpdate, ScannerWorkSource,
|
||||||
UpdateCurrentPathFn, current_path_updater, global_metrics,
|
UpdateCurrentPathFn, current_path_updater, global_metrics,
|
||||||
};
|
};
|
||||||
|
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, trace_emit, trace_subscriber_count};
|
||||||
use rustfs_filemeta::{MetaCacheEntries, MetaCacheEntry, MetadataResolutionParams};
|
use rustfs_filemeta::{MetaCacheEntries, MetaCacheEntry, MetadataResolutionParams};
|
||||||
use rustfs_utils::path::{SLASH_SEPARATOR, path_join_buf};
|
use rustfs_utils::path::{SLASH_SEPARATOR, path_join_buf};
|
||||||
use s3s::dto::{BucketLifecycleConfiguration, ObjectLockConfiguration};
|
use s3s::dto::{BucketLifecycleConfiguration, ObjectLockConfiguration};
|
||||||
@@ -97,6 +99,101 @@ const METRIC_SCANNER_EXCESS_FOLDERS_TOTAL: &str = "rustfs_scanner_excess_folders
|
|||||||
const METRIC_SCANNER_PENDING_HEAL_PRUNE_TOTAL: &str = "rustfs_scanner_pending_heal_prune_total";
|
const METRIC_SCANNER_PENDING_HEAL_PRUNE_TOTAL: &str = "rustfs_scanner_pending_heal_prune_total";
|
||||||
const METRIC_SCANNER_PENDING_HEAL_MALFORMED_TOTAL: &str = "rustfs_scanner_pending_heal_malformed_total";
|
const METRIC_SCANNER_PENDING_HEAL_MALFORMED_TOTAL: &str = "rustfs_scanner_pending_heal_malformed_total";
|
||||||
const MAX_PENDING_SCANNER_HEAL_RETRIES_PER_BUCKET: usize = 128;
|
const MAX_PENDING_SCANNER_HEAL_RETRIES_PER_BUCKET: usize = 128;
|
||||||
|
|
||||||
|
// --- scanner excess alerts as S3 notification events (rustfs/backlog#1868) --
|
||||||
|
//
|
||||||
|
// The excess-versions / excess-version-size / excess-folders alerts were
|
||||||
|
// metrics-and-logs only; subscribers (consoles, external auditors) had no way
|
||||||
|
// to hear them. MinIO emits s3:ObjectManyVersions / s3:ObjectLargeVersions /
|
||||||
|
// s3:PrefixManyFolders for the same conditions — RustFS carries those as
|
||||||
|
// EventName::Scanner* with the wire names below. Without a cooldown a single
|
||||||
|
// over-threshold object would re-emit on every scan cycle (~a minute), so
|
||||||
|
// emissions are edge-held per (kind, bucket, object) for 24h.
|
||||||
|
|
||||||
|
/// `s3:Scanner:ManyVersions` (MinIO `s3:ObjectManyVersions`).
|
||||||
|
pub const EVENT_SCANNER_MANY_VERSIONS: &str = "s3:Scanner:ManyVersions";
|
||||||
|
/// `s3:Scanner:LargeVersions` (MinIO `s3:ObjectLargeVersions`).
|
||||||
|
pub const EVENT_SCANNER_LARGE_VERSIONS: &str = "s3:Scanner:LargeVersions";
|
||||||
|
/// `s3:Scanner:BigPrefix` (MinIO `s3:PrefixManyFolders`).
|
||||||
|
pub const EVENT_SCANNER_BIG_PREFIX: &str = "s3:Scanner:BigPrefix";
|
||||||
|
const ENV_SCANNER_ALERT_COOLDOWN_SECS: &str = "RUSTFS_SCANNER_ALERT_COOLDOWN_SECS";
|
||||||
|
const DEFAULT_SCANNER_ALERT_COOLDOWN_SECS: u64 = 86_400;
|
||||||
|
/// Hard cap on distinct cooldown keys; a pathological number of over-threshold
|
||||||
|
/// objects clears the map wholesale instead of growing without bound (the
|
||||||
|
/// worst case is one re-emission per still-hot key per scan cycle).
|
||||||
|
const MAX_SCANNER_ALERT_COOLDOWN_KEYS: usize = 4096;
|
||||||
|
|
||||||
|
/// Distinct alert kinds sharing one cooldown map.
|
||||||
|
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
|
||||||
|
enum ScannerAlertKind {
|
||||||
|
ManyVersions,
|
||||||
|
LargeVersions,
|
||||||
|
BigPrefix,
|
||||||
|
}
|
||||||
|
|
||||||
|
type ScannerAlertCooldownKey = (ScannerAlertKind, String, String);
|
||||||
|
type ScannerAlertCooldownMap = HashMap<ScannerAlertCooldownKey, Instant>;
|
||||||
|
|
||||||
|
static SCANNER_ALERT_EMISSION_COOLDOWN: Mutex<Option<ScannerAlertCooldownMap>> = Mutex::new(None);
|
||||||
|
|
||||||
|
fn scanner_alert_cooldown() -> Duration {
|
||||||
|
let raw = std::env::var(ENV_SCANNER_ALERT_COOLDOWN_SECS)
|
||||||
|
.ok()
|
||||||
|
.and_then(|v| v.parse::<u64>().ok());
|
||||||
|
Duration::from_secs(raw.unwrap_or(DEFAULT_SCANNER_ALERT_COOLDOWN_SECS))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Edge-held emission gate: returns `true` (and records the cooldown) only
|
||||||
|
/// when this (kind, bucket, object) last fired longer than the cooldown ago —
|
||||||
|
/// or never. Metrics and logs stay level-triggered every cycle; only the
|
||||||
|
/// notification events are held back.
|
||||||
|
fn scanner_alert_emission_allows(kind: ScannerAlertKind, bucket: &str, object: &str, cooldown: Duration) -> bool {
|
||||||
|
let key = (kind, bucket.to_string(), object.to_string());
|
||||||
|
let mut guard = SCANNER_ALERT_EMISSION_COOLDOWN
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(|poison| poison.into_inner());
|
||||||
|
let guard = guard.get_or_insert_with(ScannerAlertCooldownMap::new);
|
||||||
|
let now = Instant::now();
|
||||||
|
// Expired entries leave first; the cap is still exceeded only when live
|
||||||
|
// keys alone overflow it, in which case a wholesale clear trades one
|
||||||
|
// extra emission per hot key for a hard memory bound.
|
||||||
|
if guard.len() >= MAX_SCANNER_ALERT_COOLDOWN_KEYS {
|
||||||
|
guard.retain(|_, fired_at| now.duration_since(*fired_at) < cooldown);
|
||||||
|
if guard.len() >= MAX_SCANNER_ALERT_COOLDOWN_KEYS {
|
||||||
|
guard.clear();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
match guard.get(&key) {
|
||||||
|
Some(fired_at) if now.duration_since(*fired_at) < cooldown => false,
|
||||||
|
_ => {
|
||||||
|
guard.insert(key, now);
|
||||||
|
true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Emit a scanner alert as an S3 notification event through the standard
|
||||||
|
/// dispatch pipeline. Fire-and-forget: the notify layer owns delivery,
|
||||||
|
/// retry, and target filtering; the scanner never waits on it.
|
||||||
|
fn emit_scanner_alert_event(event_name: &str, bucket: &str, object: &str, size: i64, details: &[(&str, String)]) {
|
||||||
|
let mut req_params = HashMap::with_capacity(details.len());
|
||||||
|
for (key, value) in details {
|
||||||
|
req_params.insert((*key).to_string(), value.clone());
|
||||||
|
}
|
||||||
|
ecstore_send_event(EcstoreEventArgs {
|
||||||
|
event_name: event_name.to_string(),
|
||||||
|
bucket_name: bucket.to_string(),
|
||||||
|
object: crate::ScannerObjectInfo {
|
||||||
|
bucket: bucket.to_string(),
|
||||||
|
name: object.to_string(),
|
||||||
|
size,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
req_params,
|
||||||
|
user_agent: "Scanner".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
});
|
||||||
|
}
|
||||||
const MAX_PENDING_SCANNER_HEALS_PER_BUCKET: usize = 10_000;
|
const MAX_PENDING_SCANNER_HEALS_PER_BUCKET: usize = 10_000;
|
||||||
|
|
||||||
static SCANNER_INLINE_HEAL_WARN_ONCE: Once = Once::new();
|
static SCANNER_INLINE_HEAL_WARN_ONCE: Once = Once::new();
|
||||||
@@ -430,6 +527,113 @@ fn non_negative_i64_to_u64(value: i64) -> u64 {
|
|||||||
value.max(0) as u64
|
value.max(0) as u64
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn trace_start_instant() -> Option<Instant> {
|
||||||
|
(trace_subscriber_count() > 0).then(Instant::now)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn emit_scanner_folder_trace(root: &str, folder: &str, objects: u64, started_at: Option<Instant>, state: &'static str) {
|
||||||
|
let Some(started_at) = started_at else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
trace_emit(|| {
|
||||||
|
let (bucket, prefix) = path2_bucket_object_with_base_path(root, folder);
|
||||||
|
TraceEvent::new(TraceKind::Scanner, TraceFunc::ScannerFolder)
|
||||||
|
.with_bucket(bucket)
|
||||||
|
.with_object(prefix)
|
||||||
|
.with_duration(started_at.elapsed())
|
||||||
|
.with_attr("state", state)
|
||||||
|
.with_attr("objects", objects)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
fn emit_scanner_ilm_action_trace(
|
||||||
|
bucket: &str,
|
||||||
|
object: &str,
|
||||||
|
action: IlmAction,
|
||||||
|
count: u64,
|
||||||
|
queued: bool,
|
||||||
|
started_at: Option<Instant>,
|
||||||
|
) {
|
||||||
|
let Some(started_at) = started_at else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
|
||||||
|
let state = if queued { "queued" } else { "not_queued" };
|
||||||
|
trace_emit(|| {
|
||||||
|
TraceEvent::new(TraceKind::Scanner, TraceFunc::ScannerIlmAction)
|
||||||
|
.with_bucket(bucket)
|
||||||
|
.with_object(object)
|
||||||
|
.with_duration(started_at.elapsed())
|
||||||
|
.with_attr("state", state)
|
||||||
|
.with_attr("action", action.as_str())
|
||||||
|
.with_attr("count", count)
|
||||||
|
.with_attr("queued", queued)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
struct ScannerHealCandidateTraceContext {
|
||||||
|
bucket: String,
|
||||||
|
object: Option<String>,
|
||||||
|
version_id: Option<String>,
|
||||||
|
scan_mode: Option<HealScanMode>,
|
||||||
|
started_at: Instant,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn scanner_heal_candidate_trace_context(request: &HealChannelRequest) -> Option<ScannerHealCandidateTraceContext> {
|
||||||
|
let started_at = trace_start_instant()?;
|
||||||
|
Some(ScannerHealCandidateTraceContext {
|
||||||
|
bucket: request.bucket.clone(),
|
||||||
|
object: request.object_prefix.clone(),
|
||||||
|
version_id: request.object_version_id.clone(),
|
||||||
|
scan_mode: request.scan_mode,
|
||||||
|
started_at,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
struct ScannerHealCandidateTrace<'a> {
|
||||||
|
candidate_type: &'static str,
|
||||||
|
bucket: &'a str,
|
||||||
|
object: Option<&'a str>,
|
||||||
|
version_id: Option<&'a str>,
|
||||||
|
priority: HealChannelPriority,
|
||||||
|
scan_mode: Option<HealScanMode>,
|
||||||
|
result: Result<HealAdmissionResult, &'a str>,
|
||||||
|
started_at: Instant,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn emit_scanner_heal_candidate_trace(trace: ScannerHealCandidateTrace<'_>) {
|
||||||
|
trace_emit(|| {
|
||||||
|
let (state, admission, error) = match trace.result {
|
||||||
|
Ok(result) if result.is_admitted() => ("admitted", describe_heal_admission(result), None),
|
||||||
|
Ok(result) => ("not_admitted", describe_heal_admission(result), None),
|
||||||
|
Err(error) => ("submit_failed", "channel_error".to_string(), Some(error)),
|
||||||
|
};
|
||||||
|
let mut event = TraceEvent::new(TraceKind::Scanner, TraceFunc::ScannerHealCandidate)
|
||||||
|
.with_bucket(trace.bucket)
|
||||||
|
.with_duration(trace.started_at.elapsed())
|
||||||
|
.with_attr("state", state)
|
||||||
|
.with_attr("candidate_type", trace.candidate_type)
|
||||||
|
.with_attr("priority", heal_priority_label(trace.priority))
|
||||||
|
.with_attr("admission", admission);
|
||||||
|
|
||||||
|
if let Some(object) = trace.object {
|
||||||
|
event = event.with_object(object);
|
||||||
|
}
|
||||||
|
if let Some(version_id) = trace.version_id {
|
||||||
|
event = event.with_attr("version_id", version_id);
|
||||||
|
}
|
||||||
|
if let Some(scan_mode) = trace.scan_mode {
|
||||||
|
event = event.with_attr("scan_mode", scan_mode.as_str());
|
||||||
|
}
|
||||||
|
if let Some(error) = error {
|
||||||
|
event = event.with_attr("error", error);
|
||||||
|
}
|
||||||
|
|
||||||
|
event
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
fn apply_scanner_size_summary(into: &mut DataUsageEntry, summary: &SizeSummary) {
|
fn apply_scanner_size_summary(into: &mut DataUsageEntry, summary: &SizeSummary) {
|
||||||
into.size = into.size.saturating_add(summary.total_size);
|
into.size = into.size.saturating_add(summary.total_size);
|
||||||
into.versions = into.versions.saturating_add(summary.versions);
|
into.versions = into.versions.saturating_add(summary.versions);
|
||||||
@@ -677,9 +881,22 @@ async fn send_scanner_heal_request(
|
|||||||
request: HealChannelRequest,
|
request: HealChannelRequest,
|
||||||
) -> Result<HealAdmissionResult, ScannerError> {
|
) -> Result<HealAdmissionResult, ScannerError> {
|
||||||
let priority = request.priority;
|
let priority = request.priority;
|
||||||
|
let trace_context = scanner_heal_candidate_trace_context(&request);
|
||||||
match send_heal_request_with_admission(request).await {
|
match send_heal_request_with_admission(request).await {
|
||||||
Ok(result) => {
|
Ok(result) => {
|
||||||
record_heal_candidate_admission(candidate_type, priority, result);
|
record_heal_candidate_admission(candidate_type, priority, result);
|
||||||
|
if let Some(trace_context) = trace_context.as_ref() {
|
||||||
|
emit_scanner_heal_candidate_trace(ScannerHealCandidateTrace {
|
||||||
|
candidate_type,
|
||||||
|
bucket: &trace_context.bucket,
|
||||||
|
object: trace_context.object.as_deref(),
|
||||||
|
version_id: trace_context.version_id.as_deref(),
|
||||||
|
priority,
|
||||||
|
scan_mode: trace_context.scan_mode,
|
||||||
|
result: Ok(result),
|
||||||
|
started_at: trace_context.started_at,
|
||||||
|
});
|
||||||
|
}
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
Err(err) => {
|
Err(err) => {
|
||||||
@@ -690,6 +907,18 @@ async fn send_scanner_heal_request(
|
|||||||
"result" => "channel_error".to_string()
|
"result" => "channel_error".to_string()
|
||||||
)
|
)
|
||||||
.increment(1);
|
.increment(1);
|
||||||
|
if let Some(trace_context) = trace_context.as_ref() {
|
||||||
|
emit_scanner_heal_candidate_trace(ScannerHealCandidateTrace {
|
||||||
|
candidate_type,
|
||||||
|
bucket: &trace_context.bucket,
|
||||||
|
object: trace_context.object.as_deref(),
|
||||||
|
version_id: trace_context.version_id.as_deref(),
|
||||||
|
priority,
|
||||||
|
scan_mode: trace_context.scan_mode,
|
||||||
|
result: Err(err.as_str()),
|
||||||
|
started_at: trace_context.started_at,
|
||||||
|
});
|
||||||
|
}
|
||||||
Err(ScannerError::Other(err))
|
Err(ScannerError::Other(err))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -905,7 +1134,9 @@ impl ScannerItem {
|
|||||||
"Scanner lifecycle action dispatched"
|
"Scanner lifecycle action dispatched"
|
||||||
);
|
);
|
||||||
let done_ilm = Metrics::time_ilm(event.action);
|
let done_ilm = Metrics::time_ilm(event.action);
|
||||||
|
let trace_started_at = trace_start_instant();
|
||||||
let queued = apply_expiry_rule(event, &LcEventSrc::Scanner, oi).await;
|
let queued = apply_expiry_rule(event, &LcEventSrc::Scanner, oi).await;
|
||||||
|
emit_scanner_ilm_action_trace(&self.bucket, &oi.name, event.action, 1, queued, trace_started_at);
|
||||||
if record_scanner_ilm_action_if_queued(global_metrics(), event.action, 1, queued) {
|
if record_scanner_ilm_action_if_queued(global_metrics(), event.action, 1, queued) {
|
||||||
done_ilm(1)();
|
done_ilm(1)();
|
||||||
remaining_versions = 0;
|
remaining_versions = 0;
|
||||||
@@ -957,7 +1188,9 @@ impl ScannerItem {
|
|||||||
"Scanner lifecycle action dispatched"
|
"Scanner lifecycle action dispatched"
|
||||||
);
|
);
|
||||||
let done_ilm = Metrics::time_ilm(event.action);
|
let done_ilm = Metrics::time_ilm(event.action);
|
||||||
|
let trace_started_at = trace_start_instant();
|
||||||
let queued = apply_expiry_rule(event, &LcEventSrc::Scanner, oi).await;
|
let queued = apply_expiry_rule(event, &LcEventSrc::Scanner, oi).await;
|
||||||
|
emit_scanner_ilm_action_trace(&self.bucket, &oi.name, event.action, 1, queued, trace_started_at);
|
||||||
if record_scanner_ilm_action_if_queued(global_metrics(), event.action, 1, queued) {
|
if record_scanner_ilm_action_if_queued(global_metrics(), event.action, 1, queued) {
|
||||||
done_ilm(1)();
|
done_ilm(1)();
|
||||||
if !versioning_config.prefix_enabled(&self.object_path()) && event.action == IlmAction::DeleteAction {
|
if !versioning_config.prefix_enabled(&self.object_path()) && event.action == IlmAction::DeleteAction {
|
||||||
@@ -995,7 +1228,9 @@ impl ScannerItem {
|
|||||||
"Scanner lifecycle action dispatched"
|
"Scanner lifecycle action dispatched"
|
||||||
);
|
);
|
||||||
let done_ilm = Metrics::time_ilm(event.action);
|
let done_ilm = Metrics::time_ilm(event.action);
|
||||||
|
let trace_started_at = trace_start_instant();
|
||||||
let queued = apply_transition_rule(event, &LcEventSrc::Scanner, oi).await;
|
let queued = apply_transition_rule(event, &LcEventSrc::Scanner, oi).await;
|
||||||
|
emit_scanner_ilm_action_trace(&self.bucket, &oi.name, event.action, 1, queued, trace_started_at);
|
||||||
if record_scanner_ilm_action_if_queued(global_metrics(), event.action, 1, queued) {
|
if record_scanner_ilm_action_if_queued(global_metrics(), event.action, 1, queued) {
|
||||||
done_ilm(1)();
|
done_ilm(1)();
|
||||||
}
|
}
|
||||||
@@ -1019,7 +1254,21 @@ impl ScannerItem {
|
|||||||
let action = event.action;
|
let action = event.action;
|
||||||
let count = u64::try_from(to_delete_objs.len()).unwrap_or(u64::MAX);
|
let count = u64::try_from(to_delete_objs.len()).unwrap_or(u64::MAX);
|
||||||
let done_ilm = Metrics::time_ilm(action);
|
let done_ilm = Metrics::time_ilm(action);
|
||||||
|
let trace_started_at = trace_start_instant();
|
||||||
let queued = enqueue_runtime_newer_noncurrent(&self.bucket, to_delete_objs, event, &LcEventSrc::Scanner).await;
|
let queued = enqueue_runtime_newer_noncurrent(&self.bucket, to_delete_objs, event, &LcEventSrc::Scanner).await;
|
||||||
|
if let Some(trace_started_at) = trace_started_at {
|
||||||
|
let state = if queued { "queued" } else { "not_queued" };
|
||||||
|
trace_emit(|| {
|
||||||
|
TraceEvent::new(TraceKind::Scanner, TraceFunc::ScannerIlmAction)
|
||||||
|
.with_bucket(self.bucket.as_str())
|
||||||
|
.with_object(self.object_path())
|
||||||
|
.with_duration(trace_started_at.elapsed())
|
||||||
|
.with_attr("state", state)
|
||||||
|
.with_attr("action", action.as_str())
|
||||||
|
.with_attr("count", count)
|
||||||
|
.with_attr("queued", queued)
|
||||||
|
});
|
||||||
|
}
|
||||||
if record_scanner_ilm_action_if_queued(global_metrics(), action, count, queued) {
|
if record_scanner_ilm_action_if_queued(global_metrics(), action, count, queued) {
|
||||||
done_ilm(count)();
|
done_ilm(count)();
|
||||||
remaining_versions = remaining_versions.saturating_sub(noncurrent_accounting.len());
|
remaining_versions = remaining_versions.saturating_sub(noncurrent_accounting.len());
|
||||||
@@ -1197,6 +1446,7 @@ impl ScannerItem {
|
|||||||
fn alert_excessive_versions(&self, remaining_versions: usize, cumulative_size: i64) {
|
fn alert_excessive_versions(&self, remaining_versions: usize, cumulative_size: i64) {
|
||||||
ensure_scanner_alert_metrics_registered();
|
ensure_scanner_alert_metrics_registered();
|
||||||
let (too_many_versions, too_large_versions) = should_alert_excessive_versions(remaining_versions, cumulative_size);
|
let (too_many_versions, too_large_versions) = should_alert_excessive_versions(remaining_versions, cumulative_size);
|
||||||
|
let object_path = self.object_path();
|
||||||
if too_many_versions {
|
if too_many_versions {
|
||||||
global_metrics().record_scanner_source_executed(ScannerWorkSource::Alerts, 1);
|
global_metrics().record_scanner_source_executed(ScannerWorkSource::Alerts, 1);
|
||||||
counter!(
|
counter!(
|
||||||
@@ -1204,13 +1454,26 @@ impl ScannerItem {
|
|||||||
"bucket" => self.bucket.clone()
|
"bucket" => self.bucket.clone()
|
||||||
)
|
)
|
||||||
.increment(1);
|
.increment(1);
|
||||||
|
if scanner_alert_emission_allows(ScannerAlertKind::ManyVersions, &self.bucket, &object_path, scanner_alert_cooldown())
|
||||||
|
{
|
||||||
|
emit_scanner_alert_event(
|
||||||
|
EVENT_SCANNER_MANY_VERSIONS,
|
||||||
|
&self.bucket,
|
||||||
|
&object_path,
|
||||||
|
cumulative_size,
|
||||||
|
&[
|
||||||
|
("versions", remaining_versions.to_string()),
|
||||||
|
("threshold", scanner_excess_versions_threshold().to_string()),
|
||||||
|
],
|
||||||
|
);
|
||||||
|
}
|
||||||
warn!(
|
warn!(
|
||||||
target: "rustfs::scanner::folder",
|
target: "rustfs::scanner::folder",
|
||||||
event = EVENT_SCANNER_ALERT_STATE,
|
event = EVENT_SCANNER_ALERT_STATE,
|
||||||
component = LOG_COMPONENT_SCANNER,
|
component = LOG_COMPONENT_SCANNER,
|
||||||
subsystem = LOG_SUBSYSTEM_FOLDER,
|
subsystem = LOG_SUBSYSTEM_FOLDER,
|
||||||
bucket = %self.bucket,
|
bucket = %self.bucket,
|
||||||
object = %self.object_path(),
|
object = %object_path,
|
||||||
versions = remaining_versions,
|
versions = remaining_versions,
|
||||||
threshold = scanner_excess_versions_threshold(),
|
threshold = scanner_excess_versions_threshold(),
|
||||||
state = "excess_versions",
|
state = "excess_versions",
|
||||||
@@ -1224,13 +1487,31 @@ impl ScannerItem {
|
|||||||
"bucket" => self.bucket.clone()
|
"bucket" => self.bucket.clone()
|
||||||
)
|
)
|
||||||
.increment(1);
|
.increment(1);
|
||||||
|
if scanner_alert_emission_allows(
|
||||||
|
ScannerAlertKind::LargeVersions,
|
||||||
|
&self.bucket,
|
||||||
|
&object_path,
|
||||||
|
scanner_alert_cooldown(),
|
||||||
|
) {
|
||||||
|
emit_scanner_alert_event(
|
||||||
|
EVENT_SCANNER_LARGE_VERSIONS,
|
||||||
|
&self.bucket,
|
||||||
|
&object_path,
|
||||||
|
cumulative_size,
|
||||||
|
&[
|
||||||
|
("versions", remaining_versions.to_string()),
|
||||||
|
("cumulativeSize", cumulative_size.to_string()),
|
||||||
|
("threshold", scanner_excess_version_size_threshold().to_string()),
|
||||||
|
],
|
||||||
|
);
|
||||||
|
}
|
||||||
warn!(
|
warn!(
|
||||||
target: "rustfs::scanner::folder",
|
target: "rustfs::scanner::folder",
|
||||||
event = EVENT_SCANNER_ALERT_STATE,
|
event = EVENT_SCANNER_ALERT_STATE,
|
||||||
component = LOG_COMPONENT_SCANNER,
|
component = LOG_COMPONENT_SCANNER,
|
||||||
subsystem = LOG_SUBSYSTEM_FOLDER,
|
subsystem = LOG_SUBSYSTEM_FOLDER,
|
||||||
bucket = %self.bucket,
|
bucket = %self.bucket,
|
||||||
object = %self.object_path(),
|
object = %object_path,
|
||||||
versions = remaining_versions,
|
versions = remaining_versions,
|
||||||
cumulative_size,
|
cumulative_size,
|
||||||
threshold = scanner_excess_version_size_threshold(),
|
threshold = scanner_excess_version_size_threshold(),
|
||||||
@@ -1611,6 +1892,15 @@ impl FolderScanner {
|
|||||||
"root" => self.root.clone()
|
"root" => self.root.clone()
|
||||||
)
|
)
|
||||||
.increment(1);
|
.increment(1);
|
||||||
|
if scanner_alert_emission_allows(ScannerAlertKind::BigPrefix, &self.root, folder, scanner_alert_cooldown()) {
|
||||||
|
emit_scanner_alert_event(
|
||||||
|
EVENT_SCANNER_BIG_PREFIX,
|
||||||
|
&self.root,
|
||||||
|
folder,
|
||||||
|
0,
|
||||||
|
&[("folders", total_folders.to_string()), ("threshold", threshold.to_string())],
|
||||||
|
);
|
||||||
|
}
|
||||||
warn!(
|
warn!(
|
||||||
target: "rustfs::scanner::folder",
|
target: "rustfs::scanner::folder",
|
||||||
event = EVENT_SCANNER_ALERT_STATE,
|
event = EVENT_SCANNER_ALERT_STATE,
|
||||||
@@ -1830,6 +2120,7 @@ impl FolderScanner {
|
|||||||
into: &mut DataUsageEntry,
|
into: &mut DataUsageEntry,
|
||||||
) -> Result<(), ScannerError> {
|
) -> Result<(), ScannerError> {
|
||||||
let done_folder = Metrics::time(Metric::ScanFolder);
|
let done_folder = Metrics::time(Metric::ScanFolder);
|
||||||
|
let trace_started_at = trace_start_instant();
|
||||||
|
|
||||||
if ctx.is_cancelled() {
|
if ctx.is_cancelled() {
|
||||||
return Err(ScannerError::Other("Operation cancelled".to_string()));
|
return Err(ScannerError::Other("Operation cancelled".to_string()));
|
||||||
@@ -2187,6 +2478,15 @@ impl FolderScanner {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if let GetSizeFailureAction::HealMetadata { object } = failure_action {
|
if let GetSizeFailureAction::HealMetadata { object } = failure_action {
|
||||||
|
// MRF journal intent: durable High-priority Metadata
|
||||||
|
// heal across restarts (HS-01); the scanner heal
|
||||||
|
// request below stays as the immediate path.
|
||||||
|
rustfs_common::mrf_channel::try_send_mrf_intent(
|
||||||
|
rustfs_common::mrf_channel::MrfKind::MetadataCorruption,
|
||||||
|
&item.bucket,
|
||||||
|
&object,
|
||||||
|
None,
|
||||||
|
);
|
||||||
self.send_required_scanner_heal_request(
|
self.send_required_scanner_heal_request(
|
||||||
PendingScannerHealKind::Object,
|
PendingScannerHealKind::Object,
|
||||||
item.bucket.clone(),
|
item.bucket.clone(),
|
||||||
@@ -2895,6 +3195,8 @@ impl FolderScanner {
|
|||||||
}
|
}
|
||||||
|
|
||||||
done_folder();
|
done_folder();
|
||||||
|
let scanned_objects = u64::try_from(into.objects).unwrap_or(u64::MAX);
|
||||||
|
emit_scanner_folder_trace(&self.root, &folder.name, scanned_objects, trace_started_at, "completed");
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
@@ -3076,6 +3378,90 @@ mod tests {
|
|||||||
#[cfg(unix)]
|
#[cfg(unix)]
|
||||||
use std::os::unix::fs::{PermissionsExt, symlink};
|
use std::os::unix::fs::{PermissionsExt, symlink};
|
||||||
use std::sync::Mutex;
|
use std::sync::Mutex;
|
||||||
|
|
||||||
|
/// Reset the process-global alert cooldown map; test-only.
|
||||||
|
fn reset_alert_cooldowns() {
|
||||||
|
*SCANNER_ALERT_EMISSION_COOLDOWN
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(|poison| poison.into_inner()) = Some(ScannerAlertCooldownMap::new());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The emitted event-name strings must be exactly what `EventName`
|
||||||
|
/// serializes, or a bucket notification subscribed to the documented name
|
||||||
|
/// would silently never match (rustfs/backlog#1868).
|
||||||
|
#[test]
|
||||||
|
fn scanner_alert_wire_names_match_canonical_event_names() {
|
||||||
|
use rustfs_s3_types::EventName;
|
||||||
|
assert_eq!(EVENT_SCANNER_MANY_VERSIONS, EventName::ScannerManyVersions.to_string());
|
||||||
|
assert_eq!(EVENT_SCANNER_LARGE_VERSIONS, EventName::ScannerLargeVersions.to_string());
|
||||||
|
assert_eq!(EVENT_SCANNER_BIG_PREFIX, EventName::ScannerBigPrefix.to_string());
|
||||||
|
}
|
||||||
|
|
||||||
|
fn cooldown_map_len() -> usize {
|
||||||
|
SCANNER_ALERT_EMISSION_COOLDOWN
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(|poison| poison.into_inner())
|
||||||
|
.as_ref()
|
||||||
|
.map(|map| map.len())
|
||||||
|
.unwrap_or(0)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Backdate every recorded cooldown so the next check fires again.
|
||||||
|
fn expire_all_alert_cooldowns(cooldown: Duration) {
|
||||||
|
let now = Instant::now();
|
||||||
|
let mut guard = SCANNER_ALERT_EMISSION_COOLDOWN
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(|poison| poison.into_inner());
|
||||||
|
if let Some(map) = guard.as_mut() {
|
||||||
|
for fired_at in map.values_mut() {
|
||||||
|
if let Some(expired) = now.checked_sub(cooldown + Duration::from_secs(1)) {
|
||||||
|
*fired_at = expired;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The emission gate is the only thing standing between an over-threshold
|
||||||
|
/// object and one S3 event per scan cycle, so its edge semantics get
|
||||||
|
/// pinned directly. All scenarios share one #[test] because the cooldown
|
||||||
|
/// map is process-global and parallel tests would read each other's
|
||||||
|
/// firings.
|
||||||
|
#[test]
|
||||||
|
fn scanner_alert_emission_is_edge_held_per_key_and_bounded() {
|
||||||
|
reset_alert_cooldowns();
|
||||||
|
let cooldown = Duration::from_secs(3600);
|
||||||
|
|
||||||
|
// First firing allows, an immediate re-check is held.
|
||||||
|
assert!(scanner_alert_emission_allows(ScannerAlertKind::ManyVersions, "bkt", "obj", cooldown));
|
||||||
|
assert!(!scanner_alert_emission_allows(ScannerAlertKind::ManyVersions, "bkt", "obj", cooldown));
|
||||||
|
|
||||||
|
// Different kind, object, and bucket are independent keys.
|
||||||
|
assert!(scanner_alert_emission_allows(ScannerAlertKind::LargeVersions, "bkt", "obj", cooldown));
|
||||||
|
assert!(scanner_alert_emission_allows(ScannerAlertKind::ManyVersions, "bkt", "other", cooldown));
|
||||||
|
assert!(scanner_alert_emission_allows(ScannerAlertKind::ManyVersions, "other", "obj", cooldown));
|
||||||
|
assert_eq!(cooldown_map_len(), 4);
|
||||||
|
|
||||||
|
// After the cooldown elapses the same key fires again.
|
||||||
|
expire_all_alert_cooldowns(cooldown);
|
||||||
|
assert!(scanner_alert_emission_allows(ScannerAlertKind::ManyVersions, "bkt", "obj", cooldown));
|
||||||
|
|
||||||
|
// A zero cooldown degenerates to always-emit (operators may want that).
|
||||||
|
assert!(scanner_alert_emission_allows(ScannerAlertKind::BigPrefix, "bkt", "dir", Duration::ZERO));
|
||||||
|
assert!(scanner_alert_emission_allows(ScannerAlertKind::BigPrefix, "bkt", "dir", Duration::ZERO));
|
||||||
|
|
||||||
|
// Hard bound: overflow the cap with zero-cooldown keys and confirm the
|
||||||
|
// map clears rather than growing past it.
|
||||||
|
reset_alert_cooldowns();
|
||||||
|
for index in 0..=(MAX_SCANNER_ALERT_COOLDOWN_KEYS + 8) {
|
||||||
|
let _ = scanner_alert_emission_allows(ScannerAlertKind::BigPrefix, "bkt", &format!("dir-{index}"), Duration::ZERO);
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
cooldown_map_len() <= MAX_SCANNER_ALERT_COOLDOWN_KEYS,
|
||||||
|
"cooldown map must stay bounded, got {}",
|
||||||
|
cooldown_map_len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
|
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
|
||||||
use temp_env::{with_var, with_var_unset};
|
use temp_env::{with_var, with_var_unset};
|
||||||
use tracing_subscriber::fmt::MakeWriter;
|
use tracing_subscriber::fmt::MakeWriter;
|
||||||
@@ -4400,6 +4786,104 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn scanner_trace_helpers_emit_expected_events() {
|
||||||
|
let mut trace = rustfs_common::trace_bus::subscribe_trace_events();
|
||||||
|
|
||||||
|
emit_scanner_folder_trace(
|
||||||
|
"/tmp/rustfs-scanner-trace",
|
||||||
|
"/tmp/rustfs-scanner-trace/bucket-a/folder-a",
|
||||||
|
7,
|
||||||
|
Some(Instant::now()),
|
||||||
|
"completed",
|
||||||
|
);
|
||||||
|
let folder = recv_scanner_trace_event(
|
||||||
|
&mut trace,
|
||||||
|
TraceFunc::ScannerFolder,
|
||||||
|
Some("bucket-a"),
|
||||||
|
Some("folder-a"),
|
||||||
|
Some("completed"),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
assert_eq!(trace_attr_string(&folder, "objects").as_deref(), Some("7"));
|
||||||
|
|
||||||
|
emit_scanner_ilm_action_trace("bucket-a", "object-a", IlmAction::DeleteAction, 2, true, Some(Instant::now()));
|
||||||
|
let ilm = recv_scanner_trace_event(
|
||||||
|
&mut trace,
|
||||||
|
TraceFunc::ScannerIlmAction,
|
||||||
|
Some("bucket-a"),
|
||||||
|
Some("object-a"),
|
||||||
|
Some("queued"),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
assert_eq!(trace_attr_string(&ilm, "action").as_deref(), Some("delete"));
|
||||||
|
assert_eq!(trace_attr_string(&ilm, "count").as_deref(), Some("2"));
|
||||||
|
assert_eq!(trace_attr_string(&ilm, "queued").as_deref(), Some("true"));
|
||||||
|
|
||||||
|
emit_scanner_heal_candidate_trace(ScannerHealCandidateTrace {
|
||||||
|
candidate_type: "object",
|
||||||
|
bucket: "bucket-a",
|
||||||
|
object: Some("object-a"),
|
||||||
|
version_id: Some("version-a"),
|
||||||
|
priority: HealChannelPriority::High,
|
||||||
|
scan_mode: Some(HealScanMode::Deep),
|
||||||
|
result: Ok(HealAdmissionResult::Merged),
|
||||||
|
started_at: Instant::now(),
|
||||||
|
});
|
||||||
|
let heal_candidate = recv_scanner_trace_event(
|
||||||
|
&mut trace,
|
||||||
|
TraceFunc::ScannerHealCandidate,
|
||||||
|
Some("bucket-a"),
|
||||||
|
Some("object-a"),
|
||||||
|
Some("admitted"),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
assert_eq!(trace_attr_string(&heal_candidate, "candidate_type").as_deref(), Some("object"));
|
||||||
|
assert_eq!(trace_attr_string(&heal_candidate, "priority").as_deref(), Some("high"));
|
||||||
|
assert_eq!(trace_attr_string(&heal_candidate, "scan_mode").as_deref(), Some("deep"));
|
||||||
|
assert_eq!(trace_attr_string(&heal_candidate, "version_id").as_deref(), Some("version-a"));
|
||||||
|
assert_eq!(trace_attr_string(&heal_candidate, "admission").as_deref(), Some("merged"));
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn recv_scanner_trace_event(
|
||||||
|
trace: &mut rustfs_common::trace_bus::TraceSubscription,
|
||||||
|
func: TraceFunc,
|
||||||
|
bucket: Option<&str>,
|
||||||
|
object: Option<&str>,
|
||||||
|
state: Option<&str>,
|
||||||
|
) -> TraceEvent {
|
||||||
|
for _ in 0..32 {
|
||||||
|
let event = tokio::time::timeout(Duration::from_secs(1), trace.recv())
|
||||||
|
.await
|
||||||
|
.expect("scanner trace event should arrive")
|
||||||
|
.expect("trace bus should stay open");
|
||||||
|
if event.kind == TraceKind::Scanner
|
||||||
|
&& event.func == func
|
||||||
|
&& event.bucket.as_deref() == bucket
|
||||||
|
&& event.object.as_deref() == object
|
||||||
|
&& state.is_none_or(|state| trace_attr_string(&event, "state").as_deref() == Some(state))
|
||||||
|
{
|
||||||
|
return (*event).clone();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
panic!("expected scanner trace event {func:?} for bucket {bucket:?} object {object:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
fn trace_attr_string(event: &TraceEvent, key: &str) -> Option<String> {
|
||||||
|
event.attrs.iter().find_map(|attr| {
|
||||||
|
if attr.key != key {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(match &attr.value {
|
||||||
|
rustfs_common::trace_bus::TraceVal::Bool(value) => value.to_string(),
|
||||||
|
rustfs_common::trace_bus::TraceVal::U64(value) => value.to_string(),
|
||||||
|
rustfs_common::trace_bus::TraceVal::I64(value) => value.to_string(),
|
||||||
|
rustfs_common::trace_bus::TraceVal::Str(value) => value.to_string(),
|
||||||
|
})
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_build_high_priority_heal_admission_error_contains_context() {
|
fn test_build_high_priority_heal_admission_error_contains_context() {
|
||||||
let err = build_high_priority_heal_admission_error(
|
let err = build_high_priority_heal_admission_error(
|
||||||
|
|||||||
@@ -231,6 +231,10 @@ pub fn record_dirty_usage_bucket(bucket: &str) {
|
|||||||
dirty_buckets.len()
|
dirty_buckets.len()
|
||||||
};
|
};
|
||||||
global_metrics().record_scanner_dirty_usage_pending(usize_to_u64_saturated(pending_buckets));
|
global_metrics().record_scanner_dirty_usage_pending(usize_to_u64_saturated(pending_buckets));
|
||||||
|
// A write invalidates this bucket's prefix-usage answers on the spot so
|
||||||
|
// admin/console consumers never ride the full TTL after a change
|
||||||
|
// (rustfs/backlog#1872).
|
||||||
|
crate::prefix_usage::invalidate_prefix_usage_cache(bucket);
|
||||||
DIRTY_USAGE_BUCKET_NOTIFY.notify_one();
|
DIRTY_USAGE_BUCKET_NOTIFY.notify_one();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user