Merge remote-tracking branch 'upstream/main' into test/heal-chaos-restart-recovery

# Conflicts:
#	.config/e2e-nightly-selection.txt
This commit is contained in:
marshawcoco
2026-08-31 14:26:42 +08:00
94 changed files with 19165 additions and 1555 deletions
+1
View File
@@ -100,6 +100,7 @@ aws-sdk-s3 = { workspace = true, default-features = false, features = ["sigv4a",
aws-sdk-sts = { workspace = true, default-features = false, features = ["default-https-client", "rt-tokio"] }
aws-config = { workspace = true }
aws-smithy-http-client = { workspace = true, default-features = false, features = ["rustls-aws-lc"] }
aws-smithy-types.workspace = true
async-compression = { workspace = true, features = ["tokio", "bzip2", "xz"] }
async-trait = { workspace = true }
flate2.workspace = true
@@ -27,8 +27,10 @@
//! Readiness is established by the harness's `start()` handshake (TCP reachability
//! plus an S3 `ListBuckets` poll) — there are no fixed sleeps.
//!
//! Out of scope for this block (tracked separately): network fault injection
//! (toxiproxy / socket proxy) and 5GiB large-object budgets.
//! The volume-proxy smoke below also proves that the socket-level fault proxy
//! can be installed before startup without changing the client-facing node URL.
//! A full lock-plane partition matrix and 5GiB large-object budget remain
//! tracked separately.
use crate::common::{ClusterTopology, RustFSTestClusterEnvironment};
@@ -76,6 +78,28 @@ async fn cluster_multidrive_single_pool_smoke() -> TestResult {
Ok(())
}
/// 4 nodes x 4 drives, single pool: exercise the maximum local erasure layout
/// supported by the cluster harness. This remains in the nightly lane because
/// it starts four real server processes and sixteen data directories.
#[tokio::test]
async fn cluster_four_node_four_drive_single_pool_smoke() -> TestResult {
crate::common::init_logging();
let mut cluster = RustFSTestClusterEnvironment::with_topology(ClusterTopology::single_pool_multidrive(4, 4)).await?;
let volumes = cluster.rustfs_volumes_arg();
assert_eq!(volumes.split(' ').count(), 16, "expected 16 explicit endpoints, got: {volumes}");
assert!(!volumes.contains('{'), "single-pool layout must not use ellipses: {volumes}");
assert!(cluster.nodes.iter().all(|node| node.data_dirs.len() == 4));
cluster.start().await?;
cluster.create_test_bucket(BUCKET).await?;
let payload = vec![0x3Cu8; 1024 * 1024];
put_get_roundtrip(&cluster, "multidrive-4/object", &payload).await?;
Ok(())
}
/// Two single-node pools, 2 drives each: the multi-pool layout boots and
/// round-trips. Every pool is a distinct erasure pool (`pool_idx` 0 and 1).
#[tokio::test]
@@ -103,3 +127,27 @@ async fn cluster_two_pool_smoke() -> TestResult {
put_get_roundtrip(&cluster, "twopool/object", &payload).await?;
Ok(())
}
/// A real cluster smoke for the volume FaultProxy wiring. The proxy target is
/// not listening yet when it is created; cluster startup must still converge
/// once the target node starts, and peer disk/RPC traffic must traverse it.
#[tokio::test]
async fn cluster_volume_fault_proxy_pass_smoke() -> TestResult {
crate::common::init_logging();
let mut cluster = RustFSTestClusterEnvironment::with_topology(ClusterTopology::single_pool_multidrive(2, 2)).await?;
let proxy = cluster.start_volume_proxy_for_node(0).await?;
let proxied = proxy.local_addr().to_string();
assert!(cluster.rustfs_volumes_arg().contains(&proxied));
let result: TestResult = async {
cluster.start().await?;
cluster.create_test_bucket(BUCKET).await?;
let payload = vec![0x6Du8; 256 * 1024];
put_get_roundtrip(&cluster, "volume-proxy/object", &payload).await
}
.await;
proxy.shutdown().await;
result
}
+69 -4
View File
@@ -34,6 +34,7 @@ use serde_json;
use std::ffi::OsStr;
use std::fs as stdfs;
use std::io::ErrorKind;
use std::net::SocketAddr;
use std::path::{Path, PathBuf};
use std::process::{Child, Command, Stdio};
use std::sync::Once;
@@ -1214,6 +1215,9 @@ pub struct RustFSTestClusterEnvironment {
pub node_extra_env: Vec<Vec<(String, String)>>,
pub node_capture_log_paths: Vec<Option<String>>,
pub topology: ClusterTopology,
/// Optional socket proxies used for the corresponding node's volume
/// endpoints. Proxies must be installed before [`Self::start`].
volume_proxy_addresses: Vec<Option<SocketAddr>>,
}
impl RustFSTestClusterEnvironment {
@@ -1305,6 +1309,7 @@ impl RustFSTestClusterEnvironment {
extra_env.push(("RUSTFS_UNSAFE_BYPASS_DISK_CHECK".to_string(), "true".to_string()));
}
let node_count = topology.node_count;
Ok(Self {
nodes,
temp_dir,
@@ -1314,6 +1319,7 @@ impl RustFSTestClusterEnvironment {
node_extra_env: vec![Vec::new(); topology.node_count],
node_capture_log_paths: vec![None; topology.node_count],
topology,
volume_proxy_addresses: vec![None; node_count],
})
}
@@ -1381,6 +1387,34 @@ impl RustFSTestClusterEnvironment {
self.build_volumes_arg()
}
/// Start a socket proxy for one node's volume endpoints and route all
/// subsequent `RUSTFS_VOLUMES` references for that node through it.
///
/// Call this before [`Self::start`], then use the returned proxy's
/// [`crate::fault_proxy::FaultProxy::set_mode`] to inject latency,
/// blackhole, or one-way partition faults. The node's own listen address
/// remains direct, so S3 clients can still reach it while peer disk/RPC
/// traffic is steered through the proxy.
pub async fn start_volume_proxy_for_node(
&mut self,
node_idx: usize,
) -> Result<crate::fault_proxy::FaultProxy, Box<dyn std::error::Error + Send + Sync>> {
self.ensure_node_index(node_idx)?;
if self.volume_proxy_addresses[node_idx].is_some() {
return Err(format!("a volume proxy is already configured for node {node_idx}").into());
}
let target = self.nodes[node_idx].address.parse::<SocketAddr>()?;
let proxy = crate::fault_proxy::FaultProxy::start(target).await?;
self.volume_proxy_addresses[node_idx] = Some(proxy.local_addr());
Ok(proxy)
}
fn volume_address(&self, node_idx: usize) -> String {
self.volume_proxy_addresses[node_idx]
.map(|address| address.to_string())
.unwrap_or_else(|| self.nodes[node_idx].address.clone())
}
fn build_volumes_arg(&self) -> String {
let pools = self.topology.normalized_pools();
@@ -1389,7 +1423,11 @@ impl RustFSTestClusterEnvironment {
return self
.nodes
.iter()
.flat_map(|n| n.data_dirs.iter().map(move |dir| format!("http://{}{}", n.address, dir)))
.enumerate()
.flat_map(|(node_idx, n)| {
let address = self.volume_address(node_idx);
n.data_dirs.iter().map(move |dir| format!("http://{}{}", address, dir))
})
.collect::<Vec<_>>()
.join(" ");
}
@@ -1400,13 +1438,19 @@ impl RustFSTestClusterEnvironment {
pools
.iter()
.map(|nodes| {
let node = &self.nodes[nodes[0]];
let node_idx = nodes[0];
let node = &self.nodes[node_idx];
let base = node
.data_dirs
.first()
.and_then(|d| d.rsplit_once('/').map(|(parent, _)| parent))
.unwrap_or(&node.data_dir);
format!("http://{}{}/drive{{0...{}}}", node.address, base, self.topology.drives_per_node - 1)
format!(
"http://{}{}/drive{{0...{}}}",
self.volume_address(node_idx),
base,
self.topology.drives_per_node - 1
)
})
.collect::<Vec<_>>()
.join(" ")
@@ -2000,7 +2044,7 @@ mod tests {
}
let multidrive = topology.drives_per_node > 1;
let nodes = (0..topology.node_count)
let nodes: Vec<ClusterNode> = (0..topology.node_count)
.map(|i| {
let address = format!("127.0.0.1:{}", 9000 + i);
let data_dirs: Vec<String> = if multidrive {
@@ -2021,6 +2065,7 @@ mod tests {
})
.collect();
let node_count = nodes.len();
RustFSTestClusterEnvironment {
nodes,
temp_dir,
@@ -2030,6 +2075,7 @@ mod tests {
node_extra_env: vec![Vec::new(); topology.node_count],
node_capture_log_paths: vec![None; topology.node_count],
topology,
volume_proxy_addresses: vec![None; node_count],
}
}
@@ -2114,6 +2160,25 @@ mod tests {
assert!(ClusterTopology::single_pool_multidrive(1, 1).validate().is_ok());
}
#[tokio::test]
async fn volume_proxy_rewrites_cluster_volume_endpoint() {
let mut env = RustFSTestClusterEnvironment::new(1)
.await
.expect("cluster environment should allocate a node");
let direct = env.nodes[0].address.clone();
let proxy = env
.start_volume_proxy_for_node(0)
.await
.expect("volume proxy should bind before the target server starts");
let proxied = proxy.local_addr().to_string();
let volumes = env.rustfs_volumes_arg();
assert!(volumes.contains(&proxied), "volumes must use the proxy address: {volumes}");
assert!(!volumes.contains(&direct), "volumes must not retain the direct address: {volumes}");
proxy.shutdown().await;
}
#[test]
fn cluster_node_env_supports_per_node_overrides() {
let mut env = fake_cluster(ClusterTopology::single_pool(4));
+1
View File
@@ -21,5 +21,6 @@ mod head_tls_bodyless_test;
mod lifecycle;
mod lock;
mod node_interact_test;
mod s3_select_compression;
mod sql;
mod tiering;
@@ -0,0 +1,351 @@
#![cfg(test)]
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use crate::common::{RustFSTestEnvironment, init_logging};
use async_compression::tokio::write::BzEncoder;
use aws_sdk_s3::{
Client,
error::ProvideErrorMetadata,
operation::select_object_content::{SelectObjectContentOutput, builders::SelectObjectContentFluentBuilder},
types::{
CompressionType, CsvInput, CsvOutput, ExpressionType, FileHeaderInfo, InputSerialization, JsonInput, JsonOutput,
JsonType, OutputSerialization, SelectObjectContentEventStream,
},
};
use aws_smithy_types::event_stream::RawMessage;
use bytes::Bytes;
use flate2::{Compression, write::GzEncoder};
use std::{error::Error, io::Cursor, time::Duration};
use tokio::io::AsyncWriteExt;
const BUCKET: &str = "s3-select-compression";
const SELECT_RESPONSE_TIMEOUT: Duration = Duration::from_secs(30);
type TestResult<T> = Result<T, Box<dyn Error + Send + Sync>>;
async fn create_test_environment(extra_env: &[(&str, &str)]) -> TestResult<(RustFSTestEnvironment, Client)> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
env.start_rustfs_server_with_env(vec![], extra_env).await?;
let client = env.create_s3_client();
client.create_bucket().bucket(BUCKET).send().await?;
Ok((env, client))
}
async fn put_object(client: &Client, key: &str, body: &[u8]) -> TestResult<()> {
client
.put_object()
.bucket(BUCKET)
.key(key)
.body(Bytes::copy_from_slice(body).into())
.send()
.await?;
Ok(())
}
fn gzip(input: &[u8]) -> TestResult<Vec<u8>> {
let mut encoder = GzEncoder::new(Vec::new(), Compression::default());
std::io::Write::write_all(&mut encoder, input)?;
Ok(encoder.finish()?)
}
async fn bzip2(input: &[u8]) -> TestResult<Vec<u8>> {
let mut encoder = BzEncoder::new(Cursor::new(Vec::new()));
encoder.write_all(input).await?;
encoder.shutdown().await?;
Ok(encoder.into_inner().into_inner())
}
fn csv_select_request(
client: &Client,
key: &str,
compression: CompressionType,
expression: &str,
) -> SelectObjectContentFluentBuilder {
client
.select_object_content()
.bucket(BUCKET)
.key(key)
.expression(expression)
.expression_type(ExpressionType::Sql)
.input_serialization(
InputSerialization::builder()
.compression_type(compression)
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
.build(),
)
.output_serialization(OutputSerialization::builder().csv(CsvOutput::builder().build()).build())
}
fn json_select_request(
client: &Client,
key: &str,
compression: CompressionType,
json_type: JsonType,
) -> SelectObjectContentFluentBuilder {
client
.select_object_content()
.bucket(BUCKET)
.key(key)
.expression("SELECT name FROM S3Object")
.expression_type(ExpressionType::Sql)
.input_serialization(
InputSerialization::builder()
.compression_type(compression)
.json(JsonInput::builder().set_type(Some(json_type)).build())
.build(),
)
.output_serialization(OutputSerialization::builder().json(JsonOutput::builder().build()).build())
}
async fn collect_success(
mut response: SelectObjectContentOutput,
compressed_bytes: usize,
processed_bytes: usize,
) -> TestResult<Vec<u8>> {
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async move {
let mut records = Vec::new();
let mut stats = None;
let mut saw_end = false;
while let Some(event) = response.payload.recv().await? {
assert!(!saw_end, "Select emitted an event after End");
match event {
SelectObjectContentEventStream::Records(event) => {
assert!(stats.is_none(), "Select emitted Records after Stats");
if let Some(payload) = event.payload {
records.extend_from_slice(payload.as_ref());
}
}
SelectObjectContentEventStream::Stats(event) => {
assert!(stats.is_none(), "Select emitted more than one Stats event");
stats = event.details;
}
SelectObjectContentEventStream::End(_) => {
assert!(stats.is_some(), "Select emitted End before Stats");
saw_end = true;
}
_ => assert!(stats.is_none(), "Select emitted a non-terminal event after Stats"),
}
}
let stats = stats.ok_or("Select response ended without a Stats event")?;
assert_eq!(stats.bytes_scanned(), Some(i64::try_from(compressed_bytes)?));
assert_eq!(stats.bytes_processed(), Some(i64::try_from(processed_bytes)?));
assert_eq!(stats.bytes_returned(), Some(i64::try_from(records.len())?));
assert!(saw_end, "Select response ended without an End event");
Ok::<_, Box<dyn Error + Send + Sync>>(records)
})
.await
.map_err(|_| -> Box<dyn Error + Send + Sync> { "Select response timed out".into() })?
}
async fn assert_truncated_stream_failure(mut response: SelectObjectContentOutput) -> TestResult<()> {
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async move {
loop {
match response.payload.recv().await {
Err(error) => {
// S3 Select request-level errors use `error` frames, which this SDK version exposes as raw response errors.
if let Some(code) = error.code() {
assert_eq!(code, "TruncatedInput", "unexpected modeled event-stream error: {error:?}");
} else if let aws_sdk_s3::error::SdkError::ResponseError(context) = &error
&& let RawMessage::Decoded(message) = context.raw()
{
let header = |name: &str| {
message
.headers()
.iter()
.find(|header| header.name().as_str() == name)
.and_then(|header| header.value().as_string().ok())
.map(|value| value.as_str())
};
assert_eq!(header(":message-type"), Some("error"));
assert_eq!(header(":error-code"), Some("TruncatedInput"));
} else {
panic!("unexpected event-stream error: {error:?}");
}
return Ok(());
}
Ok(Some(SelectObjectContentEventStream::Stats(_))) | Ok(Some(SelectObjectContentEventStream::End(_))) => {
return Err("truncated compressed input reached a success terminal event".into());
}
Ok(Some(_)) => {}
Ok(None) => return Err("truncated compressed input ended without an error event".into()),
}
}
})
.await
.map_err(|_| -> Box<dyn Error + Send + Sync> { "truncated Select response timed out".into() })?
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn test_select_object_content_compressed_csv_and_json() -> TestResult<()> {
const CSV: &[u8] = b"name,age\nAlice,30\nBob,25\n";
const JSON_LINES: &[u8] = b"{\"name\":\"Alice\"}\n{\"name\":\"Bob\"}\n";
const JSON_DOCUMENT: &[u8] = br#"[{"name":"Alice"},{"name":"Bob"}]"#;
let (_env, client) = create_test_environment(&[]).await?;
let gzip_csv = gzip(CSV)?;
put_object(&client, "records.csv.gz", &gzip_csv).await?;
let gzip_csv_records = collect_success(
csv_select_request(&client, "records.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
.send()
.await?,
gzip_csv.len(),
CSV.len(),
)
.await?;
assert_eq!(gzip_csv_records, b"Alice,30\nBob,25\n");
let bzip_csv = bzip2(CSV).await?;
put_object(&client, "records.csv.bz2", &bzip_csv).await?;
let bzip_csv_records = collect_success(
csv_select_request(&client, "records.csv.bz2", CompressionType::Bzip2, "SELECT * FROM S3Object")
.send()
.await?,
bzip_csv.len(),
CSV.len(),
)
.await?;
assert_eq!(bzip_csv_records, gzip_csv_records);
let gzip_json_lines = gzip(JSON_LINES)?;
put_object(&client, "json-lines", &gzip_json_lines).await?;
let gzip_json_records = collect_success(
json_select_request(&client, "json-lines", CompressionType::Gzip, JsonType::Lines)
.send()
.await?,
gzip_json_lines.len(),
JSON_LINES.len(),
)
.await?;
assert_eq!(gzip_json_records, JSON_LINES);
let bzip_json_lines = bzip2(JSON_LINES).await?;
put_object(&client, "records.jsonl.bz2", &bzip_json_lines).await?;
let bzip_json_records = collect_success(
json_select_request(&client, "records.jsonl.bz2", CompressionType::Bzip2, JsonType::Lines)
.send()
.await?,
bzip_json_lines.len(),
JSON_LINES.len(),
)
.await?;
assert_eq!(bzip_json_records, gzip_json_records);
let gzip_json_document = gzip(JSON_DOCUMENT)?;
put_object(&client, "document.json.gz", &gzip_json_document).await?;
let document_records = collect_success(
json_select_request(&client, "document.json.gz", CompressionType::Gzip, JsonType::Document)
.send()
.await?,
gzip_json_document.len(),
JSON_DOCUMENT.len(),
)
.await?;
assert_eq!(document_records, JSON_LINES);
Ok(())
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn test_select_object_content_invalid_compressed_stream_fails() -> TestResult<()> {
const CSV: &[u8] = b"name\nAlice\n";
let (_env, client) = create_test_environment(&[]).await?;
put_object(&client, "invalid.csv.gz", CSV).await?;
let invalid = csv_select_request(&client, "invalid.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
.send()
.await
.expect_err("invalid GZIP header must fail before streaming");
assert_eq!(
invalid.as_service_error().and_then(ProvideErrorMetadata::code),
Some("InvalidCompressionFormat")
);
put_object(&client, "empty.csv.gz", b"").await?;
let empty = csv_select_request(&client, "empty.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
.send()
.await
.expect_err("empty GZIP input must fail as truncated");
assert_eq!(empty.as_service_error().and_then(ProvideErrorMetadata::code), Some("TruncatedInput"));
let mut truncated = bzip2(CSV).await?;
truncated.pop();
put_object(&client, "truncated.csv.bz2", &truncated).await?;
let truncated = csv_select_request(&client, "truncated.csv.bz2", CompressionType::Bzip2, "SELECT * FROM S3Object")
.send()
.await?;
assert_truncated_stream_failure(truncated).await?;
Ok(())
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn test_select_object_content_compressed_disconnect_releases_query() -> TestResult<()> {
const OBJECT: &str = "disconnect.csv.gz";
const ROWS: usize = 16 * 1024;
const RELEASE_ATTEMPTS: usize = 20;
const RELEASE_BACKOFF: Duration = Duration::from_millis(25);
let (_env, client) = create_test_environment(&[("RUSTFS_S3SELECT_MAX_CONCURRENT_QUERIES", "1")]).await?;
let row = format!("{}\n", "x".repeat(1023));
let mut body = Vec::with_capacity("value\n".len() + ROWS * row.len());
body.extend_from_slice(b"value\n");
for _ in 0..ROWS {
body.extend_from_slice(row.as_bytes());
}
let compressed = gzip(&body)?;
put_object(&client, OBJECT, &compressed).await?;
let first = csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
.send()
.await?;
let saturated = csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
.send()
.await
.expect_err("the unread compressed response should retain the only query permit");
assert_eq!(saturated.as_service_error().and_then(ProvideErrorMetadata::code), Some("SlowDown"));
drop(first);
let second = tokio::time::timeout(Duration::from_secs(5), async {
for attempt in 0..RELEASE_ATTEMPTS {
match csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
.send()
.await
{
Ok(response) => return Ok::<_, Box<dyn Error + Send + Sync>>(response),
Err(error)
if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown")
&& attempt + 1 < RELEASE_ATTEMPTS =>
{
tokio::time::sleep(RELEASE_BACKOFF).await;
}
Err(error) if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown") => {
return Err("disconnected compressed Select retained its query permit".into());
}
Err(error) => return Err(format!("unexpected Select error after disconnect: {error}").into()),
}
}
Err("query permit release retry loop ended unexpectedly".into())
})
.await
.map_err(|_| -> Box<dyn Error + Send + Sync> { "compressed Select did not release its query permit".into() })??;
drop(second);
Ok(())
}
+110 -25
View File
@@ -122,6 +122,24 @@ async fn select_json_document(client: &Client, key: &str, expression: &str) -> T
process_select_response(response).await
}
fn csv_select_request(
client: &Client,
key: &str,
) -> aws_sdk_s3::operation::select_object_content::builders::SelectObjectContentFluentBuilder {
client
.select_object_content()
.bucket(BUCKET)
.key(key)
.expression("SELECT * FROM S3Object")
.expression_type(ExpressionType::Sql)
.input_serialization(
InputSerialization::builder()
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
.build(),
)
.output_serialization(OutputSerialization::builder().csv(CsvOutput::builder().build()).build())
}
async fn process_select_response(
mut event_stream: aws_sdk_s3::operation::select_object_content::SelectObjectContentOutput,
) -> TestResult<String> {
@@ -188,30 +206,42 @@ async fn assert_input_byte_stats(
let mut last_progress: Option<aws_sdk_s3::types::Progress> = None;
let mut stats = None;
let mut saw_end = false;
while let Some(event) = payload.recv().await? {
match event {
aws_sdk_s3::types::SelectObjectContentEventStream::Records(records) => {
if let Some(bytes) = records.payload {
records_len = records_len.saturating_add(u64::try_from(bytes.as_ref().len())?);
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async {
// The AWS SDK validates both event-stream CRCs before yielding an event.
while let Some(event) = payload.recv().await? {
assert!(!saw_end, "Select emitted an event after End");
match event {
aws_sdk_s3::types::SelectObjectContentEventStream::Records(records) => {
assert!(stats.is_none(), "Select emitted Records after Stats");
if let Some(bytes) = records.payload {
records_len = records_len.saturating_add(u64::try_from(bytes.as_ref().len())?);
}
}
}
aws_sdk_s3::types::SelectObjectContentEventStream::Progress(event) => {
let details = event.details.ok_or("Progress event did not contain details")?;
if let Some(previous) = last_progress.as_ref() {
assert!(details.bytes_scanned() >= previous.bytes_scanned());
assert!(details.bytes_processed() >= previous.bytes_processed());
assert!(details.bytes_returned() >= previous.bytes_returned());
aws_sdk_s3::types::SelectObjectContentEventStream::Progress(event) => {
assert!(stats.is_none(), "Select emitted Progress after Stats");
let details = event.details.ok_or("Progress event did not contain details")?;
if let Some(previous) = last_progress.as_ref() {
assert!(details.bytes_scanned() >= previous.bytes_scanned());
assert!(details.bytes_processed() >= previous.bytes_processed());
assert!(details.bytes_returned() >= previous.bytes_returned());
}
last_progress = Some(details);
}
last_progress = Some(details);
aws_sdk_s3::types::SelectObjectContentEventStream::Stats(event) => {
assert!(stats.is_none(), "Select emitted more than one Stats event");
stats = event.details;
}
aws_sdk_s3::types::SelectObjectContentEventStream::End(_) => {
assert!(stats.is_some(), "Select emitted End before Stats");
saw_end = true;
}
_ => assert!(stats.is_none(), "Select emitted a non-terminal event after Stats"),
}
aws_sdk_s3::types::SelectObjectContentEventStream::Stats(event) => stats = event.details,
aws_sdk_s3::types::SelectObjectContentEventStream::End(_) => {
saw_end = true;
break;
}
_ => {}
}
}
Ok::<(), Box<dyn Error + Send + Sync>>(())
})
.await
.map_err(|_| -> Box<dyn Error + Send + Sync> { "Select response timed out".into() })??;
let stats = stats.ok_or("Select response ended without a Stats event")?;
let input_len = i64::try_from(body.len())?;
@@ -219,10 +249,11 @@ async fn assert_input_byte_stats(
assert_eq!(stats.bytes_processed(), Some(input_len));
assert_eq!(stats.bytes_returned(), Some(i64::try_from(records_len)?));
if progress_enabled {
let progress = last_progress.ok_or("Select response ended without a Progress event")?;
assert_eq!(progress.bytes_scanned(), stats.bytes_scanned());
assert_eq!(progress.bytes_processed(), stats.bytes_processed());
assert_eq!(progress.bytes_returned(), stats.bytes_returned());
if let Some(progress) = last_progress {
assert!(stats.bytes_scanned() >= progress.bytes_scanned());
assert!(stats.bytes_processed() >= progress.bytes_processed());
assert!(stats.bytes_returned() >= progress.bytes_returned());
}
} else {
assert!(last_progress.is_none(), "disabled request progress emitted a Progress event");
}
@@ -231,7 +262,7 @@ async fn assert_input_byte_stats(
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn test_select_object_content_reports_input_byte_stats() -> TestResult<()> {
async fn test_select_object_content_http_event_order_crc_and_input_byte_stats() -> TestResult<()> {
const CSV_BODY: &[u8] = b"name,age\nAlice,30\nBob,25\n";
const JSON_LINES_BODY: &[u8] = b"{\"name\":\"Alice\"}\n{\"name\":\"Bob\"}\n";
const JSON_DOCUMENT_BODY: &[u8] = b"[{\"name\":\"Alice\"},{\"name\":\"Bob\"}]";
@@ -289,6 +320,60 @@ async fn test_select_object_content_reports_input_byte_stats() -> TestResult<()>
Ok(())
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn test_select_object_content_http_disconnect_releases_query() -> TestResult<()> {
const OBJECT: &str = "disconnect.csv";
const ROWS: usize = 16 * 1024;
const RELEASE_BACKOFF: Duration = Duration::from_millis(25);
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
env.start_rustfs_server_with_env(vec![], &[("RUSTFS_S3SELECT_MAX_CONCURRENT_QUERIES", "1")])
.await?;
let client = env.create_s3_client();
setup_test_bucket(&client).await?;
let row = format!("{}\n", "x".repeat(1023));
let mut body = Vec::with_capacity("value\n".len() + ROWS * row.len());
body.extend_from_slice(b"value\n");
for _ in 0..ROWS {
body.extend_from_slice(row.as_bytes());
}
client
.put_object()
.bucket(BUCKET)
.key(OBJECT)
.body(Bytes::from(body).into())
.send()
.await?;
// Leaving this response body unread fills the bounded HTTP/event channels before the query can finish.
let first = csv_select_request(&client, OBJECT).send().await?;
let saturated = csv_select_request(&client, OBJECT)
.send()
.await
.expect_err("the first HTTP stream should retain the only query permit");
assert_eq!(saturated.as_service_error().and_then(ProvideErrorMetadata::code), Some("SlowDown"));
drop(first);
let second = tokio::time::timeout(Duration::from_secs(5), async {
loop {
match csv_select_request(&client, OBJECT).send().await {
Ok(response) => return Ok::<_, Box<dyn Error + Send + Sync>>(response),
Err(error) if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown") => {
tokio::time::sleep(RELEASE_BACKOFF).await;
}
Err(error) => return Err(format!("unexpected Select error after disconnect: {error}").into()),
}
}
})
.await
.map_err(|_| -> Box<dyn Error + Send + Sync> { "disconnected Select did not release its query permit".into() })??;
drop(second);
Ok(())
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn test_select_object_content_csv_basic() -> TestResult<()> {
let (_env, client) = create_test_environment().await?;
+5 -5
View File
@@ -461,8 +461,8 @@ pub mod rpc {
tonic_boot_epoch_challenge, tonic_boot_epoch_response_headers, tonic_rpc_auth_failure_reason,
verify_ns_scanner_capability, verify_ns_scanner_capability_with_tier_registry_generation, verify_put_file_auth_trailer,
verify_put_file_capability, verify_rpc_signature, verify_tonic_boot_epoch_response, verify_tonic_canonical_body_digest,
verify_tonic_mutation_body_digest, verify_tonic_rpc_response_proof, verify_tonic_rpc_signature,
verify_tonic_rpc_signature_with_bootstrap,
verify_tonic_mutation_body_digest, verify_tonic_mutation_body_digest_reject_unsigned, verify_tonic_rpc_response_proof,
verify_tonic_rpc_signature, verify_tonic_rpc_signature_with_bootstrap,
};
}
@@ -489,9 +489,9 @@ pub mod storage {
pub use crate::core::pools::HealLifecycleExpiryContext;
pub use crate::store::HealWalkVersion;
pub use crate::store::{
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, all_local_disk, all_local_disk_path, find_local_disk_by_ref, init_local_disks,
init_local_disks_with_instance_ctx, init_lock_clients, prewarm_local_disk_id_map,
prewarm_local_disk_id_map_with_instance_ctx,
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, ScannerDataMovementPauseStatus, all_local_disk, all_local_disk_path,
find_local_disk_by_ref, init_local_disks, init_local_disks_with_instance_ctx, init_lock_clients,
prewarm_local_disk_id_map, prewarm_local_disk_id_map_with_instance_ctx,
};
}
+200 -5
View File
@@ -24,6 +24,7 @@ use crate::runtime::sources as runtime_sources;
use aws_credential_types::Credentials as SdkCredentials;
use aws_credential_types::provider::{ProvideCredentials, error::CredentialsError, future};
use aws_sdk_s3::config::Region as SdkRegion;
use aws_sdk_s3::config::RequestChecksumCalculation;
use aws_sdk_s3::config::SharedHttpClient;
use aws_sdk_s3::error::ProvideErrorMetadata;
use aws_sdk_s3::error::SdkError;
@@ -39,6 +40,7 @@ use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::Tagging as SdkTagging;
use aws_sdk_s3::types::{
ChecksumMode, CompletedMultipartUpload, CompletedPart, ObjectLockLegalHoldStatus, ObjectLockRetentionMode,
ServerSideEncryption,
};
use aws_sdk_s3::{Client as S3Client, Config as S3Config, operation::head_object::HeadObjectOutput};
use aws_sdk_s3::{config::SharedCredentialsProvider, types::BucketVersioningStatus};
@@ -1071,7 +1073,8 @@ impl BucketTargetSys {
.endpoint_url(endpoint.clone())
.credentials_provider(SharedCredentialsProvider::new(RemoteTargetCredentialsProvider { credentials: creds }))
.region(SdkRegion::new(target.region.clone()))
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest());
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest())
.request_checksum_calculation(replication_request_checksum_calculation());
if should_force_path_style(target) {
config_builder = config_builder.force_path_style(true);
@@ -1367,6 +1370,25 @@ fn loopback_replication_targets_allowed() -> bool {
.unwrap_or(false)
}
const REPLICATION_STREAMING_CHECKSUMS_ENV: &str = "RUSTFS_REPLICATION_STREAMING_CHECKSUMS";
/// Streaming trailer checksums make the SDK frame request bodies as
/// `aws-chunked`; a target that does not decode that framing stores the frames
/// verbatim, silently corrupting every replica while the transfer itself
/// succeeds (#6853). Plain signed payloads are the compatible default; the env
/// knob restores trailer checksums for fleets whose targets are all known to
/// decode them.
fn replication_request_checksum_calculation() -> RequestChecksumCalculation {
if std::env::var(REPLICATION_STREAMING_CHECKSUMS_ENV)
.map(|v| v.eq_ignore_ascii_case("true") || v == "1")
.unwrap_or(false)
{
RequestChecksumCalculation::WhenSupported
} else {
RequestChecksumCalculation::WhenRequired
}
}
fn validate_replication_target_endpoint(url: &Url) -> Result<(), OutboundUrlError> {
validate_replication_target_endpoint_inner(url, loopback_replication_targets_allowed())
}
@@ -1746,6 +1768,17 @@ impl Default for AdvancedPutOptions {
}
}
/// The subset of the target's PutObject response replication audits.
#[derive(Debug, Clone)]
pub struct RemotePutObjectResponse {
/// Version id the target assigned (`x-amz-version-id`).
pub version_id: Option<String>,
/// ETag of what the target stored; `None` when the target withheld it or
/// when its encryption mode (SSE-KMS / SSE-C) makes it incomparable to
/// the source ETag. `None` is therefore "not decidable", never evidence.
pub etag: Option<String>,
}
#[derive(Clone)]
pub struct PutObjectOptions {
pub user_metadata: HashMap<String, String>,
@@ -2291,7 +2324,9 @@ impl TargetClient {
/// On success returns the version id the target assigned (from
/// `x-amz-version-id`), letting callers audit the version-identity
/// contract — a target that adopts the source version echoes it back.
/// contract — a target that adopts the source version echoes it back
/// together with the ETag of what the target actually stored, so callers
/// can detect a target that persisted transformed bytes (#6853).
pub async fn put_object(
&self,
bucket: &str,
@@ -2299,7 +2334,7 @@ impl TargetClient {
size: i64,
body: ByteStream,
opts: &PutObjectOptions,
) -> Result<Option<String>, S3ClientError> {
) -> Result<RemotePutObjectResponse, S3ClientError> {
let mut headers = opts.header();
let builder = self.client.put_object();
@@ -2334,7 +2369,25 @@ impl TargetClient {
.send()
.await
{
Ok(output) => Ok(output.version_id().map(ToOwned::to_owned)),
Ok(output) => {
// Under SSE-KMS/DSSE or SSE-C the target's ETag is not the MD5
// of the stored plaintext, so it cannot be compared against the
// source ETag; withhold it rather than let a caller conclude
// corruption from an opaque value.
let etag_comparable = output.sse_customer_algorithm().is_none()
&& !matches!(
output.server_side_encryption(),
Some(ServerSideEncryption::AwsKms) | Some(ServerSideEncryption::AwsKmsDsse)
);
Ok(RemotePutObjectResponse {
version_id: output.version_id().map(ToOwned::to_owned),
etag: if etag_comparable {
output.e_tag().map(ToOwned::to_owned)
} else {
None
},
})
}
Err(e) => match e {
SdkError::ServiceError(service_err) => {
let err = service_err.into_err();
@@ -2673,6 +2726,145 @@ mod tests {
}
}
type RecordedHeaders = Arc<std::sync::Mutex<Vec<Vec<(String, String)>>>>;
/// Records full request headers and answers with canned response headers,
/// for asserting wire framing and response parsing.
#[derive(Clone, Debug)]
struct RecordingHeaderConnector {
request_headers: RecordedHeaders,
response_headers: Vec<(String, String)>,
}
impl SmithyHttpConnector for RecordingHeaderConnector {
fn call(&self, request: HttpRequest) -> HttpConnectorFuture {
self.request_headers
.lock()
.expect("recorded header lock should not be poisoned")
.push(
request
.headers()
.iter()
.map(|(k, v)| (k.to_string(), v.to_string()))
.collect(),
);
let mut response = HttpResponse::new(
aws_smithy_runtime_api::http::StatusCode::try_from(200_u16).expect("200 should be a valid response status"),
SdkBody::empty(),
);
for (name, value) in &self.response_headers {
response.headers_mut().insert(name.clone(), value.clone());
}
HttpConnectorFuture::ready(Ok(response))
}
}
fn header_recording_target_client(response_headers: Vec<(String, String)>) -> (TargetClient, RecordedHeaders) {
let request_headers: RecordedHeaders = Arc::new(std::sync::Mutex::new(Vec::new()));
let connector = SharedHttpConnector::new(RecordingHeaderConnector {
request_headers: Arc::clone(&request_headers),
response_headers,
});
let http_client = http_client_fn(move |_settings, _components| connector.clone());
let client = s3_client_for_test(443, Some(http_client));
(
TargetClient {
endpoint: "https://localhost:443".to_string(),
credentials: None,
bucket: "target-bucket".to_string(),
storage_class: String::new(),
disable_proxy: false,
arn: "arn:rustfs:replication:us-east-1:target:bucket".to_string(),
reset_id: String::new(),
secure: true,
health_check_duration: Duration::from_secs(5),
replicate_sync: false,
client: Arc::new(client),
},
request_headers,
)
}
fn streaming_test_body(payload: &'static [u8]) -> ByteStream {
let stream = tokio_util::io::ReaderStream::new(std::io::Cursor::new(payload));
let body = http_body_util::StreamBody::new(futures::StreamExt::map(stream, |r| r.map(http_body::Frame::data)));
ByteStream::new(SdkBody::from_body_1_x(body))
}
#[test]
fn replication_checksums_default_to_plain_payloads() {
assert!(matches!(
replication_request_checksum_calculation(),
RequestChecksumCalculation::WhenRequired
));
}
#[tokio::test]
async fn replication_put_object_sends_plain_signed_payloads_by_default() {
let (client, recorded) = header_recording_target_client(Vec::new());
client
.put_object("target-bucket", "object", 4, streaming_test_body(b"data"), &PutObjectOptions::default())
.await
.expect("recorded put_object should succeed");
let recorded = recorded.lock().expect("recorded header lock should not be poisoned");
let headers = &recorded[0];
let header = |name: &str| {
headers
.iter()
.find(|(k, _)| k.eq_ignore_ascii_case(name))
.map(|(_, v)| v.as_str())
};
// The #6853 regression shape: trailer checksums force aws-chunked
// framing, which a non-decoding target stores verbatim as the object.
assert_eq!(header("x-amz-trailer"), None, "streaming uploads must not carry a trailer checksum");
assert!(
header("content-encoding").is_none_or(|v| !v.contains("aws-chunked")),
"streaming uploads must not be aws-chunked framed"
);
assert_eq!(header("x-amz-decoded-content-length"), None);
assert_eq!(header("content-length"), Some("4"));
}
#[tokio::test]
async fn put_object_returns_the_etag_the_target_stored() {
let (client, _) =
header_recording_target_client(vec![("etag".to_string(), "\"9a0364b9e99bb480dd25e1f0284c8555\"".to_string())]);
let response = client
.put_object(
"target-bucket",
"object",
4,
ByteStream::from_static(b"data"),
&PutObjectOptions::default(),
)
.await
.expect("recorded put_object should succeed");
assert_eq!(response.etag.as_deref(), Some("\"9a0364b9e99bb480dd25e1f0284c8555\""));
}
#[tokio::test]
async fn put_object_withholds_the_etag_under_target_side_kms() {
let (client, _) = header_recording_target_client(vec![
("etag".to_string(), "\"9a0364b9e99bb480dd25e1f0284c8555\"".to_string()),
("x-amz-server-side-encryption".to_string(), "aws:kms".to_string()),
]);
let response = client
.put_object(
"target-bucket",
"object",
4,
ByteStream::from_static(b"data"),
&PutObjectOptions::default(),
)
.await
.expect("recorded put_object should succeed");
assert!(
response.etag.is_none(),
"a KMS-encrypted replica's etag is not the content MD5 and must be withheld"
);
}
#[derive(Clone, Debug)]
struct RecordingAuthConnector {
signed_requests: Arc<std::sync::Mutex<Vec<(bool, bool)>>>,
@@ -2969,7 +3161,10 @@ mod tests {
.credentials_provider(SharedCredentialsProvider::new(credentials))
.region(SdkRegion::new("us-east-1"))
.force_path_style(true)
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest());
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest())
// Mirror the production remote-target builder so recorded requests
// exercise the same checksum/framing behavior (#6853).
.request_checksum_calculation(replication_request_checksum_calculation());
if let Some(http_client) = http_client {
config = config.http_client(http_client);
}
@@ -12195,7 +12195,7 @@ mod tests {
#[tokio::test]
#[serial]
async fn tier_free_version_recovery_continues_after_deleted_marker_bucket() {
let (_paths, ecstore) = setup_test_env().await;
let (disk_paths, ecstore) = setup_test_env().await;
let suffix = Uuid::new_v4().simple();
let earlier_bucket = format!("zzzz-recovery-{suffix}-a");
let deleted_marker = format!("zzzz-recovery-{suffix}-m");
@@ -12203,11 +12203,7 @@ mod tests {
let later_object = "a-before-stale-marker";
create_test_bucket(&ecstore, &earlier_bucket).await;
create_test_bucket(&ecstore, &later_bucket).await;
let mut reader = PutObjReader::from_vec(b"cursor reset probe".to_vec());
ecstore
.put_object(&later_bucket, later_object, &mut reader, &ObjectOptions::default())
.await
.expect("successor bucket object should be created");
seed_recoverable_free_version(&disk_paths, &later_bucket, later_object, None, None).await;
let page = list_tier_free_versions(
Arc::clone(&ecstore),
@@ -12220,14 +12216,10 @@ mod tests {
.expect("recovery should resume at the first bucket after a deleted marker bucket");
assert_eq!(page.buckets_scanned, 1, "the later bucket must not be skipped");
assert_eq!(
page.scanned_entries, 1,
"the deleted bucket's object marker must not skip objects in the successor bucket"
);
ecstore
.delete_object(&later_bucket, later_object, ObjectOptions::default())
.await
.expect("successor bucket object should be removed");
assert_eq!(page.items.len(), 1, "the successor bucket's recoverable object must be returned");
assert_eq!(page.items[0].bucket, later_bucket);
assert_eq!(page.items[0].name, later_object);
remove_seeded_free_version(&disk_paths, &later_bucket, later_object).await;
for bucket in [&earlier_bucket, &later_bucket] {
ecstore
.delete_bucket(bucket, &DeleteBucketOptions::default())
+158
View File
@@ -2512,11 +2512,169 @@ pub(crate) mod test_support {
mod tests {
use super::test_support::isolated_store_over_temp_disks;
use super::*;
use crate::bucket::metadata::{
BUCKET_ACCELERATE_CONFIG, BUCKET_CORS_CONFIG, BUCKET_LIFECYCLE_CONFIG, BUCKET_LOGGING_CONFIG, BUCKET_NOTIFICATION_CONFIG,
BUCKET_POLICY_CONFIG, BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, BUCKET_REPLICATION_CONFIG, BUCKET_REQUEST_PAYMENT_CONFIG,
BUCKET_SSECONFIG, BUCKET_TAGGING_CONFIG, BUCKET_VERSIONING_CONFIG, BUCKET_WEBSITE_CONFIG, OBJECT_LOCK_CONFIG,
};
use crate::bucket::target::{BucketTarget, BucketTargetType, Credentials};
use crate::config::com::read_config;
use crate::storage_api_contracts::bucket::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions};
use byteorder::{ByteOrder as _, LittleEndian};
use serial_test::serial;
use tokio::time::timeout;
const NEW_WRITER_REPLICATION_XML: &[u8] = br#"<ReplicationConfiguration xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><Role>arn:aws:iam::111122223333:role/replication-role</Role><Rule><ID>rollback</ID><Priority>1</Priority><Filter><Prefix>documents/</Prefix></Filter><Status>Enabled</Status><Destination><Bucket>arn:aws:s3:::replica-bucket</Bucket></Destination><DeleteMarkerReplication><Status>Disabled</Status></DeleteMarkerReplication></Rule></ReplicationConfiguration>"#;
const NEW_WRITER_CONFIGS: [(&str, &[u8]); 14] = [
(BUCKET_POLICY_CONFIG, br#"{"Version":"2012-10-17","Statement":[]}"#),
(BUCKET_NOTIFICATION_CONFIG, br#"<NotificationConfiguration/>"#),
(
BUCKET_LIFECYCLE_CONFIG,
br#"<LifecycleConfiguration><Rule><ID>expire</ID><Status>Enabled</Status><Filter><Prefix>logs/</Prefix></Filter><Expiration><Days>30</Days></Expiration></Rule></LifecycleConfiguration>"#,
),
(
OBJECT_LOCK_CONFIG,
br#"<ObjectLockConfiguration><ObjectLockEnabled>Enabled</ObjectLockEnabled><Rule><DefaultRetention><Mode>GOVERNANCE</Mode><Days>7</Days></DefaultRetention></Rule></ObjectLockConfiguration>"#,
),
(
BUCKET_VERSIONING_CONFIG,
br#"<VersioningConfiguration><Status>Enabled</Status></VersioningConfiguration>"#,
),
(
BUCKET_SSECONFIG,
br#"<ServerSideEncryptionConfiguration><Rule><ApplyServerSideEncryptionByDefault><SSEAlgorithm>AES256</SSEAlgorithm></ApplyServerSideEncryptionByDefault></Rule></ServerSideEncryptionConfiguration>"#,
),
(
BUCKET_TAGGING_CONFIG,
r#"<Tagging><TagSet><Tag><Key>environment</Key><Value>测试-🦀</Value></Tag></TagSet></Tagging>"#.as_bytes(),
),
(BUCKET_REPLICATION_CONFIG, NEW_WRITER_REPLICATION_XML),
(
BUCKET_CORS_CONFIG,
br#"<CORSConfiguration><CORSRule><AllowedMethod>GET</AllowedMethod><AllowedOrigin>https://example.test</AllowedOrigin></CORSRule></CORSConfiguration>"#,
),
(BUCKET_LOGGING_CONFIG, br#"<BucketLoggingStatus/>"#),
(
BUCKET_WEBSITE_CONFIG,
br#"<WebsiteConfiguration><IndexDocument><Suffix>index.html</Suffix></IndexDocument></WebsiteConfiguration>"#,
),
(
BUCKET_ACCELERATE_CONFIG,
br#"<AccelerateConfiguration><Status>Enabled</Status></AccelerateConfiguration>"#,
),
(
BUCKET_REQUEST_PAYMENT_CONFIG,
br#"<RequestPaymentConfiguration><Payer>Requester</Payer></RequestPaymentConfiguration>"#,
),
(
BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG,
br#"<PublicAccessBlockConfiguration><BlockPublicAcls>true</BlockPublicAcls><IgnorePublicAcls>true</IgnorePublicAcls><BlockPublicPolicy>true</BlockPublicPolicy><RestrictPublicBuckets>false</RestrictPublicBuckets></PublicAccessBlockConfiguration>"#,
),
];
#[tokio::test]
async fn g_d3_003_new_writer_replication_loads_without_fail_closed_state() {
let (dirs, store) = isolated_store_over_temp_disks().await;
let bucket = "rollback-new-replication";
for dir in &dirs {
std::fs::create_dir_all(dir.path().join(bucket)).expect("rollback fixture bucket should be created");
}
let writer = BucketMetadataSys::new(store.clone());
let mut metadata = BucketMetadata::new(bucket);
metadata
.update_config(BUCKET_REPLICATION_CONFIG, NEW_WRITER_REPLICATION_XML.to_vec())
.expect("new-writer replication XML should be accepted before persistence");
writer
.persist_new_and_set(metadata)
.await
.expect("new-writer replication metadata should persist");
let old_reader = BucketMetadataSys::new(store);
let (loaded, _) = old_reader
.get_replication_config(bucket)
.await
.expect("old metadata_sys must not classify new-writer replication XML as invalid");
assert_eq!(loaded.role, "arn:aws:iam::111122223333:role/replication-role");
assert_eq!(loaded.rules.len(), 1);
assert_eq!(loaded.rules[0].id.as_deref(), Some("rollback"));
}
#[tokio::test]
async fn g_d3_004_new_writer_metadata_blob_keeps_legacy_header_and_configs() {
let (dirs, store) = isolated_store_over_temp_disks().await;
let bucket = "rollback-new-metadata";
for dir in &dirs {
std::fs::create_dir_all(dir.path().join(bucket)).expect("rollback fixture bucket should be created");
}
let writer = BucketMetadataSys::new(store.clone());
let mut metadata = BucketMetadata::new(bucket);
for (config_file, bytes) in NEW_WRITER_CONFIGS {
metadata
.update_config(config_file, bytes.to_vec())
.unwrap_or_else(|err| panic!("new-writer {config_file} fixture must be valid: {err}"));
}
writer
.persist_new_and_set(metadata)
.await
.expect("new-writer metadata should persist");
let path = BucketMetadata::new(bucket).save_file_path();
let blob = read_config(store.clone(), &path)
.await
.expect("persisted .metadata.bin should be readable");
assert_eq!(
LittleEndian::read_u16(&blob[0..2]),
1,
"bucket metadata format must stay rollback-readable"
);
assert_eq!(
LittleEndian::read_u16(&blob[2..4]),
1,
"bucket metadata version must stay rollback-readable"
);
let loaded = load_bucket_metadata(store, bucket)
.await
.expect("old read_bucket_metadata path must load the new-writer blob");
let loaded_configs: [(&str, &[u8]); 14] = [
(BUCKET_POLICY_CONFIG, &loaded.policy_config_json),
(BUCKET_NOTIFICATION_CONFIG, &loaded.notification_config_xml),
(BUCKET_LIFECYCLE_CONFIG, &loaded.lifecycle_config_xml),
(OBJECT_LOCK_CONFIG, &loaded.object_lock_config_xml),
(BUCKET_VERSIONING_CONFIG, &loaded.versioning_config_xml),
(BUCKET_SSECONFIG, &loaded.encryption_config_xml),
(BUCKET_TAGGING_CONFIG, &loaded.tagging_config_xml),
(BUCKET_REPLICATION_CONFIG, &loaded.replication_config_xml),
(BUCKET_CORS_CONFIG, &loaded.cors_config_xml),
(BUCKET_LOGGING_CONFIG, &loaded.logging_config_xml),
(BUCKET_WEBSITE_CONFIG, &loaded.website_config_xml),
(BUCKET_ACCELERATE_CONFIG, &loaded.accelerate_config_xml),
(BUCKET_REQUEST_PAYMENT_CONFIG, &loaded.request_payment_config_xml),
(BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, &loaded.public_access_block_config_xml),
];
for ((expected_name, expected), (loaded_name, actual)) in NEW_WRITER_CONFIGS.into_iter().zip(loaded_configs) {
assert_eq!(loaded_name, expected_name);
assert_eq!(actual, expected, "old read_bucket_metadata changed {expected_name} bytes");
}
assert!(loaded.policy_config.is_some());
assert!(loaded.notification_config.is_some());
assert!(loaded.lifecycle_config.is_some());
assert!(loaded.object_lock_config.is_some());
assert!(loaded.versioning_config.is_some());
assert!(loaded.sse_config.is_some());
assert!(loaded.tagging_config.is_some());
assert!(loaded.replication_config.is_some());
assert!(loaded.cors_config.is_some());
assert!(loaded.logging_config.is_some());
assert!(loaded.website_config.is_some());
assert!(loaded.accelerate_config.is_some());
assert!(loaded.request_payment_config.is_some());
assert!(loaded.public_access_block_config.is_some());
}
#[tokio::test]
async fn malformed_delete_configs_are_not_treated_as_absent() {
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
@@ -20,8 +20,8 @@ pub use rustfs_replication::{
pub(crate) use rustfs_replication::{
ReplicationDeleteSource, ReplicationMultipartPartInput, ReplicationResyncTargetObject, delete_marker_purge_mrf_entry,
delete_marker_purge_version_id, delete_replication_creates_marker, delete_replication_missing_source_decision,
delete_replication_object_opts, heal_uses_delete_replication_path, is_retryable_delete_replication_head_error,
is_version_delete_replication, replicate_delete_outcome, replication_etags_match, replication_multipart_complete_actual_size,
replication_multipart_part_plan, resync_existing_delete_replication_info, resync_target_for_object,
should_retry_delete_marker_purge, target_delete_version_id,
delete_replication_object_opts, heal_uses_delete_replication_path, is_object_lock_denied_delete,
is_retryable_delete_replication_head_error, is_version_delete_replication, replicate_delete_outcome, replication_etags_match,
replication_multipart_complete_actual_size, replication_multipart_part_plan, resync_existing_delete_replication_info,
resync_target_for_object, should_retry_delete_marker_purge, single_part_replica_etag_mismatch, target_delete_version_id,
};
@@ -3177,6 +3177,19 @@ pub(crate) async fn queue_replication_heal_internal(
}
}
ReplicationHealQueueAction::QueueDelete(dv) => {
// A purge the peer denied under object lock cannot succeed until
// the lock lapses (#6850); requeuing it every heal cycle only
// burns bandwidth and failure counters. The backoff expires on
// its own, so the purge is probed again — and converges — once
// the retention window has a chance of being over.
if super::replication_object_decision_boundary::is_version_delete_replication(&dv.delete_object)
&& super::replication_resyncer::object_lock_denied_purge_backoff_active(&dv)
{
return ReplicationHealQueueResult {
object_info: roi,
admission: ReplicationQueueAdmission::Skipped,
};
}
let admission = if let Some(pool) = runtime_sources::replication_pool() {
pool.queue_replica_delete_task(dv).await
} else {
@@ -30,10 +30,10 @@ use super::replication_msgp_boundary::ReplicationMsgpCodec;
use super::replication_object_config::{ReplicationConfig, get_replication_config, must_replicate};
use super::replication_object_decision_boundary::{
MustReplicateOptions, ReplicationMultipartPartInput, delete_marker_purge_mrf_entry, delete_marker_purge_version_id,
delete_replication_creates_marker, heal_uses_delete_replication_path, is_retryable_delete_replication_head_error,
is_version_delete_replication, replicate_delete_outcome, replication_etags_match, replication_multipart_complete_actual_size,
replication_multipart_part_plan, resync_existing_delete_replication_info, should_retry_delete_marker_purge,
target_delete_version_id,
delete_replication_creates_marker, heal_uses_delete_replication_path, is_object_lock_denied_delete,
is_retryable_delete_replication_head_error, is_version_delete_replication, replicate_delete_outcome, replication_etags_match,
replication_multipart_complete_actual_size, replication_multipart_part_plan, resync_existing_delete_replication_info,
should_retry_delete_marker_purge, single_part_replica_etag_mismatch, target_delete_version_id,
};
use super::replication_queue_boundary::{DeletedObjectReplicationInfo, ReplicationQueueAdmission};
use super::replication_resync_boundary::ResyncStatusType;
@@ -54,7 +54,7 @@ use super::replication_storage_boundary::{
};
use super::replication_target_boundary::{
ERR_REPLICATION_SSEC_PASSTHROUGH_UNSUPPORTED, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions,
ReplicationTargetStore, S3ClientError, SsecPassthroughCapability, SsecPassthroughGate, TargetClient,
RemotePutObjectResponse, ReplicationTargetStore, S3ClientError, SsecPassthroughCapability, SsecPassthroughGate, TargetClient,
is_replication_target_offline_error, replication_action_for_target_head, replication_complete_multipart_options,
replication_delete_marker_purge_remove_options, replication_delete_remove_options, replication_force_delete_remove_options,
replication_object_is_ssec_encrypted, replication_put_object_header_size, replication_put_object_options,
@@ -96,7 +96,7 @@ use tokio::task::{JoinHandle, JoinSet};
use tokio::time::Duration as TokioDuration;
use tokio_util::io::ReaderStream;
use tokio_util::sync::CancellationToken;
use tracing::{debug, error, instrument, trace, warn};
use tracing::{debug, error, info, instrument, trace, warn};
const BACKGROUND_WALKDIR_TIMEOUT: TokioDuration = TokioDuration::from_secs(60);
const ENV_REPL_RESYNC_MAX_JOBS: &str = "RUSTFS_REPL_RESYNC_MAX_JOBS";
@@ -112,11 +112,13 @@ const EVENT_REPLICATION_DELETE_SKIPPED: &str = "replication_delete_skipped";
const EVENT_REPLICATION_FORCE_DELETE_SKIPPED: &str = "replication_force_delete_skipped";
const EVENT_RESYNC_TASK_FAILED: &str = "replication_resync_task_failed";
const EVENT_RESYNC_TARGET_OPERATION_FAILED: &str = "replication_resync_target_operation_failed";
const EVENT_REPLICATION_ABORT_RETRY_RESOLVED: &str = "replication_abort_retry_resolved";
const EVENT_RESYNC_RUNTIME_CHANNEL_FAILED: &str = "replication_resync_runtime_channel_failed";
const EVENT_DELETE_MARKER_PURGE_FAILED: &str = "replication_delete_marker_purge_failed";
const EVENT_DELETE_MARKER_PURGE_MRF: &str = "replication_delete_marker_purge_mrf";
const METRIC_DELETE_MARKER_PURGE_TOTAL: &str = "rustfs_replication_delete_marker_purge_total";
const EVENT_REPLICATION_VERSION_IDENTITY_DRIFT: &str = "replication_version_identity_drift";
const EVENT_REPLICATION_PURGE_OBJECT_LOCK_DENIED: &str = "replication_purge_object_lock_denied";
#[allow(
dead_code,
@@ -194,6 +196,123 @@ const METRIC_VERSION_IDENTITY_DRIFT_TOTAL: &str = "rustfs_replication_version_id
/// after a restart is acceptable.
static VERSION_IDENTITY_WARNED_ARNS: LazyLock<StdMutex<HashSet<String>>> = LazyLock::new(|| StdMutex::new(HashSet::new()));
/// Version purges the peer denied under object lock (#6850). Replication
/// carries no governance bypass, so such a purge cannot succeed until the
/// lock on the replica lapses — retrying every heal cycle only burns
/// bandwidth and failure counters. Entries suppress heal requeues for the
/// backoff window; after it expires one probe runs again, so the purge still
/// converges on its own once retention ends. In-process only: a restart
/// costs at most one extra probe per entry.
const OBJECT_LOCK_DENIED_PURGE_BACKOFF: std::time::Duration = std::time::Duration::from_secs(60 * 60);
const OBJECT_LOCK_DENIED_PURGE_CACHE_MAX: usize = 4096;
type ObjectLockDeniedPurgeKey = (String, String, String);
struct ObjectLockDeniedPurge {
denied_at: std::time::Instant,
denied_arns: HashSet<String>,
}
static OBJECT_LOCK_DENIED_PURGES: LazyLock<StdMutex<HashMap<ObjectLockDeniedPurgeKey, ObjectLockDeniedPurge>>> =
LazyLock::new(|| StdMutex::new(HashMap::new()));
fn object_lock_denied_purge_key(dobj: &DeletedObjectReplicationInfo) -> ObjectLockDeniedPurgeKey {
let version_id = dobj
.delete_object
.delete_marker_version_id
.or(dobj.delete_object.version_id)
.unwrap_or_default();
(dobj.bucket.clone(), dobj.delete_object.object_name.clone(), version_id.to_string())
}
fn record_object_lock_denied_purge(dobj: &DeletedObjectReplicationInfo, arn: &str) {
let mut denied = OBJECT_LOCK_DENIED_PURGES
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
if denied.len() >= OBJECT_LOCK_DENIED_PURGE_CACHE_MAX {
denied.retain(|_, entry| entry.denied_at.elapsed() < OBJECT_LOCK_DENIED_PURGE_BACKOFF);
}
let key = object_lock_denied_purge_key(dobj);
if denied.len() < OBJECT_LOCK_DENIED_PURGE_CACHE_MAX || denied.contains_key(&key) {
let entry = denied.entry(key).or_insert_with(|| ObjectLockDeniedPurge {
denied_at: std::time::Instant::now(),
denied_arns: HashSet::new(),
});
entry.denied_at = std::time::Instant::now();
entry.denied_arns.insert(arn.to_string());
}
// Still full after dropping expired entries: skip recording — the purge
// then simply keeps retrying, which is the pre-#6850 behavior.
}
/// Whether a heal requeue of this delete can only reach targets that denied
/// it under object lock within the backoff window. A target the entry does
/// not cover (another peer, or one whose denial expired) keeps the requeue
/// flowing — suppressing it would delay a purge that could succeed there.
pub(crate) fn object_lock_denied_purge_backoff_active(dobj: &DeletedObjectReplicationInfo) -> bool {
let key = object_lock_denied_purge_key(dobj);
let mut denied = OBJECT_LOCK_DENIED_PURGES
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
match denied.get(&key) {
Some(entry) if entry.denied_at.elapsed() < OBJECT_LOCK_DENIED_PURGE_BACKOFF => {
let admitted = dobj.admitted_target_arns();
!admitted.is_empty() && admitted.iter().all(|arn| entry.denied_arns.contains(arn))
}
Some(_) => {
denied.remove(&key);
false
}
None => false,
}
}
const REPLICA_ETAG_VERIFY_ENV: &str = "RUSTFS_REPLICATION_REPLICA_ETAG_VERIFY";
/// Escape hatch for a target whose 32-hex ETags are legitimately not the
/// content MD5 (e.g. a gateway hashing its own ciphertext without announcing
/// SSE in the response) — such a target would otherwise fail every object.
fn replica_etag_verification_enabled() -> bool {
std::env::var(REPLICA_ETAG_VERIFY_ENV)
.map(|v| !(v.eq_ignore_ascii_case("false") || v == "0"))
.unwrap_or(true)
}
/// A 200 from the target is not proof the replica holds the source bytes: a
/// target that stores a transformed payload (e.g. undecoded `aws-chunked`
/// frames, #6853) returns the ETag of what it actually wrote. Reporting
/// COMPLETED over such a replica is silent corruption, so a decidable
/// mismatch fails the replication instead. An SSE-C ciphertext passthrough
/// transfer is exempt: the wire bytes are ciphertext while the source ETag is
/// the plaintext MD5, and that path has its own HEAD-back audit.
fn verify_single_part_replica(
object_info: &ObjectInfo,
response: &RemotePutObjectResponse,
ciphertext_passthrough: bool,
) -> std::result::Result<(), std::io::Error> {
if ciphertext_passthrough || !replica_etag_verification_enabled() {
return Ok(());
}
if single_part_replica_etag_mismatch(object_info.etag.as_deref(), response.etag.as_deref()) {
// The differing ETags go into the structured log; the error message
// stays constant so same-cause failures bucket together downstream.
warn!(
event = EVENT_RESYNC_TARGET_OPERATION_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
bucket = %object_info.bucket,
object = %object_info.name,
source_etag = ?object_info.etag,
replica_etag = ?response.etag,
operation = "verify_replica_etag",
"Replication target operation failed"
);
return Err(std::io::Error::other(REPLICA_ETAG_MISMATCH_ERROR));
}
Ok(())
}
const REPLICA_ETAG_MISMATCH_ERROR: &str = "replica etag mismatch: the target persisted different bytes than were sent";
fn audit_target_version_identity(tgt_client: &TargetClient, source_version_id: &str, assigned_version_id: Option<&str>) {
if !version_identity_drifted(source_version_id, assigned_version_id) {
return;
@@ -2708,19 +2827,42 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli
}
}
Err(e) => {
warn!(
event = EVENT_RESYNC_TARGET_OPERATION_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
bucket = tgt_client.bucket,
object = dobj.delete_object.object_name,
version_id = ?version_id,
delete_marker = dobj.delete_object.delete_marker,
is_version_purge,
error = %e,
operation = "replicate_delete_to_target",
"Replication target operation failed"
);
let object_lock_denied = is_version_purge && is_object_lock_denied_delete(e.code.as_deref(), e.message.as_deref());
if object_lock_denied {
// Terminal for as long as the lock holds: the peer retains
// this version and replication carries no governance bypass
// (#6850), so the sites stay diverged until the retention or
// legal hold on the replica lapses. Surface it loudly instead
// of letting a silent failed counter and a hot heal-retry
// loop stand in for the divergence.
record_object_lock_denied_purge(dobj, &tgt_client.arn);
error!(
event = EVENT_REPLICATION_PURGE_OBJECT_LOCK_DENIED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
bucket = tgt_client.bucket,
object = dobj.delete_object.object_name,
version_id = ?version_id,
arn = %tgt_client.arn,
error = %e,
operation = "replicate_delete_to_target",
"Replicated version purge denied by object lock on the target; the sites stay diverged until the lock lapses"
);
} else {
warn!(
event = EVENT_RESYNC_TARGET_OPERATION_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
bucket = tgt_client.bucket,
object = dobj.delete_object.object_name,
version_id = ?version_id,
delete_marker = dobj.delete_object.delete_marker,
is_version_purge,
error = %e,
operation = "replicate_delete_to_target",
"Replication target operation failed"
);
}
rinfo.error = Some(e.to_string());
if !is_version_purge {
rinfo.replication_status = ReplicationStatusType::Failed;
@@ -3274,14 +3416,15 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo {
let result = tgt_client
.put_object(&tgt_client.bucket, &object, transfer_size, byte_stream, &put_opts)
.await
.map(|assigned_version_id| {
.map_err(|e| std::io::Error::other(e.to_string()))
.and_then(|response| {
audit_target_version_identity(
&tgt_client,
&put_opts.internal.source_version_id,
assigned_version_id.as_deref(),
)
})
.map_err(|e| std::io::Error::other(e.to_string()));
response.version_id.as_deref(),
);
verify_single_part_replica(&object_info, &response, obj_opts.raw_data_movement_read)
});
result.err()
} {
rinfo.replication_status = ReplicationStatusType::Failed;
@@ -3942,14 +4085,15 @@ async fn replicate_all_payload_to_target<S: ReplicationObjectIO>(
.tgt_client
.put_object(&ctx.tgt_client.bucket, ctx.object, ctx.transfer_size, byte_stream, &ctx.put_opts)
.await
.map(|assigned_version_id| {
.map_err(|e| std::io::Error::other(e.to_string()))
.and_then(|response| {
audit_target_version_identity(
ctx.tgt_client,
&ctx.put_opts.internal.source_version_id,
assigned_version_id.as_deref(),
)
})
.map_err(|e| std::io::Error::other(e.to_string()));
response.version_id.as_deref(),
);
verify_single_part_replica(ctx.object_info, &response, ctx.obj_opts.raw_data_movement_read)
});
result.err()
}
}
@@ -4036,28 +4180,132 @@ async fn replicate_object_with_multipart<S: ReplicationObjectIO>(ctx: MultipartR
let arn = ctx.arn;
let result = replicate_multipart_parts_and_complete(ctx, &upload_id).await;
abort_multipart_on_failure(result, dst_bucket, object, &upload_id, arn, || async {
cli.abort_multipart_upload(dst_bucket, object, &upload_id).await
})
abort_multipart_on_failure(
result,
dst_bucket,
object,
&upload_id,
arn,
|| async { cli.abort_multipart_upload(dst_bucket, object, &upload_id).await },
|| {
schedule_replication_abort_retry(
cli.clone(),
dst_bucket.to_string(),
object.to_string(),
upload_id.clone(),
arn.to_string(),
)
},
)
.await
}
const REPLICATION_ABORT_RETRY_ATTEMPTS: u32 = 5;
const REPLICATION_ABORT_RETRY_INITIAL_DELAY_SECS: u64 = 30;
/// The immediate abort usually fails for the same reason the transfer did —
/// the target is unreachable — and MRF only retries the *object*: every replay
/// mints a fresh upload id, so a failed abort would leak its upload on the
/// target forever (#6854). Retry the abort on a detached, bounded backoff
/// (~30s..8m) so it lands once the target comes back; an upload the target no
/// longer knows counts as cleaned up.
fn schedule_replication_abort_retry(cli: Arc<TargetClient>, dst_bucket: String, object: String, upload_id: String, arn: String) {
tokio::spawn(async move {
let mut delay_secs = REPLICATION_ABORT_RETRY_INITIAL_DELAY_SECS;
for attempt in 1..=REPLICATION_ABORT_RETRY_ATTEMPTS {
tokio::time::sleep(tokio::time::Duration::from_secs(delay_secs)).await;
delay_secs = delay_secs.saturating_mul(2);
match cli.abort_multipart_upload(&dst_bucket, &object, &upload_id).await {
Ok(()) => {
info!(
event = EVENT_REPLICATION_ABORT_RETRY_RESOLVED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
target_bucket = %dst_bucket,
object = %object,
arn = %arn,
upload_id = %upload_id,
operation = "abort_multipart_upload_retry",
attempt,
"Replication abort retry cleaned up the orphaned upload"
);
return;
}
Err(err) if target_upload_already_removed(&err) => {
info!(
event = EVENT_REPLICATION_ABORT_RETRY_RESOLVED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
target_bucket = %dst_bucket,
object = %object,
arn = %arn,
upload_id = %upload_id,
operation = "abort_multipart_upload_retry",
attempt,
"Replication abort retry found the upload already removed"
);
return;
}
Err(err) => {
warn!(
event = EVENT_RESYNC_TARGET_OPERATION_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
target_bucket = %dst_bucket,
object = %object,
arn = %arn,
upload_id = %upload_id,
operation = "abort_multipart_upload_retry",
attempt,
error = %err,
"Replication target operation failed"
);
}
}
}
// Terminal: the upload id stays in the log so an operator can reap it
// with list-multipart-uploads/abort by hand (the #6840 contract).
warn!(
event = EVENT_RESYNC_TARGET_OPERATION_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC,
target_bucket = %dst_bucket,
object = %object,
arn = %arn,
upload_id = %upload_id,
operation = "abort_multipart_upload_retry",
result = "gave_up",
"Replication abort retries exhausted; the incomplete upload remains on the target"
);
});
}
/// AWS answers an abort for an unknown upload with `NoSuchUpload`; that means
/// the orphan is gone (aborted elsewhere or expired), which is the goal state.
fn target_upload_already_removed(err: &S3ClientError) -> bool {
err.code.as_deref() == Some("NoSuchUpload")
}
/// Best-effort abort of the target-side multipart upload once the transfer has
/// failed past CreateMultipartUpload; without it every failed attempt leaves an
/// invisible incomplete upload on the target that keeps billing for its parts.
/// The abort outcome never replaces the transfer error: an abort failure is
/// only logged and `result` is returned as-is.
async fn abort_multipart_on_failure<F, Fut>(
async fn abort_multipart_on_failure<F, Fut, R>(
result: std::io::Result<()>,
dst_bucket: &str,
object: &str,
upload_id: &str,
arn: &str,
abort: F,
schedule_abort_retry: R,
) -> std::io::Result<()>
where
F: FnOnce() -> Fut,
Fut: std::future::Future<Output = std::result::Result<(), S3ClientError>>,
R: FnOnce(),
{
if result.is_ok() {
return result;
@@ -4075,6 +4323,9 @@ where
error = %abort_err,
"Replication target operation failed"
);
if !target_upload_already_removed(&abort_err) {
schedule_abort_retry();
}
}
result
}
@@ -5354,27 +5605,72 @@ mod tests {
assert!(!resync_state_accepts_update(&current, &stale));
}
#[test]
fn object_lock_denied_purge_backoff_tracks_version_and_target() {
let denied = DeletedObjectReplicationInfo {
bucket: "worm-backoff-test-bucket".to_string(),
target_arn: "arn:rustfs:replication::worm-test:t1".to_string(),
delete_object: ReplicationDeletedObject {
object_name: "locked-object".to_string(),
version_id: Some(uuid::Uuid::new_v4()),
..Default::default()
},
..Default::default()
};
assert!(!object_lock_denied_purge_backoff_active(&denied));
record_object_lock_denied_purge(&denied, "arn:rustfs:replication::worm-test:t1");
assert!(object_lock_denied_purge_backoff_active(&denied));
// A requeue that can also reach a target this denial does not cover
// must keep flowing: the purge may succeed there.
let mut other_target = denied.clone();
other_target.target_arn = "arn:rustfs:replication::worm-test:t2".to_string();
assert!(!object_lock_denied_purge_backoff_active(&other_target));
// A different version of the same object must not be suppressed.
let mut other_version = denied;
other_version.delete_object.version_id = Some(uuid::Uuid::new_v4());
assert!(!object_lock_denied_purge_backoff_active(&other_version));
}
#[tokio::test]
async fn abort_multipart_on_failure_skips_abort_when_transfer_succeeded() {
let aborted = Arc::new(AtomicBool::new(false));
let flag = aborted.clone();
let retry_scheduled = Arc::new(AtomicBool::new(false));
let retry_flag = retry_scheduled.clone();
let result = abort_multipart_on_failure(Ok(()), "dst-bucket", "obj", "upload-1", "arn:dest", move || async move {
flag.store(true, Ordering::SeqCst);
Ok(())
})
let result = abort_multipart_on_failure(
Ok(()),
"dst-bucket",
"obj",
"upload-1",
"arn:dest",
move || async move {
flag.store(true, Ordering::SeqCst);
Ok(())
},
move || retry_flag.store(true, Ordering::SeqCst),
)
.await;
assert!(result.is_ok());
assert!(!aborted.load(Ordering::SeqCst));
assert!(!retry_scheduled.load(Ordering::SeqCst));
}
#[tokio::test]
async fn abort_multipart_on_failure_aborts_and_keeps_transfer_error() {
let aborted = Arc::new(AtomicBool::new(false));
let flag = aborted.clone();
let retry_scheduled = Arc::new(AtomicBool::new(false));
let retry_flag = retry_scheduled.clone();
// The abort itself failing must not mask the transfer error.
// The abort itself failing must not mask the transfer error, and a
// failed abort must hand the upload id to the retry schedule (#6854):
// the object itself is re-replicated under a fresh upload id, so
// nothing else will ever abort this one.
let result = abort_multipart_on_failure(
Err(std::io::Error::other("transfer failed")),
"dst-bucket",
@@ -5385,10 +5681,34 @@ mod tests {
flag.store(true, Ordering::SeqCst);
Err(S3ClientError::new("abort failed"))
},
move || retry_flag.store(true, Ordering::SeqCst),
)
.await;
assert!(aborted.load(Ordering::SeqCst));
assert!(retry_scheduled.load(Ordering::SeqCst));
assert_eq!(result.unwrap_err().to_string(), "transfer failed");
}
#[tokio::test]
async fn abort_multipart_on_failure_does_not_retry_a_gone_upload() {
let retry_scheduled = Arc::new(AtomicBool::new(false));
let retry_flag = retry_scheduled.clone();
let result = abort_multipart_on_failure(
Err(std::io::Error::other("transfer failed")),
"dst-bucket",
"obj",
"upload-1",
"arn:dest",
|| async { Err(S3ClientError::with_metadata("gone", None, Some("NoSuchUpload".to_string()), None)) },
move || retry_flag.store(true, Ordering::SeqCst),
)
.await;
// NoSuchUpload means the orphan no longer exists; retrying would only
// produce noise.
assert!(!retry_scheduled.load(Ordering::SeqCst));
assert_eq!(result.unwrap_err().to_string(), "transfer failed");
}
}
@@ -36,8 +36,8 @@ use time::OffsetDateTime;
use time::format_description::well_known::Rfc3339;
pub(crate) use crate::bucket::bucket_target_sys::{
AdvancedPutOptions, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, RemoveObjectOptions, S3ClientError,
TargetClient, resolve_read_api_version_id,
AdvancedPutOptions, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, RemotePutObjectResponse, RemoveObjectOptions,
S3ClientError, TargetClient, resolve_read_api_version_id,
};
#[cfg(test)]
pub(crate) use crate::bucket::target::BucketTarget;
+1
View File
@@ -73,6 +73,7 @@ pub fn check_valid_bucket_name_strict(bucket_name: &str) -> Result<()> {
check_bucket_name_common(bucket_name, true)
}
// RUSTFS_COMPAT_TODO(s3gate-metadata-xml): the s3s codec reads persisted XML during migration. Remove after every supported writer uses the gateway codec and every retained metadata object and backup archive is verified or rewritten.
pub fn deserialize<T>(input: &[u8]) -> xml::DeResult<T>
where
T: for<'xml> xml::Deserialize<'xml>,
@@ -1307,6 +1307,32 @@ pub fn verify_tonic_mutation_body_digest<T>(request: &tonic::Request<T>, canonic
verify_tonic_mutation_body_digest_with_strictness(request, canonical_body, internode_rpc_body_digest_strict())
}
/// Verify a non-disk mutation without accepting a newly-generated unsigned v2 body.
///
/// The disk mutation lane has a rolling-upgrade exception for `UNSIGNED-PAYLOAD`
/// while peer replay-cache capability is being discovered. Historical v2 peers
/// used the fixed `unsigned` nonce before body-digest rollout; preserve that
/// exact marker for mixed-version compatibility, but reject unsigned v2
/// requests that omit it or present a different nonce.
pub fn verify_tonic_mutation_body_digest_reject_unsigned<T>(
request: &tonic::Request<T>,
canonical_body: &[u8],
) -> std::io::Result<()> {
let version = request
.metadata()
.get(RPC_AUTH_VERSION_HEADER)
.and_then(|value| value.to_str().ok());
let digest = request
.metadata()
.get(RPC_CONTENT_SHA256_HEADER)
.and_then(|value| value.to_str().ok());
let nonce = request.metadata().get(RPC_NONCE_HEADER).and_then(|value| value.to_str().ok());
if version == Some(RPC_AUTH_VERSION_V2) && digest == Some(UNSIGNED_PAYLOAD) && nonce != Some("unsigned") {
return Err(std::io::Error::other("RPC mutation requires a body-bound v2 signature"));
}
verify_tonic_mutation_body_digest(request, canonical_body)
}
/// [`verify_tonic_mutation_body_digest`] with the strict gate injected as a parameter, so both
/// rollout postures are unit-testable without racing on process-global environment variables.
fn verify_tonic_mutation_body_digest_with_strictness<T>(
+2 -2
View File
@@ -39,8 +39,8 @@ pub use http_auth::{
sign_tonic_rpc_response_proof, tonic_boot_epoch_challenge, tonic_boot_epoch_response_headers, tonic_rpc_auth_failure_reason,
verify_ns_scanner_capability, verify_ns_scanner_capability_with_tier_registry_generation, verify_put_file_auth_trailer,
verify_put_file_capability, verify_rpc_signature, verify_tonic_boot_epoch_response, verify_tonic_canonical_body_digest,
verify_tonic_mutation_body_digest, verify_tonic_rpc_response_proof, verify_tonic_rpc_signature,
verify_tonic_rpc_signature_with_bootstrap,
verify_tonic_mutation_body_digest, verify_tonic_mutation_body_digest_reject_unsigned, verify_tonic_rpc_response_proof,
verify_tonic_rpc_signature, verify_tonic_rpc_signature_with_bootstrap,
};
#[cfg(test)]
pub(crate) use internode_data_transport::TcpHttpInternodeDataTransport;
File diff suppressed because it is too large Load Diff
+163 -17
View File
@@ -109,7 +109,10 @@ static USAGE_MEMORY_GENERATION: AtomicU64 = AtomicU64::new(0);
/// strictly tighter than beta.11 (usage treated as 0) and strictly more
/// available than a blanket 503. The fallback applies to any window without
/// authoritative usage, not only pre-v2 upgrades; the values always come from
/// the last persisted scanner output. Loads go through the TTL-bounded
/// the last persisted scanner output — pre-discard sizes of the
/// authoritative snapshot first, backfilled per bucket from the observed
/// (nonconverged) snapshot for buckets no authoritative cycle has covered
/// yet (issue #6852). Loads go through the TTL-bounded
/// snapshot cache, so the quota path adds at most one backend read per
/// [`DATA_USAGE_CACHE_TTL_SECS`] window. Returns `None` for buckets absent
/// from every persisted snapshot — those still fail closed.
@@ -168,7 +171,7 @@ fn fresh_cached_data_usage_snapshot(
fn cache_data_usage_snapshot_result(
cache: &mut Option<CachedDataUsageSnapshot>,
result: Result<(DataUsageInfo, HashMap<String, u64>), Error>,
result: Result<LoadedUsageBaseline, Error>,
loaded_at: tokio::time::Instant,
refresh_generation: u64,
current_generation: u64,
@@ -178,7 +181,19 @@ fn cache_data_usage_snapshot_result(
}
Some(match result {
Ok((info, degraded_baseline)) => {
Ok(LoadedUsageBaseline {
info,
mut degraded_baseline,
observed_unavailable,
}) => {
// A flaky observed read must not shrink quota coverage for a TTL
// window: carry the previous refresh's baseline entries forward,
// letting the fresh (authoritative) values win where they exist.
if observed_unavailable && let Some(previous) = cache.as_ref() {
for (bucket, size) in &previous.degraded_baseline {
degraded_baseline.entry(bucket.clone()).or_insert(*size);
}
}
*cache = Some(CachedDataUsageSnapshot {
info: Some(info.clone()),
loaded_at,
@@ -1113,24 +1128,78 @@ async fn load_data_usage_snapshot(store: Arc<ECStore>) -> Result<(DataUsageInfo,
/// Load data usage info from backend storage
#[instrument(skip(store))]
pub async fn load_data_usage_from_backend(store: Arc<ECStore>) -> Result<DataUsageInfo, Error> {
Ok(load_data_usage_from_backend_with_baseline(store).await?.0)
Ok(load_data_usage_from_backend_with_baseline(store).await?.info)
}
/// One refresh of the persisted usage snapshot plus the quota-admission
/// baseline derived from it.
struct LoadedUsageBaseline {
info: DataUsageInfo,
degraded_baseline: HashMap<String, u64>,
/// True when the observed snapshot could not be read (a transport error,
/// not absence): the cached loader then carries the previous refresh's
/// baseline entries forward instead of shrinking quota coverage for a
/// whole TTL window over one flaky read.
observed_unavailable: bool,
}
/// Like [`load_data_usage_from_backend`], but also returns the pre-discard
/// per-bucket sizes so the cached loader can retain them as the degraded
/// quota-admission baseline (issue #5716).
async fn load_data_usage_from_backend_with_baseline(store: Arc<ECStore>) -> Result<(DataUsageInfo, HashMap<String, u64>), Error> {
let (data_usage_info, source) = load_data_usage_snapshot(store).await?;
Ok(normalize_loaded_data_usage(data_usage_info, source.is_authoritative()).await)
async fn load_data_usage_from_backend_with_baseline(store: Arc<ECStore>) -> Result<LoadedUsageBaseline, Error> {
let (loaded_snapshot, source) = load_data_usage_snapshot(store.clone()).await?;
// The observed-newness gate below compares against the snapshot as
// persisted, before normalization demotes or discards anything.
let authoritative_as_persisted = loaded_snapshot.clone();
let (info, mut degraded_baseline) = normalize_loaded_data_usage(loaded_snapshot, source.is_authoritative()).await;
// A bucket without a converged scanner cycle behind it — a freshly joined
// replica whose every cycle is superseded by the sustained replication
// write stream, or a bucket created after the last converged cycle on a
// busy site (#6852) — has no authoritative size, and quota admission
// fails its writes closed indefinitely. The observed (nonconverged)
// snapshot those superseded cycles still publish is the only grounded
// usage in that window, so it backfills buckets the loaded baseline does
// not cover; a value already in the baseline always wins. The newness
// gate ties the observation to this exact authoritative snapshot, so a
// stale observed object left behind by an earlier incarnation (e.g. a
// deleted and recreated bucket) cannot inject ghost usage. Loads sit
// behind the same TTL cache as the snapshot itself, so this adds at most
// one backend read per TTL window.
let mut observed_unavailable = false;
match load_observed_data_usage_snapshot(store).await {
Ok(Some(observed)) if observed_data_usage_is_newer(&observed, &authoritative_as_persisted) => {
backfill_degraded_baseline_from_observed(&mut degraded_baseline, &observed);
}
Ok(_) => {}
Err(_) => observed_unavailable = true,
}
Ok(LoadedUsageBaseline {
info,
degraded_baseline,
observed_unavailable,
})
}
async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Option<DataUsageInfo> {
/// Fill quota-baseline gaps from an observed (nonconverged) snapshot without
/// overriding any bucket the authoritative baseline already covers.
fn backfill_degraded_baseline_from_observed(degraded_baseline: &mut HashMap<String, u64>, observed: &DataUsageInfo) {
for (bucket, usage) in &observed.buckets_usage {
degraded_baseline.entry(bucket.clone()).or_insert(usage.size);
}
}
/// `Ok(None)` means the observed snapshot is absent or invalid (a settled
/// answer); `Err` means it could not be read at all, so the caller may keep
/// using what it learned from a previous read.
async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Result<Option<DataUsageInfo>, Error> {
let data = match read_config_preserve_empty(store, &DATA_USAGE_OBSERVED_OBJ_NAME_PATH).await {
Ok(data) => data,
Err(Error::ConfigNotFound) => return None,
Err(Error::ConfigNotFound) => return Ok(None),
Err(err) => {
record_usage_snapshot_failure("read_observed", DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str(), &err);
return None;
return Err(err);
}
};
@@ -1139,7 +1208,7 @@ async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Option<DataUs
if info.usage_snapshot_converged == Some(false)
&& (info.is_complete_bucket_usage_snapshot() || info.is_valid_partial_snapshot()) =>
{
Some(info)
Ok(Some(info))
}
Ok(_) => {
error!(
@@ -1150,11 +1219,11 @@ async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Option<DataUs
object = %DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str(),
"observed data usage snapshot was not a structurally complete nonconverged view"
);
None
Ok(None)
}
Err(err) => {
record_usage_snapshot_decode_failure("parse_observed", DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str(), &err);
None
Ok(None)
}
}
}
@@ -1212,7 +1281,9 @@ fn merge_partial_observation_for_admin(mut authoritative: DataUsageInfo, observe
async fn load_admin_data_usage_from_backend(store: Arc<ECStore>) -> Result<DataUsageInfo, Error> {
let (authoritative, source) = load_data_usage_snapshot(store.clone()).await?;
let observed = load_observed_data_usage_snapshot(store).await;
// For the one-shot admin view a failed observed read degrades to "no
// observation", same as before the read was fallible.
let observed = load_observed_data_usage_snapshot(store).await.ok().flatten();
let (selected, selected_is_current_format) =
select_admin_data_usage_snapshot(authoritative, source.is_authoritative(), observed);
Ok(normalize_loaded_data_usage(selected, selected_is_current_format).await.0)
@@ -1375,7 +1446,11 @@ pub async fn load_admin_data_usage_from_backend_cached(store: Arc<ECStore>) -> R
let refresh_generation = admin_data_usage_snapshot_generation();
let result = load_admin_data_usage_from_backend(store.clone())
.await
.map(|info| (info, HashMap::new()));
.map(|info| LoadedUsageBaseline {
info,
degraded_baseline: HashMap::new(),
observed_unavailable: false,
});
let loaded_at = tokio::time::Instant::now();
let mut cache = admin_data_usage_snapshot_cache().write().await;
if let Some(result) = cache_data_usage_snapshot_result(
@@ -2526,6 +2601,37 @@ mod tests {
use std::sync::Arc;
use tokio::{io::AsyncReadExt, sync::Mutex};
#[test]
fn observed_snapshot_only_backfills_baseline_gaps() {
let mut baseline = HashMap::from([("covered".to_string(), 111_u64)]);
let observed = DataUsageInfo {
buckets_usage: HashMap::from([
(
"covered".to_string(),
BucketUsageInfo {
size: 999,
..Default::default()
},
),
(
"replica-only".to_string(),
BucketUsageInfo {
size: 42,
..Default::default()
},
),
]),
..Default::default()
};
backfill_degraded_baseline_from_observed(&mut baseline, &observed);
// The authoritative value must win; only the uncovered bucket (#6852:
// a replica that never landed a converged cycle) is filled in.
assert_eq!(baseline.get("covered"), Some(&111));
assert_eq!(baseline.get("replica-only"), Some(&42));
}
#[derive(Debug, Default)]
struct UsageCasState {
object: Option<(Vec<u8>, u64)>,
@@ -3479,7 +3585,11 @@ mod tests {
let first = cache_data_usage_snapshot_result(
&mut cache,
Ok((expected, HashMap::new())),
Ok(LoadedUsageBaseline {
info: expected,
degraded_baseline: HashMap::new(),
observed_unavailable: false,
}),
loaded_at,
refresh_generation,
data_usage_snapshot_generation(),
@@ -3494,6 +3604,38 @@ mod tests {
assert_snapshot_bucket(&cached, "bucket");
}
#[test]
#[serial]
fn unavailable_observed_read_keeps_previous_baseline_coverage() {
let loaded_at = tokio::time::Instant::now();
let refresh_generation = data_usage_snapshot_generation();
let mut cache = Some(CachedDataUsageSnapshot {
info: Some(data_usage_info_for_test("bucket", 1, 42, SystemTime::UNIX_EPOCH)),
loaded_at,
degraded_baseline: HashMap::from([("observed-only".to_string(), 7_u64), ("covered".to_string(), 1)]),
});
cache_data_usage_snapshot_result(
&mut cache,
Ok(LoadedUsageBaseline {
info: data_usage_info_for_test("bucket", 1, 42, SystemTime::UNIX_EPOCH),
degraded_baseline: HashMap::from([("covered".to_string(), 2_u64)]),
observed_unavailable: true,
}),
loaded_at,
refresh_generation,
data_usage_snapshot_generation(),
)
.expect("an uninterrupted refresh should populate the cache")
.expect("successful load must be returned");
let baseline = &cache.as_ref().expect("cache must be populated").degraded_baseline;
// The bucket only the (now unreadable) observed snapshot covered must
// survive the refresh; the freshly loaded value wins where it exists.
assert_eq!(baseline.get("observed-only"), Some(&7));
assert_eq!(baseline.get("covered"), Some(&2));
}
#[test]
#[serial]
fn cache_invalidation_during_refresh_prevents_stale_snapshot_resurrection() {
@@ -3508,7 +3650,11 @@ mod tests {
let stale_result = cache_data_usage_snapshot_result(
&mut cache,
Ok((data_usage_info_for_test("stale", 1, 42, SystemTime::UNIX_EPOCH), HashMap::new())),
Ok(LoadedUsageBaseline {
info: data_usage_info_for_test("stale", 1, 42, SystemTime::UNIX_EPOCH),
degraded_baseline: HashMap::new(),
observed_unavailable: false,
}),
loaded_at,
refresh_generation,
data_usage_snapshot_generation(),
@@ -16,10 +16,143 @@ use rustfs_filemeta::{MetacacheReader, MetacacheWriter};
use std::io::Cursor;
use std::path::PathBuf;
use std::sync::Arc;
use std::sync::atomic::{AtomicBool, Ordering};
use tokio::fs;
use tokio::io::AsyncReadExt;
use tokio::sync::RwLock;
/// Test-only lock client whose refresh path can be rejected independently of
/// every other lock operation. The observed event is awaitable so lock-loss
/// tests do not depend on sleeps or scheduler timing.
#[derive(Debug)]
pub(crate) struct RefreshLossLockClient {
inner: rustfs_lock::LocalClient,
reject_refresh: AtomicBool,
rejected_refresh: AtomicBool,
rejected_refresh_notify: tokio::sync::Notify,
}
impl RefreshLossLockClient {
pub(crate) fn with_manager(manager: Arc<rustfs_lock::GlobalLockManager>) -> Self {
Self {
inner: rustfs_lock::LocalClient::with_manager(manager),
reject_refresh: AtomicBool::new(false),
rejected_refresh: AtomicBool::new(false),
rejected_refresh_notify: tokio::sync::Notify::new(),
}
}
pub(crate) fn reject_refreshes(&self) {
self.reject_refresh.store(true, Ordering::Release);
}
pub(crate) fn refreshes_rejected(&self) -> bool {
self.rejected_refresh.load(Ordering::Acquire)
}
pub(crate) async fn wait_for_rejected_refresh(
&self,
timeout: std::time::Duration,
) -> std::result::Result<(), tokio::time::error::Elapsed> {
tokio::time::timeout(timeout, async {
loop {
let notified = self.rejected_refresh_notify.notified();
if self.refreshes_rejected() {
return;
}
notified.await;
}
})
.await
}
}
#[async_trait::async_trait]
impl rustfs_lock::LockClient for RefreshLossLockClient {
async fn acquire_lock(&self, request: &rustfs_lock::LockRequest) -> rustfs_lock::Result<rustfs_lock::LockResponse> {
rustfs_lock::LockClient::acquire_lock(&self.inner, request).await
}
async fn release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
rustfs_lock::LockClient::release(&self.inner, lock_id).await
}
async fn refresh(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
if self.reject_refresh.load(Ordering::Acquire) {
self.rejected_refresh.store(true, Ordering::Release);
self.rejected_refresh_notify.notify_waiters();
return Ok(false);
}
rustfs_lock::LockClient::refresh(&self.inner, lock_id).await
}
async fn force_release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
rustfs_lock::LockClient::force_release(&self.inner, lock_id).await
}
async fn check_status(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<Option<rustfs_lock::LockInfo>> {
rustfs_lock::LockClient::check_status(&self.inner, lock_id).await
}
async fn list_lock_leases(&self) -> Vec<rustfs_lock::LockLeaseInfo> {
rustfs_lock::LockClient::list_lock_leases(&self.inner).await
}
async fn get_stats(&self) -> rustfs_lock::Result<rustfs_lock::LockStats> {
rustfs_lock::LockClient::get_stats(&self.inner).await
}
async fn close(&self) -> rustfs_lock::Result<()> {
rustfs_lock::LockClient::close(&self.inner).await
}
async fn is_online(&self) -> bool {
rustfs_lock::LockClient::is_online(&self.inner).await
}
async fn is_local(&self) -> bool {
rustfs_lock::LockClient::is_local(&self.inner).await
}
}
#[tokio::test]
async fn refresh_loss_lock_client_keeps_rejection_observable_for_late_waiters() {
let manager = Arc::new(rustfs_lock::GlobalLockManager::Enabled(Arc::new(
rustfs_lock::FastObjectLockManager::new(),
)));
let client = RefreshLossLockClient::with_manager(manager);
let resource = rustfs_lock::ObjectKey::new("bucket", "object");
let response = rustfs_lock::LockClient::acquire_lock(
&client,
&rustfs_lock::LockRequest::new(resource, rustfs_lock::LockType::Shared, "refresh-loss-harness"),
)
.await
.expect("acquire should reach the inner local client");
let lock_id = response.lock_info.expect("the inner local client should acquire the lock").id;
assert_eq!(
rustfs_lock::LockClient::list_lock_leases(&client).await.len(),
1,
"lease diagnostics must remain transparent through the refresh wrapper"
);
client.reject_refreshes();
assert!(
!rustfs_lock::LockClient::refresh(&client, &lock_id)
.await
.expect("refresh should return a response")
);
client
.wait_for_rejected_refresh(std::time::Duration::from_millis(50))
.await
.expect("a waiter registered after rejection must still observe the event");
assert!(client.refreshes_rejected());
assert!(
rustfs_lock::LockClient::release(&client, &lock_id)
.await
.expect("release should reach the inner local client")
);
}
/// Returns the backing [`tempfile::TempDir`]s alongside the set so callers keep
/// them alive for the test's duration and the directories are removed on drop.
pub(crate) async fn make_local_set_disks(drive_count: usize, parity_count: usize) -> (Vec<tempfile::TempDir>, Arc<SetDisks>) {
+239 -60
View File
@@ -46,6 +46,7 @@ type ShardReadFuture<'a> = Pin<Box<dyn Future<Output = (usize, ShardReadCost, Re
type OwnedShardReadFuture<'a, R> =
Pin<Box<dyn Future<Output = (usize, ShardReadCost, Result<Vec<u8>, Error>, Option<BitrotReader<R>>, bool)> + Send + 'a>>;
pub(crate) type DeferredReaderReopener<R> = Arc<dyn Fn(usize) -> Option<BitrotReader<R>> + Send + Sync>;
pub(crate) type DecodeOutcome = (usize, Option<std::io::Error>, bool);
type ShardIndexes = SmallVec<[usize; INLINE_SHARD_SLOTS]>;
type ActiveReaders = SmallVec<[bool; INLINE_SHARD_SLOTS]>;
@@ -574,6 +575,7 @@ pub(crate) struct ParallelReader<R> {
read_timeout: Duration,
verify_reconstruction: bool,
locality_preference_enabled: bool,
demand_bound_lockstep: bool,
// Request-scoped shard buffers keyed by shard index. Keeping ownership in
// `ParallelReader` avoids dropping unused parity/backup slot buffers between stripes.
buffers: ShardBufferPool,
@@ -585,10 +587,8 @@ pub(crate) struct ParallelReader<R> {
// it to the current stripe when it is engaged mid-object (backlog#923).
engaged: SmallVec<[bool; INLINE_SHARD_SLOTS]>,
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
// Copy-source hedges use a fresh deferred reader so cancelling a hedge
// never consumes the unopened reader reserved for a later stripe. The
// vector is empty for callers that do not provide a reopen factory (tests
// and the ordinary GET path retain the handle-based behavior).
// Demand-bound hedges use a fresh deferred reader so cancelling a hedge
// never consumes the unopened reader reserved for a later stripe.
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
stripe_index: usize,
}
@@ -777,9 +777,9 @@ where
// reads all live readers on every stripe — the pre-backlog#923
// behavior. With the gate on, only data slots start engaged; parity is
// engaged on demand, stripe-aligned through its deferred handle.
let data_shards_only = get_lockstep_data_shards_only_enabled();
let demand_bound_lockstep = get_lockstep_data_shards_only_enabled();
let engaged: SmallVec<_> = (0..readers.len())
.map(|index| !data_shards_only || index < e.data_shards)
.map(|index| !demand_bound_lockstep || index < e.data_shards)
.collect();
ParallelReader {
readers,
@@ -793,6 +793,7 @@ where
read_timeout,
verify_reconstruction,
locality_preference_enabled: get_shard_locality_preference_enabled(),
demand_bound_lockstep,
buffers: ShardBufferPool::new(e.data_shards + e.parity_shards),
stripe_state: None,
engaged,
@@ -1275,7 +1276,7 @@ where
/// realigned (no pending deferred handle) is likewise retired instead of
/// being read out of position.
async fn read_lockstep(&mut self, state: &mut StripeReadState) {
if matches!(decode_read_policy(), DecodeReadPolicy::DemandBound) {
if self.demand_bound_lockstep {
self.read_lockstep_demand_bound(state).await;
return;
}
@@ -1531,17 +1532,18 @@ where
}
}
/// Demand-bound lockstep stripe read used by server-side copy sources.
/// Demand-bound data-shards-only lockstep stripe read.
///
/// The ordinary lockstep path can cancel every in-flight reader once it
/// has a quorum because all of its parity readers are already engaged.
/// Copy sources keep parity unopened until a data reader is missing. A
/// hedge therefore has to race the deferred parity reads against the
/// original data reads and may retire the latter only after the parity has
/// produced an actual decode-plus-verification quorum. The futures own
/// their readers so disjoint data/parity slots can be admitted while the
/// other group is still pending; dropping an abandoned future retires its
/// stream without leaving a borrowed slot behind.
/// Copy sources and the data-shards-only rollout gate keep parity unopened
/// until a data reader is missing. A hedge therefore has to race the
/// deferred parity reads against the original data reads and may retire the
/// latter only after parity has produced an actual decode-plus-verification
/// quorum. The futures own their readers so disjoint data/parity slots can
/// be admitted while the other group is still pending; dropping an
/// abandoned future retires its stream without leaving a borrowed slot
/// behind.
async fn read_lockstep_demand_bound(&mut self, state: &mut StripeReadState) {
let num_readers = self.readers.len();
state.reset(num_readers, self.data_shards);
@@ -1576,14 +1578,14 @@ where
let mut completed = 0usize;
let mut failed = 0usize;
let mut first_shard_recorded = false;
let mut active = vec![false; num_readers];
let mut temporary_parity = vec![false; num_readers];
let mut active: ActiveReaders = smallvec![false; num_readers];
let mut temporary_parity: ActiveReaders = smallvec![false; num_readers];
// A deferred parity slot is attempted at most once per stripe. A
// failed disposable hedge keeps its unopened reserve for the next
// stripe, but must not be relaunched in a tight same-stripe retry
// loop (which would defeat the bounded fan-out and amplify a remote
// outage).
let mut attempted_parity = vec![false; num_readers];
let mut attempted_parity: ActiveReaders = smallvec![false; num_readers];
// Once a data reader has returned an error (or was already missing at
// setup), the loss is permanent for lockstep alignment. Use the
// deferred handle and keep parity engaged across subsequent stripes;
@@ -2189,8 +2191,10 @@ impl Erasure {
W: AsyncWrite + Send + Sync + Unpin,
R: crate::erasure::coding::ShardSource,
{
self.decode_inner(writer, readers, offset, length, total_length, None, Vec::new(), Vec::new())
.await
let (written, error, _) = self
.decode_inner(writer, readers, offset, length, total_length, None, Vec::new(), Vec::new())
.await;
(written, error)
}
#[allow(dead_code, reason = "read-cost decode path asserted by this file's tests (backlog#1823)")]
@@ -2207,8 +2211,10 @@ impl Erasure {
W: AsyncWrite + Send + Sync + Unpin,
R: crate::erasure::coding::ShardSource,
{
self.decode_inner(writer, readers, offset, length, total_length, Some(read_costs), Vec::new(), Vec::new())
.await
let (written, error, _) = self
.decode_inner(writer, readers, offset, length, total_length, Some(read_costs), Vec::new(), Vec::new())
.await;
(written, error)
}
/// GET decode entry point that also carries the deferred-parity stripe
@@ -2261,6 +2267,37 @@ impl Erasure {
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
) -> (usize, Option<std::io::Error>)
where
W: AsyncWrite + Send + Sync + Unpin,
R: crate::erasure::coding::ShardSource,
{
let (written, error, _) = self
.decode_inner(
writer,
readers,
offset,
length,
total_length,
read_costs,
deferred_handles,
deferred_reopeners,
)
.await;
(written, error)
}
#[allow(clippy::too_many_arguments)]
pub(crate) async fn decode_with_stripe_handles_and_reopeners_with_diagnostics<W, R>(
&self,
writer: &mut W,
readers: Vec<Option<BitrotReader<R>>>,
offset: usize,
length: usize,
total_length: usize,
read_costs: Option<Vec<ShardReadCost>>,
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
) -> DecodeOutcome
where
W: AsyncWrite + Send + Sync + Unpin,
R: crate::erasure::coding::ShardSource,
@@ -2298,6 +2335,7 @@ impl Erasure {
written: &mut usize,
ret_err: &mut Option<std::io::Error>,
stage_metrics_enabled: bool,
require_surplus_source: bool,
) -> StripeFlow
where
W: AsyncWrite + Send + Sync + Unpin,
@@ -2335,7 +2373,12 @@ impl Erasure {
// missing data shard and an extra source shard was available, verify
// the reconstructed data against that source before streaming bytes.
let reconstruct_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
if let Err(e) = self.decode_data_with_reconstruction_verification(shards) {
let decode_result = if require_surplus_source {
self.decode_data_with_reconstruction_verification_for_lockstep(shards)
} else {
self.decode_data_with_reconstruction_verification(shards)
};
if let Err(e) = decode_result {
record_get_stage_duration_if_enabled(GET_OBJECT_PATH_LEGACY_DUPLEX, GET_STAGE_RECONSTRUCT, reconstruct_stage_start);
let reason = GetObjectFailureReason::DecodeError;
error!(
@@ -2404,36 +2447,48 @@ impl Erasure {
read_costs: Option<Vec<ShardReadCost>>,
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
) -> (usize, Option<std::io::Error>)
) -> DecodeOutcome
where
W: AsyncWrite + Send + Sync + Unpin,
R: crate::erasure::coding::ShardSource,
{
if readers.len() != self.data_shards + self.parity_shards {
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers")));
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers")), false);
}
// block_size/data_shards come from on-disk metadata; a corrupt FileInfo with a
// zero here must surface as an error, not a divide-by-zero panic on every GET.
if self.block_size == 0 || self.data_shards == 0 {
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid erasure coding parameters")));
return (
0,
Some(io::Error::new(ErrorKind::InvalidInput, "Invalid erasure coding parameters")),
false,
);
}
let Some(end_offset) = offset.checked_add(length) else {
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")));
return (
0,
Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")),
false,
);
};
if end_offset > total_length {
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")));
return (
0,
Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")),
false,
);
}
let mut ret_err = None;
if length == 0 {
return (0, ret_err);
return (0, ret_err, false);
}
let mut written = 0;
@@ -2473,6 +2528,7 @@ impl Erasure {
}
};
let mut exact_quorum = false;
if legacy_stripe_prefetch_enabled() {
// Depth-1 stripe prefetch (backlog#930 HP-9 step 2): while the current
// stripe is reconstructed and emitted, the next stripe's shard reads
@@ -2515,6 +2571,7 @@ impl Erasure {
let Some((mut shards, errs)) = current.take() else {
break;
};
exact_quorum |= shards.iter().filter(|shard| shard.is_some()).count() == self.data_shards;
if idx + 1 < blocks.len() {
// Overlap: read stripe idx+1 while reconstructing/emitting idx.
@@ -2546,6 +2603,7 @@ impl Erasure {
// `shards` are borrowed again below. In the `Stop` case that
// drop is what cancels the still-in-flight prefetch read.
let (flow, next): (Option<StripeFlow>, Option<StripeReadOutput>) = {
let require_surplus_source = reader.demand_bound_lockstep;
let read_fut = read_stripe_timed(&mut reader, stage_metrics_enabled);
let emit_fut = self.emit_decoded_stripe(
writer,
@@ -2556,6 +2614,7 @@ impl Erasure {
&mut written,
&mut ret_err,
stage_metrics_enabled,
require_surplus_source,
);
tokio::pin!(read_fut);
tokio::pin!(emit_fut);
@@ -2603,6 +2662,7 @@ impl Erasure {
&mut written,
&mut ret_err,
stage_metrics_enabled,
reader.demand_bound_lockstep,
)
.await
{
@@ -2626,6 +2686,7 @@ impl Erasure {
let stage_metrics_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
let stripe_read_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let (mut shards, errs) = reader.read().await;
exact_quorum |= shards.iter().filter(|shard| shard.is_some()).count() == self.data_shards;
record_get_stage_duration_if_enabled(
GET_OBJECT_PATH_LEGACY_DUPLEX,
GET_STAGE_STRIPE_READ,
@@ -2642,6 +2703,7 @@ impl Erasure {
&mut written,
&mut ret_err,
stage_metrics_enabled,
reader.demand_bound_lockstep,
)
.await
{
@@ -2654,14 +2716,14 @@ impl Erasure {
}
if ret_err.is_some() {
return (written, ret_err);
return (written, ret_err, exact_quorum);
}
if written < length {
ret_err = Some(Error::LessData.into());
}
(written, ret_err)
(written, ret_err, exact_quorum)
}
}
@@ -2866,6 +2928,7 @@ mod tests {
cursor: Cursor<Vec<u8>>,
stall: Duration,
sleep: Option<Pin<Box<Sleep>>>,
stall_polls: Arc<AtomicUsize>,
},
}
@@ -2904,7 +2967,12 @@ mod tests {
TestShardReader::TerminalFileNotFound => {
Poll::Ready(Err(crate::disk::error::terminal_read_error_to_io(Error::FileNotFound)))
}
TestShardReader::PrefixThenSlow { cursor, stall, sleep } => {
TestShardReader::PrefixThenSlow {
cursor,
stall,
sleep,
stall_polls,
} => {
let before = buf.filled().len();
match Pin::new(cursor).poll_read(cx, buf) {
// Cursor still has bytes for the current stripe: serve them.
@@ -2914,6 +2982,7 @@ mod tests {
// the task cleanly (no busy `wake_by_ref` spin), letting the
// `#[tokio::test(start_paused = true)]` clock auto-advance.
Poll::Ready(Ok(())) => {
stall_polls.fetch_add(1, Ordering::SeqCst);
let stall = *stall;
let sleeper = sleep.get_or_insert_with(|| Box::pin(tokio::time::sleep(stall)));
let _ = sleeper.as_mut().poll(cx);
@@ -2942,6 +3011,29 @@ mod tests {
}
}
struct YieldOnceThenFailWriter {
yielded: bool,
}
impl AsyncWrite for YieldOnceThenFailWriter {
fn poll_write(mut self: Pin<&mut Self>, cx: &mut Context<'_>, _buf: &[u8]) -> Poll<io::Result<usize>> {
if !self.yielded {
self.yielded = true;
cx.waker().wake_by_ref();
return Poll::Pending;
}
Poll::Ready(Err(io::Error::new(ErrorKind::BrokenPipe, "injected emit failure after prefetch poll")))
}
fn poll_flush(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Ok(()))
}
fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Ok(()))
}
}
struct DownstreamClosedWriter;
impl AsyncWrite for DownstreamClosedWriter {
@@ -3878,6 +3970,7 @@ mod tests {
(rustfs_config::ENV_OBJECT_DISK_READ_TIMEOUT, Some(READ_TIMEOUT_SECS)),
];
temp_env::async_with_vars(vars, async {
let stall_polls = Arc::new(AtomicUsize::new(0));
let readers: Vec<Option<BitrotReader<TestShardReader>>> = shard_bufs
.iter()
.map(|buf| {
@@ -3887,12 +3980,13 @@ mod tests {
cursor: Cursor::new(prefix),
stall: STALL,
sleep: None,
stall_polls: Arc::clone(&stall_polls),
};
Some(BitrotReader::new(reader, shard_size, hash_algo.clone(), false))
})
.collect();
let mut writer = FailingEmitWriter;
let mut writer = YieldOnceThenFailWriter { yielded: false };
let start = TokioInstant::now();
let (written, err) = erasure.decode(&mut writer, readers, 0, total_len, total_len).await;
let elapsed = start.elapsed();
@@ -3900,6 +3994,10 @@ mod tests {
// Emit failed on stripe 0, so the GET fails with no bytes emitted.
assert!(err.is_some(), "emit failure must surface as an error");
assert_eq!(written, 0, "the failing writer accepts no bytes");
assert!(
stall_polls.load(Ordering::SeqCst) > 0,
"the speculative next-stripe read must be in flight before emit fails"
);
// The decisive assertion: the prefetch read was cancelled rather than
// awaited. Without cancel-safety this would take READ_TIMEOUT_SECS.
assert!(
@@ -4911,6 +5009,24 @@ mod tests {
/// read timeout even though both parity readers were available to engage.
#[tokio::test]
async fn test_demand_bound_lockstep_hedges_to_deferred_parity_quorum() {
with_decode_read_policy(DecodeReadPolicy::DemandBound, assert_deferred_parity_hedges_slow_data()).await;
}
/// The ordinary GET rollout gate must use the same bounded parity race as
/// CopySource. Leaving it on the legacy lockstep loop deadlocks the hedge:
/// that loop waits for a parity success before cancelling the slow data
/// read, but does not admit deferred parity until after the data read ends.
#[tokio::test]
#[serial_test::serial]
async fn test_data_shards_only_gate_hedges_to_deferred_parity_quorum() {
temp_env::async_with_vars(
[(ENV_RUSTFS_GET_LOCKSTEP_DATA_SHARDS_ONLY_ENABLE, Some("true"))],
assert_deferred_parity_hedges_slow_data(),
)
.await;
}
async fn assert_deferred_parity_hedges_slow_data() {
const NUM_SHARDS: usize = 1;
const BLOCK_SIZE: usize = 64;
const DATA_SHARDS: usize = 2;
@@ -4951,33 +5067,27 @@ mod tests {
];
let erasure = Erasure::new(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE);
let (bufs, errs, engaged, readers_remaining) = with_decode_read_policy(DecodeReadPolicy::DemandBound, async {
let mut parallel_reader = ParallelReader::new_with_metrics_path_read_costs_timeout_and_reconstruction_verification(
readers,
erasure,
0,
NUM_SHARDS * BLOCK_SIZE,
None,
vec![ShardReadCost::Unknown; DATA_SHARDS + PARITY_SHARDS],
Duration::from_secs(60),
true,
);
let (bufs, errs) = tokio::time::timeout(Duration::from_secs(2), parallel_reader.read())
.await
.expect("deferred parity must cover a hedged data shard without waiting for read_timeout");
(
bufs,
errs,
parallel_reader.engaged.clone(),
parallel_reader.readers.iter().map(Option::is_some).collect::<Vec<_>>(),
)
})
.await;
let mut parallel_reader = ParallelReader::new_with_metrics_path_read_costs_timeout_and_reconstruction_verification(
readers,
erasure,
0,
NUM_SHARDS * BLOCK_SIZE,
None,
vec![ShardReadCost::Unknown; DATA_SHARDS + PARITY_SHARDS],
Duration::from_secs(60),
true,
);
let (bufs, errs) = tokio::time::timeout(Duration::from_secs(2), parallel_reader.read())
.await
.expect("deferred parity must cover a hedged data shard without waiting for read_timeout");
assert!(matches!(&errs[0], Some(DiskError::Io(err)) if err.kind() == ErrorKind::TimedOut));
assert_eq!(bufs.iter().filter(|buf| buf.is_some()).count(), DATA_SHARDS + 1);
assert_eq!(engaged.as_slice(), &[true, true, true, true]);
assert_eq!(readers_remaining, vec![false, true, true, true]);
assert_eq!(parallel_reader.engaged.as_slice(), &[true, true, true, true]);
assert_eq!(
parallel_reader.readers.iter().map(Option::is_some).collect::<Vec<_>>(),
vec![false, true, true, true]
);
}
/// A fast data failure must admit deferred parity immediately. There is
@@ -5046,6 +5156,24 @@ mod tests {
#[tokio::test]
async fn test_demand_bound_canceled_hedge_preserves_deferred_parity_for_next_stripe() {
with_decode_read_policy(
DecodeReadPolicy::DemandBound,
assert_canceled_hedge_preserves_deferred_parity_for_next_stripe(),
)
.await;
}
#[tokio::test]
#[serial_test::serial]
async fn test_data_shards_only_gate_canceled_hedge_preserves_deferred_parity_for_next_stripe() {
temp_env::async_with_vars(
[(ENV_RUSTFS_GET_LOCKSTEP_DATA_SHARDS_ONLY_ENABLE, Some("true"))],
assert_canceled_hedge_preserves_deferred_parity_for_next_stripe(),
)
.await;
}
async fn assert_canceled_hedge_preserves_deferred_parity_for_next_stripe() {
const BLOCK_SIZE: usize = 64;
const DATA_SHARDS: usize = 2;
const PARITY_SHARDS: usize = 2;
@@ -5094,7 +5222,7 @@ mod tests {
Some(BitrotReader::new(TestShardReader::Pending, SHARD_SIZE, hash_algo, false)),
];
let (first_parity_reserved, second_result) = with_decode_read_policy(DecodeReadPolicy::DemandBound, async {
let (first_parity_reserved, second_result) = {
let erasure = Erasure::new(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE);
let mut parallel_reader = ParallelReader::new_with_metrics_path_read_timeout_and_reconstruction_verification(
readers,
@@ -5155,8 +5283,7 @@ mod tests {
parallel_reader.readers[2].is_some() && parallel_reader.readers[3].is_some(),
(third_buffers, third_errors),
)
})
.await;
};
assert!(first_parity_reserved);
assert_eq!(parity_calls.load(Ordering::SeqCst), PARITY_SHARDS * 2);
@@ -5240,6 +5367,58 @@ mod tests {
assert!(error.is_none(), "a failed disposable hedge must not fail a recovered stripe: {error:?}");
}
/// Rollout guard for backlog#1308: when a data shard and the first parity
/// hedge both fail, the gate-on path must not settle at decode quorum and
/// emit an unverified body. The second parity can restore decode quorum but
/// cannot provide the extra source required for reconstruction verification,
/// so the stripe must fail before exposing bytes.
#[tokio::test]
#[serial_test::serial]
async fn test_data_shards_only_gate_data_and_parity_failure_fails_before_output() {
const BLOCK_SIZE: usize = 64;
const DATA_SHARDS: usize = 2;
const PARITY_SHARDS: usize = 2;
temp_env::async_with_vars([(ENV_RUSTFS_GET_LOCKSTEP_DATA_SHARDS_ONLY_ENABLE, Some("true"))], async {
let erasure = Erasure::new(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE);
let payload = (0..BLOCK_SIZE).map(|value| value as u8).collect::<Vec<_>>();
let shards = erasure.encode_data(&payload).expect("test payload should encode");
let shard_size = erasure.shard_size();
let readers = vec![
Some(BitrotReader::new(TestShardReader::TimedOut, shard_size, HashAlgorithm::None, false)),
Some(BitrotReader::new(
TestShardReader::Ready(Cursor::new(shards[1].to_vec())),
shard_size,
HashAlgorithm::None,
false,
)),
Some(BitrotReader::new(
TestShardReader::TerminalFileNotFound,
shard_size,
HashAlgorithm::None,
false,
)),
Some(BitrotReader::new(
TestShardReader::Ready(Cursor::new(shards[3].to_vec())),
shard_size,
HashAlgorithm::None,
false,
)),
];
let mut output = Vec::new();
let (written, error) = erasure.decode(&mut output, readers, 0, payload.len(), payload.len()).await;
assert_eq!(written, 0, "an unverified stripe must not report body bytes");
assert!(output.is_empty(), "an unverified stripe must not expose a clean short body");
let error = error.expect("data plus parity loss must fail closed");
assert_eq!(error.kind(), ErrorKind::InvalidData);
assert!(error.to_string().contains("insufficient source shards"));
})
.await;
}
/// Lockstep verification-quorum regression (backlog#1156). When a data shard is
/// missing, the hedge must settle only at `data_shards + 1` (decode quorum plus
/// a reconstruction-verification source), never at exactly `data_shards` — that
@@ -933,8 +933,29 @@ impl Erasure {
}
pub(crate) fn decode_data_with_reconstruction_verification(&self, shards: &mut [Option<Vec<u8>>]) -> io::Result<()> {
self.decode_data_with_reconstruction_verification_policy(shards, false)
}
pub(crate) fn decode_data_with_reconstruction_verification_for_lockstep(
&self,
shards: &mut [Option<Vec<u8>>],
) -> io::Result<()> {
self.decode_data_with_reconstruction_verification_policy(shards, true)
}
fn decode_data_with_reconstruction_verification_policy(
&self,
shards: &mut [Option<Vec<u8>>],
require_surplus_source: bool,
) -> io::Result<()> {
let missing_data_source = shards.iter().take(self.data_shards).any(|shard| shard.is_none());
let available_shards = shards.iter().filter(|shard| shard.is_some()).count();
if require_surplus_source && missing_data_source && available_shards == self.data_shards {
return Err(io::Error::new(
io::ErrorKind::InvalidData,
"insufficient source shards to verify reconstructed data",
));
}
let source_parity = if missing_data_source && available_shards > self.data_shards {
shards
.iter()
@@ -1868,6 +1889,31 @@ mod tests {
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
}
#[test]
fn decode_data_with_verification_scopes_exact_quorum_to_lockstep() {
for uses_legacy in [false, true] {
let erasure = Erasure::new_with_options(3, 2, 128, uses_legacy);
let data = b"verified reads must not accept reconstruction without a surplus source";
let encoded = erasure.encode_data(data).expect("encode should succeed");
let mut exact_quorum = optional_shards(&encoded);
exact_quorum[0] = None;
exact_quorum[erasure.total_shard_count() - 1] = None;
let mut default_shards = exact_quorum.clone();
erasure
.decode_data_with_reconstruction_verification(&mut default_shards)
.expect("default decode must preserve exact-quorum reconstruction");
assert_eq!(default_shards[0].as_deref(), Some(encoded[0].as_ref()));
let err = erasure
.decode_data_with_reconstruction_verification_for_lockstep(&mut exact_quorum)
.expect_err("data-shards-only lockstep must reject an exact decode quorum");
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
assert!(err.to_string().contains("insufficient source shards"));
}
}
#[test]
fn verify_data_and_parity_rejects_missing_and_mismatched_shards() {
let erasure = Erasure::new(4, 2, 128);
+16 -7
View File
@@ -185,6 +185,12 @@ pub enum StorageError {
DecommissionAlreadyRunning,
#[error("Rebalance already running")]
RebalanceAlreadyRunning,
#[error("{operation}: stale pool metadata update rejected for pool {pool_index}; {reason}")]
StalePoolMetadataUpdate {
operation: String,
pool_index: usize,
reason: &'static str,
},
#[error("Operation canceled")]
OperationCanceled,
#[error("No heal required")]
@@ -564,6 +570,15 @@ impl Clone for StorageError {
StorageError::DoneForNow => StorageError::DoneForNow,
StorageError::DecommissionAlreadyRunning => StorageError::DecommissionAlreadyRunning,
StorageError::RebalanceAlreadyRunning => StorageError::RebalanceAlreadyRunning,
StorageError::StalePoolMetadataUpdate {
operation,
pool_index,
reason,
} => StorageError::StalePoolMetadataUpdate {
operation: operation.clone(),
pool_index: *pool_index,
reason,
},
StorageError::OperationCanceled => StorageError::OperationCanceled,
StorageError::ErasureReadQuorum => StorageError::ErasureReadQuorum,
StorageError::ErasureWriteQuorum => StorageError::ErasureWriteQuorum,
@@ -667,6 +682,7 @@ impl StorageError {
StorageError::DoneForNow => StorageErrorCode::DoneForNow,
StorageError::DecommissionAlreadyRunning => StorageErrorCode::DecommissionAlreadyRunning,
StorageError::RebalanceAlreadyRunning => StorageErrorCode::RebalanceAlreadyRunning,
StorageError::StalePoolMetadataUpdate { .. } => StorageErrorCode::InvalidArgument,
StorageError::OperationCanceled => StorageErrorCode::OperationCanceled,
StorageError::ErasureReadQuorum => StorageErrorCode::ErasureReadQuorum,
StorageError::ErasureWriteQuorum => StorageErrorCode::ErasureWriteQuorum,
@@ -948,10 +964,6 @@ pub fn is_err_data_movement_overwrite(err: &Error) -> bool {
matches!(err, &StorageError::DataMovementOverwriteErr(_, _, _))
}
pub fn is_err_decommission_running(err: &Error) -> bool {
matches!(err, &StorageError::DecommissionAlreadyRunning)
}
#[allow(dead_code, reason = "predicate asserted by this file's tests (backlog#1823)")]
pub fn is_err_rebalance_running(err: &Error) -> bool {
matches!(err, &StorageError::RebalanceAlreadyRunning)
@@ -1347,9 +1359,6 @@ mod tests {
#[test]
fn test_error_running_state_helpers() {
assert!(is_err_decommission_running(&StorageError::DecommissionAlreadyRunning));
assert!(!is_err_decommission_running(&StorageError::RebalanceAlreadyRunning));
assert!(is_err_rebalance_running(&StorageError::RebalanceAlreadyRunning));
assert!(!is_err_rebalance_running(&StorageError::DecommissionAlreadyRunning));
assert!(is_err_operation_canceled(&StorageError::OperationCanceled));
+28 -4
View File
@@ -368,6 +368,19 @@ impl InstanceContext {
Arc::clone(&self.data_movement_generation_notify)
}
pub(crate) fn observe_durable_data_movement_generation(&self, generation: u64) {
if generation == 0 || self.data_movement_generation_exhausted.load(Ordering::Acquire) {
return;
}
let previous = self.data_movement_generation.fetch_max(generation, Ordering::AcqRel);
if generation == u64::MAX {
self.data_movement_generation_exhausted.store(true, Ordering::Release);
}
if generation > previous {
self.data_movement_generation_notify.notify_waiters();
}
}
pub(crate) fn scanner_publication_state_allowed(&self) -> bool {
!self.data_movement_operation_epoch_exhausted()
&& !self.data_movement_generation_exhausted()
@@ -386,6 +399,20 @@ impl InstanceContext {
}
pub(crate) fn advance_data_movement_operation_epoch(&self) -> u64 {
let (previous, result) = self.advance_data_movement_operation_epoch_only();
if result != previous {
let _ = self.advance_data_movement_generation();
}
result
}
pub(crate) fn advance_data_movement_operation_epoch_to_durable_generation(&self, generation: u64) -> u64 {
let (_, result) = self.advance_data_movement_operation_epoch_only();
self.observe_durable_data_movement_generation(generation);
result
}
fn advance_data_movement_operation_epoch_only(&self) -> (u64, u64) {
self.scanner_publication_state
.store(SCANNER_PUBLICATION_STATE_UNKNOWN, Ordering::Release);
let previous = self.data_movement_operation_epoch.load(Ordering::Acquire);
@@ -396,10 +423,7 @@ impl InstanceContext {
if result == u64::MAX {
self.data_movement_operation_epoch_exhausted.store(true, Ordering::Release);
}
if result != previous {
let _ = self.advance_data_movement_generation();
}
result
(previous, result)
}
/// Advance the movement generation after a durable movement transition.
@@ -845,7 +845,7 @@ impl ECStore {
let mut pool_stats = Vec::with_capacity(self.pools.len());
let now = OffsetDateTime::now_utc();
let now = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
for disk_stat in disk_stats.iter() {
let mut pool_stat = RebalanceStats {
@@ -868,8 +868,10 @@ impl ECStore {
pool_stats.push(pool_stat);
}
let has_participating_pool = pool_stats.iter().any(|pool_stat| pool_stat.participating);
let meta = RebalanceMeta {
id: Uuid::new_v4().to_string(),
stopped_at: (!has_participating_pool).then_some(now),
percent_free_goal,
pool_stats,
..Default::default()
@@ -963,6 +965,18 @@ impl ECStore {
)));
}
if meta.stopped_at.is_some() {
if !is_rebalance_conflicting_with_decommission(meta) {
debug!(
event = EVENT_REBALANCE_STATE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REBALANCE,
state = "start_skipped",
reason = "not_started_terminal",
rebalance_id = %expected_id,
"Skipped rebalance start because metadata is already terminal"
);
return Ok(());
}
return Err(Error::other(format!("rebalance {expected_id} was stopped before start")));
}
}
@@ -1214,11 +1228,11 @@ impl ECStore {
};
let movement_gate = self.ctx.data_movement_operation_gate();
let _movement_guard = movement_gate.write().await;
let stopped_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
let (previous_meta, meta_to_save) = {
let mut rebalance_meta = self.rebalance_meta.write().await;
let previous_meta = rebalance_meta.clone();
let meta_to_save =
stop_rebalance_meta_snapshot_for_id(rebalance_meta.as_mut(), OffsetDateTime::now_utc(), expected_id)?;
let meta_to_save = stop_rebalance_meta_snapshot_for_id(rebalance_meta.as_mut(), stopped_at, expected_id)?;
(previous_meta, meta_to_save)
};
@@ -1250,14 +1264,10 @@ impl ECStore {
.await?;
let movement_gate = self.ctx.data_movement_operation_gate();
let _movement_guard = movement_gate.write().await;
let failed_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
let meta_to_save = {
let mut rebalance_meta = self.rebalance_meta.write().await;
rollback_rebalance_start_meta_snapshot_for_id(
rebalance_meta.as_mut(),
OffsetDateTime::now_utc(),
expected_id,
start_error,
)
rollback_rebalance_start_meta_snapshot_for_id(rebalance_meta.as_mut(), failed_at, expected_id, start_error)
};
if let Some(meta_to_save) = meta_to_save {
@@ -1402,6 +1412,62 @@ mod tests {
assert!(cancel.is_cancelled());
}
#[tokio::test]
#[serial_test::serial]
async fn equal_free_ratio_admin_no_participant_rebalance_succeeds_and_persists_terminal_generation_after_restart() {
let (_temp_dirs, store, restarted) =
crate::services::rebalance::test_two_pool_stores_with_isolated_node_contexts(None).await;
let movement_floor = OffsetDateTime::from_unix_timestamp(4_100_000_000).expect("future test timestamp should be valid");
*store.rebalance_meta.write().await = Some(RebalanceMeta {
id: "previous-terminal-rebalance".to_string(),
stopped_at: Some(movement_floor),
..Default::default()
});
set_rebalance_disk_stats_override_for_test(
store.id,
vec![
DiskStat {
total_space: 100,
available_space: 50,
},
DiskStat {
total_space: 100,
available_space: 50,
},
],
);
let rebalance_id = store
.init_and_start_rebalance(vec!["equal-ratio-no-op".to_string()])
.await
.expect("equal free ratio admin rebalance should succeed as a terminal no-op");
let stopped_at = {
let local = store.rebalance_meta.read().await;
let local = local.as_ref().expect("no-op rebalance metadata should remain available");
assert_eq!(local.id, rebalance_id);
assert!(local.pool_stats.iter().all(|pool_stat| !pool_stat.participating));
let stopped_at = local.stopped_at.expect("no-op rebalance must persist a terminal timestamp");
assert_eq!(stopped_at, movement_floor + time::Duration::nanoseconds(1));
stopped_at
};
let stopped_generation =
u64::try_from(stopped_at.unix_timestamp_nanos()).expect("terminal timestamp should map to scanner generation");
let live_status = store.scanner_data_movement_pause_status().await;
assert!(!live_status.paused);
assert_eq!(live_status.movement_generation, stopped_generation);
restarted
.load_rebalance_meta()
.await
.expect("restarted store should load the persisted no-op rebalance metadata");
let status = restarted.scanner_data_movement_pause_status().await;
assert!(!status.paused);
assert_eq!(status.movement_generation, stopped_generation);
assert_eq!(restarted.scanner_data_movement_generation(), stopped_generation);
}
#[tokio::test]
#[serial_test::serial]
async fn rebalance_activation_rejects_initialized_cluster_with_all_pool_meta_missing() {
@@ -161,6 +161,7 @@ impl ECStore {
let cancel_tx = CancellationToken::new();
let rx = cancel_tx.clone();
let activation_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
let activation_outcome;
let candidate;
let expected_cancel;
@@ -185,12 +186,8 @@ impl ECStore {
return Ok(false);
}
expected_cancel = meta.cancel.clone();
(candidate, activation_outcome, must_persist) = stage_local_rebalance_worker_activation(
meta,
expected_id.as_ref(),
cancel_tx.clone(),
OffsetDateTime::now_utc(),
)?;
(candidate, activation_outcome, must_persist) =
stage_local_rebalance_worker_activation(meta, expected_id.as_ref(), cancel_tx.clone(), activation_at)?;
if let Err(err) = activation_fence.ensure_held() {
cancel_tx.cancel();
return Err(err);
@@ -384,11 +381,11 @@ impl ECStore {
tokio::select! {
result = done_rx.recv() => {
quit = true;
let now = OffsetDateTime::now_utc();
let terminal_event = classify_rebalance_terminal_event(result, now);
msg = terminal_event.message().to_string();
let movement_gate = store.ctx.data_movement_operation_gate();
let movement_guard = movement_gate.write().await;
let terminal_at = store.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
let terminal_event = classify_rebalance_terminal_event(result, terminal_at);
msg = terminal_event.message().to_string();
let previous_meta = store.rebalance_meta.read().await.clone();
let terminal_state_present = {
let mut rebalance_meta = store.rebalance_meta.write().await;
@@ -405,7 +402,7 @@ impl ECStore {
{
pool_stat.info.stopping = false;
pool_stat.info.status = RebalStatus::Failed;
pool_stat.info.end_time = Some(now);
pool_stat.info.end_time = Some(terminal_at);
pool_stat.info.last_error = Some(
pool_stat
.cleanup_warnings
@@ -433,7 +430,7 @@ impl ECStore {
&mut pool_stat.info.end_time,
&mut pool_stat.info.last_error,
terminal_event,
now,
terminal_at,
);
}
true
@@ -835,6 +832,10 @@ impl ECStore {
opt: RebalSaveOpt,
expected_id: Option<&str>,
) -> Result<()> {
let now = match opt {
RebalSaveOpt::Stats => OffsetDateTime::now_utc(),
RebalSaveOpt::StoppedAt => self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await,
};
let meta_to_save = {
let mut rebalance_meta = self.rebalance_meta.write().await;
if let Some(expected_id) = expected_id {
@@ -844,7 +845,6 @@ impl ECStore {
return Ok(());
};
let now = OffsetDateTime::now_utc();
apply_rebalance_save_option(meta, pool_idx, opt, now);
meta.clone()
};
+344 -64
View File
@@ -37,6 +37,8 @@ use super::super::ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_DISKS;
#[cfg(test)]
use super::super::ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_OBJECT_PREFIX;
#[cfg(test)]
use super::super::ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE;
#[cfg(test)]
use super::super::get_metadata_slowtail_fault_delay;
use super::super::{
Bytes, CHECK_PART_DISK_NOT_FOUND, DeleteOptions, DiskError, DiskStore, EVENT_SET_DISK_RENAME_TAIL_DRAIN_FAILED,
@@ -46,13 +48,14 @@ use super::super::{
ObjectPartInfo, OffsetDateTime, RUSTFS_META_BUCKET, RUSTFS_META_MULTIPART_BUCKET, RawFileInfo, ReadMultipleReq,
ReadMultipleResp, ReadOptions, Result, SLASH_SEPARATOR, STORAGE_FORMAT_FILE, SetDisks, SnapshotLeaseToken, StorageError,
UpdateMetadataOpts, Uuid, build_inline_bitrot_readers_from_refs, can_try_inline_data_shards_direct,
capacity_scope_from_disks, coding, collect_inline_data_shard_fileinfos_by_index_or_reason, current_dirty_generation, debug,
disk, file_info_is_valid_for_metadata, get_metadata_slowtail_fault_request, info, inline_erasure_shard_file_offset,
inline_erasure_shard_size, is_err_object_not_found, is_err_version_not_found, is_get_metadata_data_read_early_stop_enabled,
is_get_metadata_early_stop_bounded_fanout_enabled, is_get_metadata_early_stop_enabled,
is_get_metadata_two_phase_read_plan_enabled, is_object_dangling, is_version_early_stop_enabled, issue3031_diag_enabled,
join_all, join_errs, log_multipart_write_quorum_failure, merge_file_meta_versions, path_join_buf, record_global_dirty_scope,
reduce_read_quorum_errs, reduce_write_quorum_errs, send_heal_request_with_admission, should_prevent_write, to_object_err,
capacity_scope_from_disks, codec_streaming_rollout_applies, coding, collect_inline_data_shard_fileinfos_by_index_or_reason,
current_dirty_generation, debug, disk, file_info_is_valid_for_metadata, get_metadata_slowtail_fault_request, info,
inline_erasure_shard_file_offset, inline_erasure_shard_size, is_err_object_not_found, is_err_version_not_found,
is_get_metadata_data_read_early_stop_enabled, is_get_metadata_early_stop_bounded_fanout_enabled,
is_get_metadata_early_stop_enabled, is_get_metadata_non_inline_data_read_early_stop_enabled, is_object_dangling,
is_version_early_stop_enabled, issue3031_diag_enabled, join_all, join_errs, log_multipart_write_quorum_failure,
merge_file_meta_versions, object_fits_single_block, path_join_buf, record_global_dirty_scope, reduce_read_quorum_errs,
reduce_write_quorum_errs, send_heal_request_with_admission, should_prevent_write, to_object_err,
try_read_inline_data_shards_direct, warn,
};
#[cfg(test)]
@@ -450,6 +453,22 @@ use tokio::io::{AsyncRead, ReadBuf};
use tokio::sync::{Mutex, RwLock, oneshot};
use tokio::task::JoinSet;
struct AbortOnDropJoinHandle<T>(tokio::task::JoinHandle<T>);
impl<T> Future for AbortOnDropJoinHandle<T> {
type Output = std::result::Result<T, tokio::task::JoinError>;
fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll<Self::Output> {
Pin::new(&mut self.0).poll(cx)
}
}
impl<T> Drop for AbortOnDropJoinHandle<T> {
fn drop(&mut self) {
self.0.abort();
}
}
pub(in crate::set_disk) const EVENT_SET_DISK_READ: &str = "set_disk_read";
pub(in crate::set_disk) const ENV_RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP: &str = "RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP";
const ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE: &str = "RUSTFS_GET_METADATA_READ_VERSION_COALESCE";
@@ -688,6 +707,10 @@ pub(in crate::set_disk) struct MetadataQuorumAccumulator {
pub(in crate::set_disk) hard_errors: usize,
pub(in crate::set_disk) candidate: Option<FileInfo>,
pub(in crate::set_disk) candidate_votes: usize,
// Bitset of shard indexes whose metadata matches the candidate. Erasure
// layouts are capped at 16 shards, so this stays allocation-free on the
// GET metadata hot path.
candidate_shard_mask: u16,
pub(in crate::set_disk) conflicting_metadata: bool,
pub(in crate::set_disk) delete_marker_seen: bool,
pub(in crate::set_disk) delete_marker_candidates: Vec<(FileInfo, usize)>,
@@ -709,6 +732,7 @@ impl MetadataQuorumAccumulator {
hard_errors: 0,
candidate: None,
candidate_votes: 0,
candidate_shard_mask: 0,
conflicting_metadata: false,
delete_marker_seen: false,
delete_marker_candidates: Vec::new(),
@@ -724,6 +748,14 @@ impl MetadataQuorumAccumulator {
}
pub(in crate::set_disk) fn observe_file_info(&mut self, file_info: &FileInfo) {
self.observe_file_info_with_index(None, file_info);
}
pub(in crate::set_disk) fn observe_file_info_at(&mut self, disk_index: usize, file_info: &FileInfo) {
self.observe_file_info_with_index(Some(disk_index), file_info);
}
fn observe_file_info_with_index(&mut self, disk_index: Option<usize>, file_info: &FileInfo) {
if !file_info_is_valid_for_metadata(file_info) {
self.hard_errors = self.hard_errors.saturating_add(1);
return;
@@ -763,6 +795,11 @@ impl MetadataQuorumAccumulator {
match &self.candidate {
Some(candidate) if metadata_early_stop_candidate_matches(candidate, file_info) => {
self.candidate_votes = self.candidate_votes.saturating_add(1);
if let Some(disk_index) = disk_index
&& let Some(bit) = Self::candidate_shard_bit(candidate, file_info, disk_index)
{
self.candidate_shard_mask |= bit;
}
}
Some(_) => {
self.conflicting_metadata = true;
@@ -770,10 +807,38 @@ impl MetadataQuorumAccumulator {
None => {
self.candidate = Some(file_info.clone());
self.candidate_votes = 1;
if let Some(disk_index) = disk_index
&& let Some(bit) = Self::candidate_shard_bit(file_info, file_info, disk_index)
{
self.candidate_shard_mask |= bit;
}
}
}
}
fn candidate_shard_bit(candidate: &FileInfo, file_info: &FileInfo, disk_index: usize) -> Option<u16> {
let &erasure_index = candidate.erasure.distribution.get(disk_index)?;
if erasure_index == 0 || erasure_index > u16::BITS as usize || file_info.erasure.index != erasure_index {
return None;
}
Some(1u16 << (erasure_index - 1))
}
pub(in crate::set_disk) fn candidate_has_read_reserve(&self) -> bool {
self.candidate_read_reserve_target()
.is_some_and(|required| self.candidate_shard_mask.count_ones() as usize >= required)
}
pub(in crate::set_disk) fn candidate_read_reserve_target(&self) -> Option<usize> {
let candidate = self.candidate.as_ref()?;
Some(
candidate
.erasure
.data_blocks
.saturating_add(usize::from(candidate.erasure.parity_blocks > 0)),
)
}
pub(in crate::set_disk) fn observe_error(&mut self, err: &DiskError) {
match err {
DiskError::FileNotFound | DiskError::VolumeNotFound => {
@@ -1084,11 +1149,7 @@ fn data_read_early_stop_inline_candidate_miss_reason(candidate: &FileInfo) -> Op
None
}
fn non_inline_data_read_candidate_is_safe(
candidate: &FileInfo,
parts_metadata: &[FileInfo],
disks: &[Option<DiskStore>],
) -> bool {
pub(in crate::set_disk) fn non_inline_data_read_candidate_is_safe(candidate: &FileInfo) -> bool {
if candidate.inline_data()
|| candidate.is_compressed()
|| candidate.is_remote()
@@ -1100,34 +1161,21 @@ fn non_inline_data_read_candidate_is_safe(
{
return false;
}
let Ok(erasure) = coding::Erasure::try_new_with_options(
candidate.erasure.data_blocks,
candidate.erasure.parity_blocks,
candidate.erasure.block_size,
candidate.uses_legacy_checksum,
) else {
return false;
};
// The regular reader setup can reconstruct missing data shards from any
// `data_shards` matching metadata entries. Requiring every data slot here
// would unnecessarily wait for one slow data disk even when parity and
// the remaining data shards already form a read quorum.
let mut available_shards = vec![false; erasure.data_shards + erasure.parity_shards];
for ((file_info, disk), &erasure_index) in parts_metadata
.iter()
.zip(disks.iter())
.zip(candidate.erasure.distribution.iter())
{
if erasure_index == 0 || erasure_index > available_shards.len() || disk.is_none() {
continue;
}
if metadata_early_stop_candidate_matches(file_info, candidate) && file_info.erasure.index == erasure_index {
available_shards[erasure_index - 1] = true;
}
}
available_shards.into_iter().filter(|present| *present).count() >= erasure.data_shards
candidate.has_valid_erasure_geometry()
}
pub(in crate::set_disk) fn late_materialization_candidate_is_safe(candidate: &FileInfo) -> bool {
non_inline_data_read_candidate_is_safe(candidate)
&& candidate.size > 512 * 1024
&& object_fits_single_block(candidate.size, candidate.erasure.block_size)
}
pub(in crate::set_disk) fn non_inline_data_read_early_stop_allowed(read_data: bool, bucket: &str, object: &str) -> bool {
read_data && is_get_metadata_non_inline_data_read_early_stop_enabled() && !codec_streaming_rollout_applies(bucket, object)
}
const NON_INLINE_SINGLE_PENDING_HEDGE_DELAY: Duration = Duration::from_millis(100);
fn data_read_inline_missing_shards_are_pending(
candidate: &FileInfo,
parts_metadata: &[FileInfo],
@@ -1974,14 +2022,10 @@ pub(in crate::set_disk) fn fill_deferred_bitrot_readers(
return;
}
// Only CopySource uses disposable, stripe-aligned reopeners. Ordinary GET
// readers use the existing deferred handle and should not retain one
// heap-allocated closure (plus cloned path/disk state) for every parity
// slot.
let copy_source_demand_bound = matches!(
crate::set_disk::get_object_read_policy(),
crate::set_disk::GetObjectReadPolicy::CopySource
);
// Every demand-bound lockstep reader needs a disposable, stripe-aligned
// reopener. Otherwise a recovered slow data read can cancel and consume
// the only parity reserve needed by a later degraded stripe.
let demand_bound_lockstep = crate::erasure::coding::decode::get_lockstep_data_shards_only_enabled();
for idx in 0..disks.len() {
if setup.attempted[idx] {
@@ -1996,7 +2040,7 @@ pub(in crate::set_disk) fn fill_deferred_bitrot_readers(
let disk = disks[idx].clone();
let data_dir = files[idx].data_dir.unwrap_or_default();
let path = format!("{object}/{data_dir}/part.{part_number}");
let reopener = copy_source_demand_bound.then(|| {
let reopener = demand_bound_lockstep.then(|| {
deferred_reader_reopener(
inline_data.clone(),
disk.clone(),
@@ -2037,7 +2081,7 @@ pub(in crate::set_disk) fn fill_deferred_bitrot_readers(
// ready/error bookkeeping that quorum decisions rely on is left untouched.
// Gate off (default): keep the eagerly opened parity readers exactly as
// before — the lockstep path reads them on every stripe.
if !crate::erasure::coding::decode::get_lockstep_data_shards_only_enabled() {
if !demand_bound_lockstep {
return;
}
for idx in data_shards..disks.len() {
@@ -2049,7 +2093,7 @@ pub(in crate::set_disk) fn fill_deferred_bitrot_readers(
let disk = disks[idx].clone();
let data_dir = files[idx].data_dir.unwrap_or_default();
let path = format!("{object}/{data_dir}/part.{part_number}");
let reopener = copy_source_demand_bound.then(|| {
let reopener = demand_bound_lockstep.then(|| {
deferred_reader_reopener(
inline_data.clone(),
disk.clone(),
@@ -2914,7 +2958,7 @@ impl SetDisks {
read_data,
healing,
incl_free_versions,
read_data && is_get_metadata_two_phase_read_plan_enabled(),
non_inline_data_read_early_stop_allowed(read_data, bucket, object),
default_parity_count,
allow_coalescing,
)
@@ -2980,7 +3024,7 @@ impl SetDisks {
let object = object.clone();
let version_id = version_id.clone();
let slowtail_fault = slowtail_fault.clone();
tokio::spawn(async move {
AbortOnDropJoinHandle(tokio::spawn(async move {
let response_start = observe.then(Instant::now);
let result = if let Some(disk) = disk {
Self::record_read_version_call(&object, disk_index);
@@ -2995,7 +3039,7 @@ impl SetDisks {
};
let elapsed = response_start.map(|start| start.elapsed());
(result, elapsed)
})
}))
});
// Wait for all futures to complete
@@ -3085,6 +3129,8 @@ impl SetDisks {
let mut scheduled_count = 0usize;
let mut force_full_wait = false;
let mut final_miss_reason_override = None;
let mut non_inline_candidate_eligible = None;
let mut single_pending_hedge_deadline = None;
let slowtail_fault = get_metadata_slowtail_fault_request(bucket.as_ref(), object.as_ref(), read_data);
let spawn_read_version =
|join_set: &mut JoinSet<(usize, disk::error::Result<FileInfo>, Duration)>, index: usize, disk: Option<DiskStore>| {
@@ -3132,18 +3178,54 @@ impl SetDisks {
}
}
while let Some(result) = join_set.join_next().await {
loop {
let mut defer_pending_inline_data_shard = false;
let result = if let Some(deadline) = single_pending_hedge_deadline.take() {
tokio::select! {
result = join_set.join_next() => result,
_ = tokio::time::sleep_until(deadline) => {
if bounded_fanout
&& !force_full_wait
&& join_set.len() == 1
&& non_inline_candidate_eligible == Some(true)
&& !accumulator.candidate_has_read_reserve()
&& next_fanout_index < disks.len()
{
while next_fanout_index < disks.len() {
let disk_index = fanout_order[next_fanout_index];
next_fanout_index = next_fanout_index.saturating_add(1);
if let Some(disk) = disks.get(disk_index).cloned() {
spawn_read_version(&mut join_set, disk_index, disk);
scheduled_count = scheduled_count.saturating_add(1);
break;
}
}
}
continue;
}
}
} else {
join_set.join_next().await
};
let Some(result) = result else { break };
match result {
Ok((index, res, elapsed)) => match res {
Ok(file_info) => {
observations.push(MetadataFanoutObservation::from_file_info(&file_info, elapsed));
accumulator.observe_file_info(&file_info);
if allow_non_inline_data_read_early_stop {
accumulator.observe_file_info_at(index, &file_info);
} else {
accumulator.observe_file_info(&file_info);
}
if allow_non_inline_data_read_early_stop && non_inline_candidate_eligible.is_none() {
non_inline_candidate_eligible =
accumulator.candidate.as_ref().map(non_inline_data_read_candidate_is_safe);
}
if bounded_fanout
&& read_data
&& !force_full_wait
&& let Some(reason) = data_read_early_stop_inline_candidate_miss_reason(&file_info)
&& !(allow_non_inline_data_read_early_stop
&& !(non_inline_candidate_eligible == Some(true)
&& reason == GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_NOT_INLINE)
{
force_full_wait = true;
@@ -3175,11 +3257,8 @@ impl SetDisks {
{
let should_return_early = if read_data {
match accumulator.candidate.as_ref() {
Some(candidate)
if allow_non_inline_data_read_early_stop
&& non_inline_data_read_candidate_is_safe(candidate, &ress, disks) =>
{
true
Some(_candidate) if non_inline_candidate_eligible == Some(true) => {
accumulator.candidate_has_read_reserve()
}
Some(candidate) => match data_read_early_stop_inline_body_miss_reason(
bucket.as_ref(),
@@ -3251,12 +3330,37 @@ impl SetDisks {
}
let pending_responses = join_set.len();
let should_hedge_single_pending_data_read = read_data
// Inline verification can still depend on a missing data shard;
// issue one immediate spare when only that shard remains. The
// non-inline path keeps its delayed hedge below to avoid healthy
// reads paying speculative I/O before the candidate is classified.
let should_hedge_single_pending_inline_read = read_data
&& !force_full_wait
&& !defer_pending_inline_data_shard
&& pending_responses == 1
&& non_inline_candidate_eligible != Some(true)
&& accumulator.can_still_reach_early_stop_with_pending(pending_responses);
if bounded_fanout && force_full_wait {
// A non-inline plan must retain one extra matching shard as a
// reconstruction reserve. Schedule that reserve only after the
// candidate is known to be eligible, so inline GETs do not pay an
// extra fanout and the healthy path remains allocation-free.
let needs_non_inline_read_reserve = non_inline_candidate_eligible == Some(true)
&& !accumulator.candidate_has_read_reserve()
&& accumulator
.candidate_read_reserve_target()
.is_some_and(|reserve_target| scheduled_count < reserve_target || pending_responses == 0);
if bounded_fanout
&& !force_full_wait
&& (needs_non_inline_read_reserve || should_hedge_single_pending_inline_read)
&& next_fanout_index < disks.len()
{
let disk_index = fanout_order[next_fanout_index];
if let Some(disk) = disks.get(disk_index).cloned() {
spawn_read_version(&mut join_set, disk_index, disk);
scheduled_count = scheduled_count.saturating_add(1);
}
next_fanout_index = next_fanout_index.saturating_add(1);
} else if bounded_fanout && force_full_wait {
while next_fanout_index < disks.len() {
let disk_index = fanout_order[next_fanout_index];
if let Some(disk) = disks.get(disk_index).cloned() {
@@ -3268,8 +3372,7 @@ impl SetDisks {
} else if bounded_fanout
&& !defer_pending_inline_data_shard
&& next_fanout_index < disks.len()
&& (!accumulator.can_still_reach_early_stop_with_pending(pending_responses)
|| should_hedge_single_pending_data_read)
&& !accumulator.can_still_reach_early_stop_with_pending(pending_responses)
{
let disk_index = fanout_order[next_fanout_index];
if let Some(disk) = disks.get(disk_index).cloned() {
@@ -3278,6 +3381,17 @@ impl SetDisks {
}
next_fanout_index = next_fanout_index.saturating_add(1);
}
if bounded_fanout
&& !force_full_wait
&& !defer_pending_inline_data_shard
&& join_set.len() == 1
&& non_inline_candidate_eligible == Some(true)
&& !accumulator.candidate_has_read_reserve()
&& accumulator.can_still_reach_early_stop_with_pending(join_set.len())
&& next_fanout_index < disks.len()
{
single_pending_hedge_deadline = Some(tokio::time::Instant::now() + NON_INLINE_SINGLE_PENDING_HEDGE_DELAY);
}
}
let accumulator_miss_reason = accumulator.final_miss_reason();
@@ -6935,6 +7049,27 @@ mod tests {
use tempfile::TempDir;
use tokio::io::AsyncReadExt;
#[test]
#[serial_test::serial(codec_streaming_env)]
fn non_inline_early_stop_is_mutually_exclusive_with_codec_rollout() {
temp_env::with_vars(
[
(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("true")),
("RUSTFS_GET_CODEC_STREAMING_ROLLOUT", Some("on")),
("RUSTFS_GET_CODEC_STREAMING_BODY_COMPAT_CONFIRMED", Some("true")),
("RUSTFS_GET_CODEC_STREAMING_HEADER_COMPAT_CONFIRMED", Some("true")),
],
|| assert!(!non_inline_data_read_early_stop_allowed(true, "bucket", "object")),
);
temp_env::with_vars(
[
(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("true")),
("RUSTFS_GET_CODEC_STREAMING_ROLLOUT", Some("off")),
],
|| assert!(non_inline_data_read_early_stop_allowed(true, "bucket", "object")),
);
}
#[tokio::test]
async fn scanner_delete_owner_survives_waiter_cancellation() {
let movement_gate = Arc::new(tokio::sync::RwLock::new(()));
@@ -7362,6 +7497,90 @@ mod tests {
drop(dirs);
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn metadata_slowtail_fault_gate_stops_before_unneeded_tail() {
const DISKS: usize = 4;
let bucket = "metadata-slowtail-gated-bucket";
let object = "objects/metadata-slowtail-gated-object";
let (dirs, disks) = call_counter_local_disks(bucket, DISKS).await;
install_mapped_metadata_fanout_fileinfo(&disks, bucket, object).await;
let order = bounded_metadata_fanout_order(bucket, object, DISKS, 2);
let slow_disk = *order.get(3).expect("four-disk fanout should have a deferred tail disk");
let slow_disk_env = slow_disk.to_string();
temp_env::async_with_vars(
[
(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("true")),
(ENV_RUSTFS_GET_METADATA_EARLY_STOP_ENABLE, Some("true")),
(ENV_RUSTFS_GET_METADATA_DATA_READ_EARLY_STOP_ENABLE, Some("true")),
(ENV_RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT, Some("true")),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_DELAY_MS, Some("150")),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_DISKS, Some(slow_disk_env.as_str())),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_BUCKET, Some(bucket)),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_OBJECT_PREFIX, Some("objects/")),
],
async {
let calls = disk_call_counters::observe(object);
let read_with_data =
SetDisks::read_all_fileinfo_observed(&disks, bucket, bucket, object, "", true, false, false, true, 2);
let (parts_metadata, errs, diagnostics) = tokio::time::timeout(Duration::from_millis(500), read_with_data)
.await
.expect("gated metadata read should stop before the deferred slow tail")
.expect("gated metadata fanout should resolve");
assert!(parts_metadata.iter().filter(|fi| fi.name == object).count() >= 3);
assert!(errs.iter().all(Option::is_none));
assert!(diagnostics.total_responses() < DISKS);
assert_eq!(calls.total(disk_call_counters::KIND_METADATA_SLOWTAIL_FAULT), 0);
},
)
.await;
drop(dirs);
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn metadata_slowtail_fault_gate_hedges_an_initial_slow_data_shard() {
const DISKS: usize = 4;
let bucket = "metadata-slowtail-gated-initial-bucket";
let object = "objects/metadata-slowtail-gated-initial-object";
let (dirs, disks) = call_counter_local_disks(bucket, DISKS).await;
install_mapped_metadata_fanout_fileinfo(&disks, bucket, object).await;
let order = bounded_metadata_fanout_order(bucket, object, DISKS, 2);
let slow_disk = *order.get(1).expect("four-disk fanout should have an initial data disk");
let spare_disk = *order.get(3).expect("four-disk fanout should have a spare disk");
let slow_disk_env = slow_disk.to_string();
temp_env::async_with_vars(
[
(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("true")),
(ENV_RUSTFS_GET_METADATA_EARLY_STOP_ENABLE, Some("true")),
(ENV_RUSTFS_GET_METADATA_DATA_READ_EARLY_STOP_ENABLE, Some("true")),
(ENV_RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT, Some("true")),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_DELAY_MS, Some("500")),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_DISKS, Some(slow_disk_env.as_str())),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_BUCKET, Some(bucket)),
(ENV_RUSTFS_GET_METADATA_SLOWTAIL_FAULT_OBJECT_PREFIX, Some("objects/")),
],
async {
let calls = disk_call_counters::observe(object);
let read_with_data =
SetDisks::read_all_fileinfo_observed(&disks, bucket, bucket, object, "", true, false, false, true, 2);
let (parts_metadata, errs, diagnostics) = tokio::time::timeout(Duration::from_millis(300), read_with_data)
.await
.expect("gated metadata read should hedge the initial slow shard")
.expect("gated metadata fanout should resolve");
assert!(parts_metadata.iter().filter(|fi| fi.name == object).count() >= 3);
assert!(errs.iter().all(Option::is_none));
assert!(diagnostics.total_responses() < DISKS);
assert_eq!(calls.for_disk(disk_call_counters::KIND_METADATA_SLOWTAIL_FAULT, slow_disk), 1);
assert_eq!(calls.for_disk(disk_call_counters::KIND_READ_VERSION, spare_disk), 1);
},
)
.await;
drop(dirs);
}
/// Demo / regression guard for the backlog#1325 per-disk call counters.
///
/// The metadata fan-out issues each `read_version` inside its own
@@ -7775,6 +7994,32 @@ mod tests {
}
}
async fn install_mapped_metadata_fanout_fileinfo(disks: &[Option<DiskStore>], bucket: &str, object: &str) {
let version_id = Uuid::new_v4();
let data_dir = Uuid::new_v4();
let mod_time = OffsetDateTime::now_utc();
let distribution = FileInfo::new(&metadata_distribution_key(bucket, object), 2, 2)
.erasure
.distribution;
for (index, disk) in disks
.iter()
.enumerate()
.filter_map(|(index, disk)| disk.as_ref().map(|disk| (index, disk)))
{
disk.write_all(bucket, &format!("{object}/{data_dir}/part.1"), Bytes::from_static(b"x"))
.await
.expect("part data should be installed on every disk");
let mut file_info = valid_metadata_fanout_fileinfo(bucket, object, version_id, data_dir, mod_time);
file_info.erasure.distribution = distribution.clone();
file_info.erasure.index = *distribution
.get(index)
.expect("mapped metadata distribution should cover every disk");
disk.write_metadata(bucket, bucket, object, file_info)
.await
.expect("mapped metadata should be installed on every disk");
}
}
async fn inline_metadata_fanout_fileinfos_with_mode(
bucket: &str,
object: &str,
@@ -10397,6 +10642,41 @@ mod tests {
assert_eq!(accumulator.candidate_latest_quorum(&impossible_parity), None);
}
#[test]
fn metadata_quorum_accumulator_tracks_mapped_shards_and_requires_a_reserve() {
let version_id = Uuid::new_v4();
let data_dir = Uuid::new_v4();
let base = valid_metadata_fanout_fileinfo("bucket", "object", version_id, data_dir, OffsetDateTime::now_utc());
let distribution = base.erasure.distribution.clone();
let mut accumulator = MetadataQuorumAccumulator::new(4, 2, true);
for (disk_index, &erasure_index) in distribution.iter().take(2).enumerate() {
let mut file_info = base.clone();
file_info.erasure.index = erasure_index;
accumulator.observe_file_info_at(disk_index, &file_info);
}
assert!(
!accumulator.candidate_has_read_reserve(),
"data quorum without parity reserve must not early-stop"
);
let mut mismatched = base.clone();
mismatched.erasure.index = distribution[3];
accumulator.observe_file_info_at(2, &mismatched);
assert!(
!accumulator.candidate_has_read_reserve(),
"mapped index mismatch must not count as a reserve"
);
let mut reserve = base;
reserve.erasure.index = distribution[2];
accumulator.observe_file_info_at(2, &reserve);
assert!(
accumulator.candidate_has_read_reserve(),
"one matching reserve shard should complete the read reserve"
);
}
#[test]
fn metadata_quorum_accumulator_treats_invalid_default_parity_as_full_fanout() {
let accumulator = MetadataQuorumAccumulator::new(2, 2, true);
+569 -30
View File
@@ -773,10 +773,11 @@ const DEFAULT_RUSTFS_GET_METADATA_VERSION_EARLY_STOP_ENABLE: bool = false;
const ENV_RUSTFS_GET_METADATA_DATA_READ_EARLY_STOP_ENABLE: &str = "RUSTFS_GET_METADATA_DATA_READ_EARLY_STOP_ENABLE";
const DEFAULT_RUSTFS_GET_METADATA_DATA_READ_EARLY_STOP_ENABLE: bool = true;
// Two-phase metadata/read-plan rollout (backlog#1309). The first phase reads
// metadata without inline payloads and only fetches inline data from the
// selected data-shard slots. Keep this opt-in until the Linux multi-node
// slow-tail and small-inline cost gates are complete.
// Opt-in non-inline data-read quorum early-stop rollout (backlog#1309). The
// existing metadata fanout still reads data-bearing metadata; this gate only
// permits a safe plain single-part candidate to stop before the full fanout.
// Keep it opt-in until the Linux multi-node slow-tail and small-inline cost
// gates are complete. The environment name is retained for compatibility.
const ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE: &str = "RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE";
const DEFAULT_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE: bool = false;
@@ -893,8 +894,11 @@ struct OwnedGetObjectFileInfo {
fi: FileInfo,
parts_metadata: Vec<FileInfo>,
online_disks: Vec<Option<DiskStore>>,
late_metadata_fanout_disks: Option<Vec<Option<DiskStore>>>,
}
type OwnedGetObjectFileInfoParts = (FileInfo, Vec<FileInfo>, Vec<Option<DiskStore>>, Option<Vec<Option<DiskStore>>>);
impl GetObjectFileInfo {
fn owned(fi: FileInfo, parts_metadata: Vec<FileInfo>, online_disks: Vec<Option<DiskStore>>) -> Self {
Self {
@@ -902,6 +906,24 @@ impl GetObjectFileInfo {
fi,
parts_metadata,
online_disks,
late_metadata_fanout_disks: None,
}),
shared: None,
}
}
fn owned_with_late_metadata_fanout(
fi: FileInfo,
parts_metadata: Vec<FileInfo>,
online_disks: Vec<Option<DiskStore>>,
late_metadata_fanout_disks: Vec<Option<DiskStore>>,
) -> Self {
Self {
owned: Some(OwnedGetObjectFileInfo {
fi,
parts_metadata,
online_disks,
late_metadata_fanout_disks: Some(late_metadata_fanout_disks),
}),
shared: None,
}
@@ -938,19 +960,28 @@ impl GetObjectFileInfo {
}
}
fn has_late_metadata_fanout(&self) -> bool {
self.owned
.as_ref()
.is_some_and(|snapshot| snapshot.late_metadata_fanout_disks.is_some())
}
fn into_owned(self) -> (FileInfo, Vec<FileInfo>, Vec<Option<DiskStore>>) {
let (fi, parts_metadata, online_disks, _) = self.into_owned_with_late_metadata_fanout();
(fi, parts_metadata, online_disks)
}
fn into_owned_with_late_metadata_fanout(self) -> OwnedGetObjectFileInfoParts {
match (self.owned, self.shared) {
(Some(snapshot), None) => {
let OwnedGetObjectFileInfo {
fi,
parts_metadata,
online_disks,
} = snapshot;
(fi, parts_metadata, online_disks)
}
(Some(snapshot), None) => (
snapshot.fi,
snapshot.parts_metadata,
snapshot.online_disks,
snapshot.late_metadata_fanout_disks,
),
(None, Some(entry)) => match Arc::try_unwrap(entry) {
Ok(entry) => (entry.fi, entry.parts_metadata, entry.online_disks),
Err(entry) => (entry.fi.clone(), entry.parts_metadata.clone(), entry.online_disks.clone()),
Ok(entry) => (entry.fi, entry.parts_metadata, entry.online_disks, None),
Err(entry) => (entry.fi.clone(), entry.parts_metadata.clone(), entry.online_disks.clone(), None),
},
_ => unreachable!("GET metadata snapshot representation must be exclusive"),
}
@@ -1037,14 +1068,19 @@ mod prepared_get_object_metadata_tests {
const READ_VERSION_BARRIER_GUARD: std::time::Duration = std::time::Duration::from_secs(10);
fn object_with_initial_data_shards(bucket: &str, prefix: &str) -> String {
object_with_initial_data_shards_for_geometry(bucket, prefix, 4, 2)
}
fn object_with_initial_data_shards_for_geometry(bucket: &str, prefix: &str, total_disks: usize, parity: usize) -> String {
(0..1000)
.map(|index| format!("{prefix}-{index}.bin"))
.find(|name| {
let order = bounded_metadata_fanout_order(bucket, name, 4, 2);
let distribution = FileInfo::new(&[bucket, name].join("/"), 2, 2).erasure.distribution;
let mut seen = [false; 2];
for disk_index in order.into_iter().take(3) {
if let Some(block_index @ 1..=2) = distribution.get(disk_index).copied() {
let order = bounded_metadata_fanout_order(bucket, name, total_disks, parity);
let data = total_disks.saturating_sub(parity);
let distribution = FileInfo::new(&[bucket, name].join("/"), data, parity).erasure.distribution;
let mut seen = vec![false; data];
for disk_index in order.into_iter().take(total_disks.saturating_sub(parity).saturating_add(1)) {
if let Some(block_index) = distribution.get(disk_index).copied().filter(|index| *index <= data) {
seen[block_index - 1] = true;
}
}
@@ -1177,9 +1213,9 @@ mod prepared_get_object_metadata_tests {
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn two_phase_read_plan_uses_metadata_only_for_non_inline_get() {
async fn non_inline_data_read_early_stop_uses_quorum_plan() {
let (_dirs, set_disks) = make_local_set_disks(4, 2).await;
let bucket = "two-phase-read-plan";
let bucket = "non-inline-read-plan";
let object = object_with_initial_data_shards(bucket, "non-inline-object");
let payload = vec![0x5a; 2 * 1024 * 1024];
let opts = ObjectOptions {
@@ -1209,17 +1245,17 @@ mod prepared_get_object_metadata_tests {
let mut reader = set_disks
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
.await
.expect("two-phase GET reader should open");
.expect("quorum GET reader should open");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("two-phase GET body should stream");
.expect("quorum GET body should stream");
assert_eq!(restored, payload);
assert!(
calls.total(disk_call_counters::KIND_READ_VERSION) < 4,
"non-inline two-phase GET should stop metadata fanout at a quorum"
"non-inline quorum GET should retain a reconstruction reserve"
);
},
)
@@ -1228,11 +1264,334 @@ mod prepared_get_object_metadata_tests {
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn two_phase_read_plan_preserves_inline_early_stop_path() {
async fn non_inline_two_phase_read_fetches_late_parity_after_two_selected_shards_fail() {
let (dirs, set_disks) = make_local_set_disks(4, 2).await;
let bucket = "non-inline-read-late-parity";
let object = object_with_initial_data_shards(bucket, "late-parity-object");
let payload = vec![0x5a; 1024 * 1024];
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut put_reader = PutObjReader::from_vec(payload.clone());
set_disks
.put_object(bucket, &object, &mut put_reader, &opts)
.await
.expect("object should be written");
let order = bounded_metadata_fanout_order(bucket, &object, 4, 2);
let distribution = FileInfo::new(&[bucket, object.as_str()].join("/"), 2, 2).erasure.distribution;
assert!(
order.iter().take(2).all(|disk_index| distribution[*disk_index] <= 2),
"the two failed selected shards must be data shards"
);
assert!(
distribution[order[3]] > 2,
"the metadata shard omitted by the plan must be healthy parity"
);
for disk_index in order.iter().take(2) {
let object_dir = dirs[*disk_index].path().join(bucket).join(&object);
let data_dir = std::fs::read_dir(&object_dir)
.expect("object directory should be readable")
.find_map(|entry| {
let entry = entry.expect("object directory entry should be readable");
entry
.file_type()
.expect("object directory entry type should be readable")
.is_dir()
.then(|| entry.path())
})
.expect("object data directory should exist");
let part_path = data_dir.join("part.1");
let mut shard = std::fs::read(&part_path).expect("selected data shard should be readable before corruption");
shard[0] ^= 0xff;
std::fs::write(part_path, shard).expect("selected data shard should be corrupted after metadata was written");
}
temp_env::async_with_vars(
[
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
],
async {
let calls = disk_call_counters::observe(&object);
let mut reader = set_disks
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
.await
.expect("two-phase GET should recover using late parity");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("late parity should restore the exact GET body");
assert_eq!(restored, payload);
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 7);
},
)
.await;
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn non_inline_two_phase_read_fetches_late_parity_when_selected_parts_are_missing() {
let (dirs, set_disks) = make_local_set_disks(4, 2).await;
let bucket = "non-inline-read-late-parity-missing";
let object = object_with_initial_data_shards(bucket, "late-parity-missing-object");
let payload = vec![0x3c; 1024 * 1024];
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut put_reader = PutObjReader::from_vec(payload.clone());
set_disks
.put_object(bucket, &object, &mut put_reader, &opts)
.await
.expect("object should be written");
let order = bounded_metadata_fanout_order(bucket, &object, 4, 2);
for disk_index in order.iter().take(2) {
let object_dir = dirs[*disk_index].path().join(bucket).join(&object);
let data_dir = std::fs::read_dir(&object_dir)
.expect("object directory should be readable")
.find_map(|entry| {
let entry = entry.expect("object directory entry should be readable");
entry
.file_type()
.expect("entry type should be readable")
.is_dir()
.then(|| entry.path())
})
.expect("object data directory should exist");
std::fs::remove_file(data_dir.join("part.1")).expect("selected data shard should be removed");
}
temp_env::async_with_vars(
[
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
],
async {
let calls = disk_call_counters::observe(&object);
let mut reader = set_disks
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
.await
.expect("two-phase GET should recover using late parity");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("late parity should restore the exact GET body");
assert_eq!(restored, payload);
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 7);
},
)
.await;
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn four_data_two_parity_two_phase_read_recovers_one_failed_data_shard() {
let (dirs, set_disks) = make_local_set_disks(6, 2).await;
let bucket = "four-data-two-parity-late-read";
let object = object_with_initial_data_shards_for_geometry(bucket, "one-failed-data", 4, 2);
let payload = vec![0x7a; 1024 * 1024];
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut put_reader = PutObjReader::from_vec(payload.clone());
set_disks
.put_object(bucket, &object, &mut put_reader, &opts)
.await
.expect("object should be written");
let order = bounded_metadata_fanout_order(bucket, &object, 6, 2);
let distribution = FileInfo::new(&[bucket, object.as_str()].join("/"), 4, 2).erasure.distribution;
let failed_disk = *order
.iter()
.take(4)
.find(|disk_index| distribution[**disk_index] <= 4)
.expect("initial fanout should include a data shard");
assert!(
order.iter().take(4).all(|disk_index| distribution[*disk_index] <= 4),
"initial fanout should cover all four data shards"
);
assert!(distribution[order[5]] > 4, "the final deferred metadata shard should be parity");
let object_dir = dirs[failed_disk].path().join(bucket).join(&object);
let data_dir = std::fs::read_dir(&object_dir)
.expect("object directory should be readable")
.find_map(|entry| {
let entry = entry.expect("object directory entry should be readable");
entry
.file_type()
.expect("object directory entry type should be readable")
.is_dir()
.then(|| entry.path())
})
.expect("object data directory should exist");
let part_path = data_dir.join("part.1");
let mut shard = std::fs::read(&part_path).expect("selected data shard should be readable before corruption");
shard[0] ^= 0xff;
std::fs::write(part_path, shard).expect("selected data shard should be corrupted after metadata was written");
temp_env::async_with_vars(
[
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
],
async {
let calls = disk_call_counters::observe(&object);
let mut reader = set_disks
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
.await
.expect("two-phase GET should recover with one failed data shard");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("late parity should restore the exact GET body");
assert_eq!(restored, payload);
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 11);
},
)
.await;
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn four_data_two_parity_two_phase_read_rejects_below_read_quorum() {
let (dirs, set_disks) = make_local_set_disks(6, 2).await;
let bucket = "four-data-two-parity-quorum-minus-one";
let object = object_with_initial_data_shards_for_geometry(bucket, "quorum-minus-one", 4, 2);
let payload = vec![0x4b; 1024 * 1024];
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut put_reader = PutObjReader::from_vec(payload);
set_disks
.put_object(bucket, &object, &mut put_reader, &opts)
.await
.expect("object should be written");
let order = bounded_metadata_fanout_order(bucket, &object, 6, 2);
for disk_index in order.iter().take(3) {
let object_dir = dirs[*disk_index].path().join(bucket).join(&object);
let data_dir = std::fs::read_dir(&object_dir)
.expect("object directory should be readable")
.find_map(|entry| {
let entry = entry.expect("object directory entry should be readable");
entry
.file_type()
.expect("entry type should be readable")
.is_dir()
.then(|| entry.path())
})
.expect("object data directory should exist");
std::fs::remove_file(data_dir.join("part.1")).expect("selected shard should be removed");
}
temp_env::async_with_vars(
[
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
],
async {
let result = set_disks
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
.await;
assert!(result.is_err(), "quorum-minus-one read must fail closed without exposing a body");
},
)
.await;
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn non_inline_data_read_early_stop_keeps_reserve_on_unequal_layout() {
let (_dirs, set_disks) = make_local_set_disks(6, 2).await;
let bucket = "non-inline-read-reserve";
let object = object_with_initial_data_shards_for_geometry(bucket, "reserve-object", 6, 2);
let payload = vec![0x5a; 2 * 1024 * 1024];
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut put_reader = PutObjReader::from_vec(payload.clone());
set_disks
.put_object(bucket, &object, &mut put_reader, &opts)
.await
.expect("object should be written");
temp_env::async_with_vars(
[
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
],
async {
let calls = disk_call_counters::observe(&object);
let mut reader = set_disks
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
.await
.expect("quorum GET reader should open");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("quorum GET body should stream");
assert_eq!(restored, payload);
assert_eq!(
calls.total(disk_call_counters::KIND_READ_VERSION),
5,
"the unequal layout should schedule exactly one reserve beyond its data quorum"
);
},
)
.await;
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn non_inline_data_read_early_stop_preserves_inline_path() {
let (_dirs, set_disks) = make_local_set_disks(4, 2).await;
let bucket = "two-phase-read-plan-inline";
let bucket = "non-inline-read-plan-inline";
let object = object_with_initial_data_shards(bucket, "inline-object");
let payload = b"two-phase inline payload".repeat(256);
let payload = b"quorum inline payload".repeat(256);
let opts = ObjectOptions {
no_lock: true,
..Default::default()
@@ -1259,13 +1618,13 @@ mod prepared_get_object_metadata_tests {
let mut reader = set_disks
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
.await
.expect("two-phase inline GET reader should open");
.expect("inline GET reader should open");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("two-phase inline GET body should stream");
.expect("inline GET body should stream");
assert_eq!(restored, payload);
assert_eq!(
test_get_object_reader_path_id(),
@@ -1278,6 +1637,69 @@ mod prepared_get_object_metadata_tests {
.await;
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn non_inline_data_read_early_stop_does_not_add_inline_fanout_on_unequal_layout() {
let (_dirs, set_disks) = make_local_set_disks(6, 2).await;
let bucket = "inline-read-plan-unequal";
let object = object_with_initial_data_shards_for_geometry(bucket, "inline-object", 6, 2);
let payload = b"inline quorum payload".repeat(256);
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut put_reader = PutObjReader::from_vec(payload.clone());
set_disks
.put_object(bucket, &object, &mut put_reader, &opts)
.await
.expect("inline object should be written");
let read_once = |enabled: bool| {
let set_disks = Arc::clone(&set_disks);
let bucket = bucket.to_string();
let object = object.clone();
let payload = payload.clone();
let opts = opts.clone();
async move {
temp_env::async_with_vars(
[
(
"RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE",
Some(if enabled { "true" } else { "false" }),
),
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
],
async {
let calls = disk_call_counters::observe(&object);
let mut reader = set_disks
.get_object_reader(&bucket, &object, None, HeaderMap::new(), &opts)
.await
.expect("inline GET reader should open");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("inline GET body should stream");
assert_eq!(restored, payload);
calls.total(disk_call_counters::KIND_READ_VERSION)
},
)
.await
}
};
let gate_off_calls = read_once(false).await;
let gate_on_calls = read_once(true).await;
assert_eq!(gate_on_calls, gate_off_calls, "inline gate must not add reserve fanout");
}
#[test]
#[serial_test::serial(body_cache_hook)]
fn inline_data_read_early_stop_defaults_return_exact_body() {
@@ -2024,7 +2446,7 @@ fn is_get_metadata_data_read_early_stop_enabled() -> bool {
}
}
fn is_get_metadata_two_phase_read_plan_enabled() -> bool {
fn is_get_metadata_non_inline_data_read_early_stop_enabled() -> bool {
#[cfg(test)]
{
rustfs_utils::get_env_bool(
@@ -2220,6 +2642,15 @@ fn should_use_codec_streaming(config: GetCodecStreamingConfig, bucket: &str, obj
is_optimization_enabled_for_request(config.enabled, config.rollout_pct, bucket, object)
}
pub(in crate::set_disk) fn codec_streaming_rollout_applies(bucket: &str, object: &str) -> bool {
let config = get_codec_streaming_config();
config.enabled
&& config.body_compat_confirmed
&& config.header_compat_confirmed
&& config.rollout.is_opted_in()
&& should_use_codec_streaming(config, bucket, object)
}
/// Should this specific request use metadata early-stop?
#[allow(
dead_code,
@@ -6371,6 +6802,7 @@ mod tests {
use crate::object_api::BLOCK_SIZE_V2;
use crate::object_api::ObjectInfo;
use crate::set_disk::core::io_primitives::rename_fanout_barrier;
use crate::set_disk::ops::object::{PutObjectCommitBarrier, PutObjectCommitPause};
use crate::storage_api_contracts::{
heal::HealOperations as _, lifecycle::TransitionedObject, list::ListOperations as _, multipart::CompletePart,
object::ObjectOperations as _,
@@ -12129,6 +12561,7 @@ mod tests {
0,
true,
false,
false,
GET_OBJECT_PATH_LEGACY_DUPLEX,
GET_CODEC_STREAMING_OBJECT_CLASS_PLAIN_SINGLE_PART,
metrics_size_bucket,
@@ -12241,6 +12674,7 @@ mod tests {
0,
true,
false,
false,
GET_OBJECT_PATH_LEGACY_DUPLEX,
GET_CODEC_STREAMING_OBJECT_CLASS_PLAIN_SINGLE_PART,
metrics_size_bucket,
@@ -12811,6 +13245,111 @@ mod tests {
.await;
}
#[tokio::test(flavor = "multi_thread")]
#[serial]
async fn multipart_streaming_get_blocks_overwrite_across_part_boundary() {
temp_env::async_with_vars(
[
(rustfs_config::ENV_OBJECT_LOCK_OPTIMIZATION_ENABLE, Some("true")),
(ENV_RUSTFS_GET_MULTIPART_READER_SETUP_PREFETCH, Some("false")),
],
async {
let set_disks = make_local_bucket_test_set_disks().await;
let bucket = "snapshot-multipart-overwrite";
let object = "object";
let part_size = usize::try_from(GLOBAL_MIN_PART_SIZE.as_u64()).expect("minimum part size should fit usize");
let first_part = vec![0x41; part_size];
let second_part = vec![0x42; part_size];
let replacement = vec![0x43; first_part.len() + second_part.len()];
let opts = ObjectOptions::default();
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let upload = set_disks
.new_multipart_upload(bucket, object, &opts)
.await
.expect("multipart upload should be created");
let mut completed_parts = Vec::with_capacity(2);
for (part_num, body) in [(1, &first_part), (2, &second_part)] {
let mut reader = PutObjReader::from_vec(body.clone());
let part = set_disks
.put_object_part(bucket, object, &upload.upload_id, part_num, &mut reader, &opts)
.await
.expect("multipart part should be written");
completed_parts.push(CompletePart {
part_num,
etag: part.etag,
..Default::default()
});
}
let completed = Arc::clone(&set_disks)
.complete_multipart_upload(bucket, object, &upload.upload_id, completed_parts, &opts)
.await
.expect("multipart upload should complete");
assert!(completed.is_multipart());
let mut snapshot = set_disks
.get_object_reader(bucket, object, None, HeaderMap::new(), &opts)
.await
.expect("multipart snapshot reader should open");
let overwrite_set = Arc::clone(&set_disks);
let overwrite_opts = opts.clone();
let overwrite_body = replacement.clone();
let commit_barrier = PutObjectCommitBarrier::install(bucket, object, PutObjectCommitPause::BeforeNamespace);
let overwrite = tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(overwrite_body);
overwrite_set.put_object(bucket, object, &mut reader, &overwrite_opts).await
});
commit_barrier.wait_until_paused().await;
commit_barrier.release_and_wait_until_namespace_pending().await;
assert!(
!commit_barrier.namespace_acquired(),
"overwrite must wait for the multipart response's read lock"
);
let mut restored_first = vec![0; first_part.len()];
snapshot
.stream
.read_exact(&mut restored_first)
.await
.expect("the first multipart part should stream");
assert_eq!(restored_first, first_part);
assert!(
!commit_barrier.namespace_acquired() && !overwrite.is_finished(),
"overwrite must remain blocked at the first/second part boundary"
);
let mut restored_second = Vec::new();
snapshot
.stream
.read_to_end(&mut restored_second)
.await
.expect("the second multipart part should stream");
assert_eq!(restored_second, second_part);
tokio::time::timeout(Duration::from_secs(5), overwrite)
.await
.expect("overwrite should proceed after multipart EOF")
.expect("overwrite task should join")
.expect("overwrite should succeed");
let mut latest = set_disks
.get_object_reader(bucket, object, None, HeaderMap::new(), &opts)
.await
.expect("replacement reader should open");
let mut latest_body = Vec::new();
latest
.stream
.read_to_end(&mut latest_body)
.await
.expect("replacement should stream");
assert_eq!(latest_body, replacement);
},
)
.await;
}
#[tokio::test(flavor = "multi_thread")]
#[serial]
async fn streaming_get_blocks_concurrent_delete_until_eof() {
+37 -14
View File
@@ -2767,6 +2767,26 @@ mod heal_result_report_tests {
}
}
async fn remove_current_object_part(temp_dir: &TempDir, bucket: &str, object: &str) -> std::io::Result<()> {
let object_dir = temp_dir.path().join(bucket).join(object);
let mut entries = tokio::fs::read_dir(&object_dir).await?;
while let Some(entry) = entries.next_entry().await? {
if !entry.file_type().await?.is_dir() {
continue;
}
let part = entry.path().join("part.1");
match tokio::fs::remove_file(&part).await {
Ok(()) => return Ok(()),
Err(err) if err.kind() == std::io::ErrorKind::NotFound => continue,
Err(err) => return Err(err),
}
}
Err(std::io::Error::new(
std::io::ErrorKind::NotFound,
format!("no current part.1 found under {}", object_dir.display()),
))
}
#[test]
fn heal_writer_error_summary_redacts_io_message() {
let error = DiskError::Io(std::io::Error::new(std::io::ErrorKind::PermissionDenied, "/sensitive/storage/path"));
@@ -2799,21 +2819,13 @@ mod heal_result_report_tests {
.read_version("", &bucket, object, "", &ReadOptions::default())
.await
.expect("source metadata should be readable");
let data_dir = source.data_dir.expect("non-inline source should have a data directory");
let mut target_slots = [source.erasure.distribution[0] - 1, source.erasure.distribution[1] - 1];
target_slots.sort_unstable();
for index in [0, 1] {
tokio::fs::remove_file(
temp_dirs[index]
.path()
.join(&bucket)
.join(object)
.join(data_dir.to_string())
.join("part.1"),
)
.await
.expect("target shard should be removed before heal");
remove_current_object_part(&temp_dirs[index], &bucket, object)
.await
.expect("target shard should be removed before heal");
}
let failed_slots = &target_slots[..failed_target_count];
@@ -3069,9 +3081,20 @@ mod heal_result_report_tests {
let payload = vec![0x5a; 1024 * 1024];
let mut reader = PutObjReader::from_vec(payload);
set.put_object(&bucket, object, &mut reader, &ObjectOptions::default())
.await
.expect("source object should be written");
// This fixture removes physical shards immediately after PUT. A
// lock-owning PUT may quorum-ack before its rename tail drains, so
// keep the isolated setup on the full-fanout commit path.
set.put_object(
&bucket,
object,
&mut reader,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("source object should be written");
let source = disks[2]
.read_version("", &bucket, object, "", &ReadOptions::default())
.await
+167
View File
@@ -1379,6 +1379,14 @@ fn data_read_metadata_early_stop_request_shape_allowed(range: &Option<HTTPRangeS
&& !crate::object_api::restore_request_active(opts)
}
fn prepare_late_materialized_retry(initial_result: &Result<()>, output: &mut Vec<u8>, expected_size: usize) -> bool {
if initial_result.is_ok() && output.len() == expected_size {
return false;
}
output.clear();
true
}
#[cfg(test)]
mod data_read_metadata_early_stop_request_shape_tests {
use super::*;
@@ -1434,6 +1442,14 @@ mod data_read_metadata_early_stop_request_shape_tests {
restore_opts.transition.restore_request.days = Some(1);
assert!(!data_read_metadata_early_stop_request_shape_allowed(&None, &restore_opts));
}
#[test]
fn late_materialized_retry_clears_partial_buffer_after_error() {
let mut output = b"partial-prefix".to_vec();
let result = Err(Error::FileCorrupt);
assert!(prepare_late_materialized_retry(&result, &mut output, 1024));
assert!(output.is_empty());
}
}
/// Length of the full plaintext body when — and only when — this read's output
@@ -2012,6 +2028,88 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
}
}
if snapshot.has_late_metadata_fanout() {
// Keep refresh plus the second decode off the default GET poll stack.
// The allocation is limited to the opt-in late-materialization path.
return Box::pin(async move {
let object_size = usize::try_from(object_info.size)
.map_err(|_| to_object_err(Error::other("two-phase GET object size is invalid"), vec![bucket, object]))?;
let mut output = Vec::with_capacity(object_size);
let (fi, files, disks, late_metadata_fanout_disks) = snapshot.into_owned_with_late_metadata_fanout();
let expected_identity = super::super::read::LateMetadataIdentity::from_file_info(&fi);
let late_metadata_fanout_disks = late_metadata_fanout_disks.ok_or_else(|| {
to_object_err(Error::other("two-phase GET fallback context is missing"), vec![bucket, object])
})?;
let initial_result = Self::get_object_with_fileinfo(
bucket,
object,
Arc::clone(&self.erasure_cache),
0,
object_info.size,
&mut output,
fi,
files,
&disks,
self.set_index,
self.pool_index,
opts.skip_verify_bitrot,
true,
true,
GET_OBJECT_PATH_LEGACY_DUPLEX,
object_class.as_str(),
size_bucket,
)
.await;
if prepare_late_materialized_retry(&initial_result, &mut output, object_size) {
let (full_fi, full_parts_metadata, full_online_disks) = Self::refresh_late_metadata_fanout(
&late_metadata_fanout_disks,
bucket,
object,
&expected_identity,
GET_OBJECT_PATH_LEGACY_DUPLEX,
)
.await?;
Self::get_object_with_fileinfo(
bucket,
object,
Arc::clone(&self.erasure_cache),
0,
object_info.size,
&mut output,
full_fi,
full_parts_metadata,
&full_online_disks,
self.set_index,
self.pool_index,
opts.skip_verify_bitrot,
true,
false,
GET_OBJECT_PATH_LEGACY_DUPLEX,
object_class.as_str(),
size_bucket,
)
.await?;
}
if output.len() != object_size {
return Err(to_object_err(Error::other("two-phase GET decoded length mismatch"), vec![bucket, object]));
}
record_get_object_reader_path_observation(GET_OBJECT_PATH_LEGACY_DUPLEX, object_class, size_bucket);
let body = Bytes::from(output);
let reader = GetObjectReader {
stream: Box::new(Cursor::new(body.clone())),
object_info,
buffered_body: Some(body),
body_source,
};
if lock_optimization_enabled {
release_materialized_read_lock(bucket, object, read_lock_guard.take());
}
Ok(reader)
})
.await;
}
let direct_memory_decision = get_small_object_direct_memory_decision_with_threshold_and_plan(
&range,
&object_info,
@@ -2073,6 +2171,7 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
self.pool_index,
opts.skip_verify_bitrot,
true,
false,
GET_OBJECT_PATH_DIRECT_MEMORY,
object_class.as_str(),
size_bucket,
@@ -2272,6 +2371,7 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
pool_index,
skip_verify,
false,
false,
GET_OBJECT_PATH_LEGACY_DUPLEX,
object_class.as_str(),
size_bucket,
@@ -7821,6 +7921,7 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
pool_index,
skip_verify,
false,
false,
GET_OBJECT_PATH_LEGACY_DUPLEX,
GET_CODEC_STREAMING_OBJECT_CLASS_PLAIN_SINGLE_PART,
metrics_size_bucket,
@@ -9791,6 +9892,9 @@ mod inline_put_commit_path_tests {
use super::*;
use crate::config::storageclass::{INLINE_BLOCK_ENV, lookup_config_for_pools, lookup_config_for_pools_without_env};
use crate::disk::ReadOptions;
use crate::ecstore_validation_blackbox::make_local_set_disks;
use crate::set_disk::disk_call_counters;
use crate::storage_api_contracts::bucket::{BucketOperations, MakeBucketOptions};
use rustfs_config::server_config::KVS;
use serial_test::serial;
use tokio::io::AsyncReadExt;
@@ -9963,6 +10067,69 @@ mod inline_put_commit_path_tests {
.await;
}
#[tokio::test]
#[serial]
async fn get_object_reader_codec_rollout_excludes_late_metadata_refresh() {
let (_temp_dirs, set_disks) = make_local_set_disks(4, 2).await;
let bucket = "one-mib-codec-reader";
let object = "object.bin";
let payload = vec![0x6b; 1024 * 1024];
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("codec bucket should be created");
let storage_class = temp_env::with_var(INLINE_BLOCK_ENV, Some("1KiB"), || lookup_config_for_pools(&KVS::new(), &[4]))
.expect("test storage class should resolve");
set_disks.set_test_storage_class_config(storage_class);
let mut writer = PutObjReader::from_vec(payload.clone());
temp_env::async_with_vars(
[
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
(ENV_RUSTFS_GET_MID_SIZE_STREAMING_ENABLE, Some("false")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_ENABLE, Some("true")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_MIN_SIZE, Some("1")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_ENGINE, Some("legacy")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_MULTIPART_ENABLE, Some("true")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_DATA_BLOCKS_FIRST_ENABLE, Some("false")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_BODY_COMPAT_CONFIRMED, Some("true")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_HEADER_COMPAT_CONFIRMED, Some("true")),
(crate::set_disk::ENV_RUSTFS_GET_CODEC_STREAMING_ROLLOUT, Some("on")),
(rustfs_config::ENV_OBJECT_LOCK_OPTIMIZATION_ENABLE, Some("true")),
],
async {
set_disks
.put_object(bucket, object, &mut writer, &ObjectOptions::default())
.await
.expect("codec fixture should commit");
crate::set_disk::reset_test_get_object_reader_path();
let calls = disk_call_counters::observe(object);
let mut reader = set_disks
.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default())
.await
.expect("codec GET should succeed");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("codec GET should stream");
assert_eq!(restored, payload);
assert_eq!(
crate::set_disk::test_get_object_reader_path_id(),
5,
"codec path must win over late refresh"
);
assert_eq!(
calls.total(disk_call_counters::KIND_READ_VERSION),
4,
"codec path must use full metadata fanout and must not trigger a second late refresh"
);
},
)
.await;
}
#[tokio::test]
async fn repeated_gets_reuse_the_set_erasure_shell() {
let (_temp_dirs, disk_stores, set_disks) = hermetic_set_disks(4).await;
+180 -27
View File
@@ -128,7 +128,7 @@ use super::is_get_metadata_early_stop_bounded_fanout_enabled;
#[cfg(test)]
use super::is_get_metadata_early_stop_enabled;
#[cfg(test)]
use super::is_get_metadata_two_phase_read_plan_enabled;
use super::is_get_metadata_non_inline_data_read_early_stop_enabled;
#[cfg(test)]
use super::is_version_early_stop_enabled;
#[cfg(test)]
@@ -609,7 +609,20 @@ impl SetDisks {
// let online_disks: Vec<Option<DiskStore>> = op_online_disks.iter().filter(|v| v.is_some()).cloned().collect();
Ok(GetObjectFileInfo::owned(fi, parts_metadata, op_online_disks))
if !metadata_fanout_complete
&& allow_early_stop
&& non_inline_data_read_early_stop_allowed(read_data, bucket, object)
&& late_materialization_candidate_is_safe(&fi)
{
Ok(GetObjectFileInfo::owned_with_late_metadata_fanout(
fi,
parts_metadata,
op_online_disks,
disks,
))
} else {
Ok(GetObjectFileInfo::owned(fi, parts_metadata, op_online_disks))
}
}
#[hotpath::measure(impl_type = "SetDisks")]
@@ -819,6 +832,7 @@ impl SetDisks {
pool_index: usize,
skip_verify_bitrot: bool,
prefer_data_blocks_first_reader_setup: bool,
require_reconstruction_surplus: bool,
metrics_path: &'static str,
metrics_object_class: &'static str,
metrics_size_bucket: &'static str,
@@ -1083,6 +1097,9 @@ impl SetDisks {
}
let nil_count = reader_setup.available_shards();
if require_reconstruction_surplus && nil_count <= erasure.data_shards {
return Err(Error::other("insufficient reconstruction surplus for two-phase read"));
}
if nil_count < erasure.data_shards {
if let Some(read_err) = reduce_read_quorum_errs(&reader_setup.errors, OBJECT_OP_IGNORED_ERRS, erasure.data_shards)
{
@@ -1190,18 +1207,34 @@ impl SetDisks {
let readers = reader_setup.readers;
let deferred_stripe_handles = reader_setup.deferred_stripe_handles;
let deferred_reopeners = reader_setup.deferred_reopeners;
let (written, err) = erasure
.decode_with_stripe_handles_and_reopeners(
writer,
readers,
part_offset,
part_length,
part_size,
read_costs,
deferred_stripe_handles,
deferred_reopeners,
)
.await;
let (written, err, exact_quorum) = if require_reconstruction_surplus {
erasure
.decode_with_stripe_handles_and_reopeners_with_diagnostics(
writer,
readers,
part_offset,
part_length,
part_size,
read_costs,
deferred_stripe_handles,
deferred_reopeners,
)
.await
} else {
let (written, err) = erasure
.decode_with_stripe_handles_and_reopeners(
writer,
readers,
part_offset,
part_length,
part_size,
read_costs,
deferred_stripe_handles,
deferred_reopeners,
)
.await;
(written, err, false)
};
let decode_elapsed = decode_stage_start.elapsed();
rustfs_io_metrics::record_get_object_decode_duration(decode_elapsed.as_secs_f64());
rustfs_io_metrics::record_get_object_stage_duration_by_size(
@@ -1211,6 +1244,9 @@ impl SetDisks {
metrics_size_bucket,
decode_elapsed.as_secs_f64(),
);
if exact_quorum && err.is_none() {
return Err(Error::other("two-phase read completed with exact reconstruction quorum"));
}
if decode_elapsed >= SLOW_OBJECT_READ_LOG_THRESHOLD && err.is_none() {
warn!(
event = EVENT_SET_DISK_READ,
@@ -1758,6 +1794,102 @@ fn multipart_reader_setup_prefetch_enabled(policy: GetObjectReadPolicy) -> bool
policy.allows_multipart_setup_prefetch() && is_multipart_reader_setup_prefetch_enabled()
}
pub(super) struct LateMetadataIdentity {
volume: String,
name: String,
algorithm: String,
block_size: usize,
uses_legacy_checksum: bool,
quorum_hash: [u8; 32],
distribution: Vec<usize>,
parity_blocks: usize,
}
impl LateMetadataIdentity {
pub(super) fn from_file_info(file_info: &FileInfo) -> Self {
Self {
volume: file_info.volume.clone(),
name: file_info.name.clone(),
algorithm: file_info.erasure.algorithm.clone(),
block_size: file_info.erasure.block_size,
uses_legacy_checksum: file_info.uses_legacy_checksum,
quorum_hash: SetDisks::file_info_quorum_hash(file_info),
distribution: file_info.erasure.distribution.clone(),
parity_blocks: file_info.erasure.parity_blocks,
}
}
}
fn late_metadata_read_identity_matches(expected: &LateMetadataIdentity, actual: &FileInfo) -> bool {
expected.volume == actual.volume
&& expected.name == actual.name
&& expected.algorithm == actual.erasure.algorithm
&& expected.block_size == actual.erasure.block_size
&& expected.uses_legacy_checksum == actual.uses_legacy_checksum
&& expected.quorum_hash == SetDisks::file_info_quorum_hash(actual)
}
fn late_metadata_shard_matches(expected: &LateMetadataIdentity, actual: &FileInfo, disk_index: usize) -> bool {
expected
.distribution
.get(disk_index)
.is_some_and(|mapped_index| *mapped_index == actual.erasure.index)
&& late_metadata_read_identity_matches(expected, actual)
}
impl SetDisks {
pub(super) async fn refresh_late_metadata_fanout(
fallback_disks: &[Option<DiskStore>],
bucket: &str,
object: &str,
expected: &LateMetadataIdentity,
metrics_path: &'static str,
) -> Result<(FileInfo, Vec<FileInfo>, Vec<Option<DiskStore>>)> {
let (mut parts_metadata, errs, diagnostics) = SetDisks::read_all_fileinfo_observed(
fallback_disks,
"",
bucket,
object,
"",
true,
false,
false,
false,
expected.parity_blocks,
)
.await?;
diagnostics.record(metrics_path);
let (read_quorum, write_quorum) = SetDisks::object_quorum_from_meta(&parts_metadata, &errs, expected.parity_blocks)
.map_err(|err| to_object_err(err.into(), vec![bucket, object]))?;
let read_quorum =
usize::try_from(read_quorum).map_err(|_| to_object_err(DiskError::ErasureReadQuorum.into(), vec![bucket, object]))?;
let write_quorum = usize::try_from(write_quorum)
.map_err(|_| to_object_err(DiskError::ErasureWriteQuorum.into(), vec![bucket, object]))?;
if let Some(err) = reduce_read_quorum_errs(&errs, OBJECT_OP_IGNORED_ERRS, read_quorum) {
return Err(to_object_err(err.into(), vec![bucket, object]));
}
let (mut online_disks, full_fi, _) =
SetDisks::select_valid_fileinfo(fallback_disks, &parts_metadata, &errs, "", read_quorum, write_quorum)?;
if !late_metadata_read_identity_matches(expected, &full_fi) {
return Err(to_object_err(DiskError::ErasureReadQuorum.into(), vec![bucket, object]));
}
for (disk_index, (metadata, disk)) in parts_metadata.iter_mut().zip(online_disks.iter_mut()).enumerate() {
if !late_metadata_shard_matches(expected, metadata, disk_index) {
*metadata = FileInfo::default();
*disk = None;
}
}
if online_disks.iter().filter(|disk| disk.is_some()).count() < read_quorum {
return Err(to_object_err(DiskError::ErasureReadQuorum.into(), vec![bucket, object]));
}
Ok((full_fi, parts_metadata, online_disks))
}
}
/// Run one part's bitrot reader setup and measure its wall-clock duration.
///
/// Shared by the synchronous path and the prefetch task in
@@ -2349,6 +2481,7 @@ mod metadata_cache_tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"plain",
"small",
@@ -2380,6 +2513,7 @@ mod metadata_cache_tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"plain",
"small",
@@ -2404,6 +2538,7 @@ mod metadata_cache_tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"plain",
"small",
@@ -2426,6 +2561,7 @@ mod metadata_cache_tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"plain",
"small",
@@ -2450,6 +2586,7 @@ mod metadata_cache_tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"plain",
"small",
@@ -2488,6 +2625,7 @@ mod metadata_cache_tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"plain",
"empty",
@@ -2521,6 +2659,7 @@ mod metadata_cache_tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"plain",
"small",
@@ -4272,15 +4411,16 @@ mod tests {
}
#[test]
fn two_phase_read_plan_gate_defaults_off_and_honors_override() {
#[serial(body_cache_hook)]
fn non_inline_data_read_early_stop_gate_defaults_off_and_honors_override() {
temp_env::with_var(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, None::<&str>, || {
assert!(!is_get_metadata_two_phase_read_plan_enabled());
assert!(!is_get_metadata_non_inline_data_read_early_stop_enabled());
});
temp_env::with_var(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("true"), || {
assert!(is_get_metadata_two_phase_read_plan_enabled());
assert!(is_get_metadata_non_inline_data_read_early_stop_enabled());
});
temp_env::with_var(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("false"), || {
assert!(!is_get_metadata_two_phase_read_plan_enabled());
assert!(!is_get_metadata_non_inline_data_read_early_stop_enabled());
});
}
@@ -4881,6 +5021,7 @@ mod tests {
0,
false,
false,
false,
GET_OBJECT_PATH_SET_DISK,
"test-object-class",
"test-size-bucket",
@@ -5564,9 +5705,10 @@ mod tests {
/// backlog#923: with the data-shards-only lockstep gate on, every retained
/// parity reader must be an unopened deferred reader carrying a stripe
/// handle, so the decode path can realign it to a mid-object stripe. With
/// the gate off (default), eagerly opened parity readers are kept exactly
/// as before and carry no handles.
/// handle and disposable reopener, so the decode path can realign it to a
/// mid-object stripe without consuming the later-stripe reserve. With the
/// gate off (default), eagerly opened parity readers are kept exactly as
/// before and carry neither.
#[tokio::test]
#[serial_test::serial]
async fn bitrot_reader_setup_gates_parity_stripe_handle_conversion() {
@@ -5595,6 +5737,11 @@ mod tests {
enabled.is_some(),
"parity slot {idx} stripe handle must match the gate (enabled={enabled:?})"
);
assert_eq!(
setup.deferred_reopeners[idx].is_some(),
enabled.is_some(),
"parity slot {idx} reopener must match the gate (enabled={enabled:?})"
);
}
if enabled.is_some() {
@@ -5617,13 +5764,17 @@ mod tests {
}
#[tokio::test]
#[serial_test::serial]
async fn bitrot_reader_setup_data_blocks_first_keeps_deferred_fallback_readers() {
let mut setup = setup_inline_bitrot_readers_with_env(
vec![Some(b"aaaa"), Some(b"bbbb"), Some(b"cccc"), Some(b"dddd")],
2,
2,
BitrotReaderSetupMode::ReadQuorum,
true,
let mut setup = temp_env::async_with_vars(
[("RUSTFS_GET_LOCKSTEP_DATA_SHARDS_ONLY_ENABLE", Some("true"))],
setup_inline_bitrot_readers_with_env(
vec![Some(b"aaaa"), Some(b"bbbb"), Some(b"cccc"), Some(b"dddd")],
2,
2,
BitrotReaderSetupMode::ReadQuorum,
true,
),
)
.await;
@@ -5631,6 +5782,8 @@ mod tests {
assert_eq!(setup.available_shards(), 2);
assert_eq!(setup.scheduled_shards(), 2);
assert_eq!(setup.readers.iter().filter(|reader| reader.is_some()).count(), 4);
assert!(setup.deferred_reopeners[2].is_some());
assert!(setup.deferred_reopeners[3].is_some());
let fallback_index = setup
.attempted
+128 -105
View File
@@ -17,18 +17,17 @@ use crate::core::pools::{
PoolMetaReplicaState, PoolMetaWriteState, load_pool_meta_identity_observing, local_decommission_queue_prefix,
persist_pool_meta_identity_for_startup, pool_meta_has_active_decommission,
};
use crate::error::is_err_decommission_running;
use crate::runtime::instance::InstanceContext;
use crate::runtime::sources as runtime_sources;
use crate::storage_api_contracts::object::EcstoreObjectIO;
use rustfs_config::server_config::KVS;
use rustfs_credentials::{RPC_SECRET_REQUIRED_OPERATOR_MESSAGE, try_get_rpc_token};
use std::future::Future;
use tracing::{debug, error, info, warn};
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
const LOG_SUBSYSTEM_STORE_INIT: &str = "store_init";
const EVENT_DECOMMISSION_RESUME_RETRY: &str = "decommission_resume_retry";
const EVENT_DECOMMISSION_RESUME_FAILED: &str = "decommission_resume_failed";
const EVENT_STORE_FORMAT_RETRY: &str = "store_format_retry";
const EVENT_ECSTORE_INIT_STATUS: &str = "ecstore_init_status";
const EVENT_STORE_RPC_SECRET_PREFLIGHT_FAILED: &str = "store_rpc_secret_preflight_failed";
@@ -96,16 +95,13 @@ fn preflight_startup_rpc_secret_with(
}
}
const LOCAL_DECOMMISSION_RESUME_MAX_CONFIG_RETRIES: usize = 6;
const LOCAL_DECOMMISSION_INITIAL_RESUME_DELAY: Duration = Duration::from_secs(60 * 3);
const LOCAL_DECOMMISSION_RESUME_RETRY_DELAY: Duration = Duration::from_secs(30);
const LOCAL_DECOMMISSION_WATCHDOG_INTERVAL: Duration = Duration::from_secs(30);
const LOCAL_DECOMMISSION_WATCHDOG_MAX_RETRY_DELAY: Duration = Duration::from_secs(60 * 5);
const REBALANCE_INITIAL_RESUME_DELAY: Duration = Duration::from_secs(10);
const REBALANCE_RESUME_RETRY_DELAY: Duration = Duration::from_secs(10);
fn should_retry_local_decommission_resume(err: &Error, attempt: usize) -> bool {
matches!(err, Error::ConfigNotFound) && attempt < LOCAL_DECOMMISSION_RESUME_MAX_CONFIG_RETRIES
}
fn should_retry_format_load(err: &Error) -> bool {
!matches!(err, Error::CorruptedFormat)
}
@@ -118,14 +114,6 @@ fn should_defer_rebalance_auto_start(distributed: bool, fleet_proof_available: b
distributed && !fleet_proof_available
}
fn should_schedule_local_decommission_resume(
pool_indices: &[usize],
pool_meta_replica_state: PoolMetaReplicaState,
pool_meta_write_safe: bool,
) -> bool {
!pool_indices.is_empty() && pool_meta_replica_state.repair_write_safe && pool_meta_write_safe
}
async fn wait_for_local_decommission_resume_delay(rx: &CancellationToken, delay: Duration) -> bool {
tokio::select! {
_ = rx.cancelled() => false,
@@ -133,6 +121,13 @@ async fn wait_for_local_decommission_resume_delay(rx: &CancellationToken, delay:
}
}
fn local_decommission_watchdog_retry_delay(consecutive_failures: u32) -> Duration {
let exponent = consecutive_failures.saturating_sub(1).min(4);
LOCAL_DECOMMISSION_RESUME_RETRY_DELAY
.saturating_mul(1_u32 << exponent)
.min(LOCAL_DECOMMISSION_WATCHDOG_MAX_RETRY_DELAY)
}
async fn wait_for_rebalance_resume_delay(rx: &CancellationToken, delay: Duration) -> bool {
tokio::select! {
_ = rx.cancelled() => false,
@@ -235,71 +230,63 @@ where
Ok(committed)
}
async fn resume_local_decommission_after_init(store: Arc<ECStore>, rx: CancellationToken, pool_indices: Vec<usize>) {
for attempt in 0..=LOCAL_DECOMMISSION_RESUME_MAX_CONFIG_RETRIES {
async fn run_local_decommission_watchdog<F, Fut>(rx: CancellationToken, mut reconcile: F)
where
F: FnMut() -> Fut,
Fut: Future<Output = Result<()>>,
{
let mut consecutive_failures = 0_u32;
loop {
if rx.is_cancelled() {
return;
}
let result = if pool_indices.len() > 1 {
store
.spawn_decommission_routines(store.clone(), rx.clone(), pool_indices.clone())
.await
} else {
store.decommission(rx.clone(), pool_indices.clone()).await
};
match result {
Ok(()) => return,
Err(err) if is_err_decommission_running(&err) => {
if let Err(spawn_err) = store
.spawn_decommission_routines(store.clone(), rx.clone(), pool_indices.clone())
.await
{
error!(
event = EVENT_DECOMMISSION_RESUME_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_STORE_INIT,
pool_indices = ?pool_indices,
error = %spawn_err,
reason = "spawn_workers_failed",
"Failed to resume decommission workers"
);
}
return;
let delay = match reconcile().await {
Ok(()) => {
consecutive_failures = 0;
LOCAL_DECOMMISSION_WATCHDOG_INTERVAL
}
Err(err) if should_retry_local_decommission_resume(&err, attempt) => {
Err(err) => {
consecutive_failures = consecutive_failures.saturating_add(1);
let retry_delay = local_decommission_watchdog_retry_delay(consecutive_failures);
warn!(
event = EVENT_DECOMMISSION_RESUME_RETRY,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_STORE_INIT,
pool_indices = ?pool_indices,
retry_count = attempt + 1,
retry_limit = LOCAL_DECOMMISSION_RESUME_MAX_CONFIG_RETRIES + 1,
consecutive_failures,
retry_delay_secs = retry_delay.as_secs(),
error = %err,
"Retrying decommission resume after missing config"
"Retrying decommission worker recovery"
);
tokio::select! {
_ = rx.cancelled() => return,
_ = tokio::time::sleep(LOCAL_DECOMMISSION_RESUME_RETRY_DELAY) => {}
}
}
Err(err) => {
error!(
event = EVENT_DECOMMISSION_RESUME_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_STORE_INIT,
pool_indices = ?pool_indices,
error = %err,
reason = "resume_failed",
"Failed to resume decommission"
);
return;
retry_delay
}
};
if !wait_for_local_decommission_resume_delay(&rx, delay).await {
return;
}
}
}
async fn supervise_local_decommission_after_init(store: Arc<ECStore>, rx: CancellationToken) {
run_local_decommission_watchdog(rx.clone(), || {
let store = store.clone();
let worker_rx = rx.clone();
async move {
store
.ensure_pool_meta_side_effects_safe("decommission worker recovery blocked while pool metadata requires recovery")
.await?;
if store.has_active_local_decommission_worker().await {
return Ok(());
}
store.refresh_pool_status_meta().await?;
store.spawn_missing_local_decommission_routines_with_token(worker_rx).await
}
})
.await;
}
async fn resume_rebalance_after_init(store: Arc<ECStore>, rx: CancellationToken) {
if !wait_for_rebalance_resume_delay(&rx, REBALANCE_INITIAL_RESUME_DELAY).await {
return;
@@ -729,31 +716,32 @@ impl ECStore {
}
let local_pool_indices = local_decommission_queue_prefix(&endpoints, &pool_indices)?;
let has_local_decommission_leadership = endpoints.as_ref().iter().any(pool_first_endpoint_is_local);
let pool_meta_write_safe = self
.ensure_pool_meta_side_effects_safe("decommission resume blocked while pool metadata requires recovery")
.await
.is_ok();
if should_schedule_local_decommission_resume(&local_pool_indices, pool_meta_replica_state, pool_meta_write_safe) {
let store = self.clone();
let decommission_rx = rx.clone();
tokio::spawn(async move {
if !wait_for_local_decommission_resume_delay(&decommission_rx, LOCAL_DECOMMISSION_INITIAL_RESUME_DELAY).await {
return;
}
resume_local_decommission_after_init(store, decommission_rx, local_pool_indices).await;
});
} else if !local_pool_indices.is_empty() {
error!(
event = EVENT_DECOMMISSION_RESUME_FAILED,
if !pool_meta_replica_state.repair_write_safe || !pool_meta_write_safe {
warn!(
event = EVENT_DECOMMISSION_RESUME_RETRY,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_STORE_INIT,
state = "blocked",
pool_indices = ?local_pool_indices,
reason = "pool_meta_write_blocked",
"Decommission resume blocked until pool metadata replicas are readable and consistent"
"Decommission watchdog waiting for pool metadata recovery"
);
}
if has_local_decommission_leadership {
let store = self.clone();
let decommission_rx = rx.clone();
tokio::spawn(async move {
if !wait_for_local_decommission_resume_delay(&decommission_rx, LOCAL_DECOMMISSION_INITIAL_RESUME_DELAY).await {
return;
}
supervise_local_decommission_after_init(store, decommission_rx).await;
});
}
runtime_sources::init_bucket_monitor_for_current_endpoints();
crate::bucket::bucket_target_sys::BucketTargetSys::get().start_heartbeat();
@@ -786,12 +774,12 @@ impl ECStore {
#[cfg(test)]
mod tests {
use super::{
LOCAL_DECOMMISSION_RESUME_MAX_CONFIG_RETRIES, PoolMetaWriteState, establish_pool_meta_bootstrap_identity_if_proven,
load_pool_meta_for_startup, persist_pool_meta_for_startup_if_safe, pool_first_endpoint_is_local,
pool_meta_has_active_decommission, preflight_startup_rpc_secret_with, resolve_startup_pool_defaults_with,
resolve_store_init_stage_result, save_validated_pool_meta_for_startup, should_auto_start_rebalance_after_init,
should_defer_rebalance_auto_start, should_retry_format_load, should_retry_local_decommission_resume,
wait_for_local_decommission_resume_delay,
LOCAL_DECOMMISSION_RESUME_RETRY_DELAY, LOCAL_DECOMMISSION_WATCHDOG_MAX_RETRY_DELAY, PoolMetaWriteState,
establish_pool_meta_bootstrap_identity_if_proven, load_pool_meta_for_startup, local_decommission_watchdog_retry_delay,
persist_pool_meta_for_startup_if_safe, pool_first_endpoint_is_local, pool_meta_has_active_decommission,
preflight_startup_rpc_secret_with, resolve_startup_pool_defaults_with, resolve_store_init_stage_result,
run_local_decommission_watchdog, save_validated_pool_meta_for_startup, should_auto_start_rebalance_after_init,
should_defer_rebalance_auto_start, should_retry_format_load, wait_for_local_decommission_resume_delay,
};
#[cfg(feature = "test-util")]
use crate::disk::DiskAPI;
@@ -1481,15 +1469,6 @@ mod tests {
assert!(err.to_string().contains("cannot overwrite an unreadable replica"));
assert!(!valid.wrote_without_lock.load(Ordering::SeqCst));
assert!(!unreadable.wrote_without_lock.load(Ordering::SeqCst));
assert!(!super::should_schedule_local_decommission_resume(&[0], replica_state, true));
assert!(!super::should_schedule_local_decommission_resume(
&[0],
crate::core::pools::PoolMetaReplicaState {
needs_repair: false,
repair_write_safe: true,
},
false,
));
}
#[tokio::test]
@@ -1517,7 +1496,6 @@ mod tests {
.contains("restart after all replicas are readable and consistent")
);
assert!(!repaired.wrote_without_lock.load(Ordering::SeqCst));
assert!(!super::should_schedule_local_decommission_resume(&[0], replica_state, false));
}
#[test]
@@ -1551,21 +1529,66 @@ mod tests {
}
#[test]
fn test_should_retry_local_decommission_resume_accepts_config_not_found_before_retry_limit() {
assert!(should_retry_local_decommission_resume(&StorageError::ConfigNotFound, 0));
fn test_local_decommission_watchdog_retry_delay_is_bounded() {
assert_eq!(local_decommission_watchdog_retry_delay(1), LOCAL_DECOMMISSION_RESUME_RETRY_DELAY);
assert_eq!(
local_decommission_watchdog_retry_delay(u32::MAX),
LOCAL_DECOMMISSION_WATCHDOG_MAX_RETRY_DELAY
);
}
#[test]
fn test_should_retry_local_decommission_resume_rejects_config_not_found_at_retry_limit() {
assert!(!should_retry_local_decommission_resume(
&StorageError::ConfigNotFound,
LOCAL_DECOMMISSION_RESUME_MAX_CONFIG_RETRIES
));
#[tokio::test(start_paused = true)]
async fn test_local_decommission_watchdog_retries_general_failures_until_cancelled() {
let rx = CancellationToken::new();
let attempts = Arc::new(AtomicUsize::new(0));
let task = tokio::spawn(run_local_decommission_watchdog(rx.clone(), {
let attempts = attempts.clone();
let rx = rx.clone();
move || {
let attempts = attempts.clone();
let rx = rx.clone();
async move {
if attempts.fetch_add(1, Ordering::SeqCst) == 0 {
Err(StorageError::SlowDown)
} else {
rx.cancel();
Ok(())
}
}
}
}));
tokio::task::yield_now().await;
assert_eq!(attempts.load(Ordering::SeqCst), 1);
tokio::time::advance(LOCAL_DECOMMISSION_RESUME_RETRY_DELAY).await;
task.await.expect("watchdog task should exit after cancellation");
assert_eq!(attempts.load(Ordering::SeqCst), 2);
}
#[test]
fn test_should_retry_local_decommission_resume_rejects_non_config_errors() {
assert!(!should_retry_local_decommission_resume(&StorageError::SlowDown, 0));
#[tokio::test(start_paused = true)]
async fn test_local_decommission_watchdog_rescans_after_success() {
let rx = CancellationToken::new();
let attempts = Arc::new(AtomicUsize::new(0));
let task = tokio::spawn(run_local_decommission_watchdog(rx.clone(), {
let attempts = attempts.clone();
let rx = rx.clone();
move || {
let attempts = attempts.clone();
let rx = rx.clone();
async move {
if attempts.fetch_add(1, Ordering::SeqCst) == 1 {
rx.cancel();
}
Ok(())
}
}
}));
tokio::task::yield_now().await;
assert_eq!(attempts.load(Ordering::SeqCst), 1);
tokio::time::advance(super::LOCAL_DECOMMISSION_WATCHDOG_INTERVAL).await;
task.await.expect("watchdog task should exit after cancellation");
assert_eq!(attempts.load(Ordering::SeqCst), 2);
}
#[test]
+852 -15
View File
@@ -44,7 +44,7 @@ use crate::error::{
use crate::runtime::global::DISK_RESERVE_FRACTION;
use crate::runtime::instance::InstanceContext;
use crate::runtime::sources as runtime_sources;
use crate::services::rebalance::{RebalanceMeta, is_rebalance_conflicting_with_decommission};
use crate::services::rebalance::{RebalStatus, RebalanceMeta, is_rebalance_conflicting_with_decommission};
use crate::storage_api_contracts::{
bucket::{BucketInfo, BucketOperations, BucketOptions, DeleteBucketOptions, MakeBucketOptions},
list::{StorageListObjectVersionsInfo, StorageListObjectsV2Info, StorageObjectInfoOrErr, StorageWalkOptions},
@@ -273,6 +273,215 @@ pub struct ECStore {
pub(crate) bucket_fence_registry: Arc<bucket_fence::BucketFenceRegistry>,
}
const METRIC_SCANNER_DATA_MOVEMENT_PAUSED: &str = "rustfs_scanner_data_movement_paused";
const METRIC_SCANNER_DATA_MOVEMENT_PAUSE_DURATION_SECONDS: &str = "rustfs_scanner_data_movement_pause_duration_seconds";
const METRIC_SCANNER_DATA_MOVEMENT_BACKLOG_WORK_ITEMS: &str = "rustfs_scanner_data_movement_backlog_work_items";
const SCANNER_DATA_MOVEMENT_PAUSE_POLICY: &str = "global_pause";
#[derive(Clone, Copy, Debug, PartialEq, Eq, serde::Serialize)]
#[serde(rename_all = "snake_case")]
pub enum ScannerDataMovementPauseReason {
OperationEpochExhausted,
MovementGenerationExhausted,
DecommissionActive,
DecommissionFailed,
DecommissionCanceled,
RebalanceActive,
}
#[derive(Clone, Debug, PartialEq, Eq, serde::Serialize)]
pub struct ScannerDataMovementPauseStatus {
pub paused: bool,
pub policy: &'static str,
pub reasons: Vec<ScannerDataMovementPauseReason>,
pub started_at_unix_secs: u64,
pub duration_seconds: u64,
pub operation_epoch: u64,
pub movement_generation: u64,
pub movement_backlog_work_items: u64,
pub movement_backlog_estimated: bool,
}
impl Default for ScannerDataMovementPauseStatus {
fn default() -> Self {
Self {
paused: false,
policy: SCANNER_DATA_MOVEMENT_PAUSE_POLICY,
reasons: Vec::new(),
started_at_unix_secs: 0,
duration_seconds: 0,
operation_epoch: 0,
movement_generation: 0,
movement_backlog_work_items: 0,
movement_backlog_estimated: false,
}
}
}
fn offset_unix_seconds(value: OffsetDateTime) -> u64 {
u64::try_from(value.unix_timestamp()).unwrap_or(0)
}
fn earliest_timestamp(current: Option<OffsetDateTime>, candidate: Option<OffsetDateTime>) -> Option<OffsetDateTime> {
match (current, candidate) {
(Some(current), Some(candidate)) => Some(current.min(candidate)),
(Some(current), None) => Some(current),
(None, candidate) => candidate,
}
}
fn usize_to_u64(value: usize) -> u64 {
u64::try_from(value).unwrap_or(u64::MAX)
}
fn metric_u64(value: u64) -> f64 {
f64::from(u32::try_from(value).unwrap_or(u32::MAX))
}
pub(crate) fn scanner_data_movement_timestamp_generation(value: OffsetDateTime) -> u64 {
let timestamp = value.unix_timestamp_nanos();
if timestamp <= 0 {
0
} else {
u64::try_from(timestamp).unwrap_or(u64::MAX)
}
}
fn valid_scanner_data_movement_timestamp_generation(value: OffsetDateTime) -> Option<u64> {
let generation = scanner_data_movement_timestamp_generation(value);
(generation != 0 && generation != u64::MAX).then_some(generation)
}
fn durable_scanner_data_movement_generation(pool_meta: &PoolMeta, rebalance_meta: Option<&RebalanceMeta>) -> u64 {
let mut generation = 0;
for pool in pool_meta.pools.iter().filter(|pool| pool.decommission.is_some()) {
let Some(pool_generation) = valid_scanner_data_movement_timestamp_generation(pool.last_update) else {
return u64::MAX;
};
generation = generation.max(pool_generation);
}
for movement_timestamp in rebalance_meta.into_iter().flat_map(|meta| {
meta.stopped_at.into_iter().chain(
meta.pool_stats
.iter()
.flat_map(|pool| [pool.info.start_time, pool.info.end_time])
.flatten(),
)
}) {
let Some(rebalance_generation) = valid_scanner_data_movement_timestamp_generation(movement_timestamp) else {
return u64::MAX;
};
generation = generation.max(rebalance_generation);
}
if generation == 0
&& rebalance_meta.is_some_and(|meta| !meta.id.is_empty() || !meta.pool_stats.is_empty() || meta.stopped_at.is_some())
{
u64::MAX
} else {
generation
}
}
#[derive(Clone, Copy)]
struct ScannerDataMovementSequenceState {
operation_epoch: u64,
operation_epoch_exhausted: bool,
movement_generation: u64,
movement_generation_exhausted: bool,
}
fn resolve_scanner_data_movement_pause_status(
pool_meta: &PoolMeta,
rebalance_meta: Option<&RebalanceMeta>,
decommission_worker_active: bool,
sequence: ScannerDataMovementSequenceState,
now: OffsetDateTime,
) -> ScannerDataMovementPauseStatus {
let mut decommission_active = decommission_worker_active;
let mut decommission_failed = false;
let mut decommission_canceled = false;
let mut rebalance_active = false;
let mut started_at = None;
let mut movement_backlog_work_items = 0_u64;
for pool in &pool_meta.pools {
let Some(info) = pool.decommission.as_ref() else {
continue;
};
let active = info.has_decommission_state() && !info.complete && !info.failed && !info.canceled;
let failed = !info.queued && info.failed;
let canceled = !info.queued && info.canceled;
if !(active || failed || canceled) {
continue;
}
decommission_active |= active;
decommission_failed |= failed;
decommission_canceled |= canceled;
started_at = earliest_timestamp(started_at, info.start_time.or(Some(pool.last_update)));
let queued = usize_to_u64(info.queued_buckets.len());
let current_bucket = if info.bucket.is_empty() { 0 } else { 1 };
movement_backlog_work_items = movement_backlog_work_items.saturating_add(queued.max(current_bucket));
}
if let Some(rebalance_meta) = rebalance_meta {
for pool in &rebalance_meta.pool_stats {
let active = (pool.participating && pool.info.status == RebalStatus::Started) || pool.info.stopping;
if !active {
continue;
}
rebalance_active = true;
started_at = earliest_timestamp(started_at, pool.info.start_time);
movement_backlog_work_items = movement_backlog_work_items.saturating_add(usize_to_u64(pool.buckets.len()));
}
}
let mut reasons = Vec::with_capacity(6);
if sequence.operation_epoch_exhausted {
reasons.push(ScannerDataMovementPauseReason::OperationEpochExhausted);
}
if sequence.movement_generation_exhausted {
reasons.push(ScannerDataMovementPauseReason::MovementGenerationExhausted);
}
if decommission_active {
reasons.push(ScannerDataMovementPauseReason::DecommissionActive);
}
if decommission_failed {
reasons.push(ScannerDataMovementPauseReason::DecommissionFailed);
}
if decommission_canceled {
reasons.push(ScannerDataMovementPauseReason::DecommissionCanceled);
}
if rebalance_active {
reasons.push(ScannerDataMovementPauseReason::RebalanceActive);
}
let started_at_unix_secs = started_at.map(offset_unix_seconds).unwrap_or(0);
let duration_seconds = started_at
.and_then(|started_at| u64::try_from((now - started_at).whole_seconds()).ok())
.unwrap_or(0);
let paused = !reasons.is_empty();
ScannerDataMovementPauseStatus {
paused,
policy: SCANNER_DATA_MOVEMENT_PAUSE_POLICY,
reasons,
started_at_unix_secs,
duration_seconds,
operation_epoch: sequence.operation_epoch,
movement_generation: sequence.movement_generation,
movement_backlog_work_items,
movement_backlog_estimated: paused,
}
}
fn record_scanner_data_movement_pause_status(status: &ScannerDataMovementPauseStatus) {
metrics::gauge!(METRIC_SCANNER_DATA_MOVEMENT_PAUSED).set(if status.paused { 1.0 } else { 0.0 });
metrics::gauge!(METRIC_SCANNER_DATA_MOVEMENT_PAUSE_DURATION_SECONDS).set(metric_u64(status.duration_seconds));
metrics::gauge!(METRIC_SCANNER_DATA_MOVEMENT_BACKLOG_WORK_ITEMS).set(metric_u64(status.movement_backlog_work_items));
}
impl std::fmt::Debug for ECStore {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
let disk_slot_count: usize = self.disk_map.values().map(Vec::len).sum();
@@ -300,6 +509,28 @@ impl ECStore {
self.pools.iter().flat_map(|pool| pool.disk_set.iter().cloned()).collect()
}
/// Erasure sets that may receive scanner pause-backlog replicas.
///
/// An actively decommissioning or already decommissioned source pool is
/// excluded so an operational record acknowledged during movement always
/// has a copy on storage that remains in the cluster. The record is kept
/// separate from pool and rebalance metadata.
pub async fn scanner_pause_backlog_writable_set_disks(&self) -> Vec<Arc<crate::set_disk::SetDisks>> {
let pool_meta = self.pool_meta.read().await;
self.pools
.iter()
.enumerate()
.filter(|(pool_index, _)| {
!pool_meta.pools.get(*pool_index).is_some_and(|pool| {
pool.decommission
.as_ref()
.is_some_and(|info| info.has_decommission_state() && !info.failed && !info.canceled)
})
})
.flat_map(|(_, pool)| pool.disk_set.iter().cloned())
.collect()
}
/// Get server configuration (delegates to global)
pub fn get_server_config(&self) -> Option<Config> {
runtime_sources::server_config()
@@ -454,14 +685,14 @@ impl ECStore {
self.scanner_data_usage_publication_snapshot_blocked().await
}
pub async fn scanner_data_movement_pause_status(&self) -> ScannerDataMovementPauseStatus {
let operation_gate = self.ctx.data_movement_operation_gate();
let _operation_guard = operation_gate.read_owned().await;
self.scanner_data_movement_pause_snapshot().await
}
async fn scanner_data_usage_publication_snapshot_blocked(&self) -> bool {
if self.ctx.data_movement_operation_epoch_exhausted() || self.ctx.data_movement_generation_exhausted() {
self.ctx.set_scanner_publication_state(true);
return true;
}
let (_, blocked) = self.scanner_data_movement_snapshot_locked().await;
self.ctx.set_scanner_publication_state(blocked);
blocked
self.scanner_data_movement_pause_snapshot().await.paused
}
async fn scanner_data_movement_snapshot_locked(&self) -> (bool, bool) {
@@ -481,19 +712,56 @@ impl ECStore {
.as_ref()
.is_some_and(|info| !info.queued && (info.failed || info.canceled))
});
drop(pool_meta);
let rebalance_active = self
.rebalance_meta
.read()
.await
let rebalance_meta = self.rebalance_meta.read().await;
let rebalance_active = rebalance_meta
.as_ref()
.is_some_and(is_rebalance_conflicting_with_decommission);
self.ctx
.observe_durable_data_movement_generation(durable_scanner_data_movement_generation(
&pool_meta,
rebalance_meta.as_ref(),
));
let blocked = decommission_active || decommission_terminal || rebalance_active;
(decommission_active || rebalance_active, blocked)
}
async fn scanner_data_movement_pause_snapshot(&self) -> ScannerDataMovementPauseStatus {
let decommission_active = {
let decommission_cancelers = self.decommission_cancelers.read().await;
decommission_cancelers
.iter()
.any(|canceler| canceler.as_ref().is_some_and(DecommissionCanceler::is_active))
};
let pool_meta = self.pool_meta.read().await.clone();
let rebalance_meta = self.rebalance_meta.read().await.clone();
self.ctx
.observe_durable_data_movement_generation(durable_scanner_data_movement_generation(
&pool_meta,
rebalance_meta.as_ref(),
));
let status = resolve_scanner_data_movement_pause_status(
&pool_meta,
rebalance_meta.as_ref(),
decommission_active,
ScannerDataMovementSequenceState {
operation_epoch: self.ctx.data_movement_operation_epoch(),
operation_epoch_exhausted: self.ctx.data_movement_operation_epoch_exhausted(),
movement_generation: self.ctx.data_movement_generation(),
movement_generation_exhausted: self.ctx.data_movement_generation_exhausted(),
},
OffsetDateTime::now_utc(),
);
self.ctx.set_scanner_publication_state(status.paused);
record_scanner_data_movement_pause_status(&status);
status
}
#[cfg(test)]
pub(crate) async fn scanner_data_movement_pause_snapshot_for_test(&self) -> ScannerDataMovementPauseStatus {
self.scanner_data_movement_pause_snapshot().await
}
/// Admit one short data-usage publication commit under the same
/// per-instance gate used by decommission side effects and transitions.
/// The epoch is sampled while the read guard is held, so a transition
@@ -1196,7 +1464,7 @@ impl crate::storage_api_contracts::admin::StorageAdminApi for ECStore {
#[cfg(test)]
mod tests {
use super::*;
use crate::core::pools::{PoolDecommissionInfo, PoolStatus};
use crate::core::pools::{PoolDecommissionInfo, PoolSpaceInfo, PoolStatus};
use crate::layout::endpoints::{Endpoints, PoolEndpoints, SetupType};
use crate::object_api::ObjectOptions;
use crate::runtime::global::reset_local_disk_test_state;
@@ -1205,6 +1473,23 @@ mod tests {
use serial_test::serial;
use tempfile::TempDir;
#[test]
fn g_d2_008_default_versioning_config_keeps_persisted_bytes() {
let bytes = crate::bucket::utils::serialize::<VersioningConfiguration>(&ENABLED_VERSIONING_CONFIG)
.expect("the default Versioning configuration must serialize");
assert_eq!(bytes, b"<VersioningConfiguration><Status>Enabled</Status></VersioningConfiguration>");
}
#[test]
fn g_d2_009_default_object_lock_config_keeps_persisted_bytes() {
let bytes = crate::bucket::utils::serialize::<ObjectLockConfiguration>(&ENABLED_OBJECT_LOCK_CONFIG)
.expect("the default Object Lock configuration must serialize");
assert_eq!(
bytes,
b"<ObjectLockConfiguration><ObjectLockEnabled>Enabled</ObjectLockEnabled></ObjectLockConfiguration>"
);
}
#[tokio::test]
async fn test_get_disk_infos() {
let disks = vec![None, None]; // Empty disks for testing
@@ -1309,6 +1594,558 @@ mod tests {
})
}
fn scanner_sequence_state(operation_epoch: u64, movement_generation: u64) -> ScannerDataMovementSequenceState {
ScannerDataMovementSequenceState {
operation_epoch,
operation_epoch_exhausted: false,
movement_generation,
movement_generation_exhausted: false,
}
}
#[test]
fn scanner_pause_status_derives_restart_stable_decommission_fields() {
let started_at = OffsetDateTime::from_unix_timestamp(1_000).expect("fixed timestamp should be valid");
let now = OffsetDateTime::from_unix_timestamp(1_090).expect("fixed timestamp should be valid");
let pool_meta = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: started_at,
decommission: Some(PoolDecommissionInfo {
start_time: Some(started_at),
queued_buckets: vec!["bucket-a".to_string(), "bucket-b".to_string()],
bucket: "bucket-a".to_string(),
..Default::default()
}),
}],
..Default::default()
};
let status = resolve_scanner_data_movement_pause_status(&pool_meta, None, false, scanner_sequence_state(7, 11), now);
assert!(status.paused);
assert_eq!(status.policy, "global_pause");
assert_eq!(status.reasons, vec![ScannerDataMovementPauseReason::DecommissionActive]);
assert_eq!(status.started_at_unix_secs, 1_000);
assert_eq!(status.duration_seconds, 90);
assert_eq!(status.operation_epoch, 7);
assert_eq!(status.movement_generation, 11);
assert_eq!(status.movement_backlog_work_items, 2);
assert!(status.movement_backlog_estimated);
}
#[test]
fn completed_decommission_restores_durable_movement_generation() {
let completed_at = OffsetDateTime::from_unix_timestamp(1_100).expect("fixed timestamp should be valid");
let pool_meta = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: completed_at,
decommission: Some(PoolDecommissionInfo {
complete: true,
..Default::default()
}),
}],
..Default::default()
};
let durable_generation = durable_scanner_data_movement_generation(&pool_meta, None);
let ctx = InstanceContext::new();
ctx.observe_durable_data_movement_generation(durable_generation);
assert_eq!(durable_generation, 1_100_000_000_000);
assert_eq!(ctx.data_movement_generation(), durable_generation);
}
#[tokio::test]
async fn cleared_decommission_restores_durable_movement_generation_after_restart() {
let mut pool_meta = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: OffsetDateTime::UNIX_EPOCH,
decommission: Some(PoolDecommissionInfo {
failed: true,
..Default::default()
}),
}],
..Default::default()
};
assert!(pool_meta.clear_decommission(0).expect("failed decommission should clear"));
assert!(
pool_meta.pools[0]
.decommission
.as_ref()
.is_some_and(|info| !info.has_decommission_state())
);
let durable_generation = durable_scanner_data_movement_generation(&pool_meta, None);
let restarted = build_store_with_ctx(Arc::new(InstanceContext::new()));
*restarted.pool_meta.write().await = pool_meta;
let status = restarted.scanner_data_movement_pause_status().await;
assert_ne!(durable_generation, 0);
assert!(!status.paused);
assert_eq!(status.movement_generation, durable_generation);
assert_eq!(restarted.scanner_data_movement_generation(), durable_generation);
}
#[tokio::test]
async fn same_tick_cleared_decommission_tombstones_advance_durable_movement_generation() {
let same_tick = OffsetDateTime::from_unix_timestamp(1_100).expect("fixed timestamp should be valid");
let mut pool_meta = PoolMeta {
pools: vec![
PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: same_tick,
decommission: Some(PoolDecommissionInfo {
failed: true,
..Default::default()
}),
},
PoolStatus {
id: 1,
cmd_line: "pool-1".to_string(),
last_update: same_tick,
decommission: Some(PoolDecommissionInfo {
canceled: true,
..Default::default()
}),
},
],
..Default::default()
};
assert!(
pool_meta
.clear_decommission_at_for_test(0, same_tick, None)
.expect("first terminal decommission should clear")
);
let first_generation = durable_scanner_data_movement_generation(&pool_meta, None);
assert_eq!(
first_generation,
scanner_data_movement_timestamp_generation(same_tick + time::Duration::nanoseconds(1))
);
assert!(
pool_meta
.clear_decommission_at_for_test(1, same_tick, None)
.expect("second terminal decommission should clear")
);
let second_generation = durable_scanner_data_movement_generation(&pool_meta, None);
assert_eq!(
second_generation,
scanner_data_movement_timestamp_generation(same_tick + time::Duration::nanoseconds(2))
);
assert!(second_generation > first_generation);
let restarted = build_store_with_ctx(Arc::new(InstanceContext::new()));
*restarted.pool_meta.write().await = pool_meta;
let status = restarted.scanner_data_movement_pause_status().await;
assert!(!status.paused);
assert_eq!(status.movement_generation, second_generation);
assert_eq!(restarted.scanner_data_movement_generation(), second_generation);
}
#[tokio::test]
async fn terminal_decommission_transitions_advance_durable_generation_across_same_or_earlier_clocks() {
let same_tick = OffsetDateTime::from_unix_timestamp(1_200).expect("fixed timestamp should be valid");
let earlier_tick = same_tick - time::Duration::nanoseconds(10);
let rebalance_floor = same_tick + time::Duration::nanoseconds(5);
let rebalance = RebalanceMeta {
stopped_at: Some(rebalance_floor),
id: "completed-rebalance".to_string(),
..Default::default()
};
let active_decommission = |id| PoolStatus {
id,
cmd_line: format!("pool-{id}"),
last_update: same_tick,
decommission: Some(PoolDecommissionInfo {
start_time: Some(same_tick),
..Default::default()
}),
};
let mut pool_meta = PoolMeta {
pools: vec![active_decommission(0), active_decommission(1), active_decommission(2)],
..Default::default()
};
assert!(pool_meta.decommission_complete_at_for_test(0, same_tick, Some(&rebalance)));
assert_eq!(pool_meta.pools[0].last_update, rebalance_floor + time::Duration::nanoseconds(1));
assert!(pool_meta.decommission_cancel_at_for_test(1, same_tick, Some(&rebalance)));
assert_eq!(pool_meta.pools[1].last_update, rebalance_floor + time::Duration::nanoseconds(2));
assert!(pool_meta.decommission_failed_at_for_test(2, earlier_tick, Some(&rebalance)));
assert_eq!(pool_meta.pools[2].last_update, rebalance_floor + time::Duration::nanoseconds(3));
let durable_generation = durable_scanner_data_movement_generation(&pool_meta, Some(&rebalance));
assert_eq!(
durable_generation,
scanner_data_movement_timestamp_generation(rebalance_floor + time::Duration::nanoseconds(3))
);
let restarted = build_store_with_ctx(Arc::new(InstanceContext::new()));
*restarted.pool_meta.write().await = pool_meta;
*restarted.rebalance_meta.write().await = Some(rebalance);
let status = restarted.scanner_data_movement_pause_status().await;
assert_eq!(status.movement_generation, durable_generation);
assert_eq!(restarted.scanner_data_movement_generation(), durable_generation);
assert_eq!(
status.reasons,
vec![
ScannerDataMovementPauseReason::DecommissionFailed,
ScannerDataMovementPauseReason::DecommissionCanceled
]
);
}
#[tokio::test]
async fn decommission_start_after_clear_advances_durable_generation_across_clock_rollback_after_restart() {
let same_tick = OffsetDateTime::from_unix_timestamp(1_250).expect("fixed timestamp should be valid");
let earlier_tick = same_tick - time::Duration::nanoseconds(10);
let rebalance_floor = same_tick + time::Duration::nanoseconds(5);
let rebalance = RebalanceMeta {
stopped_at: Some(rebalance_floor),
id: "completed-rebalance".to_string(),
..Default::default()
};
let mut pool_meta = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: same_tick,
decommission: Some(PoolDecommissionInfo {
failed: true,
..Default::default()
}),
}],
..Default::default()
};
assert!(
pool_meta
.clear_decommission_at_for_test(0, same_tick, Some(&rebalance))
.expect("failed decommission should clear")
);
let cleared_at = rebalance_floor + time::Duration::nanoseconds(1);
assert_eq!(pool_meta.pools[0].last_update, cleared_at);
pool_meta
.decommission_at_for_test(
0,
PoolSpaceInfo {
total: 200,
free: 50,
used: 150,
},
earlier_tick,
Some(&rebalance),
)
.expect("decommission restart after clear should be allowed");
let started_at = cleared_at + time::Duration::nanoseconds(1);
assert_eq!(pool_meta.pools[0].last_update, started_at);
assert_eq!(
pool_meta.pools[0].decommission.as_ref().and_then(|info| info.start_time),
Some(started_at)
);
assert!(pool_meta.decommission_complete_at_for_test(0, earlier_tick, Some(&rebalance)));
let completed_at = started_at + time::Duration::nanoseconds(1);
assert_eq!(pool_meta.pools[0].last_update, completed_at);
let durable_generation = durable_scanner_data_movement_generation(&pool_meta, Some(&rebalance));
assert_eq!(durable_generation, scanner_data_movement_timestamp_generation(completed_at));
let restarted = build_store_with_ctx(Arc::new(InstanceContext::new()));
*restarted.pool_meta.write().await = pool_meta;
*restarted.rebalance_meta.write().await = Some(rebalance);
let status = restarted.scanner_data_movement_pause_status().await;
assert!(!status.paused);
assert_eq!(status.movement_generation, durable_generation);
assert_eq!(restarted.scanner_data_movement_generation(), durable_generation);
}
#[tokio::test]
async fn decommission_terminal_reload_failure_advances_durable_generation_across_clock_rollback_after_restart() {
let terminal_at = OffsetDateTime::from_unix_timestamp(1_280).expect("fixed timestamp should be valid");
let earlier_tick = terminal_at - time::Duration::nanoseconds(10);
let rebalance_floor = terminal_at + time::Duration::nanoseconds(5);
let rebalance = RebalanceMeta {
stopped_at: Some(rebalance_floor),
id: "completed-rebalance".to_string(),
..Default::default()
};
let mut pool_meta = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: terminal_at,
decommission: Some(PoolDecommissionInfo {
start_time: Some(terminal_at),
complete: true,
..Default::default()
}),
}],
..Default::default()
};
assert!(
pool_meta
.record_decommission_terminal_reload_failure_at_for_test(
0,
"complete_decommission",
"peer reload failed".to_string(),
earlier_tick,
Some(&rebalance),
)
.expect("reload failure should be recorded")
);
let reload_failure_at = rebalance_floor + time::Duration::nanoseconds(1);
assert_eq!(pool_meta.pools[0].last_update, reload_failure_at);
let info = pool_meta.pools[0]
.decommission
.as_ref()
.expect("decommission metadata should exist");
assert_eq!(info.terminal_reload_attempt_at, Some(reload_failure_at));
assert_eq!(
info.terminal_reload_failures,
vec!["complete_decommission: peer reload failed".to_string()]
);
let durable_generation = durable_scanner_data_movement_generation(&pool_meta, Some(&rebalance));
assert_eq!(durable_generation, scanner_data_movement_timestamp_generation(reload_failure_at));
let restarted = build_store_with_ctx(Arc::new(InstanceContext::new()));
*restarted.pool_meta.write().await = pool_meta;
*restarted.rebalance_meta.write().await = Some(rebalance);
let status = restarted.scanner_data_movement_pause_status().await;
assert!(!status.paused);
assert_eq!(status.movement_generation, durable_generation);
assert_eq!(restarted.scanner_data_movement_generation(), durable_generation);
}
#[tokio::test]
async fn rebalance_transitions_advance_durable_generation_across_same_or_earlier_clocks_after_restart() {
let same_tick = OffsetDateTime::from_unix_timestamp(1_300).expect("fixed timestamp should be valid");
let earlier_tick = same_tick - time::Duration::nanoseconds(10);
let decommission_floor = same_tick + time::Duration::nanoseconds(5);
let store = build_store_with_ctx(Arc::new(InstanceContext::new()));
*store.pool_meta.write().await = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: decommission_floor,
decommission: Some(PoolDecommissionInfo {
complete: true,
..Default::default()
}),
}],
..Default::default()
};
let started_at = store.next_scanner_data_movement_update(same_tick).await;
assert_eq!(started_at, decommission_floor + time::Duration::nanoseconds(1));
*store.rebalance_meta.write().await = Some(RebalanceMeta {
id: "rebalance-generation".to_string(),
pool_stats: vec![crate::services::rebalance::RebalanceStats {
participating: true,
info: crate::services::rebalance::RebalanceInfo {
start_time: Some(started_at),
status: RebalStatus::Started,
..Default::default()
},
..Default::default()
}],
..Default::default()
});
let completed_at = store.next_scanner_data_movement_update(same_tick).await;
assert_eq!(completed_at, decommission_floor + time::Duration::nanoseconds(2));
{
let mut rebalance_meta = store.rebalance_meta.write().await;
let meta = rebalance_meta.as_mut().expect("rebalance metadata should be present");
meta.pool_stats[0].info.status = RebalStatus::Completed;
meta.pool_stats[0].info.end_time = Some(completed_at);
}
let stopped_at = store.next_scanner_data_movement_update(earlier_tick).await;
assert_eq!(stopped_at, decommission_floor + time::Duration::nanoseconds(3));
{
let mut rebalance_meta = store.rebalance_meta.write().await;
let meta = rebalance_meta.as_mut().expect("rebalance metadata should be present");
meta.stopped_at = Some(stopped_at);
}
let pool_meta = store.pool_meta.read().await.clone();
let rebalance_meta = store.rebalance_meta.read().await.clone();
let durable_generation = durable_scanner_data_movement_generation(&pool_meta, rebalance_meta.as_ref());
assert_eq!(
durable_generation,
scanner_data_movement_timestamp_generation(decommission_floor + time::Duration::nanoseconds(3))
);
let restarted = build_store_with_ctx(Arc::new(InstanceContext::new()));
*restarted.pool_meta.write().await = pool_meta;
*restarted.rebalance_meta.write().await = rebalance_meta;
let status = restarted.scanner_data_movement_pause_status().await;
assert!(!status.paused);
assert_eq!(status.movement_generation, durable_generation);
assert_eq!(restarted.scanner_data_movement_generation(), durable_generation);
}
#[test]
fn malformed_durable_movement_timestamp_exhausts_generation_fail_closed() {
let pool_meta = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: OffsetDateTime::UNIX_EPOCH,
decommission: Some(PoolDecommissionInfo {
complete: true,
..Default::default()
}),
}],
..Default::default()
};
assert_eq!(durable_scanner_data_movement_generation(&pool_meta, None), u64::MAX);
let exhausted_generation =
OffsetDateTime::from_unix_timestamp(253_402_300_799).expect("the largest RFC 3339 timestamp should be valid");
assert_eq!(scanner_data_movement_timestamp_generation(exhausted_generation), u64::MAX);
}
#[test]
fn malformed_durable_movement_timestamp_is_not_masked_by_valid_rebalance_generation() {
let valid_rebalance_at = OffsetDateTime::from_unix_timestamp(2_400).expect("fixed timestamp should be valid");
let pool_meta = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update: OffsetDateTime::UNIX_EPOCH,
decommission: Some(PoolDecommissionInfo {
complete: true,
..Default::default()
}),
}],
..Default::default()
};
let rebalance_meta = RebalanceMeta {
id: "completed-rebalance".to_string(),
stopped_at: Some(valid_rebalance_at),
pool_stats: vec![crate::services::rebalance::RebalanceStats {
participating: true,
info: crate::services::rebalance::RebalanceInfo {
start_time: Some(valid_rebalance_at - time::Duration::nanoseconds(1)),
end_time: Some(valid_rebalance_at),
status: RebalStatus::Completed,
..Default::default()
},
..Default::default()
}],
..Default::default()
};
assert_eq!(durable_scanner_data_movement_generation(&pool_meta, Some(&rebalance_meta)), u64::MAX);
}
#[test]
fn durable_movement_generation_without_records_is_zero() {
assert_eq!(durable_scanner_data_movement_generation(&PoolMeta::default(), None), 0);
assert_eq!(
durable_scanner_data_movement_generation(&PoolMeta::default(), Some(&RebalanceMeta::default())),
0
);
}
#[test]
fn scanner_pause_status_distinguishes_terminal_rebalance_epoch_and_idle() {
let last_update = OffsetDateTime::from_unix_timestamp(2_000).expect("fixed timestamp should be valid");
let now = OffsetDateTime::from_unix_timestamp(2_030).expect("fixed timestamp should be valid");
let failed = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: "pool-0".to_string(),
last_update,
decommission: Some(PoolDecommissionInfo {
failed: true,
..Default::default()
}),
}],
..Default::default()
};
let failed_status = resolve_scanner_data_movement_pause_status(&failed, None, false, scanner_sequence_state(3, 12), now);
assert_eq!(failed_status.reasons, vec![ScannerDataMovementPauseReason::DecommissionFailed]);
assert_eq!(failed_status.started_at_unix_secs, 2_000);
assert_eq!(failed_status.duration_seconds, 30);
let rebalance = RebalanceMeta {
pool_stats: vec![crate::services::rebalance::RebalanceStats {
buckets: vec!["bucket-a".to_string(), "bucket-b".to_string()],
participating: true,
info: crate::services::rebalance::RebalanceInfo {
start_time: Some(last_update),
status: RebalStatus::Started,
..Default::default()
},
..Default::default()
}],
..Default::default()
};
let rebalance_status = resolve_scanner_data_movement_pause_status(
&PoolMeta::default(),
Some(&rebalance),
false,
scanner_sequence_state(4, 13),
now,
);
assert_eq!(rebalance_status.reasons, vec![ScannerDataMovementPauseReason::RebalanceActive]);
assert_eq!(rebalance_status.movement_backlog_work_items, 2);
let exhausted = resolve_scanner_data_movement_pause_status(
&PoolMeta::default(),
None,
false,
ScannerDataMovementSequenceState {
operation_epoch: u64::MAX,
operation_epoch_exhausted: true,
movement_generation: 14,
movement_generation_exhausted: false,
},
now,
);
assert_eq!(exhausted.reasons, vec![ScannerDataMovementPauseReason::OperationEpochExhausted]);
assert_eq!(exhausted.started_at_unix_secs, 0);
let generation_exhausted = resolve_scanner_data_movement_pause_status(
&PoolMeta::default(),
None,
false,
ScannerDataMovementSequenceState {
operation_epoch: 5,
operation_epoch_exhausted: false,
movement_generation: u64::MAX,
movement_generation_exhausted: true,
},
now,
);
assert_eq!(
generation_exhausted.reasons,
vec![ScannerDataMovementPauseReason::MovementGenerationExhausted]
);
let idle =
resolve_scanner_data_movement_pause_status(&PoolMeta::default(), None, false, scanner_sequence_state(5, 15), now);
assert!(!idle.paused);
assert!(idle.reasons.is_empty());
assert!(!idle.movement_backlog_estimated);
}
#[tokio::test]
async fn scanner_data_usage_publication_blocks_active_and_unqueued_terminal_decommission() {
let store = build_store_with_ctx(Arc::new(InstanceContext::new()));
+443 -96
View File
@@ -884,8 +884,7 @@ impl AsyncRead for SelectObjectSnapshotReader {
}
let filled_before = buf.filled().len();
let poll = Pin::new(&mut self.inner).poll_read(cx, buf);
let reached_eof = matches!(&poll, Poll::Ready(Ok(()))) && buf.filled().len() == filled_before;
if self.lease.is_lost() || (reached_eof && self.lease.check().is_err()) {
if self.lease.check().is_err() {
buf.set_filled(filled_before);
return Poll::Ready(Err(std::io::Error::other(SnapshotConsistencyError::LockLost)));
}
@@ -1757,15 +1756,22 @@ impl ECStore {
return Err(SnapshotConsistencyError::LockLost.into());
}
let pool = if self.single_pool() {
Arc::clone(&self.pools[0])
let (mut metadata, pool) = if self.single_pool() {
let pool = Arc::clone(&self.pools[0]);
let metadata = pool.prepare_get_object_reader_metadata(bucket, &object, &opts).await?;
(metadata, pool)
} else {
let (_, pool_idx) = self.get_latest_object_info_with_idx(bucket, &object, &opts).await?;
self.pools.get(pool_idx).cloned().ok_or_else(|| {
StorageError::other(format!("resolved SelectObjectContent pool index {pool_idx} is out of bounds"))
})?
// Keep the large multi-pool selection future off the caller stack.
// Debug builds otherwise exceed the common 2 MiB worker stack.
Box::pin(async {
let (metadata, pool_idx) = self.prepare_latest_object_metadata_with_idx(bucket, &object, &opts).await?;
let pool = self.pools.get(pool_idx).cloned().ok_or_else(|| {
StorageError::other(format!("resolved SelectObjectContent pool index {pool_idx} is out of bounds"))
})?;
Ok::<_, StorageError>((metadata, pool))
})
.await?
};
let mut metadata = pool.prepare_get_object_reader_metadata(bucket, &object, &opts).await?;
if read_lock_guards.iter().any(ObjectLockDiagGuard::is_lock_lost) {
return Err(SnapshotConsistencyError::LockLost.into());
}
@@ -1817,16 +1823,21 @@ impl ECStore {
let metadata = pool.prepare_get_object_reader_metadata(bucket, &object, &opts).await?;
(metadata, pool)
} else {
let (_, pool_idx) = self
.get_latest_accessible_object_info_with_idx(bucket, &object, &opts)
.await?;
let pool = self
.pools
.get(pool_idx)
.cloned()
.ok_or_else(|| Error::other(format!("resolved GET pool index {pool_idx} is out of bounds")))?;
let metadata = pool.prepare_get_object_reader_metadata(bucket, &object, &opts).await?;
(metadata, pool)
// Keep the large multi-pool selection future off the caller stack.
// Debug builds otherwise exceed the common 2 MiB worker stack.
Box::pin(async {
let (metadata, pool_idx) = self.prepare_latest_object_metadata_with_idx(bucket, &object, &opts).await?;
if let Some(error) = latest_object_access_delete_marker_error(bucket, &object, metadata.object_info(), &opts) {
return Err(error);
}
let pool = self
.pools
.get(pool_idx)
.cloned()
.ok_or_else(|| Error::other(format!("resolved GET pool index {pool_idx} is out of bounds")))?;
Ok((metadata, pool))
})
.await?
};
Ok(PreparedGetObjectReader {
@@ -2518,12 +2529,18 @@ impl ECStore {
.get_object_reader(bucket, object.as_ref(), range, h, &opts)
.await?
} else {
let (_, idx) = self
.get_latest_accessible_object_info_with_idx(bucket, &object, &opts)
.await?;
self.pools[idx]
.get_object_reader(bucket, object.as_ref(), range, h, &opts)
.await?
// Keep selection plus prepared-open state off the caller stack.
// Debug builds otherwise exceed the common 2 MiB worker stack.
Box::pin(async {
let (metadata, idx) = self.prepare_latest_object_metadata_with_idx(bucket, &object, &opts).await?;
if let Some(error) = latest_object_access_delete_marker_error(bucket, &object, metadata.object_info(), &opts) {
return Err(error);
}
self.pools[idx]
.get_object_reader_with_prepared_metadata(bucket, object.as_ref(), range, h, &opts, metadata)
.await
})
.await?
};
Ok(Self::attach_read_lock_guard(reader, read_lock_guard))
@@ -3914,8 +3931,9 @@ mod tests {
ReplicationState, ReplicationStatusType, VersionPurgeStatusType, replication_state_to_filemeta, replication_statuses_map,
version_purge_statuses_map,
};
use crate::core::pools::{PoolDecommissionInfo, PoolStatus};
use crate::core::sets::make_local_two_set_sets_with_ctx;
use crate::ecstore_validation_blackbox::{make_local_set_disks, make_local_set_disks_with_ctx};
use crate::ecstore_validation_blackbox::{RefreshLossLockClient, make_local_set_disks, make_local_set_disks_with_ctx};
use crate::layout::{
endpoints::{Endpoints, PoolEndpoints, SetupType},
format::FormatV3,
@@ -3930,7 +3948,7 @@ mod tests {
use bytes::Bytes;
use std::io::Cursor;
use std::sync::Arc;
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
use std::sync::atomic::{AtomicUsize, Ordering};
use tokio::io::AsyncReadExt;
struct WaitForLockLossReader {
@@ -3972,68 +3990,17 @@ mod tests {
calls: AtomicUsize,
}
#[derive(Debug)]
struct RefreshFailureLockClient {
inner: LocalClient,
fail_refresh: AtomicBool,
}
#[async_trait::async_trait]
impl rustfs_lock::LockClient for RefreshFailureLockClient {
async fn acquire_lock(&self, request: &rustfs_lock::LockRequest) -> rustfs_lock::Result<rustfs_lock::LockResponse> {
rustfs_lock::LockClient::acquire_lock(&self.inner, request).await
}
async fn release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
rustfs_lock::LockClient::release(&self.inner, lock_id).await
}
async fn refresh(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
if self.fail_refresh.load(Ordering::Acquire) {
return Ok(false);
}
rustfs_lock::LockClient::refresh(&self.inner, lock_id).await
}
async fn force_release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
rustfs_lock::LockClient::force_release(&self.inner, lock_id).await
}
async fn check_status(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<Option<rustfs_lock::LockInfo>> {
rustfs_lock::LockClient::check_status(&self.inner, lock_id).await
}
async fn get_stats(&self) -> rustfs_lock::Result<rustfs_lock::LockStats> {
rustfs_lock::LockClient::get_stats(&self.inner).await
}
async fn close(&self) -> rustfs_lock::Result<()> {
rustfs_lock::LockClient::close(&self.inner).await
}
async fn is_online(&self) -> bool {
rustfs_lock::LockClient::is_online(&self.inner).await
}
async fn is_local(&self) -> bool {
rustfs_lock::LockClient::is_local(&self.inner).await
}
}
async fn refresh_failure_test_guard(
owner: &'static str,
) -> (
ObjectLockDiagGuard,
Arc<rustfs_lock::distributed_lock::LockLostSignal>,
Arc<RefreshFailureLockClient>,
Arc<RefreshLossLockClient>,
) {
let manager = Arc::new(rustfs_lock::GlobalLockManager::Enabled(Arc::new(
rustfs_lock::FastObjectLockManager::new(),
)));
let client = Arc::new(RefreshFailureLockClient {
inner: LocalClient::with_manager(manager),
fail_refresh: AtomicBool::new(false),
});
let client = Arc::new(RefreshLossLockClient::with_manager(manager));
let namespace_lock = rustfs_lock::NamespaceLock::with_clients_and_quorum(
owner.to_string(),
vec![Arc::clone(&client) as Arc<dyn rustfs_lock::LockClient>],
@@ -4068,7 +4035,7 @@ mod tests {
) -> (
Arc<SelectObjectSnapshotLease>,
Arc<rustfs_lock::distributed_lock::LockLostSignal>,
Arc<RefreshFailureLockClient>,
Arc<RefreshLossLockClient>,
) {
let (guard, signal, client) = refresh_failure_test_guard(owner).await;
(Arc::new(SelectObjectSnapshotLease::new(vec![guard])), signal, client)
@@ -4201,7 +4168,11 @@ mod tests {
let release_signal = Arc::clone(&signal);
let release_task = tokio::spawn(async move {
poll_started_rx.await.expect("reader poll should start");
release_client.fail_refresh.store(true, Ordering::Release);
release_client.reject_refreshes();
release_client
.wait_for_rejected_refresh(Duration::from_secs(5))
.await
.expect("refresh rejection should be observed");
tokio::time::timeout(Duration::from_secs(5), release_signal.notified())
.await
.expect("heartbeat should observe the rejected refresh");
@@ -4237,9 +4208,13 @@ mod tests {
}
#[tokio::test(flavor = "current_thread")]
async fn select_snapshot_reader_checks_guards_at_eof_before_monitor_runs() {
async fn select_snapshot_reader_checks_guards_before_monitor_runs() {
let (guard, signal, client) = refresh_failure_test_guard("select-snapshot-eof-fence").await;
client.fail_refresh.store(true, Ordering::Release);
client.reject_refreshes();
client
.wait_for_rejected_refresh(Duration::from_secs(5))
.await
.expect("refresh rejection should be observed");
tokio::time::timeout(Duration::from_secs(5), signal.notified())
.await
.expect("heartbeat should observe the rejected refresh");
@@ -4258,7 +4233,7 @@ mod tests {
.await
.expect_err("EOF fence must reject a lease lost before its monitor is scheduled");
assert_eq!(output, b"old-generation");
assert!(output.is_empty(), "bytes from a known-lost snapshot must not escape");
assert_eq!(error.kind(), std::io::ErrorKind::Other);
}
@@ -4304,7 +4279,11 @@ mod tests {
second_started_rx
.await
.expect("second inner reader should reach Poll::Pending");
second_client.fail_refresh.store(true, Ordering::Release);
second_client.reject_refreshes();
second_client
.wait_for_rejected_refresh(Duration::from_secs(5))
.await
.expect("refresh rejection should be observed");
let (first_result, second_result) = tokio::join!(
tokio::time::timeout(Duration::from_secs(5), first_read_task),
tokio::time::timeout(Duration::from_secs(5), second_read_task),
@@ -4317,7 +4296,7 @@ mod tests {
assert_eq!(error.kind(), std::io::ErrorKind::Other);
}
assert!(!first_client.fail_refresh.load(Ordering::Acquire));
assert!(!first_client.refreshes_rejected());
assert!(!first_signal.is_lost());
assert!(second_signal.is_lost());
}
@@ -6085,13 +6064,112 @@ mod tests {
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn prepared_reader_resolves_object_from_second_pool() {
async fn prepared_reader_reuses_metadata_across_three_pools() {
let (_first_dirs, first_set) = make_local_set_disks(4, 2).await;
let (_second_dirs, second_set) = make_local_set_disks(4, 2).await;
let (_third_dirs, third_set) = make_local_set_disks(4, 2).await;
let store = new_prepared_reader_test_store(&[first_set, second_set, third_set]).await;
let bucket = "prepared-reader-three-pools";
let object = "object.bin";
let payload = b"prepared-reader-three-pool-payload-".repeat(40_000);
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
for pool in &store.pools {
pool.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created in each pool");
}
let mut put_reader = PutObjReader::from_vec(payload.clone());
store.pools[2]
.put_object(bucket, object, &mut put_reader, &opts)
.await
.expect("object should be written only to the third pool");
let calls = disk_call_counters::observe(object);
let prepared = store
.prepare_get_object_reader(bucket, object, None, HeaderMap::new(), &opts)
.await
.expect("prepared reader should resolve the third-pool object");
assert_eq!(prepared.object_info().size, payload.len() as i64);
let metadata_calls = calls.total(disk_call_counters::KIND_READ_VERSION);
assert_eq!(metadata_calls, 12, "three 4-disk pools must fan out metadata exactly once each");
let mut reader = prepared.into_reader().await.expect("prepared body reader should open");
assert_eq!(
calls.total(disk_call_counters::KIND_READ_VERSION),
metadata_calls,
"the selected pool must reuse its prepared metadata"
);
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("prepared body should stream");
assert_eq!(restored, payload);
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn select_snapshot_reuses_metadata_across_three_pools() {
let (_first_dirs, first_set) = make_local_set_disks(4, 2).await;
let (_second_dirs, second_set) = make_local_set_disks(4, 2).await;
let (_third_dirs, third_set) = make_local_set_disks(4, 2).await;
let store = new_prepared_reader_test_store(&[first_set, second_set, third_set]).await;
let bucket = "select-snapshot-three-pools";
let object = "object.bin";
let payload = b"select-snapshot-three-pool-payload-".repeat(40_000);
let write_opts = ObjectOptions {
no_lock: true,
..Default::default()
};
for pool in &store.pools {
pool.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created in each pool");
}
let mut put_reader = PutObjReader::from_vec(payload.clone());
store.pools[2]
.put_object(bucket, object, &mut put_reader, &write_opts)
.await
.expect("object should be written only to the third pool");
let calls = disk_call_counters::observe(object);
let snapshot = store
.prepare_select_object_snapshot(bucket, object, &HeaderMap::new(), &ObjectOptions::default())
.await
.expect("SelectObjectContent snapshot should resolve the third-pool object");
assert_eq!(snapshot.object_info().size, payload.len() as i64);
let metadata_calls = calls.total(disk_call_counters::KIND_READ_VERSION);
assert_eq!(metadata_calls, 12, "three 4-disk pools must fan out metadata exactly once each");
let mut reader = snapshot.open_reader(None).await.expect("snapshot body reader should open");
assert_eq!(
calls.total(disk_call_counters::KIND_READ_VERSION),
metadata_calls,
"SelectObjectContent must consume the prepared winner without a second fanout"
);
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("snapshot body should stream");
assert_eq!(restored, payload);
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn legacy_reader_reuses_selected_pool_metadata() {
let (_first_dirs, first_set) = make_local_set_disks(4, 2).await;
let (_second_dirs, second_set) = make_local_set_disks(4, 2).await;
let store = new_prepared_reader_test_store(&[first_set, second_set]).await;
let bucket = "prepared-reader-second-pool";
let bucket = "legacy-reader-second-pool";
let object = "object.bin";
let payload = b"prepared-reader-second-pool-payload-".repeat(40_000);
let payload = b"legacy-reader-second-pool-payload-".repeat(40_000);
let opts = ObjectOptions {
no_lock: true,
..Default::default()
@@ -6108,18 +6186,287 @@ mod tests {
.await
.expect("object should be written only to the second pool");
let prepared = store
.prepare_get_object_reader(bucket, object, None, HeaderMap::new(), &opts)
clear_get_object_body_cache_hook();
let hook = Arc::new(CountingMissHook {
calls: AtomicUsize::new(0),
});
register_get_object_body_cache_hook(Arc::clone(&hook) as Arc<dyn GetObjectBodyCacheHook>);
let _hook_guard = BodyCacheHookGuard;
let calls = disk_call_counters::observe(object);
let mut reader = store
.handle_get_object_reader(bucket, object, None, HeaderMap::new(), &opts)
.await
.expect("prepared reader should resolve the second-pool object");
assert_eq!(prepared.object_info().size, payload.len() as i64);
let mut reader = prepared.into_reader().await.expect("prepared body reader should open");
.expect("legacy reader should resolve the second-pool object");
assert_eq!(
hook.calls.load(Ordering::Relaxed),
1,
"legacy reader must probe the body cache exactly once"
);
assert_eq!(reader.body_source, GetObjectBodySource::HookMissed);
assert!(
calls.total(disk_call_counters::KIND_READ_VERSION) <= 8,
"legacy reader must fan out each 4-disk pool at most once"
);
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("prepared body should stream");
.expect("legacy reader body should stream");
assert_eq!(restored, payload);
}
fn prepared_pool_test_status(id: usize, suspended: bool) -> PoolStatus {
PoolStatus {
id,
cmd_line: format!("prepared-pool-{id}"),
last_update: OffsetDateTime::now_utc(),
decommission: suspended.then(|| PoolDecommissionInfo {
start_time: Some(OffsetDateTime::now_utc()),
..Default::default()
}),
}
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn prepared_reader_refetches_when_final_pool_state_changes_winner() {
let (_dirs, set_disks) = make_local_set_disks(4, 2).await;
let store = Arc::new(new_prepared_reader_test_store(&[Arc::clone(&set_disks), Arc::clone(&set_disks)]).await);
let bucket = "prepared-reader-pool-state-fallback";
let object = "object.bin";
let payload = b"pool-state fallback payload".repeat(8_000);
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut put_reader = PutObjReader::from_vec(payload.clone());
set_disks
.put_object(bucket, object, &mut put_reader, &opts)
.await
.expect("shared object should be written");
let calls = disk_call_counters::observe(object);
let barrier = crate::store::rebalance::PreparedPoolReadFallbackBarrier::install(object, false);
let read_store = Arc::clone(&store);
let read_opts = opts.clone();
let read = tokio::spawn(async move {
read_store
.prepare_get_object_reader(bucket, object, None, HeaderMap::new(), &read_opts)
.await
});
barrier.wait_after_fanout().await;
*store.pool_meta.write().await = PoolMeta {
pools: vec![prepared_pool_test_status(0, false), prepared_pool_test_status(1, true)],
..Default::default()
};
barrier.release_after_fanout();
let prepared = read
.await
.expect("prepared read task should not panic")
.expect("final active pool should be refetched");
assert!(Arc::ptr_eq(&prepared.pool, &store.pools[0]));
assert_eq!(
calls.total(disk_call_counters::KIND_READ_VERSION),
12,
"two initial 4-disk fanouts plus one fallback refetch are required"
);
let mut reader = prepared.into_reader().await.expect("fallback body reader should open");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("fallback body should stream");
assert_eq!(restored, payload);
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn prepared_reader_fallback_rejects_generation_change_before_refetch() {
let (_dirs, set_disks) = make_local_set_disks(4, 2).await;
let store = Arc::new(new_prepared_reader_test_store(&[Arc::clone(&set_disks), Arc::clone(&set_disks)]).await);
let bucket = "prepared-reader-pool-state-generation-change";
let object = "object.bin";
let opts = ObjectOptions {
no_lock: true,
..Default::default()
};
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
let mut initial_reader = PutObjReader::from_vec(b"initial generation".to_vec());
set_disks
.put_object(bucket, object, &mut initial_reader, &opts)
.await
.expect("initial object should be written");
let barrier = crate::store::rebalance::PreparedPoolReadFallbackBarrier::install(object, true);
let read_store = Arc::clone(&store);
let read_opts = opts.clone();
let read = tokio::spawn(async move {
read_store
.prepare_get_object_reader(bucket, object, None, HeaderMap::new(), &read_opts)
.await
});
barrier.wait_after_fanout().await;
*store.pool_meta.write().await = PoolMeta {
pools: vec![prepared_pool_test_status(0, false), prepared_pool_test_status(1, true)],
..Default::default()
};
barrier.release_after_fanout();
barrier.wait_before_refetch().await;
let mut replacement_reader = PutObjReader::from_vec(b"replacement generation".to_vec());
set_disks
.put_object(bucket, object, &mut replacement_reader, &opts)
.await
.expect("replacement generation should be written before fallback refetch");
barrier.release_before_refetch();
let error = match read.await.expect("prepared read task should not panic") {
Ok(_) => panic!("changed fallback generation must not be accepted"),
Err(error) => error,
};
assert_eq!(error, Error::ErasureReadQuorum);
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn prepared_reader_rejects_latest_delete_marker_without_refetching_metadata() {
let ctx = Arc::new(crate::runtime::instance::InstanceContext::new());
let (_first_dirs, first_set) = make_local_set_disks_with_ctx(4, 2, Arc::clone(&ctx)).await;
let (_second_dirs, second_set) = make_local_set_disks_with_ctx(4, 2, Arc::clone(&ctx)).await;
let store = new_prepared_reader_test_store_with_ctx(&[Arc::clone(&first_set), Arc::clone(&second_set)], ctx).await;
let bucket = "prepared-reader-latest-delete-marker";
let object = "versioned-object.bin";
let versioned_opts = ObjectOptions {
no_lock: true,
versioned: true,
object_lock_config_snapshot: Some(Arc::new(ObjectLockConfigSnapshot::new(ObjectLockConfigState::ConfirmedAbsent))),
..Default::default()
};
for set_disks in [&first_set, &second_set] {
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
}
let mut older_reader = PutObjReader::from_vec(b"older visible generation".to_vec());
first_set
.put_object(bucket, object, &mut older_reader, &versioned_opts)
.await
.expect("older object should be written");
let mut hidden_reader = PutObjReader::from_vec(b"hidden generation".to_vec());
second_set
.put_object(bucket, object, &mut hidden_reader, &versioned_opts)
.await
.expect("newer object should be written");
let marker = second_set
.delete_object(bucket, object, versioned_opts.clone())
.await
.expect("delete marker should be committed");
assert!(marker.delete_marker);
let calls = disk_call_counters::observe(object);
let error = match store
.prepare_get_object_reader(
bucket,
object,
None,
HeaderMap::new(),
&ObjectOptions {
no_lock: true,
versioned: true,
..Default::default()
},
)
.await
{
Ok(_) => panic!("latest delete marker should hide the older live object"),
Err(error) => error,
};
assert!(is_err_object_not_found(&error));
assert!(
calls.total(disk_call_counters::KIND_READ_VERSION) <= 8,
"delete-marker resolution must fan out each pool at most once"
);
}
#[tokio::test]
#[serial_test::serial(body_cache_hook)]
async fn prepared_reader_explicit_version_reuses_the_matching_pool_metadata() {
let (_first_dirs, first_set) = make_local_set_disks(4, 2).await;
let (_second_dirs, second_set) = make_local_set_disks(4, 2).await;
let store = new_prepared_reader_test_store(&[Arc::clone(&first_set), Arc::clone(&second_set)]).await;
let bucket = "prepared-reader-explicit-version";
let object = "versioned-object.bin";
let payload = b"explicit version from first pool".repeat(8_000);
let versioned_opts = ObjectOptions {
no_lock: true,
versioned: true,
object_lock_config_snapshot: Some(Arc::new(ObjectLockConfigSnapshot::new(ObjectLockConfigState::ConfirmedAbsent))),
..Default::default()
};
for set_disks in [&first_set, &second_set] {
set_disks
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created");
}
let mut first_reader = PutObjReader::from_vec(payload.clone());
let first = first_set
.put_object(bucket, object, &mut first_reader, &versioned_opts)
.await
.expect("requested version should be written to the first pool");
let mut second_reader = PutObjReader::from_vec(b"different pool version".to_vec());
second_set
.put_object(bucket, object, &mut second_reader, &versioned_opts)
.await
.expect("a different version should be written to the second pool");
let requested_version = first
.version_id
.expect("versioned PUT should return a version id")
.to_string();
let read_opts = ObjectOptions {
no_lock: true,
versioned: true,
version_id: Some(requested_version),
..Default::default()
};
let calls = disk_call_counters::observe(object);
let prepared = store
.prepare_get_object_reader(bucket, object, None, HeaderMap::new(), &read_opts)
.await
.expect("explicit version should resolve from the matching pool");
assert_eq!(prepared.object_info().version_id, first.version_id);
let metadata_calls = calls.total(disk_call_counters::KIND_READ_VERSION);
assert!(metadata_calls <= 8, "explicit-version lookup must fan out each pool at most once");
let mut reader = prepared
.into_reader()
.await
.expect("prepared explicit-version body should open");
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), metadata_calls);
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("explicit-version body should stream");
assert_eq!(restored, payload);
}
+228 -1
View File
@@ -18,18 +18,117 @@ use crate::core::pools::merge_pool_status_refresh;
use crate::layout::pool_space::{ServerPoolsAvailableSpace, build_server_pools_available_space};
use crate::runtime::sources as runtime_sources;
use crate::storage_api_contracts::{admin::StorageAdminApi, namespace::NamespaceLocking as _, object::ObjectOperations as _};
use futures::stream::{FuturesUnordered, StreamExt};
pub(in crate::store) mod support;
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
const LOG_SUBSYSTEM_POOLS: &str = "pools";
const EVENT_POOL_META_RELOAD: &str = "pool_meta_reload";
#[cfg(test)]
struct PreparedPoolReadFallbackBarrierState {
object: String,
pause_before_refetch: bool,
fanout_arrived: tokio::sync::Notify,
fanout_release: tokio::sync::Notify,
refetch_arrived: tokio::sync::Notify,
refetch_release: tokio::sync::Notify,
}
#[cfg(test)]
pub(in crate::store) struct PreparedPoolReadFallbackBarrier {
state: Arc<PreparedPoolReadFallbackBarrierState>,
}
#[cfg(test)]
static PREPARED_POOL_READ_FALLBACK_BARRIER: std::sync::OnceLock<
std::sync::Mutex<Option<Arc<PreparedPoolReadFallbackBarrierState>>>,
> = std::sync::OnceLock::new();
#[cfg(test)]
impl PreparedPoolReadFallbackBarrier {
pub(in crate::store) fn install(object: &str, pause_before_refetch: bool) -> Self {
let state = Arc::new(PreparedPoolReadFallbackBarrierState {
object: object.to_string(),
pause_before_refetch,
fanout_arrived: tokio::sync::Notify::new(),
fanout_release: tokio::sync::Notify::new(),
refetch_arrived: tokio::sync::Notify::new(),
refetch_release: tokio::sync::Notify::new(),
});
*PREPARED_POOL_READ_FALLBACK_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("prepared pool read fallback barrier must not be poisoned") = Some(Arc::clone(&state));
Self { state }
}
pub(in crate::store) async fn wait_after_fanout(&self) {
self.state.fanout_arrived.notified().await;
}
pub(in crate::store) fn release_after_fanout(&self) {
self.state.fanout_release.notify_one();
}
pub(in crate::store) async fn wait_before_refetch(&self) {
self.state.refetch_arrived.notified().await;
}
pub(in crate::store) fn release_before_refetch(&self) {
self.state.refetch_release.notify_one();
}
}
#[cfg(test)]
impl Drop for PreparedPoolReadFallbackBarrier {
fn drop(&mut self) {
self.state.fanout_release.notify_waiters();
self.state.refetch_release.notify_waiters();
if let Some(barrier) = PREPARED_POOL_READ_FALLBACK_BARRIER.get() {
*barrier
.lock()
.expect("prepared pool read fallback barrier must not be poisoned") = None;
}
}
}
#[cfg(test)]
async fn pause_prepared_pool_read_after_fanout(object: &str) {
let state = PREPARED_POOL_READ_FALLBACK_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("prepared pool read fallback barrier must not be poisoned")
.as_ref()
.filter(|state| state.object == object)
.cloned();
if let Some(state) = state {
state.fanout_arrived.notify_one();
state.fanout_release.notified().await;
}
}
#[cfg(test)]
async fn pause_prepared_pool_read_before_refetch(object: &str) {
let state = PREPARED_POOL_READ_FALLBACK_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("prepared pool read fallback barrier must not be poisoned")
.as_ref()
.filter(|state| state.object == object && state.pause_before_refetch)
.cloned();
if let Some(state) = state {
state.refetch_arrived.notify_one();
state.refetch_release.notified().await;
}
}
#[cfg(test)]
use support::resolve_latest_object_info_candidates;
use support::{
LatestObjectInfoCandidate, PoolErr, PoolObjInfo, RebalanceDeletePoolResult, pool_lookup_not_found_error,
rebalance_disk_set_lookup_error, resolve_latest_object_info_candidates_with_pool_state,
resolve_rebalance_delete_from_all_pools_result, resolve_rebalance_delete_from_all_pools_results,
resolve_store_rebalance_pool_meta_reload_result,
resolve_store_rebalance_pool_meta_reload_result, validate_prepared_pool_refetch_identity,
};
#[derive(Debug, Default, Eq, PartialEq)]
@@ -675,6 +774,134 @@ impl ECStore {
resolve_latest_object_info_candidates_with_pool_state(candidates, &suspended_pools, bucket, object, opts)
}
pub(super) async fn prepare_latest_object_metadata_with_idx(
&self,
bucket: &str,
object: &str,
opts: &ObjectOptions,
) -> Result<(crate::set_disk::PreparedGetObjectMetadata, usize)> {
let suspended_pools = if opts.skip_decommissioned {
let pool_meta = self.pool_meta.read().await;
Some(
(0..self.pools.len())
.map(|idx| pool_meta.is_suspended(idx))
.collect::<Vec<_>>(),
)
} else {
None
};
let mut futures = FuturesUnordered::new();
for (idx, pool) in self.pools.iter().enumerate() {
if suspended_pools.as_ref().is_some_and(|pools| pools[idx]) {
continue;
}
if opts.skip_rebalancing && self.is_pool_rebalancing(idx).await {
continue;
}
futures.push(async move {
let result = pool
.prepare_get_object_reader_metadata(bucket, object, opts)
.await
.map_err(|err| to_object_err(err, vec![bucket, object]));
(idx, result)
});
}
let mut candidates = (0..self.pools.len()).map(|_| None).collect::<Vec<_>>();
// Retain one provisional winner. Other pools only need their lightweight
// identity for final conflict checks; if pool state changes while the
// fanout runs, the final winner is refetched and revalidated below.
let mut latest_prepared = None;
let mut latest_mod_time = None;
let mut provisional_dynamic_pool_state = None;
while let Some((idx, result)) = futures.next().await {
match result {
Ok(metadata) => {
let mod_time = metadata.object_info().mod_time.unwrap_or(OffsetDateTime::UNIX_EPOCH);
let info = metadata.object_info().clone();
let retain = match (latest_mod_time, latest_prepared.as_ref()) {
(None, _) => true,
(Some(current), _) if mod_time > current => true,
(Some(current), _) if mod_time < current => false,
(Some(_), Some((current_idx, _))) => {
if suspended_pools.is_none() && provisional_dynamic_pool_state.is_none() {
let pool_meta = self.pool_meta.read().await;
provisional_dynamic_pool_state = Some(
(0..self.pools.len())
.map(|pool_idx| pool_meta.is_suspended(pool_idx))
.collect::<Vec<_>>(),
);
}
let provisional_pool_state = suspended_pools
.as_ref()
.or(provisional_dynamic_pool_state.as_ref())
.ok_or_else(|| Error::other("GET pool state snapshot is unavailable"))?;
let new_key = (provisional_pool_state.get(idx).copied().unwrap_or(false), std::cmp::Reverse(idx));
let current_key = (
provisional_pool_state.get(*current_idx).copied().unwrap_or(false),
std::cmp::Reverse(*current_idx),
);
new_key < current_key
}
(Some(_), None) => true,
};
if retain {
if latest_mod_time.is_none_or(|current| mod_time > current) {
latest_mod_time = Some(mod_time);
}
latest_prepared = Some((idx, metadata));
}
candidates[idx] = Some(LatestObjectInfoCandidate {
info: Some(info),
idx,
err: None,
});
}
Err(err) => {
candidates[idx] = Some(LatestObjectInfoCandidate {
info: None,
idx,
err: Some(err),
});
}
}
}
#[cfg(test)]
pause_prepared_pool_read_after_fanout(object).await;
let suspended_pools = match suspended_pools {
Some(pools) => pools,
None => {
let pool_meta = self.pool_meta.read().await;
(0..self.pools.len())
.map(|idx| pool_meta.is_suspended(idx))
.collect::<Vec<_>>()
}
};
let candidates = candidates.into_iter().flatten().collect();
let (winner_info, winner_idx) =
resolve_latest_object_info_candidates_with_pool_state(candidates, &suspended_pools, bucket, object, opts)?;
if let Some((prepared_idx, metadata)) = latest_prepared
&& prepared_idx == winner_idx
{
return Ok((metadata, winner_idx));
}
let pool = self.pools.get(winner_idx).ok_or(Error::ErasureReadQuorum)?;
#[cfg(test)]
pause_prepared_pool_read_before_refetch(object).await;
let metadata = pool
.prepare_get_object_reader_metadata(bucket, object, opts)
.await
.map_err(|err| to_object_err(err, vec![bucket, object]))?;
validate_prepared_pool_refetch_identity(&winner_info, metadata.object_info())?;
Ok((metadata, winner_idx))
}
pub(super) async fn delete_object_from_all_pools(
&self,
bucket: &str,
+31 -2
View File
@@ -218,7 +218,7 @@ fn same_user_defined_identity(left: &ObjectInfo, right: &ObjectInfo) -> bool {
/// excluded. The selected winner still carries the chosen pool's layout, while
/// the remaining read-visible fields must agree before the pool index can
/// provide a deterministic tie-break.
fn same_latest_object_info_identity(left: &ObjectInfo, right: &ObjectInfo) -> bool {
pub(super) fn same_latest_object_info_identity(left: &ObjectInfo, right: &ObjectInfo) -> bool {
let same_read_surface = left.bucket == right.bucket
&& left.name == right.name
&& left.is_dir == right.is_dir
@@ -277,6 +277,14 @@ fn same_latest_object_info_identity(left: &ObjectInfo, right: &ObjectInfo) -> bo
}
}
pub(super) fn validate_prepared_pool_refetch_identity(expected: &ObjectInfo, refetched: &ObjectInfo) -> Result<()> {
if same_latest_object_info_identity(expected, refetched) {
Ok(())
} else {
Err(Error::ErasureReadQuorum)
}
}
#[cfg(test)]
pub(super) fn resolve_latest_object_info_candidates(
candidates: Vec<LatestObjectInfoCandidate>,
@@ -328,7 +336,11 @@ pub(super) fn resolve_latest_object_info_candidates_with_pool_state(
return Err(Error::ErasureReadQuorum);
}
return Ok((winner_info.clone(), winner.idx));
let winner = latest_candidates.swap_remove(0);
let Some(winner_info) = winner.info else {
return Err(Error::ErasureReadQuorum);
};
return Ok((winner_info, winner.idx));
}
for candidate in candidates {
@@ -347,6 +359,23 @@ pub(super) fn resolve_latest_object_info_candidates_with_pool_state(
mod tests {
use super::*;
#[test]
fn prepared_pool_refetch_identity_fails_closed_on_generation_change() {
let expected = ObjectInfo {
mod_time: Some(OffsetDateTime::from_unix_timestamp(10).expect("test timestamp should be valid")),
version_id: Some(uuid::Uuid::from_u128(1)),
etag: Some("etag-a".to_string()),
..Default::default()
};
let mut refetched = expected.clone();
refetched.etag = Some("etag-b".to_string());
let error = validate_prepared_pool_refetch_identity(&expected, &refetched)
.expect_err("refetched metadata from a changed generation must fail closed");
assert_eq!(error, Error::ErasureReadQuorum);
}
#[test]
fn rebalance_delete_result_preserves_precondition_failed() {
let err = resolve_rebalance_delete_from_all_pools_result(Err(Error::PreconditionFailed), "bucket", "object")
+12 -2
View File
@@ -1537,7 +1537,7 @@ where
Ok(deleted_at)
}
pub async fn update_user_secret_key(&self, access_key: &str, secret_key: &str) -> Result<()> {
pub async fn update_user_secret_key(&self, access_key: &str, secret_key: &str) -> Result<(OffsetDateTime, AccountStatus)> {
if access_key.is_empty() || secret_key.is_empty() {
return Err(Error::InvalidArgument);
}
@@ -1552,7 +1552,16 @@ where
let mut cred = u.credentials.clone();
cred.secret_key = secret_key.to_string();
// Status is captured from the same credential snapshot the new secret
// is persisted with, so a caller replicating the rotation broadcasts
// exactly what was written rather than re-reading racily.
let status = if cred.is_valid() {
AccountStatus::Enabled
} else {
AccountStatus::Disabled
};
let u = UserIdentity::from(cred);
let updated_at = u.update_at.unwrap_or_else(OffsetDateTime::now_utc);
drop(cache);
drop(users);
@@ -1560,7 +1569,8 @@ where
.save_user_identity(access_key, UserType::Reg, u.clone(), None)
.await?;
self.update_user_with_claims(access_key, u)
self.update_user_with_claims(access_key, u)?;
Ok((updated_at, status))
}
/// Add SSH public key for a user (for SFTP authentication)
+8 -2
View File
@@ -960,7 +960,11 @@ impl<T: Store> IamSys<T> {
Ok(updated_at)
}
pub async fn set_user_secret_key(&self, access_key: &str, secret_key: &str) -> Result<()> {
pub async fn set_user_secret_key(
&self,
access_key: &str,
secret_key: &str,
) -> Result<(OffsetDateTime, rustfs_madmin::AccountStatus)> {
if !is_access_key_valid(access_key) {
return Err(IamError::InvalidAccessKeyLength);
}
@@ -969,7 +973,9 @@ impl<T: Store> IamSys<T> {
return Err(IamError::InvalidSecretKeyLength);
}
self.store.update_user_secret_key(access_key, secret_key).await
let (updated_at, status) = self.store.update_user_secret_key(access_key, secret_key).await?;
self.notify_for_user(access_key, false).await;
Ok((updated_at, status))
}
/// Add SSH public key for a user (for SFTP authentication)
+88
View File
@@ -1671,6 +1671,94 @@ mod tests {
}
}
#[test]
fn replication_writable_fields_bind_to_typed_dto_fields() {
let mut rule = replication_rule("id-marker", "arn:bucket-marker");
rule.priority = Some(37);
rule.filter = Some(s3s::dto::ReplicationRuleFilter {
prefix: Some("prefix-marker/".to_string()),
tag: Some(s3s::dto::Tag {
key: Some("tag-key-marker".to_string()),
value: Some("tag-value-marker".to_string()),
}),
and: Some(s3s::dto::ReplicationRuleAndOperator {
prefix: Some("and-prefix-marker/".to_string()),
tags: Some(vec![s3s::dto::Tag {
key: Some("and-tag-key-marker".to_string()),
value: Some("and-tag-value-marker".to_string()),
}]),
}),
..Default::default()
});
rule.delete_marker_replication = Some(DeleteMarkerReplication {
status: Some(DeleteMarkerReplicationStatus::from_static(DeleteMarkerReplicationStatus::ENABLED)),
});
rule.delete_replication = Some(DeleteReplication {
status: DeleteReplicationStatus::from_static(DeleteReplicationStatus::ENABLED),
});
rule.source_selection_criteria = Some(SourceSelectionCriteria {
replica_modifications: Some(ReplicaModifications {
status: ReplicaModificationsStatus::from_static(ReplicaModificationsStatus::ENABLED),
}),
sse_kms_encrypted_objects: None,
});
let config = ReplicationConfiguration {
role: "role-marker".to_string(),
rules: vec![rule],
};
let rule = config.rules.first().expect("fixture should contain one rule");
let filter = rule.filter.as_ref().expect("fixture should contain a rule filter");
let field_hits = [
("Role", config.role == "role-marker"),
("Rule.ID", rule.id.as_deref() == Some("id-marker")),
("Rule.Status", rule.status.as_str() == ReplicationRuleStatus::ENABLED),
("Rule.Priority", rule.priority == Some(37)),
("Rule.Filter.Prefix", filter.prefix.as_deref() == Some("prefix-marker/")),
(
"Rule.Filter.Tag",
filter.tag.as_ref().and_then(|tag| tag.key.as_deref()) == Some("tag-key-marker"),
),
(
"Rule.Filter.And",
filter.and.as_ref().and_then(|and| and.prefix.as_deref()) == Some("and-prefix-marker/"),
),
("Rule.Destination.Bucket", rule.destination.bucket == "arn:bucket-marker"),
(
"Rule.ExistingObjectReplication.Status",
rule.existing_object_replication
.as_ref()
.is_some_and(|existing| existing.status.as_str() == ExistingObjectReplicationStatus::ENABLED),
),
(
"Rule.DeleteMarkerReplication.Status",
rule.delete_marker_replication
.as_ref()
.and_then(|delete_marker| delete_marker.status.as_ref())
.is_some_and(|status| status.as_str() == DeleteMarkerReplicationStatus::ENABLED),
),
(
"Rule.DeleteReplication.Status",
rule.delete_replication
.as_ref()
.is_some_and(|delete| delete.status.as_str() == DeleteReplicationStatus::ENABLED),
),
(
"Rule.SourceSelectionCriteria.ReplicaModifications.Status",
rule.source_selection_criteria
.as_ref()
.and_then(|criteria| criteria.replica_modifications.as_ref())
.is_some_and(|modifications| modifications.status.as_str() == ReplicaModificationsStatus::ENABLED),
),
];
let bound_paths = field_hits.iter().map(|(path, _)| *path).collect::<Vec<_>>();
assert_eq!(bound_paths, REPLICATION_WRITABLE_FIELDS);
for (path, hit) in field_hits {
assert!(hit, "typed field probe did not reach {path}");
}
}
#[test]
fn invalid_replication_status_fields_are_reported_before_persistence() {
let arn = "arn:rustfs:replication:us-east-1:target:bucket";
+44 -3
View File
@@ -126,6 +126,27 @@ pub fn should_retry_delete_marker_purge(dobj: &DeletedObject) -> bool {
dobj.delete_marker_version_id.is_some()
}
/// True when the target denied a replicated delete because object-lock
/// retention or a legal hold protects that version on the replica (its
/// deletion gate answers `AccessDenied` with the lock reason, and a
/// replication request carries no governance bypass, rustfs#6850). Retrying
/// cannot succeed until the lock itself lapses, so callers treat this as a
/// policy denial rather than a transient fault.
///
/// The reason text is the RustFS deletion-gate wording; a MinIO/AWS peer
/// phrases its WORM denial differently and simply stays unclassified — the
/// caller then falls back to plain retry behavior, never a wrong state.
pub fn is_object_lock_denied_delete(code: Option<&str>, message: Option<&str>) -> bool {
if !matches!(code, Some("AccessDenied")) {
return false;
}
let Some(message) = message else {
return false;
};
let message = message.to_ascii_lowercase();
message.contains("retention") || message.contains("legal hold")
}
fn admitted_target_arns_from_replication_state(state: &ReplicationState) -> Vec<String> {
let mut target_arns = state.targets.keys().cloned().collect::<Vec<_>>();
target_arns.extend(state.purge_targets.keys().cloned());
@@ -237,9 +258,9 @@ mod tests {
use super::{
DeletedObjectReplicationInfo, delete_marker_purge_mrf_entry, delete_marker_purge_version_id,
delete_replication_creates_marker, is_retryable_delete_replication_head_error, is_version_delete_replication,
replicate_delete_outcome, resync_existing_delete_replication_info, should_retry_delete_marker_purge,
target_delete_version_id,
delete_replication_creates_marker, is_object_lock_denied_delete, is_retryable_delete_replication_head_error,
is_version_delete_replication, replicate_delete_outcome, resync_existing_delete_replication_info,
should_retry_delete_marker_purge, target_delete_version_id,
};
use crate::storage_api::DeletedObject;
use crate::{
@@ -615,4 +636,24 @@ mod tests {
corrupt.target_delete_marker_version_ids_corrupt = true;
assert_eq!(delete_marker_purge_version_id(Some(&corrupt), arn, source), None);
}
#[test]
fn object_lock_denied_delete_is_recognized_by_code_and_reason() {
// The peer's deletion gate answers AccessDenied with the lock reason.
assert!(is_object_lock_denied_delete(
Some("AccessDenied"),
Some("Object is under GOVERNANCE retention and cannot be deleted until 2026-09-01T00:00:00Z")
));
assert!(is_object_lock_denied_delete(
Some("AccessDenied"),
Some("Object has a legal hold and cannot be deleted. Remove the legal hold first.")
));
// A plain policy denial (misconfigured replicator) is not a lock denial.
assert!(!is_object_lock_denied_delete(Some("AccessDenied"), Some("Access Denied.")));
assert!(!is_object_lock_denied_delete(Some("AccessDenied"), None));
// Other errors mentioning retention must not match.
assert!(!is_object_lock_denied_delete(Some("InternalError"), Some("retention lookup failed")));
assert!(!is_object_lock_denied_delete(None, Some("legal hold")));
}
}
+5 -4
View File
@@ -41,9 +41,9 @@ pub use config::{
};
pub use delete::{
DeletedObjectReplicationInfo, delete_marker_purge_mrf_entry, delete_marker_purge_version_id,
delete_replication_creates_marker, is_retryable_delete_replication_head_error, is_version_delete_replication,
replicate_delete_outcome, resync_existing_delete_replication_info, should_retry_delete_marker_purge,
target_delete_version_id,
delete_replication_creates_marker, is_object_lock_denied_delete, is_retryable_delete_replication_head_error,
is_version_delete_replication, replicate_delete_outcome, resync_existing_delete_replication_info,
should_retry_delete_marker_purge, target_delete_version_id,
};
pub use filemeta::{
NULL_VERSION_ID, REPLICATE_EXISTING, REPLICATE_EXISTING_DELETE, REPLICATE_HEAL, REPLICATE_HEAL_DELETE, REPLICATE_INCOMING,
@@ -65,7 +65,8 @@ pub use multipart::{
pub use object::{
ReplicationSourceObject, ReplicationTargetObject, SsecPassthroughCapability, SsecPassthroughGate, content_matches_by_etag,
is_replication_target_offline_error, replication_action_for_target, replication_etags_match,
ssec_passthrough_evidence_present, ssec_passthrough_gate, target_is_newer_than_source_null_version, version_identity_drifted,
single_part_replica_etag_mismatch, ssec_passthrough_evidence_present, ssec_passthrough_gate,
target_is_newer_than_source_null_version, version_identity_drifted,
};
pub use operation::{
MustReplicateOptions, ReplicationDeleteScheduleInput, ReplicationDeleteSource, ReplicationDeleteStateSource,
+58 -2
View File
@@ -71,6 +71,32 @@ pub fn replication_etags_match(source: Option<&str>, target: Option<&str>) -> bo
source_etag.is_some() && source_etag == target_etag
}
fn is_plain_single_part_md5(etag: &str) -> bool {
etag.len() == 32 && etag.bytes().all(|b| b.is_ascii_hexdigit())
}
/// Whether the ETag the target returned for a single-part replica proves the
/// stored bytes differ from what the source sent — e.g. a target that does not
/// decode `aws-chunked` framing stores the frames verbatim and returns their
/// ETag. Only a plain single-part MD5 ETag on both sides is decidable; a
/// multipart or opaque (encrypted) ETag, or a withheld replica ETag, returns
/// `false` because no corruption can be concluded from it.
pub fn single_part_replica_etag_mismatch(source_etag: Option<&str>, replica_etag: Option<&str>) -> bool {
let Some(source) = source_etag.map(trim_etag) else {
return false;
};
if !is_plain_single_part_md5(&source) {
return false;
}
let Some(replica) = replica_etag.map(trim_etag) else {
return false;
};
if !is_plain_single_part_md5(&replica) {
return false;
}
!source.eq_ignore_ascii_case(&replica)
}
pub fn target_is_newer_than_source_null_version(
source: &ReplicationSourceObject<'_>,
target: &ReplicationTargetObject<'_>,
@@ -276,11 +302,41 @@ pub fn ssec_passthrough_evidence_present(sse_customer_algorithm: Option<&str>) -
#[cfg(test)]
mod tests {
const SOURCE_MD5: &str = "9a0364b9e99bb480dd25e1f0284c8555";
const FRAMED_MD5: &str = "0f343b0931126a20f133d67c2b018a3b";
#[test]
fn single_part_replica_mismatch_is_only_decided_on_plain_md5_pairs() {
// The #6853 shape: the target stored aws-chunked frames verbatim and
// returned the framed bytes' ETag.
assert!(single_part_replica_etag_mismatch(Some(SOURCE_MD5), Some(FRAMED_MD5)));
assert!(single_part_replica_etag_mismatch(
Some(&format!("\"{SOURCE_MD5}\"")),
Some(&format!("\"{FRAMED_MD5}\""))
));
// A faithful replica, quoted or not, passes; hex case must not matter
// (a target may return the same MD5 uppercased).
assert!(!single_part_replica_etag_mismatch(Some(SOURCE_MD5), Some(SOURCE_MD5)));
assert!(!single_part_replica_etag_mismatch(Some(&format!("\"{SOURCE_MD5}\"")), Some(SOURCE_MD5)));
assert!(!single_part_replica_etag_mismatch(
Some(SOURCE_MD5),
Some(&SOURCE_MD5.to_ascii_uppercase())
));
// Not decidable: multipart source, opaque replica ETag, or either side
// missing must never be reported as corruption.
assert!(!single_part_replica_etag_mismatch(Some(&format!("{SOURCE_MD5}-3")), Some(FRAMED_MD5)));
assert!(!single_part_replica_etag_mismatch(Some(SOURCE_MD5), Some(&format!("{FRAMED_MD5}-3"))));
assert!(!single_part_replica_etag_mismatch(Some(SOURCE_MD5), None));
assert!(!single_part_replica_etag_mismatch(None, Some(FRAMED_MD5)));
}
use super::{
ReplicationSourceObject, ReplicationTargetObject, SsecPassthroughCapability, SsecPassthroughGate,
content_matches_by_etag, is_replication_target_offline_error, replication_action_for_target, replication_etags_match,
ssec_passthrough_evidence_present, ssec_passthrough_gate, target_is_newer_than_source_null_version,
version_identity_drifted,
single_part_replica_etag_mismatch, ssec_passthrough_evidence_present, ssec_passthrough_gate,
target_is_newer_than_source_null_version, version_identity_drifted,
};
use crate::filemeta::{ReplicationAction, ReplicationType};
use crate::http::AMZ_OBJECT_LOCK_MODE;
+5
View File
@@ -60,21 +60,26 @@ hotpath-cpu = [
[dependencies]
hotpath.workspace = true
metrics = { workspace = true }
async-compression = { workspace = true, features = ["tokio", "gzip", "bzip2"] }
async-trait.workspace = true
arc-swap.workspace = true
bytes = { workspace = true, features = ["serde"] }
chrono = { workspace = true, features = ["serde"] }
crc-fast.workspace = true
rustfs-common.workspace = true
datafusion = { workspace = true, default-features = false, features = ["parquet", "recursive_protection", "sql"] }
rustfs-ecstore.workspace = true
rustfs-storage-api.workspace = true
futures = { workspace = true }
futures-core = { workspace = true }
flate2.workspace = true
http.workspace = true
s3s = { workspace = true, features = ["minio"] }
serde_json = { workspace = true, features = ["raw_value"] }
thiserror = { workspace = true }
parking_lot.workspace = true
tokio = { workspace = true, features = ["fs", "rt-multi-thread"] }
tokio-stream.workspace = true
tokio-util = { workspace = true, features = ["io", "compat"] }
tracing.workspace = true
uuid.workspace = true
File diff suppressed because it is too large Load Diff
+10
View File
@@ -23,6 +23,7 @@ use datafusion::{
use std::{error::Error as StdError, fmt::Display};
use thiserror::Error;
mod input_stream;
mod metrics;
pub mod object_store;
pub mod query;
@@ -79,6 +80,9 @@ pub enum SelectError {
#[error("The file is not in a supported compression format. Only GZIP and BZIP2 are supported.")]
InvalidCompressionFormat,
#[error("{compression} is not applicable to the queried object. Please correct the request and try again.")]
InvalidCompressionFormatForObject { compression: &'static str },
#[error("The data source type is not valid. Only CSV, JSON, and Parquet are supported.")]
InvalidDataSource,
@@ -87,6 +91,9 @@ pub enum SelectError {
)]
TruncatedInput,
#[error("Scan range queries are not supported on this type of object.")]
UnsupportedScanRangeInput,
#[error("An error occurred while parsing the CSV file. Check the file and try again.")]
CsvParsingError,
@@ -96,6 +103,9 @@ pub enum SelectError {
#[error("An error occurred while parsing the Parquet file. Check the file and try again.")]
ParquetParsingError,
#[error("The length of a record in the input or result is greater than the maxCharsPerRecord limit of 1 MB.")]
OverMaxRecordSize,
#[error("{message}")]
ParseSelectFailure { message: String },
+90 -17
View File
@@ -12,7 +12,11 @@
// See the License for the specific language governing permissions and
// limitations under the License.
use std::sync::atomic::{AtomicU64, Ordering};
use arc_swap::ArcSwap;
use std::sync::{
Arc,
atomic::{AtomicU64, Ordering},
};
#[derive(Clone, Copy, Debug, Default, Eq, PartialEq)]
pub struct SelectInputMetricsSnapshot {
@@ -20,33 +24,72 @@ pub struct SelectInputMetricsSnapshot {
pub bytes_processed: u64,
}
#[derive(Debug, Default)]
#[derive(Debug)]
pub struct SelectInputMetrics {
active: ArcSwap<SelectInputMetricBank>,
}
#[derive(Debug, Default)]
struct SelectInputMetricBank {
uncompressed_bytes: AtomicU64,
compressed_bytes_scanned: AtomicU64,
compressed_bytes_processed: AtomicU64,
}
#[derive(Clone, Debug)]
pub(crate) struct SelectInputMetricsRecorder {
bank: Arc<SelectInputMetricBank>,
}
impl Default for SelectInputMetrics {
fn default() -> Self {
Self {
active: ArcSwap::from_pointee(SelectInputMetricBank::default()),
}
}
}
impl SelectInputMetrics {
pub fn snapshot(&self) -> SelectInputMetricsSnapshot {
let uncompressed_bytes = self.uncompressed_bytes.load(Ordering::Relaxed);
let bank = self.active.load();
let uncompressed_bytes = bank.uncompressed_bytes.load(Ordering::Relaxed);
SelectInputMetricsSnapshot {
bytes_scanned: uncompressed_bytes,
bytes_processed: uncompressed_bytes,
bytes_scanned: uncompressed_bytes.saturating_add(bank.compressed_bytes_scanned.load(Ordering::Relaxed)),
bytes_processed: uncompressed_bytes.saturating_add(bank.compressed_bytes_processed.load(Ordering::Relaxed)),
}
}
pub(crate) fn record_uncompressed(&self, bytes: usize) {
let increment = u64::try_from(bytes).unwrap_or(u64::MAX);
let _ = self
.uncompressed_bytes
.fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| Some(current.saturating_add(increment)));
pub(crate) fn recorder(&self) -> SelectInputMetricsRecorder {
SelectInputMetricsRecorder {
bank: self.active.load_full(),
}
}
/// Clears planner-only reads before query execution begins.
/// Publishes a fresh bank so late planner writes remain isolated.
pub fn reset(&self) {
self.uncompressed_bytes.store(0, Ordering::Relaxed);
self.active.store(Arc::new(SelectInputMetricBank::default()));
}
}
impl SelectInputMetricsRecorder {
pub(crate) fn record_uncompressed(&self, bytes: usize) {
saturating_add(&self.bank.uncompressed_bytes, bytes);
}
pub(crate) fn record_scanned(&self, bytes: usize) {
saturating_add(&self.bank.compressed_bytes_scanned, bytes);
}
pub(crate) fn record_processed(&self, bytes: usize) {
saturating_add(&self.bank.compressed_bytes_processed, bytes);
}
}
fn saturating_add(counter: &AtomicU64, bytes: usize) {
let increment = u64::try_from(bytes).unwrap_or(u64::MAX);
let _ = counter.fetch_update(Ordering::Relaxed, Ordering::Relaxed, |current| Some(current.saturating_add(increment)));
}
#[cfg(test)]
mod tests {
use super::*;
@@ -54,7 +97,7 @@ mod tests {
#[test]
fn records_uncompressed_input_at_both_boundaries() {
let metrics = SelectInputMetrics::default();
metrics.record_uncompressed(7);
metrics.recorder().record_uncompressed(7);
assert_eq!(
metrics.snapshot(),
@@ -68,21 +111,51 @@ mod tests {
#[test]
fn counters_saturate_instead_of_wrapping() {
let metrics = SelectInputMetrics::default();
metrics.uncompressed_bytes.store(u64::MAX - 1, Ordering::Relaxed);
metrics
.active
.load()
.uncompressed_bytes
.store(u64::MAX - 1, Ordering::Relaxed);
metrics.record_uncompressed(2);
metrics.recorder().record_uncompressed(2);
assert_eq!(metrics.snapshot().bytes_scanned, u64::MAX);
assert_eq!(metrics.snapshot().bytes_processed, u64::MAX);
}
#[test]
fn compressed_boundaries_are_counted_independently() {
let metrics = SelectInputMetrics::default();
let recorder = metrics.recorder();
recorder.record_scanned(39);
recorder.record_processed(19);
assert_eq!(
metrics.snapshot(),
SelectInputMetricsSnapshot {
bytes_scanned: 39,
bytes_processed: 19,
}
);
}
#[test]
fn reset_clears_schema_inference_bytes() {
let metrics = SelectInputMetrics::default();
metrics.record_uncompressed(9);
let planning = metrics.recorder();
planning.record_uncompressed(9);
metrics.reset();
planning.record_uncompressed(5);
let execution = metrics.recorder();
execution.record_uncompressed(3);
assert_eq!(metrics.snapshot(), SelectInputMetricsSnapshot::default());
assert_eq!(
metrics.snapshot(),
SelectInputMetricsSnapshot {
bytes_scanned: 3,
bytes_processed: 3,
}
);
}
}
File diff suppressed because it is too large Load Diff
+24
View File
@@ -30,6 +30,7 @@ use datafusion::{
prelude::SessionContext,
};
use parking_lot::Mutex;
use s3s::dto::CompressionType;
use std::sync::{
Arc, Weak,
atomic::{AtomicU8, Ordering},
@@ -446,11 +447,19 @@ impl SessionCtxFactory {
let scan_range_requires_single_file_scan =
context.input.request.scan_range.is_some() && context.input.request.input_serialization.parquet.is_none();
let json_document_requires_single_file_scan = is_json_document_input(&context.input);
let compressed_input_requires_single_file_scan = context
.input
.request
.input_serialization
.compression_type
.as_ref()
.is_some_and(|compression| compression.as_str() != CompressionType::NONE);
let metered_input_requires_single_file_scan =
input_metrics.is_some() && context.input.request.input_serialization.parquet.is_none();
let config = if custom_two_byte_record_delimiter
|| scan_range_requires_single_file_scan
|| json_document_requires_single_file_scan
|| compressed_input_requires_single_file_scan
|| metered_input_requires_single_file_scan
{
config.with_repartition_file_scans(false)
@@ -847,6 +856,21 @@ mod tests {
assert!(session.inner().config().options().optimizer.repartition_file_scans);
}
#[tokio::test]
async fn compressed_input_disables_file_repartitioning_without_metrics() {
let mut context = test_context();
Arc::make_mut(&mut context.input).request.input_serialization.compression_type =
Some(CompressionType::from_static(CompressionType::GZIP));
let session = SessionCtxFactory::new(true)
.with_target_partitions(2)
.create_session_ctx(&context)
.await
.expect("compressed session should be created");
assert!(!session.inner().config().options().optimizer.repartition_file_scans);
}
#[tokio::test]
async fn json_document_disables_file_repartitioning() {
let mut context = test_context();
+28 -12
View File
@@ -53,7 +53,7 @@ use rustfs_s3select_api::{
},
},
};
use s3s::dto::{FileHeaderInfo, JSONType, SelectObjectContentInput};
use s3s::dto::{CompressionType, FileHeaderInfo, JSONType, SelectObjectContentInput};
use std::sync::LazyLock;
use tokio::{
sync::Semaphore,
@@ -72,6 +72,7 @@ use crate::{
static IGNORE: LazyLock<FileHeaderInfo> = LazyLock::new(|| FileHeaderInfo::from_static(FileHeaderInfo::IGNORE));
static NONE: LazyLock<FileHeaderInfo> = LazyLock::new(|| FileHeaderInfo::from_static(FileHeaderInfo::NONE));
static USE: LazyLock<FileHeaderInfo> = LazyLock::new(|| FileHeaderInfo::from_static(FileHeaderInfo::USE));
const EXACT_OBJECT_FILE_EXTENSION: &str = "";
#[derive(Clone)]
pub struct SimpleQueryDispatcher {
@@ -416,6 +417,13 @@ impl SimpleQueryDispatcher {
let path = format!("s3://{}/{}", self.input.bucket, self.input.key);
let table_path = ListingTableUrl::parse(path)?;
let compressed_input = self
.input
.request
.input_serialization
.compression_type
.as_ref()
.is_some_and(|compression| compression.as_str() != CompressionType::NONE);
let (listing_options, need_rename_volume_name, need_ignore_volume_name) =
if let Some(csv) = self.input.request.input_serialization.csv.as_ref() {
let mut need_rename_volume_name = false;
@@ -465,22 +473,30 @@ impl SimpleQueryDispatcher {
file_format = file_format.with_quote(quote.as_bytes().first().copied().unwrap_or_default());
}
(
ListingOptions::new(Arc::new(file_format)).with_file_extension(".csv"),
ListingOptions::new(Arc::new(file_format)).with_file_extension(if compressed_input {
EXACT_OBJECT_FILE_EXTENSION
} else {
".csv"
}),
need_rename_volume_name,
need_ignore_volume_name,
)
} else if self.input.request.input_serialization.json.is_some() {
let file_format = JsonFormat::default();
// Use the actual file extension from the object key so that files stored
// with a `.jsonl` suffix (newline-delimited JSON) are also matched by
// DataFusion's listing/schema-inference logic. Falling back to ".json"
// preserves behaviour for keys that have no extension.
let file_ext = std::path::Path::new(&self.input.key)
.extension()
.and_then(|e| e.to_str())
.map(|e| format!(".{e}"))
.unwrap_or_else(|| ".json".to_string());
(ListingOptions::new(Arc::new(file_format)).with_file_extension(file_ext), false, false)
let file_extension = if compressed_input {
EXACT_OBJECT_FILE_EXTENSION.to_string()
} else {
std::path::Path::new(&self.input.key)
.extension()
.and_then(|extension| extension.to_str())
.map(|extension| format!(".{extension}"))
.unwrap_or_else(|| ".json".to_string())
};
(
ListingOptions::new(Arc::new(file_format)).with_file_extension(file_extension),
false,
false,
)
} else {
return Err(SelectError::InvalidDataSource.into());
};
+1 -1
View File
@@ -106,7 +106,7 @@ hex-simd.workspace = true
[dev-dependencies]
tracing-subscriber = { workspace = true, features = ["json", "env-filter", "time"] }
serial_test = { workspace = true }
temp-env = { workspace = true }
temp-env = { workspace = true, features = ["async_closure"] }
tempfile = { workspace = true }
uuid = { workspace = true, features = ["v4", "serde", "fast-rng", "macro-diagnostics"] }
tokio = { workspace = true, features = ["test-util", "fs", "rt-multi-thread"] }
+5
View File
@@ -175,6 +175,11 @@ pub static DATA_USAGE_BUCKET: LazyLock<String> =
pub static DATA_USAGE_OBJ_NAME_PATH: LazyLock<String> =
LazyLock::new(|| format!("{BUCKET_META_PREFIX}{SLASH_SEPARATOR}{DATA_USAGE_OBJECT_NAME}"));
/// Durable evidence for recovery of the exact empty usage fence written by
/// rc.2/rc.3 bucket cleanup before the first authoritative scanner snapshot.
pub static DATA_USAGE_RECOVERY_PATH: LazyLock<String> =
LazyLock::new(|| format!("{}.recovery-pending.json", DATA_USAGE_OBJ_NAME_PATH.as_str()));
pub static DATA_USAGE_OBSERVED_OBJ_NAME_PATH: LazyLock<String> =
LazyLock::new(|| format!("{BUCKET_META_PREFIX}{SLASH_SEPARATOR}{DATA_USAGE_OBSERVED_OBJECT_NAME}"));
+4 -2
View File
@@ -82,8 +82,10 @@ pub use remote_scanner::{
pub use runtime_config::{apply_scanner_runtime_config, scanner_runtime_config_status, validate_scanner_runtime_config};
pub use rustfs_scanner_contracts::last_minute;
pub use scanner::{
ScannerCycleRecoveryMarker, ScannerCycleRecoveryStatus, ScannerCycleScheduleStatus, init_data_scanner,
reset_scanner_cycle_recovery, scanner_cycle_recovery_status, scanner_cycle_schedule_status, scanner_topology_digest,
ScannerCycleRecoveryMarker, ScannerCycleRecoveryStatus, ScannerCycleScheduleStatus, ScannerPauseBacklogAlertReason,
ScannerPauseBacklogPhase, ScannerPauseBacklogStatus, ScannerPauseBacklogThresholds, init_data_scanner,
reset_scanner_cycle_recovery, scanner_cycle_recovery_status, scanner_cycle_schedule_status, scanner_pause_backlog_status,
scanner_topology_digest,
};
pub use scanner_io::{
ScannerDirtyUsageAckError, ScannerDirtyUsageState, acknowledge_dirty_usage_generation, clear_dirty_usage_bucket,
+396 -19
View File
@@ -22,7 +22,8 @@ use std::sync::{Arc, LazyLock, RwLock};
use self::heal_info::{BackgroundHealInfoReadStatus, read_background_heal_info_with_epoch, save_background_heal_info_for_epoch};
use crate::data_usage_define::{
BACKGROUND_HEAL_INFO_PATH, DATA_USAGE_BLOOM_NAME_PATH, DATA_USAGE_OBJ_NAME_PATH, DATA_USAGE_OBSERVED_OBJ_NAME_PATH,
DataUsageCache, DataUsageCacheRevision, LEGACY_DATA_USAGE_OBJ_NAME_PATH, read_config_revision, read_config_with_revision,
DATA_USAGE_RECOVERY_PATH, DataUsageCache, DataUsageCacheRevision, LEGACY_DATA_USAGE_OBJ_NAME_PATH, read_config_revision,
read_config_with_revision,
};
use crate::runtime_config::{
ScannerRuntimeConfig, ScannerRuntimeConfigSource, refresh_scanner_runtime_config_from_global, scanner_bitrot_cycle,
@@ -89,6 +90,144 @@ const EVENT_SCANNER_BACKGROUND_HEAL_STATE: &str = "scanner_background_heal_state
const METRIC_SCANNER_LEADER_LOCK_TOTAL: &str = "rustfs_scanner_leader_lock_total";
const CLEAN_IDLE_MAX_INTERVAL: Duration = Duration::from_secs(24 * 60 * 60);
const MAX_SCANNER_SCHEDULE_DELAY: Duration = Duration::from_secs(365 * 24 * 60 * 60);
#[cfg(test)]
static SCANNER_STARTUP_OBSERVED_PROBE: LazyLock<StdMutex<Option<Arc<ScannerStartupObservedProbeState>>>> =
LazyLock::new(|| StdMutex::new(None));
#[cfg(test)]
struct ScannerStartupObservedProbeState {
observed: Notify,
resume: Notify,
}
#[cfg(test)]
struct ScannerObservedProbeState {
store_key: usize,
paused: bool,
notify: Notify,
}
#[cfg(test)]
pub(super) struct ScannerStartupObservedProbe {
state: Arc<ScannerStartupObservedProbeState>,
}
#[cfg(test)]
static SCANNER_RUNTIME_OBSERVED_PROBE: LazyLock<StdMutex<Option<Arc<ScannerObservedProbeState>>>> =
LazyLock::new(|| StdMutex::new(None));
#[cfg(test)]
pub(super) struct ScannerRuntimeObservedProbe {
state: Arc<ScannerObservedProbeState>,
}
#[cfg(test)]
impl ScannerStartupObservedProbe {
pub(super) fn install() -> Self {
let state = Arc::new(ScannerStartupObservedProbeState {
observed: Notify::new(),
resume: Notify::new(),
});
let mut probe = SCANNER_STARTUP_OBSERVED_PROBE
.lock()
.expect("scanner startup observed probe should not be poisoned");
assert!(probe.is_none(), "scanner startup observed probe must be unique");
*probe = Some(state.clone());
Self { state }
}
pub(super) async fn wait(&self) {
tokio::time::timeout(Duration::from_secs(5), self.state.observed.notified())
.await
.expect("scanner should complete startup pause-backlog observation");
}
pub(super) fn resume(&self) {
self.state.resume.notify_one();
}
}
#[cfg(test)]
impl ScannerRuntimeObservedProbe {
pub(super) fn install(storeapi: &Arc<ECStore>, paused: bool) -> Self {
let state = Arc::new(ScannerObservedProbeState {
store_key: scanner_observed_probe_store_key(storeapi),
paused,
notify: Notify::new(),
});
let mut probe = SCANNER_RUNTIME_OBSERVED_PROBE
.lock()
.expect("scanner runtime observed probe should not be poisoned");
assert!(probe.is_none(), "scanner runtime observed probe must be unique");
*probe = Some(state.clone());
Self { state }
}
pub(super) async fn wait(&self) {
tokio::time::timeout(Duration::from_secs(10), self.state.notify.notified())
.await
.expect("scanner should complete runtime pause-backlog observation");
}
}
#[cfg(test)]
impl Drop for ScannerStartupObservedProbe {
fn drop(&mut self) {
let mut probe = SCANNER_STARTUP_OBSERVED_PROBE
.lock()
.expect("scanner startup observed probe should not be poisoned");
if probe.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
*probe = None;
}
self.state.resume.notify_one();
}
}
#[cfg(test)]
impl Drop for ScannerRuntimeObservedProbe {
fn drop(&mut self) {
let mut probe = SCANNER_RUNTIME_OBSERVED_PROBE
.lock()
.expect("scanner runtime observed probe should not be poisoned");
if probe.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
*probe = None;
}
}
}
#[cfg(test)]
async fn notify_scanner_startup_observed_for_test() {
let probe = {
SCANNER_STARTUP_OBSERVED_PROBE
.lock()
.expect("scanner startup observed probe should not be poisoned")
.clone()
};
if let Some(probe) = probe {
probe.observed.notify_one();
probe.resume.notified().await;
}
}
#[cfg(test)]
fn scanner_observed_probe_store_key(storeapi: &Arc<ECStore>) -> usize {
Arc::as_ptr(storeapi).cast::<()>() as usize
}
#[cfg(test)]
fn notify_scanner_runtime_observed_for_test(storeapi: &Arc<ECStore>, observation: ScannerPauseBacklogObservation) {
if let Some(probe) = SCANNER_RUNTIME_OBSERVED_PROBE
.lock()
.expect("scanner runtime observed probe should not be poisoned")
.clone()
&& probe.store_key == scanner_observed_probe_store_key(storeapi)
&& probe.paused == observation.paused
{
probe.notify.notify_one();
}
}
const CLEAN_IDLE_BACKOFF_FACTOR: u32 = 2;
/// First-retry delay after a scanner cycle cannot publish authoritative usage.
///
@@ -294,6 +433,14 @@ fn record_scanner_leader_lock_state(state: &'static str) {
.increment(1);
}
async fn finish_scanner_leader_iteration(lock_lost: bool, state: &'static str, error: String) {
reset_scanner_cycle_schedule();
let liveness_already_recorded = lock_lost && !global_metrics().report().await.leader_lock_held_by_this_process;
if !liveness_already_recorded {
global_metrics().record_scanner_leader_liveness(state, false, error).await;
}
}
#[cfg(test)]
fn scanner_cycle_max_duration() -> Option<Duration> {
resolve_scanner_runtime_config().cycle_budget.max_duration
@@ -742,15 +889,15 @@ fn prepare_cycle_for_usage_floor_bootstrap(
cycle_info: &mut CurrentCycle,
usage_floor: PersistedUsageFloor,
startup: PersistedUsageFloorStartup,
) -> (bool, bool) {
) -> (bool, ScannerCycleResetPolicy) {
match startup {
PersistedUsageFloorStartup::Authoritative => (false, false),
PersistedUsageFloorStartup::Authoritative => (false, ScannerCycleResetPolicy::None),
PersistedUsageFloorStartup::Missing => {
// Cycle progress without its corresponding usage floor cannot
// prove namespace coverage. Restart from cycle zero while keeping
// the separately fenced leader epoch monotonic.
*cycle_info = CurrentCycle::default();
(true, true)
(true, ScannerCycleResetPolicy::ResetAll)
}
PersistedUsageFloorStartup::BootstrapPending => {
// An unfenced marker may have been written before an upgrade's old
@@ -759,7 +906,25 @@ fn prepare_cycle_for_usage_floor_bootstrap(
if usage_floor.leader_epoch == 0 {
*cycle_info = CurrentCycle::default();
}
(true, usage_floor.leader_epoch == 0)
(
true,
if usage_floor.leader_epoch == 0 {
ScannerCycleResetPolicy::ResetAll
} else {
ScannerCycleResetPolicy::None
},
)
}
PersistedUsageFloorStartup::RecoveredLegacyEmptyFence => {
// The legacy empty fence proves only its leader epoch, not
// namespace coverage. Clear coverage while retaining the durable
// cycle number so surviving caches cannot force a regression.
let next = cycle_info.next;
*cycle_info = CurrentCycle {
next,
..Default::default()
};
(true, ScannerCycleResetPolicy::ResetCoveragePreservingNext)
}
}
}
@@ -1280,7 +1445,15 @@ where
LockLost: Future<Output = ()>,
{
let fence_ctx = ctx.child_token();
let claim = claim_scanner_leadership(&fence_ctx, storeapi, cycle_info, cycle_revision, leader_epoch, false, false);
let claim = claim_scanner_leadership(
&fence_ctx,
storeapi,
cycle_info,
cycle_revision,
leader_epoch,
false,
ScannerCycleResetPolicy::None,
);
tokio::pin!(claim);
tokio::pin!(lock_lost);
tokio::select! {
@@ -1776,8 +1949,18 @@ async fn run_data_scanner_cycle_with_budget(
mark_scan_cycle_idle(cycle_info, &mut cycle_metrics_guard).await;
return ScannerCycleOutcome::Failed;
}
match scanner_cycle_pre_commit_outcome(scan_cycle_result.required_cycle_floor(), &usage_persist_outcome) {
let required_cycle_floor = scan_cycle_result.required_cycle_floor();
let pre_commit_outcome = scanner_cycle_pre_commit_outcome(required_cycle_floor, &usage_persist_outcome);
update_scanner_cache_cycle_recovery_status(
cycle_info.current,
leader_epoch,
required_cycle_floor,
pre_commit_outcome,
scan_cycle_result.status == ScannerCycleStatus::Complete,
);
match pre_commit_outcome {
Some(ScannerCyclePreCommitOutcome::RecoverCacheCycle(required_cycle)) => {
record_scanner_cache_cycle_recovery_attempt();
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_CYCLE_STATE,
@@ -2145,6 +2328,74 @@ pub async fn run_data_scanner(ctx: CancellationToken, storeapi: Arc<ECStore>) ->
run_data_scanner_with_maintenance_state(ctx, storeapi, maintenance_features, maintenance_generation).await
}
async fn current_scanner_pause_backlog_observation(storeapi: &Arc<ECStore>) -> ScannerPauseBacklogObservation {
let now_unix_secs = scanner_pause_backlog_now();
let pause = storeapi.scanner_data_movement_pause_status().await;
let metrics = global_metrics().report().await;
ScannerPauseBacklogObservation {
now_unix_secs,
paused: pause.paused,
movement_generation: pause.movement_generation,
movement_work_items: pause.movement_backlog_work_items,
pause_started_at_unix_secs: pause.started_at_unix_secs,
dirty_usage_buckets: metrics.usage_freshness.dirty_pending_buckets,
discovered_expiry_items: metrics
.lifecycle_expiry
.current_queued
.saturating_add(metrics.lifecycle_expiry.current_active),
discovered_transition_items: metrics
.lifecycle_transition
.current_queued
.saturating_add(metrics.lifecycle_transition.current_active)
.saturating_add(metrics.lifecycle_transition.compensation_pending)
.saturating_add(metrics.lifecycle_transition.compensation_running),
}
}
async fn wait_for_scanner_data_movement_resume(
ctx: &CancellationToken,
storeapi: &Arc<ECStore>,
guard: &NamespaceLockGuard,
pause_backlog: &mut ScannerPauseBacklogController,
) -> bool {
loop {
let observation = current_scanner_pause_backlog_observation(storeapi).await;
pause_backlog.observe(observation).await;
#[cfg(test)]
notify_scanner_runtime_observed_for_test(storeapi, observation);
if !observation.paused {
return !ctx.is_cancelled() && !guard.is_lock_lost();
}
let movement_changed = storeapi.scanner_data_movement_changed();
if storeapi.scanner_data_movement_generation() != observation.movement_generation {
continue;
}
tokio::select! {
_ = ctx.cancelled() => return false,
_ = guard.lock_lost_notified() => return false,
_ = movement_changed.notified() => {},
_ = tokio::time::sleep(SCANNER_CYCLE_RECOVERY_PAUSED_INTERVAL) => {},
}
}
}
async fn finish_scanner_pause_backlog_cycle(
pause_backlog: &mut ScannerPauseBacklogController,
storeapi: &Arc<ECStore>,
attempt: ScannerPauseBacklogAttemptDecision,
outcome: ScannerCycleOutcome,
) {
let observation = current_scanner_pause_backlog_observation(storeapi).await;
if let ScannerPauseBacklogAttemptDecision::Tracked(serial) = attempt {
pause_backlog.finish_attempt(serial, outcome, observation).await;
} else {
pause_backlog.observe_cycle_outcome(outcome, observation).await;
}
#[cfg(test)]
notify_scanner_runtime_observed_for_test(storeapi, observation);
}
async fn run_data_scanner_with_maintenance_state(
ctx: CancellationToken,
storeapi: Arc<ECStore>,
@@ -2224,6 +2475,28 @@ async fn run_data_scanner_with_maintenance_state(
return Ok(());
}
};
let pause_backlog_now = scanner_pause_backlog_now();
let mut pause_backlog = match ScannerPauseBacklogController::claim(storeapi.clone(), pause_backlog_now).await {
Ok(controller) => controller,
Err(err) => {
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
state = "pause_backlog_claim_failed",
error = %err,
"Scanner pause backlog persistence is unavailable"
);
ScannerPauseBacklogController::unavailable(storeapi.clone(), err, pause_backlog_now)
}
};
if !wait_for_scanner_data_movement_resume(&ctx, &storeapi, &guard, &mut pause_backlog).await {
global_metrics().set_cycle(None).await;
return Ok(());
}
#[cfg(test)]
notify_scanner_startup_observed_for_test().await;
let single_disk = storeapi.setup_is_erasure_sd().await;
let erasure = storeapi.setup_is_erasure().await;
let distributed = storeapi.setup_is_dist_erasure().await;
@@ -2240,6 +2513,7 @@ async fn run_data_scanner_with_maintenance_state(
{
let Some((features, generation)) = detect_stable_scanner_maintenance_features(&ctx, &storeapi).await else {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(false, "stopped", String::new()).await;
return Ok(());
};
maintenance_features = features;
@@ -2266,16 +2540,19 @@ async fn run_data_scanner_with_maintenance_state(
} => (cycle, leader_epoch, revision),
ScannerCycleStateStartup::Blocked => {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(false, "stopped", String::new()).await;
return Ok(());
}
ScannerCycleStateStartup::Transient(err) => {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(false, "stopped", String::new()).await;
return Err(err);
}
};
let (usage_floor, usage_floor_startup) = match persisted_usage_floor_for_startup(storeapi.clone(), true).await {
Ok(result) => result,
Err(err) => {
let error = err.to_string();
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
@@ -2286,18 +2563,23 @@ async fn run_data_scanner_with_maintenance_state(
error = %err,
"Scanner stopped because the persisted usage floor could not be loaded"
);
record_scanner_usage_floor_failure(error.clone());
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(false, "usage_floor_load_failed", error).await;
return Ok(());
}
};
let (allow_usage_floor_bootstrap_pending, reset_usage_floor_bootstrap_cycle_on_conflict) =
let (allow_usage_floor_bootstrap_pending, usage_floor_cycle_reset_policy) =
prepare_cycle_for_usage_floor_bootstrap(&mut cycle_info, usage_floor, usage_floor_startup);
apply_persisted_usage_floor(&mut cycle_info, &mut leader_epoch, usage_floor);
match usage_floor_startup {
PersistedUsageFloorStartup::Authoritative | PersistedUsageFloorStartup::BootstrapPending => {}
PersistedUsageFloorStartup::Authoritative
| PersistedUsageFloorStartup::BootstrapPending
| PersistedUsageFloorStartup::RecoveredLegacyEmptyFence => {}
PersistedUsageFloorStartup::Missing => {
if ctx.is_cancelled() || guard.is_lock_lost() {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
return Ok(());
}
@@ -2322,10 +2604,12 @@ async fn run_data_scanner_with_maintenance_state(
"Scanner stopped because the usage baseline bootstrap could not be initialized"
);
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(false, "usage_floor_bootstrap_failed", err.to_string()).await;
return Ok(());
}
None => {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
return Ok(());
}
}
@@ -2334,6 +2618,7 @@ async fn run_data_scanner_with_maintenance_state(
if ctx.is_cancelled() || guard.is_lock_lost() {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
return Ok(());
}
let claim_ctx = ctx.child_token();
@@ -2346,7 +2631,7 @@ async fn run_data_scanner_with_maintenance_state(
&mut cycle_revision,
&mut leader_epoch,
allow_usage_floor_bootstrap_pending,
reset_usage_floor_bootstrap_cycle_on_conflict,
usage_floor_cycle_reset_policy,
),
guard.lock_lost_notified(),
)
@@ -2355,9 +2640,23 @@ async fn run_data_scanner_with_maintenance_state(
if guard.is_lock_lost() {
record_scanner_leader_lock_lost("Scanner leader lock lost while claiming the leadership epoch").await;
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(true, "lost", String::new()).await;
return Ok(());
}
if !leadership_claimed {
let observation = current_scanner_pause_backlog_observation(&storeapi).await;
pause_backlog.observe(observation).await;
#[cfg(test)]
notify_scanner_runtime_observed_for_test(&storeapi, observation);
if observation.paused {
if wait_for_scanner_data_movement_resume(&ctx, &storeapi, &guard, &mut pause_backlog).await {
return Err(ScannerError::Other(
"scanner startup was fenced by data movement; retrying from durable state".to_string(),
));
}
global_metrics().set_cycle(None).await;
return Ok(());
}
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_LOCK_STATE,
@@ -2367,14 +2666,36 @@ async fn run_data_scanner_with_maintenance_state(
state = "epoch_claim_failed",
"Scanner stopped because the leadership epoch could not be claimed"
);
global_metrics()
.record_scanner_leader_liveness("epoch_claim_failed", false, "leadership epoch claim failed")
.await;
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(false, "epoch_claim_failed", "leadership epoch claim failed".to_string()).await;
return Ok(());
}
if usage_floor_startup == PersistedUsageFloorStartup::RecoveredLegacyEmptyFence
&& let Err(err) = complete_legacy_empty_usage_floor_recovery(storeapi.clone(), leader_epoch).await
{
let error = err.to_string();
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
state = "usage_floor_recovery_cleanup_deferred",
path = %DATA_USAGE_RECOVERY_PATH.as_str(),
error = %err,
"Scanner usage floor recovery marker cleanup was deferred"
);
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(false, "usage_floor_recovery_pending", error).await;
return Ok(());
}
if !ctx.is_cancelled() {
let initial_pause_backlog_attempt = pause_backlog.begin_attempt(scanner_pause_backlog_now()).await;
if !ctx.is_cancelled()
&& matches!(
initial_pause_backlog_attempt,
ScannerPauseBacklogAttemptDecision::Untracked | ScannerPauseBacklogAttemptDecision::Tracked(_)
)
{
// Preserve previous behavior: run one cycle immediately after lock acquisition.
let dirty_generation_before_cycle = dirty_usage_generation();
let dirty_usage_pending_before_cycle = dirty_usage_buckets_pending();
@@ -2382,6 +2703,7 @@ async fn run_data_scanner_with_maintenance_state(
if guard.is_lock_lost() {
record_scanner_leader_lock_lost("Scanner leader lock lost before the initial cycle").await;
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(true, "lost", String::new()).await;
return Ok(());
}
let cycle_ctx = ctx.child_token();
@@ -2405,10 +2727,12 @@ async fn run_data_scanner_with_maintenance_state(
ScannerCycleWaitOutcome::LockLost => {
record_scanner_leader_lock_lost("Scanner leader lock lost during the initial cycle").await;
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(true, "lost", String::new()).await;
return Ok(());
}
ScannerCycleWaitOutcome::Cancelled => {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
return Ok(());
}
ScannerCycleWaitOutcome::Deadline { worker_stopped } => {
@@ -2425,15 +2749,18 @@ async fn run_data_scanner_with_maintenance_state(
&mut guard,
)
.await;
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
return Ok(());
}
};
finish_scanner_pause_backlog_cycle(&mut pause_backlog, &storeapi, initial_pause_backlog_attempt, initial_outcome).await;
superseded_backoff.record_retryable_cycle(initial_outcome == ScannerCycleOutcome::Superseded);
deferred_backoff.record_retryable_cycle(matches!(initial_outcome, ScannerCycleOutcome::Deferred(_)));
dirty_usage_generation_seen = dirty_generation_before_cycle;
if guard.is_lock_lost() {
record_scanner_leader_lock_lost("Scanner leader lock lost during the initial cycle").await;
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(true, "lost", String::new()).await;
return Ok(());
}
let runtime_config = resolve_scanner_runtime_config();
@@ -2479,6 +2806,10 @@ async fn run_data_scanner_with_maintenance_state(
break;
}
let pause_backlog_observation = current_scanner_pause_backlog_observation(&storeapi).await;
pause_backlog.observe(pause_backlog_observation).await;
#[cfg(test)]
notify_scanner_runtime_observed_for_test(&storeapi, pause_backlog_observation);
let runtime_config = resolve_scanner_runtime_config();
if clean_idle_topology_supported && scanner_clean_idle_backoff_configured(&runtime_config) {
let current_generation = scanner_maintenance_generation();
@@ -2515,11 +2846,16 @@ async fn run_data_scanner_with_maintenance_state(
scanner_cycle_wait_plan(&runtime_config, clean_idle_backoff, backoff_enabled, randomized_cycle_delay_for);
let superseded_retry_interval = superseded_backoff.retry_interval(runtime_config.cycle_interval);
let deferred_retry_interval = deferred_backoff.retry_interval(runtime_config.cycle_interval);
let convergence_retry_interval = superseded_retry_interval.or(deferred_retry_interval);
let mut convergence_retry_interval = superseded_retry_interval.or(deferred_retry_interval);
if let Some(retry_interval) = convergence_retry_interval {
wait_plan.effective_interval = retry_interval;
wait_plan.delay = randomized_cycle_delay_for(retry_interval).min(retry_interval);
}
if let Some(pause_backlog_delay) = pause_backlog.scheduling_delay(scanner_pause_backlog_now()) {
wait_plan.effective_interval = pause_backlog_delay.max(Duration::from_secs(1));
wait_plan.delay = pause_backlog_delay;
convergence_retry_interval = Some(pause_backlog_delay.max(Duration::from_secs(1)));
}
let dirty_generation_before_wait = dirty_usage_generation();
let dirty_usage_pending_before_wait = dirty_usage_buckets_pending();
let maintenance_generation_before_wait = scanner_maintenance_generation();
@@ -2644,6 +2980,20 @@ async fn run_data_scanner_with_maintenance_state(
record_scanner_leader_lock_lost("Scanner leader lock lost before starting the next cycle").await;
break;
}
let pause_backlog_observation = current_scanner_pause_backlog_observation(&storeapi).await;
pause_backlog.observe(pause_backlog_observation).await;
#[cfg(test)]
notify_scanner_runtime_observed_for_test(&storeapi, pause_backlog_observation);
if pause_backlog_observation.paused {
continue;
}
let pause_backlog_attempt = pause_backlog.begin_attempt(scanner_pause_backlog_now()).await;
if matches!(
pause_backlog_attempt,
ScannerPauseBacklogAttemptDecision::RateLimited | ScannerPauseBacklogAttemptDecision::PersistenceUnavailable
) {
continue;
}
let dirty_generation_before_cycle = dirty_usage_generation();
let cycle_ctx = ctx.child_token();
let cycle_budget = ScannerCycleBudget::new_with_runtime_progress_tracking(&cycle_ctx, scanner_cycle_budget_config());
@@ -2666,10 +3016,12 @@ async fn run_data_scanner_with_maintenance_state(
ScannerCycleWaitOutcome::LockLost => {
record_scanner_leader_lock_lost("Scanner leader lock lost during a scanner cycle").await;
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(true, "lost", String::new()).await;
return Ok(());
}
ScannerCycleWaitOutcome::Cancelled => {
global_metrics().set_cycle(None).await;
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
return Ok(());
}
ScannerCycleWaitOutcome::Deadline { worker_stopped } => {
@@ -2686,9 +3038,11 @@ async fn run_data_scanner_with_maintenance_state(
&mut guard,
)
.await;
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
return Ok(());
}
};
finish_scanner_pause_backlog_cycle(&mut pause_backlog, &storeapi, pause_backlog_attempt, outcome).await;
superseded_backoff.record_retryable_cycle(outcome == ScannerCycleOutcome::Superseded);
deferred_backoff.record_retryable_cycle(matches!(outcome, ScannerCycleOutcome::Deferred(_)));
dirty_usage_generation_seen = dirty_generation_before_cycle;
@@ -2771,10 +3125,7 @@ async fn run_data_scanner_with_maintenance_state(
}
global_metrics().set_cycle(None).await;
reset_scanner_cycle_schedule();
if !guard.is_lock_lost() {
global_metrics().record_scanner_leader_liveness("stopped", false, "").await;
}
finish_scanner_leader_iteration(guard.is_lock_lost(), "stopped", String::new()).await;
debug!(
target: "rustfs::scanner",
@@ -2852,6 +3203,26 @@ fn scanner_cycle_pre_commit_outcome(
}
}
fn update_scanner_cache_cycle_recovery_status(
requested_cycle: u64,
leader_epoch: u64,
required_cycle_floor: Option<u64>,
pre_commit_outcome: Option<ScannerCyclePreCommitOutcome>,
cache_scope_complete: bool,
) {
match (required_cycle_floor, pre_commit_outcome) {
(Some(required_cycle), _) => {
record_scanner_cache_cycle_ahead(requested_cycle, required_cycle, leader_epoch);
}
(None, Some(ScannerCyclePreCommitOutcome::Deferred(_))) => {
// A deferred scan may not have covered the cache that established
// the existing floor, so it cannot prove recovery is complete.
}
(None, _) if cache_scope_complete => clear_scanner_cache_cycle_ahead(),
(None, _) => {}
}
}
fn scanner_cycle_completion_outcome(
scan_status: ScannerCycleStatus,
usage_persist_outcome: DataUsagePersistOutcome,
@@ -2989,12 +3360,14 @@ fn data_usage_reintroduces_missing_bucket(incoming: &DataUsageInfo, existing: Op
/// Store data usage info in backend. Will store all objects sent on the receiver until closed.
mod activity;
mod backlog;
mod cycle_state;
mod heal_info;
mod leadership;
mod usage_store;
use activity::*;
use backlog::*;
use cycle_state::*;
use leadership::*;
use usage_store::*;
@@ -3005,6 +3378,10 @@ pub(crate) use activity::{
scanner_activity_publication_lease_targets, scanner_activity_snapshot_digest, scanner_dirty_usage_acknowledgements,
};
pub(crate) use activity::{ScannerCycleOutcome, scanner_cycle_outcome_with_pending_maintenance};
pub use backlog::{
ScannerPauseBacklogAlertReason, ScannerPauseBacklogPhase, ScannerPauseBacklogStatus, ScannerPauseBacklogThresholds,
scanner_pause_backlog_status,
};
#[cfg(test)]
pub(crate) use cycle_state::encode_scanner_cycle_fence_for_test;
pub use cycle_state::{
File diff suppressed because it is too large Load Diff
+518 -11
View File
@@ -14,7 +14,7 @@
/// Scanner cycle-state codec, persisted usage floors, and cycle-state persistence.
use super::*;
use crate::ScannerGetObjectReader;
use crate::data_usage_define::DATA_USAGE_BLOOM_RECOVERY_PATH;
use crate::data_usage_define::{DATA_USAGE_BLOOM_RECOVERY_PATH, DATA_USAGE_RECOVERY_PATH};
use crate::storage_api::owner::ObjectIO as _;
use tokio::io::AsyncReadExt as _;
@@ -23,6 +23,9 @@ const MAX_SCANNER_CYCLE_STATE_BYTES: u64 = 1024 * 1024;
pub(super) const MAX_SCANNER_CYCLE_RECOVERY_RETRIES: u32 = 5;
const METRIC_SCANNER_CYCLE_RECOVERY_REQUIRED: &str = "rustfs_scanner_cycle_recovery_required";
const METRIC_SCANNER_CYCLE_RECOVERY_RETRY_COUNT: &str = "rustfs_scanner_cycle_recovery_retry_count";
const USAGE_FLOOR_LOAD_FAILED: &str = "usage_floor_load_failed";
const LEGACY_EMPTY_USAGE_FLOOR_RECOVERY: &str = "legacy_empty_usage_floor";
const CACHE_CYCLE_AHEAD: &str = "cache_cycle_ahead";
#[derive(Clone, Debug, Default, Serialize)]
pub struct ScannerCycleRecoveryStatus {
@@ -38,6 +41,7 @@ pub struct ScannerCycleRecoveryStatus {
pub first_detected_at_unix_secs: Option<u64>,
pub last_attempt_at_unix_secs: Option<u64>,
pub retry_count: u64,
/// Maximum automatic retries, or zero when the recovery is unbounded.
pub max_retries: u32,
/// Whether the scanner may retry this state automatically.
pub retryable: bool,
@@ -62,7 +66,16 @@ pub fn scanner_cycle_recovery_status() -> ScannerCycleRecoveryStatus {
}
fn set_scanner_cycle_recovery_status(status: ScannerCycleRecoveryStatus) {
let recovery_required = if matches!(status.state.as_str(), "blocked" | "paused" | "recovery-required" | "cleanup-pending") {
let recovery_required = if matches!(
status.state.as_str(),
"blocked"
| "paused"
| "recovery-required"
| "cleanup-pending"
| "usage_floor_load_failed"
| "usage_floor_recovery_pending"
| "cache_cycle_ahead"
) {
1.0
} else {
0.0
@@ -74,11 +87,120 @@ fn set_scanner_cycle_recovery_status(status: ScannerCycleRecoveryStatus) {
.unwrap_or_else(|poisoned| poisoned.into_inner()) = status;
}
pub(super) fn record_scanner_usage_floor_failure(reason: String) {
let previous = scanner_cycle_recovery_status();
let same_failure = previous.classification.as_deref() == Some(USAGE_FLOOR_LOAD_FAILED);
let now = unix_now_secs();
let (first_detected_at_unix_secs, retry_count) = if same_failure {
(previous.first_detected_at_unix_secs.or(Some(now)), previous.retry_count)
} else {
(Some(now), 0)
};
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_OBJ_NAME_PATH.clone(),
state: USAGE_FLOOR_LOAD_FAILED.to_string(),
classification: Some(USAGE_FLOOR_LOAD_FAILED.to_string()),
first_detected_at_unix_secs,
last_attempt_at_unix_secs: Some(now),
retry_count,
max_retries: MAX_SCANNER_CYCLE_RECOVERY_RETRIES,
retryable: true,
reason: Some(reason),
..Default::default()
});
}
pub(super) fn clear_scanner_usage_floor_failure() {
if scanner_cycle_recovery_status().classification.as_deref() == Some(USAGE_FLOOR_LOAD_FAILED) {
set_scanner_cycle_recovery_status(recovery_status("healthy", None, false));
}
}
pub(super) fn record_scanner_cache_cycle_ahead(requested_cycle: u64, required_cycle: u64, leader_epoch: u64) {
let previous = scanner_cycle_recovery_status();
let same_floor = previous.classification.as_deref() == Some(CACHE_CYCLE_AHEAD)
&& previous.generation == Some(required_cycle)
&& previous.leader_epoch == Some(leader_epoch);
let now = unix_now_secs();
let (first_detected_at_unix_secs, retry_count) = if same_floor {
(previous.first_detected_at_unix_secs.or(Some(now)), previous.retry_count)
} else {
(Some(now), 0)
};
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_BLOOM_NAME_PATH.clone(),
state: CACHE_CYCLE_AHEAD.to_string(),
classification: Some(CACHE_CYCLE_AHEAD.to_string()),
generation: Some(required_cycle),
leader_epoch: Some(leader_epoch),
first_detected_at_unix_secs,
last_attempt_at_unix_secs: Some(now),
retry_count,
max_retries: 0,
retryable: true,
reason: Some(format!(
"persisted scanner cache cycle {required_cycle} is ahead of requested cycle {requested_cycle}"
)),
..Default::default()
});
}
pub(super) fn record_scanner_cache_cycle_recovery_attempt() {
let mut status = scanner_cycle_recovery_status();
if status.classification.as_deref() != Some(CACHE_CYCLE_AHEAD) {
return;
}
status.retry_count = status.retry_count.saturating_add(1);
status.last_attempt_at_unix_secs = Some(unix_now_secs());
set_scanner_cycle_recovery_status(status);
}
pub(super) fn clear_scanner_cache_cycle_ahead() {
if scanner_cycle_recovery_status().classification.as_deref() == Some(CACHE_CYCLE_AHEAD) {
set_scanner_cycle_recovery_status(recovery_status("healthy", None, false));
}
}
pub(super) fn record_legacy_empty_usage_floor_recovery_pending(leader_epoch: u64) {
let previous = scanner_cycle_recovery_status();
let same_recovery = previous.classification.as_deref() == Some(LEGACY_EMPTY_USAGE_FLOOR_RECOVERY)
&& previous.leader_epoch == Some(leader_epoch);
let now = unix_now_secs();
let (first_detected_at_unix_secs, retry_count) = if same_recovery {
(previous.first_detected_at_unix_secs.or(Some(now)), previous.retry_count)
} else {
(Some(now), 0)
};
set_scanner_cycle_recovery_status(ScannerCycleRecoveryStatus {
path: DATA_USAGE_OBJ_NAME_PATH.clone(),
quarantine_path: Some(DATA_USAGE_RECOVERY_PATH.clone()),
state: "usage_floor_recovery_pending".to_string(),
classification: Some(LEGACY_EMPTY_USAGE_FLOOR_RECOVERY.to_string()),
leader_epoch: Some(leader_epoch),
first_detected_at_unix_secs,
last_attempt_at_unix_secs: Some(now),
retry_count,
max_retries: MAX_SCANNER_CYCLE_RECOVERY_RETRIES,
retryable: true,
reason: Some("legacy empty usage floor recovery is awaiting a fenced leadership claim".to_string()),
..Default::default()
});
}
pub(super) fn clear_legacy_empty_usage_floor_recovery_status() {
if scanner_cycle_recovery_status().classification.as_deref() == Some(LEGACY_EMPTY_USAGE_FLOOR_RECOVERY) {
set_scanner_cycle_recovery_status(recovery_status("healthy", None, false));
}
}
pub(super) fn record_scanner_cycle_recovery_retry(attempt: u32) -> bool {
let mut status = scanner_cycle_recovery_status();
status.retry_count = u64::from(attempt);
if status.classification.as_deref() == Some(CACHE_CYCLE_AHEAD) {
return true;
}
status.retry_count = status.retry_count.max(u64::from(attempt));
status.last_attempt_at_unix_secs = Some(unix_now_secs());
if attempt >= MAX_SCANNER_CYCLE_RECOVERY_RETRIES {
if status.max_retries != 0 && status.retry_count >= u64::from(status.max_retries) {
status.state = "paused".to_string();
status.retryable = false;
status.reason = Some("scanner cycle recovery retry budget reached; sparse backend probes continue".to_string());
@@ -1185,6 +1307,246 @@ pub(super) enum PersistedUsageFloorStartup {
Authoritative,
Missing,
BootstrapPending,
RecoveredLegacyEmptyFence,
}
#[derive(Clone, Debug)]
struct LegacyEmptyUsageFloorPrimary {
revision: DataUsageCacheRevision,
epoch: u64,
}
#[derive(Clone, Debug, Serialize, Deserialize, PartialEq, Eq)]
#[serde(deny_unknown_fields)]
struct LegacyEmptyUsageFloorRecoveryMarker {
schema_version: u16,
primary_revision: String,
leader_epoch: u64,
}
async fn read_legacy_empty_usage_floor_recovery_marker(
storeapi: Arc<impl ScannerObjectIO>,
) -> Result<Option<(LegacyEmptyUsageFloorRecoveryMarker, DataUsageCacheRevision)>, ScannerError> {
let (data, revision) = read_config_with_revision(storeapi, DATA_USAGE_RECOVERY_PATH.as_str())
.await
.map_err(|err| ScannerError::Other(format!("failed to read scanner usage recovery marker: {err}")))?;
let Some(data) = data else {
return Ok(None);
};
let marker = serde_json::from_slice::<LegacyEmptyUsageFloorRecoveryMarker>(&data)
.map_err(|err| ScannerError::Other(format!("failed to decode scanner usage recovery marker: {err}")))?;
if marker.schema_version != 1 || marker.primary_revision.is_empty() || marker.leader_epoch == 0 {
return Err(ScannerError::Other("scanner usage recovery marker is invalid".to_string()));
}
if !matches!(revision, DataUsageCacheRevision::Etag(_)) {
return Err(ScannerError::Other("scanner usage recovery marker has no revision".to_string()));
}
Ok(Some((marker, revision)))
}
async fn clear_legacy_empty_usage_floor_recovery_marker(
storeapi: Arc<impl ScannerObjectIO + ScannerConfigObjectDelete>,
marker_revision: &DataUsageCacheRevision,
expected_publication_epoch: u64,
) -> Result<(), ScannerError> {
let delete_result = delete_config_with_publication_admission_for_epoch(
storeapi.clone(),
RUSTFS_META_BUCKET,
DATA_USAGE_RECOVERY_PATH.as_str(),
ScannerObjectOptions {
delete_prefix: false,
http_preconditions: Some(marker_revision.preconditions()),
..Default::default()
},
expected_publication_epoch,
)
.await;
match delete_result {
Ok(_) => Ok(()),
Err(err) => {
let (_, revision) = read_config_with_revision(storeapi, DATA_USAGE_RECOVERY_PATH.as_str())
.await
.map_err(|read_err| {
ScannerError::Other(format!("failed to reconcile scanner usage recovery cleanup: {read_err}"))
})?;
if matches!(revision, DataUsageCacheRevision::Missing) {
Ok(())
} else {
Err(ScannerError::Other(format!("failed to clear scanner usage recovery marker: {err}")))
}
}
}
}
pub(super) async fn complete_legacy_empty_usage_floor_recovery(
storeapi: Arc<impl ScannerObjectIO + ScannerConfigObjectDelete>,
claimed_epoch: u64,
) -> Result<(), ScannerError> {
let Some((marker, marker_revision)) = read_legacy_empty_usage_floor_recovery_marker(storeapi.clone()).await? else {
return Ok(());
};
if claimed_epoch <= marker.leader_epoch {
return Err(ScannerError::Other("scanner usage recovery did not advance the leader epoch".to_string()));
}
let (primary, _) = read_config_with_revision(storeapi.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.map_err(|err| ScannerError::Other(format!("failed to verify recovered scanner usage bootstrap: {err}")))?;
let primary = primary.ok_or_else(|| ScannerError::Other("recovered scanner usage bootstrap is missing".to_string()))?;
let usage = serde_json::from_slice::<DataUsageInfo>(&primary)
.map_err(|err| ScannerError::Other(format!("failed to decode recovered scanner usage bootstrap: {err}")))?;
if !data_usage_info_is_bootstrap_pending(&usage) || usage.scanner_epoch != Some(claimed_epoch) {
return Err(ScannerError::Other(
"recovered scanner usage bootstrap does not match the claimed epoch".to_string(),
));
}
let expected_publication_epoch = scanner_publication_epoch(storeapi.clone())
.await
.ok_or_else(|| ScannerError::Other("scanner usage recovery cleanup is blocked by data movement".to_string()))?;
clear_legacy_empty_usage_floor_recovery_marker(storeapi, &marker_revision, expected_publication_epoch).await?;
clear_legacy_empty_usage_floor_recovery_status();
Ok(())
}
fn legacy_empty_usage_fence_epoch(data: &[u8], usage: &DataUsageInfo) -> Option<Option<u64>> {
if usage.last_update.is_none() || usage.scanner_cycle.is_some() {
return None;
}
if usage.scanner_epoch.is_some_and(|epoch| epoch == 0 || epoch >= u64::MAX - 1) {
return None;
}
let expected = DataUsageInfo {
last_update: usage.last_update,
scanner_epoch: usage.scanner_epoch,
..Default::default()
};
if usage != &expected {
return None;
}
let serde_json::Value::Object(fields) = serde_json::from_slice::<serde_json::Value>(data).ok()? else {
return None;
};
// RUSTFS_COMPAT_TODO(backlog-2102): accept only the exact empty usage fence serialized by rc.2/rc.3. Remove after those releases are no longer supported direct-upgrade sources.
const REQUIRED_FIELDS: &[&str] = &[
"total_capacity",
"total_used_capacity",
"total_free_capacity",
"last_update",
"objects_total_count",
"versions_total_count",
"delete_markers_total_count",
"objects_total_size",
"replication_info",
"buckets_count",
"buckets_usage",
"usage_snapshot_complete",
"bucket_sizes",
"disk_usage_status",
];
let expected_len = REQUIRED_FIELDS.len() + if usage.scanner_epoch.is_some() { 1 } else { 0 };
if fields.len() != expected_len
|| REQUIRED_FIELDS.iter().any(|field| !fields.contains_key(*field))
|| (usage.scanner_epoch.is_some() != fields.contains_key("scanner_epoch"))
{
return None;
}
Some(usage.scanner_epoch)
}
async fn recover_legacy_empty_usage_floor(
storeapi: Arc<impl ScannerObjectIO + ScannerConfigObjectDelete>,
primary: LegacyEmptyUsageFloorPrimary,
expected_publication_epoch: u64,
) -> Result<(), ScannerError> {
let DataUsageCacheRevision::Etag(primary_revision) = &primary.revision else {
return Err(ScannerError::Other("legacy empty scanner usage floor has no revision".to_string()));
};
let marker = LegacyEmptyUsageFloorRecoveryMarker {
schema_version: 1,
primary_revision: primary_revision.clone(),
leader_epoch: primary.epoch,
};
let marker_data = serde_json::to_vec(&marker)
.map_err(|err| ScannerError::Other(format!("failed to encode scanner usage recovery marker: {err}")))?;
match read_legacy_empty_usage_floor_recovery_marker(storeapi.clone()).await? {
Some((persisted, _)) if persisted != marker => {
return Err(ScannerError::Other(
"scanner usage recovery marker conflicts with the persisted empty floor".to_string(),
));
}
Some(_) => {}
None => {
let marker_save = save_config_with_publication_admission_for_epoch(
storeapi.clone(),
DATA_USAGE_RECOVERY_PATH.as_str(),
marker_data.clone(),
DataUsageCacheRevision::Missing.preconditions(),
expected_publication_epoch,
)
.await;
if !marker_save
.as_ref()
.ok()
.and_then(|info| info.etag.as_deref())
.is_some_and(|etag| !etag.is_empty())
{
let persisted = read_legacy_empty_usage_floor_recovery_marker(storeapi.clone()).await?;
if persisted.as_ref().map(|(persisted, _)| persisted) != Some(&marker) {
return Err(ScannerError::Other(match marker_save {
Ok(_) => "scanner usage recovery marker returned no ETag and could not be confirmed".to_string(),
Err(err) => format!("failed to persist scanner usage recovery marker: {err}"),
}));
}
}
}
}
let marker = DataUsageInfo {
last_update: Some(std::time::SystemTime::now()),
scanner_epoch: Some(primary.epoch),
usage_snapshot_converged: Some(false),
usage_snapshot_bootstrap_pending: true,
..Default::default()
};
let data = serde_json::to_vec(&marker)
.map_err(|err| ScannerError::Other(format!("failed to encode recovered scanner usage bootstrap: {err}")))?;
let save_result = save_config_with_publication_admission_for_epoch(
storeapi.clone(),
DATA_USAGE_OBJ_NAME_PATH.as_str(),
data.clone(),
primary.revision.preconditions(),
expected_publication_epoch,
)
.await;
if save_result
.as_ref()
.ok()
.and_then(|info| info.etag.as_deref())
.is_some_and(|etag| !etag.is_empty())
{
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
state = "legacy_empty_usage_floor_recovered",
path = %DATA_USAGE_OBJ_NAME_PATH.as_str(),
scanner_epoch = primary.epoch,
"Scanner recovered a legacy empty usage floor"
);
return Ok(());
}
let (persisted, revision) = read_config_with_revision(storeapi, DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.map_err(|err| ScannerError::Other(format!("failed to reconcile recovered scanner usage bootstrap: {err}")))?;
if persisted.as_deref() == Some(data.as_slice()) && matches!(revision, DataUsageCacheRevision::Etag(_)) {
return Ok(());
}
Err(ScannerError::Other(match save_result {
Ok(_) => "recovered scanner usage bootstrap returned no ETag and could not be confirmed".to_string(),
Err(err) => format!("failed to recover legacy empty scanner usage floor: {err}"),
}))
}
pub(super) fn encode_scanner_cycle_state(
@@ -1324,16 +1686,22 @@ pub(super) async fn persisted_usage_floor_for_startup(
let Some(read_epoch) = scanner_publication_epoch(storeapi.clone()).await else {
return Err(ScannerError::Other("scanner usage floor read is blocked by data movement".to_string()));
};
let recovery_marker = read_legacy_empty_usage_floor_recovery_marker(storeapi.clone()).await?;
let mut floor = PersistedUsageFloor::default();
let mut found_any = false;
let mut bootstrap_pending = false;
let mut recovered_bootstrap = false;
let mut bootstrap_epoch = None;
// A valid JSON object without a baseline identity is not a floor and must
// never be treated as an empty one. It can, however, be a partially
// written v2 primary left behind during an upgrade. Keep its epoch as a
// fence while looking for a durable companion snapshot; if no companion
// is new enough, the caller still fails closed below.
let mut invalid_baseline_path: Option<String> = None;
let mut invalid_baseline_epoch: Option<u64> = None;
let mut invalid_baseline_epoch = recovery_marker.as_ref().map(|(marker, _)| marker.leader_epoch);
let mut unrecoverable_baseline_path: Option<String> = None;
let mut stale_authoritative_path: Option<String> = None;
let mut legacy_empty_primary: Option<LegacyEmptyUsageFloorPrimary> = None;
let update_floor = |floor: &mut PersistedUsageFloor, usage: &DataUsageInfo, path: &str| -> Result<(), ScannerError> {
floor.leader_epoch = floor.leader_epoch.max(usage.scanner_epoch.unwrap_or_default());
if let Some(completed_cycle) = usage.scanner_cycle {
@@ -1348,8 +1716,9 @@ pub(super) async fn persisted_usage_floor_for_startup(
for primary_path in [DATA_USAGE_OBJ_NAME_PATH.as_str(), LEGACY_DATA_USAGE_OBJ_NAME_PATH.as_str()] {
let backup_path = format!("{primary_path}.bkp");
let is_v2_path = primary_path == DATA_USAGE_OBJ_NAME_PATH.as_str();
let mut recovered_primary_companion_epoch = None;
let primary_epoch = match read_config_with_revision(storeapi.clone(), primary_path).await {
Ok((Some(data), _)) => {
Ok((Some(data), revision)) => {
let usage = serde_json::from_slice::<DataUsageInfo>(&data).map_err(|err| {
ScannerError::Other(format!("failed to decode scanner usage floor from {primary_path}: {err}"))
})?;
@@ -1358,18 +1727,47 @@ pub(super) async fn persisted_usage_floor_for_startup(
return Err(ScannerError::Other("multiple scanner usage bootstrap markers were found".to_string()));
}
bootstrap_pending = true;
bootstrap_epoch = usage.scanner_epoch;
if let Some((marker, _)) = recovery_marker.as_ref() {
if usage.scanner_epoch.is_none_or(|epoch| epoch < marker.leader_epoch) {
return Err(ScannerError::Other(
"scanner usage bootstrap is older than its recovery marker".to_string(),
));
}
recovered_bootstrap = true;
}
update_floor(&mut floor, &usage, primary_path)?;
None
} else if !data_usage_info_has_persisted_baseline_identity(&usage) {
invalid_baseline_path.get_or_insert_with(|| primary_path.to_string());
invalid_baseline_epoch = invalid_baseline_epoch.max(usage.scanner_epoch);
match legacy_empty_usage_fence_epoch(&data, &usage) {
Some(Some(epoch)) if is_v2_path => {
legacy_empty_primary = Some(LegacyEmptyUsageFloorPrimary { revision, epoch });
}
Some(_) => {}
None => {
unrecoverable_baseline_path.get_or_insert_with(|| primary_path.to_string());
}
}
None
} else {
let epoch = usage.scanner_epoch.unwrap_or_default();
if recovered_bootstrap && !is_v2_path {
if epoch < floor.leader_epoch {
unrecoverable_baseline_path.get_or_insert_with(|| primary_path.to_string());
} else {
update_floor(&mut floor, &usage, primary_path)?;
recovered_primary_companion_epoch = Some(epoch);
}
None
// A legacy snapshot may be structurally valid but older
// than an incomplete v2 snapshot left by a newer leader.
// Do not let that candidate regress the startup floor.
if !is_v2_path && invalid_baseline_epoch.is_some_and(|fenced_epoch| epoch < fenced_epoch) {
} else if invalid_baseline_epoch.is_some_and(|fenced_epoch| epoch < fenced_epoch)
&& (!is_v2_path || recovery_marker.is_some())
{
stale_authoritative_path.get_or_insert_with(|| primary_path.to_string());
None
} else {
update_floor(&mut floor, &usage, primary_path)?;
@@ -1387,17 +1785,51 @@ pub(super) async fn persisted_usage_floor_for_startup(
let mut any_found = primary_epoch.is_some();
match read_config_with_revision(storeapi.clone(), &backup_path).await {
Ok((Some(data), _)) => {
let usage = serde_json::from_slice::<DataUsageInfo>(&data).map_err(|err| {
ScannerError::Other(format!("failed to decode scanner usage floor from {backup_path}: {err}"))
})?;
if bootstrap_pending {
if let Some(primary_epoch) = recovered_primary_companion_epoch {
if data_usage_info_has_persisted_baseline_identity(&usage) {
let backup_epoch = usage.scanner_epoch.unwrap_or_default();
if backup_epoch >= primary_epoch {
update_floor(&mut floor, &usage, &backup_path)?;
}
} else if let Some(epoch) = legacy_empty_usage_fence_epoch(&data, &usage) {
if let Some(epoch) = epoch {
floor.leader_epoch = floor.leader_epoch.max(epoch);
}
} else {
unrecoverable_baseline_path.get_or_insert_with(|| backup_path.clone());
}
continue;
}
let compatible_empty_fence = legacy_empty_usage_fence_epoch(&data, &usage).is_some_and(|epoch| {
epoch.is_none_or(|epoch| bootstrap_epoch.is_some_and(|bootstrap_epoch| epoch <= bootstrap_epoch))
});
if compatible_empty_fence {
continue;
}
if recovered_bootstrap && data_usage_info_has_persisted_baseline_identity(&usage) {
let epoch = usage.scanner_epoch.unwrap_or_default();
if epoch >= floor.leader_epoch {
update_floor(&mut floor, &usage, &backup_path)?;
if primary_path == DATA_USAGE_OBJ_NAME_PATH.as_str() {
break;
}
continue;
}
}
return Err(ScannerError::Other(
"scanner usage bootstrap conflicts with a persisted backup".to_string(),
));
}
let usage = serde_json::from_slice::<DataUsageInfo>(&data).map_err(|err| {
ScannerError::Other(format!("failed to decode scanner usage floor from {backup_path}: {err}"))
})?;
if !data_usage_info_has_persisted_baseline_identity(&usage) {
invalid_baseline_path.get_or_insert_with(|| backup_path.clone());
invalid_baseline_epoch = invalid_baseline_epoch.max(usage.scanner_epoch);
if legacy_empty_usage_fence_epoch(&data, &usage).is_none() {
unrecoverable_baseline_path.get_or_insert_with(|| backup_path.clone());
}
// This is still persisted state, so it must not enable a
// missing-state bootstrap. Continue to a legacy pair in
// case it contains a complete, fenced snapshot.
@@ -1411,6 +1843,8 @@ pub(super) async fn persisted_usage_floor_for_startup(
{
update_floor(&mut floor, &usage, &backup_path)?;
any_found = true;
} else {
stale_authoritative_path.get_or_insert_with(|| backup_path.clone());
}
}
}
@@ -1433,6 +1867,30 @@ pub(super) async fn persisted_usage_floor_for_startup(
}
if !found_any && !bootstrap_pending {
if allow_missing_for_bootstrap
&& unrecoverable_baseline_path.is_none()
&& stale_authoritative_path.is_none()
&& let Some(mut primary) = legacy_empty_primary
{
primary.epoch = primary.epoch.max(invalid_baseline_epoch.unwrap_or_default());
recover_legacy_empty_usage_floor(storeapi.clone(), primary.clone(), read_epoch).await?;
record_legacy_empty_usage_floor_recovery_pending(primary.epoch);
return Ok((
PersistedUsageFloor {
next_cycle: 0,
leader_epoch: primary.epoch,
},
PersistedUsageFloorStartup::RecoveredLegacyEmptyFence,
));
}
if let Some(path) = stale_authoritative_path {
return Err(ScannerError::Other(format!(
"persisted scanner usage floor from {path} is older than the required recovery fence"
)));
}
if recovery_marker.is_some() {
return Err(ScannerError::Other("scanner usage recovery marker has no matching primary".to_string()));
}
if let Some(path) = invalid_baseline_path {
return Err(ScannerError::Other(format!(
"persisted scanner usage floor from {path} has no authoritative baseline or newer valid backup"
@@ -1470,18 +1928,67 @@ pub(super) async fn persisted_usage_floor_for_startup(
}
drop(publication_admission);
}
let Some(_publication_admission) = scanner_publication_admission_for_epoch(storeapi, read_epoch).await else {
let Some(publication_admission) = scanner_publication_admission_for_epoch(storeapi.clone(), read_epoch).await else {
return Err(ScannerError::Other(
"scanner usage floor changed while its epoch proof was being confirmed".to_string(),
));
};
let state = if found_any {
PersistedUsageFloorStartup::Authoritative
} else if recovered_bootstrap {
if let Some(path) = unrecoverable_baseline_path {
return Err(ScannerError::Other(format!(
"scanner usage recovery conflicts with persisted usage state at {path}"
)));
}
let recovery_epoch = recovery_marker
.as_ref()
.map(|(marker, _)| marker.leader_epoch)
.unwrap_or(floor.leader_epoch);
record_legacy_empty_usage_floor_recovery_pending(recovery_epoch);
PersistedUsageFloorStartup::RecoveredLegacyEmptyFence
} else if bootstrap_pending {
if let Some(path) = unrecoverable_baseline_path {
return Err(ScannerError::Other(format!(
"scanner usage bootstrap conflicts with persisted usage state at {path}"
)));
}
PersistedUsageFloorStartup::BootstrapPending
} else {
PersistedUsageFloorStartup::Missing
};
if found_any && let Some((_, marker_revision)) = recovery_marker.as_ref() {
drop(publication_admission);
let marker_cleared =
match clear_legacy_empty_usage_floor_recovery_marker(storeapi.clone(), marker_revision, read_epoch).await {
Ok(()) => true,
Err(err) => {
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
state = "usage_floor_recovery_cleanup_deferred",
path = %DATA_USAGE_RECOVERY_PATH.as_str(),
error = %err,
"Scanner usage floor recovery marker cleanup was deferred"
);
false
}
};
let Some(_final_publication_admission) = scanner_publication_admission_for_epoch(storeapi.clone(), read_epoch).await
else {
return Err(ScannerError::Other(
"scanner usage floor changed after recovery marker cleanup".to_string(),
));
};
if marker_cleared {
clear_legacy_empty_usage_floor_recovery_status();
}
clear_scanner_usage_floor_failure();
return Ok((floor, state));
}
clear_scanner_usage_floor_failure();
Ok((floor, state))
}
+28 -10
View File
@@ -21,6 +21,29 @@ pub(super) enum ScannerLeadershipClaimReconcile {
Unchanged,
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(super) enum ScannerCycleResetPolicy {
None,
ResetAll,
ResetCoveragePreservingNext,
}
impl ScannerCycleResetPolicy {
fn apply(self, cycle_info: &mut CurrentCycle, attempted_next: u64) {
match self {
Self::None => {}
Self::ResetAll => *cycle_info = CurrentCycle::default(),
Self::ResetCoveragePreservingNext => {
let next = cycle_info.next.max(attempted_next);
*cycle_info = CurrentCycle {
next,
..Default::default()
};
}
}
}
}
pub(super) async fn reconcile_scanner_leadership_claim(
storeapi: Arc<impl ScannerObjectIO>,
attempted: &[u8],
@@ -329,7 +352,7 @@ pub(super) async fn claim_scanner_leadership(
revision: &mut DataUsageCacheRevision,
persisted_epoch: &mut u64,
allow_bootstrap_pending: bool,
reset_bootstrap_cycle_on_conflict: bool,
cycle_reset_policy: ScannerCycleResetPolicy,
) -> bool {
for retry in 0..=SCANNER_PERSIST_CAS_RETRIES {
if ctx.is_cancelled() {
@@ -347,6 +370,7 @@ pub(super) async fn claim_scanner_leadership(
);
return false;
};
let attempted_next = cycle_info.next;
let data = match encode_scanner_cycle_state(cycle_info, claimed_epoch) {
Ok(data) => data,
Err(err) => {
@@ -459,9 +483,7 @@ pub(super) async fn claim_scanner_leadership(
.await;
}
Ok(ScannerLeadershipClaimReconcile::Changed) if retry < SCANNER_PERSIST_CAS_RETRIES => {
if reset_bootstrap_cycle_on_conflict {
*cycle_info = CurrentCycle::default();
}
cycle_reset_policy.apply(cycle_info, attempted_next);
continue;
}
Ok(ScannerLeadershipClaimReconcile::Changed | ScannerLeadershipClaimReconcile::Unchanged) => {
@@ -517,17 +539,13 @@ pub(super) async fn claim_scanner_leadership(
Ok(ScannerLeadershipClaimReconcile::Changed)
if retry < SCANNER_PERSIST_CAS_RETRIES && !ctx.is_cancelled() =>
{
if reset_bootstrap_cycle_on_conflict {
*cycle_info = CurrentCycle::default();
}
cycle_reset_policy.apply(cycle_info, attempted_next);
continue;
}
Ok(ScannerLeadershipClaimReconcile::Unchanged)
if precondition_failed && retry < SCANNER_PERSIST_CAS_RETRIES && !ctx.is_cancelled() =>
{
if reset_bootstrap_cycle_on_conflict {
*cycle_info = CurrentCycle::default();
}
cycle_reset_policy.apply(cycle_info, attempted_next);
continue;
}
Ok(ScannerLeadershipClaimReconcile::Changed | ScannerLeadershipClaimReconcile::Unchanged) => {
File diff suppressed because it is too large Load Diff
+11 -1
View File
@@ -420,7 +420,17 @@ where
break 'updates;
};
let authoritative = match serde_json::from_slice::<DataUsageInfo>(&authoritative_data) {
Ok(info) if data_usage_info_has_persisted_baseline_identity(&info) => info,
// The bootstrap placeholder is a valid baseline identity: on a
// site that has never converged (every cycle superseded by a
// sustained write stream, #6852) it is the only authoritative
// object that will ever exist, and refusing it here means the
// observed snapshot — the only usage data such a site can
// produce — is never published at all.
Ok(info)
if data_usage_info_has_persisted_baseline_identity(&info) || data_usage_info_is_bootstrap_pending(&info) =>
{
info
}
Ok(_) => {
error!(
target: "rustfs::scanner",