mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-07 20:46:11 +00:00
fix(ilm): reconcile lifecycle rules with current evaluation
This commit is contained in:
@@ -130,6 +130,21 @@ Scanner cycle budget controls:
|
||||
- timeout returns S3 `SlowDown`, so clients should use normal SDK retry handling.
|
||||
- this is not a fdatasync or group-commit switch. Track fdatasync batching separately with `rustfs_s3_put_object_rename_fdatasync_batch_files`.
|
||||
|
||||
## Remote tier timeout environment variables
|
||||
|
||||
- `RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS`
|
||||
- remote tier TCP connect timeout.
|
||||
- default is `10`.
|
||||
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default.
|
||||
- `RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS`
|
||||
- remote tier request timeout through response headers.
|
||||
- default is `86400` so large transition uploads keep a production-safe budget.
|
||||
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default. Very large values are accepted and act as a correspondingly long budget.
|
||||
- `RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS`
|
||||
- maximum idle time between remote tier response-body chunks.
|
||||
- default is `60`; the timer resets only when non-empty body data keeps progressing.
|
||||
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default.
|
||||
|
||||
## Drive timeout environment variables
|
||||
|
||||
- `RUSTFS_DRIVE_METADATA_TIMEOUT_SECS`
|
||||
|
||||
@@ -137,6 +137,28 @@ pub const DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED: bool = false;
|
||||
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE);
|
||||
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED);
|
||||
|
||||
/// Environment variable for remote tier TCP connect timeout in seconds.
|
||||
pub const ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS";
|
||||
/// Default remote tier TCP connect timeout in seconds.
|
||||
pub const DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS: u64 = 10;
|
||||
|
||||
/// Environment variable for the remote tier request timeout in seconds.
|
||||
///
|
||||
/// This bounds upload/download request progress through response headers. The
|
||||
/// default is intentionally large so multi-TiB transition uploads keep their
|
||||
/// previous production budget while black-hole remotes no longer wait forever.
|
||||
pub const ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS";
|
||||
/// Default remote tier request timeout in seconds.
|
||||
pub const DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS: u64 = 24 * 60 * 60;
|
||||
|
||||
/// Environment variable for remote tier response-body idle timeout in seconds.
|
||||
///
|
||||
/// The timer is re-armed on every non-empty response-body chunk, so slow but
|
||||
/// progressing remotes can continue while silent response bodies are cancelled.
|
||||
pub const ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS";
|
||||
/// Default remote tier response-body idle timeout in seconds.
|
||||
pub const DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS: u64 = 60;
|
||||
|
||||
/// Request the object-transaction fencing contract used by storage-owned
|
||||
/// cleanup receipts and lock-window optimizations.
|
||||
///
|
||||
@@ -812,6 +834,16 @@ mod remote_version_state_tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_tier_timeout_env_names_are_stable() {
|
||||
assert_eq!(super::ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS, "RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS");
|
||||
assert_eq!(super::ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS, "RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS");
|
||||
assert_eq!(
|
||||
super::ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
|
||||
"RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn data_movement_part_checksum_gate_uses_stable_environment_names() {
|
||||
assert_eq!(super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE, "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE");
|
||||
|
||||
@@ -168,9 +168,10 @@ pub mod bucket {
|
||||
idle_guarded_body,
|
||||
};
|
||||
pub use crate::bucket::on_demand_migration::{
|
||||
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListThroughCursor, ListThroughMerger, ListThroughToken,
|
||||
ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MergeOutcome, MergePick, MergeSide, SOURCE_LIST_MAX_RATE_WAIT,
|
||||
SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, decode_continuation_token, source_list_plan,
|
||||
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListPageError, ListThroughCursor, ListThroughMerger,
|
||||
ListThroughToken, ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MAX_LIST_NO_PROGRESS_PAGES, MergeOutcome,
|
||||
MergePick, MergeSide, SOURCE_LIST_MAX_RATE_WAIT, SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter,
|
||||
decode_continuation_token, source_list_plan,
|
||||
};
|
||||
pub mod backfill {
|
||||
pub use crate::bucket::on_demand_migration::backfill::{
|
||||
|
||||
@@ -25,8 +25,13 @@ use parking_lot::Mutex;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
/// The only continuation-token envelope version this build reads and writes.
|
||||
/// The continuation-token version used by ordinary progressing pages.
|
||||
pub const LIST_THROUGH_TOKEN_VERSION: u32 = 1;
|
||||
const LIST_THROUGH_PROGRESS_TOKEN_VERSION: u32 = 2;
|
||||
|
||||
/// The sixteenth consecutive merged page without a key or new EOF fails.
|
||||
/// This also bounds legitimate sparse listings; it is not a cycle detector.
|
||||
pub const MAX_LIST_NO_PROGRESS_PAGES: u8 = 16;
|
||||
|
||||
/// Envelope marker. A bucket that is *not* merging hands out the local
|
||||
/// listing's own marker, so the decoder needs a positive signal before it
|
||||
@@ -111,6 +116,10 @@ pub struct ListThroughToken {
|
||||
/// common prefix compares as itself, never as its members.
|
||||
#[serde(default)]
|
||||
pub last_key: Option<String>,
|
||||
/// Consecutive empty truncated merged pages, present only in v2 tokens.
|
||||
/// Ordinary v1 tokens retain their original serialized shape.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub no_progress: Option<u8>,
|
||||
}
|
||||
|
||||
impl ListThroughToken {
|
||||
@@ -123,6 +132,7 @@ impl ListThroughToken {
|
||||
source: source.token,
|
||||
source_done: source.done,
|
||||
last_key,
|
||||
no_progress: None,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -170,7 +180,21 @@ pub fn decode_continuation_token(decoded: &str) -> Result<ListThroughCursor, Lis
|
||||
return Ok(ListThroughCursor::Local(decoded.to_string()));
|
||||
}
|
||||
match value.get("v").and_then(serde_json::Value::as_u64) {
|
||||
Some(version) if version == u64::from(LIST_THROUGH_TOKEN_VERSION) => {}
|
||||
Some(version) if version == u64::from(LIST_THROUGH_TOKEN_VERSION) => {
|
||||
// v1 readers reject this field even when it is null or zero.
|
||||
if value.get("no_progress").is_some() {
|
||||
return Err(ListThroughTokenError::Malformed);
|
||||
}
|
||||
}
|
||||
Some(version) if version == u64::from(LIST_THROUGH_PROGRESS_TOKEN_VERSION) => {
|
||||
if !value
|
||||
.get("no_progress")
|
||||
.and_then(serde_json::Value::as_u64)
|
||||
.is_some_and(|count| (1..u64::from(MAX_LIST_NO_PROGRESS_PAGES)).contains(&count))
|
||||
{
|
||||
return Err(ListThroughTokenError::Malformed);
|
||||
}
|
||||
}
|
||||
Some(version) => return Err(ListThroughTokenError::UnsupportedVersion(version.min(u64::from(u32::MAX)) as u32)),
|
||||
None => return Err(ListThroughTokenError::Malformed),
|
||||
}
|
||||
@@ -288,6 +312,8 @@ pub enum ListPageError {
|
||||
Empty,
|
||||
#[error("truncated listing repeats a continuation token")]
|
||||
Repeated,
|
||||
#[error("listing exhausted its consecutive no-progress page budget")]
|
||||
NoProgress(MergeSide),
|
||||
}
|
||||
|
||||
pub(crate) fn validate_list_page(is_truncated: bool, token: Option<&str>, next_token: Option<&str>) -> Result<(), ListPageError> {
|
||||
@@ -352,6 +378,7 @@ pub struct MergeOutcome {
|
||||
#[derive(Debug)]
|
||||
pub struct ListThroughMerger {
|
||||
max_keys: usize,
|
||||
no_progress: Option<u8>,
|
||||
last_key: Option<String>,
|
||||
local: SideState,
|
||||
source: SideState,
|
||||
@@ -371,6 +398,7 @@ impl ListThroughMerger {
|
||||
};
|
||||
Self {
|
||||
max_keys,
|
||||
no_progress: token.and_then(|token| token.no_progress),
|
||||
last_key,
|
||||
local,
|
||||
source,
|
||||
@@ -436,13 +464,18 @@ impl ListThroughMerger {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn finish(self) -> MergeOutcome {
|
||||
/// `issue_progress_tokens` allows a v1 chain to start carrying a budget.
|
||||
/// An existing v2 budget is always enforced, including on reader-only nodes.
|
||||
/// Borrowing lets a source failure re-merge the fetched local buffers.
|
||||
pub fn finish(&self, issue_progress_tokens: bool) -> Result<MergeOutcome, ListPageError> {
|
||||
let Self {
|
||||
max_keys,
|
||||
no_progress,
|
||||
last_key,
|
||||
local,
|
||||
source,
|
||||
} = self;
|
||||
let max_keys = *max_keys;
|
||||
|
||||
// A side with more pages behind it can only be trusted up to the last
|
||||
// key it handed over: past that horizon the other side's entries could
|
||||
@@ -508,12 +541,44 @@ impl ListThroughMerger {
|
||||
let source_left = !source.disabled && (!source_cursor.done || consumed_source < source.entries.len());
|
||||
let is_truncated = local_left || source_left;
|
||||
|
||||
let last_key = consumed_key.or(last_key);
|
||||
MergeOutcome {
|
||||
let reached_eof = (!local.start.done && local_cursor.done) || (!source.start.done && source_cursor.done);
|
||||
let next_no_progress = if !is_truncated || !picks.is_empty() || reached_eof {
|
||||
None
|
||||
} else if max_keys == 0 {
|
||||
// A zero-sized request cannot consume entries. Preserve an existing
|
||||
// budget without spending it or starting a new one.
|
||||
*no_progress
|
||||
} else if issue_progress_tokens || no_progress.is_some() {
|
||||
let count = no_progress.unwrap_or(0).saturating_add(1);
|
||||
if count >= MAX_LIST_NO_PROGRESS_PAGES {
|
||||
// An empty truncated side closes the merge horizon. Local
|
||||
// failure takes precedence; disabling the source cannot fix it.
|
||||
let side = if local.more && local.entries.is_empty() {
|
||||
MergeSide::Local
|
||||
} else if !source.disabled && source.more && source.entries.is_empty() {
|
||||
MergeSide::Source
|
||||
} else {
|
||||
MergeSide::Local
|
||||
};
|
||||
return Err(ListPageError::NoProgress(side));
|
||||
}
|
||||
Some(count)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let last_key = consumed_key.or_else(|| last_key.clone());
|
||||
Ok(MergeOutcome {
|
||||
picks,
|
||||
is_truncated,
|
||||
next_token: is_truncated.then(|| ListThroughToken::new(local_cursor, source_cursor, last_key)),
|
||||
}
|
||||
next_token: is_truncated.then(|| {
|
||||
let mut token = ListThroughToken::new(local_cursor, source_cursor, last_key);
|
||||
if let Some(count) = next_no_progress {
|
||||
token.v = LIST_THROUGH_PROGRESS_TOKEN_VERSION;
|
||||
token.no_progress = Some(count);
|
||||
}
|
||||
token
|
||||
}),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -641,7 +706,7 @@ mod tests {
|
||||
.push_page(fetch.side, kept, truncated, next)
|
||||
.expect("reference provider pages must advance");
|
||||
}
|
||||
let outcome = merger.finish();
|
||||
let outcome = merger.finish(false).expect("valid merge outcome");
|
||||
assert_eq!(outcome.is_truncated, outcome.next_token.is_some());
|
||||
if outcome.is_truncated {
|
||||
assert_ne!(outcome.next_token, token, "every truncated merged page must make progress");
|
||||
@@ -724,7 +789,7 @@ mod tests {
|
||||
.push_page(MergeSide::Local, vec![ListEntryKey::object("a")], false, None)
|
||||
.expect("local EOF is valid");
|
||||
assert_eq!(merger.next_fetch(), None);
|
||||
let outcome = merger.finish();
|
||||
let outcome = merger.finish(false).expect("valid merge outcome");
|
||||
assert_eq!(outcome.picks.len(), 1);
|
||||
assert!(!outcome.is_truncated);
|
||||
assert!(outcome.next_token.is_none());
|
||||
@@ -740,6 +805,7 @@ mod tests {
|
||||
source: Some("source-1".to_string()),
|
||||
source_done: false,
|
||||
last_key: Some("a".to_string()),
|
||||
no_progress: None,
|
||||
};
|
||||
let mut merger = ListThroughMerger::new(1, Some(&resume));
|
||||
merger.disable_source();
|
||||
@@ -751,7 +817,7 @@ mod tests {
|
||||
Some("local-2".to_string()),
|
||||
)
|
||||
.expect("local cursor advances");
|
||||
let outcome = merger.finish();
|
||||
let outcome = merger.finish(false).expect("valid merge outcome");
|
||||
assert!(outcome.is_truncated);
|
||||
let token = outcome.next_token.expect("truncated page carries a token");
|
||||
assert_eq!(token.source.as_deref(), Some("source-1"), "the source cursor must not move");
|
||||
@@ -830,7 +896,7 @@ mod tests {
|
||||
.expect("opaque cursor advances regardless of sort order");
|
||||
}
|
||||
assert!(merger.next_fetch().is_none(), "two source fetches exhaust the request budget");
|
||||
let outcome = merger.finish();
|
||||
let outcome = merger.finish(false).expect("valid merge outcome");
|
||||
assert!(outcome.picks.is_empty());
|
||||
assert!(outcome.is_truncated);
|
||||
let token = outcome.next_token.expect("empty progressing page has a cursor");
|
||||
@@ -840,7 +906,7 @@ mod tests {
|
||||
merger
|
||||
.push_page(MergeSide::Source, vec![ListEntryKey::object("result")], false, None)
|
||||
.expect("source EOF");
|
||||
let outcome = merger.finish();
|
||||
let outcome = merger.finish(false).expect("valid merge outcome");
|
||||
assert_eq!(
|
||||
outcome.picks,
|
||||
vec![MergePick {
|
||||
@@ -887,7 +953,7 @@ mod tests {
|
||||
Err(ListPageError::Repeated)
|
||||
);
|
||||
merger.disable_source();
|
||||
let outcome = merger.finish();
|
||||
let outcome = merger.finish(false).expect("valid merge outcome");
|
||||
assert_eq!(
|
||||
outcome.picks,
|
||||
vec![MergePick {
|
||||
@@ -979,8 +1045,8 @@ mod tests {
|
||||
let encoded = token.encode();
|
||||
assert_eq!(decode_continuation_token(&encoded), Ok(ListThroughCursor::Merged(Box::new(token))));
|
||||
|
||||
let bumped = encoded.replace("\"v\":1", "\"v\":2");
|
||||
assert_eq!(decode_continuation_token(&bumped), Err(ListThroughTokenError::UnsupportedVersion(2)));
|
||||
let bumped = encoded.replace("\"v\":1", "\"v\":3");
|
||||
assert_eq!(decode_continuation_token(&bumped), Err(ListThroughTokenError::UnsupportedVersion(3)));
|
||||
|
||||
let extra = encoded.replace("{", "{\"x\":1,");
|
||||
assert_eq!(decode_continuation_token(&extra), Err(ListThroughTokenError::Malformed));
|
||||
@@ -992,6 +1058,257 @@ mod tests {
|
||||
assert_eq!(decode_continuation_token(no_version), Err(ListThroughTokenError::Malformed));
|
||||
}
|
||||
|
||||
fn progress_token(count: Option<u8>, local_done: bool, source_done: bool) -> ListThroughToken {
|
||||
let mut token = ListThroughToken::new(
|
||||
SideCursor {
|
||||
token: None,
|
||||
done: local_done,
|
||||
},
|
||||
SideCursor {
|
||||
token: Some("A".into()),
|
||||
done: source_done,
|
||||
},
|
||||
Some("last-key".into()),
|
||||
);
|
||||
if let Some(count) = count {
|
||||
token.v = LIST_THROUGH_PROGRESS_TOKEN_VERSION;
|
||||
token.no_progress = Some(count);
|
||||
}
|
||||
token
|
||||
}
|
||||
|
||||
fn push_empty_pages(merger: &mut ListThroughMerger, side: MergeSide) {
|
||||
for _ in 0..MAX_LIST_FETCHES_PER_SIDE {
|
||||
let fetch = merger.next_fetch().expect("empty truncated side must be fetched");
|
||||
assert_eq!(fetch.side, side);
|
||||
let next = format!("{}:next", fetch.token.unwrap_or_default());
|
||||
merger
|
||||
.push_page(side, vec![], true, Some(next))
|
||||
.expect("opaque cursor advances");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn progress_tokens_preserve_v1_bytes_and_validate_v2_counts() {
|
||||
let token = progress_token(None, true, false);
|
||||
assert_eq!(
|
||||
token.encode(),
|
||||
r#"{"t":"odm-list","v":1,"local":null,"local_done":true,"source":"A","source_done":false,"last_key":"last-key"}"#
|
||||
);
|
||||
for count in 1..MAX_LIST_NO_PROGRESS_PAGES {
|
||||
let token = progress_token(Some(count), true, false);
|
||||
assert_eq!(decode_continuation_token(&token.encode()), Ok(ListThroughCursor::Merged(Box::new(token))));
|
||||
}
|
||||
for version in [1, 2] {
|
||||
for value in ["null", "0", "16", "-1", "1.5", "256", "18446744073709551616", "\"1\""] {
|
||||
let encoded = format!(r#"{{"t":"odm-list","v":{version},"no_progress":{value}}}"#);
|
||||
assert_eq!(decode_continuation_token(&encoded), Err(ListThroughTokenError::Malformed), "{encoded}");
|
||||
}
|
||||
}
|
||||
for encoded in [
|
||||
r#"{"t":"odm-list","v":1,"no_progress":1}"#,
|
||||
r#"{"t":"odm-list","v":2}"#,
|
||||
r#"{"t":"odm-list","v":2,"no_progress":1,"extra":true}"#,
|
||||
] {
|
||||
assert_eq!(decode_continuation_token(encoded), Err(ListThroughTokenError::Malformed), "{encoded}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reader_only_nodes_do_not_start_a_budget_but_mixed_readers_preserve_one() {
|
||||
let mut token = progress_token(None, true, false);
|
||||
for _ in 0..MAX_LIST_NO_PROGRESS_PAGES {
|
||||
let mut merger = ListThroughMerger::new(2, Some(&token));
|
||||
push_empty_pages(&mut merger, MergeSide::Source);
|
||||
token = merger
|
||||
.finish(false)
|
||||
.expect("reader-only v1 behavior")
|
||||
.next_token
|
||||
.expect("truncated cursor");
|
||||
assert_eq!(token.v, 1);
|
||||
assert_eq!(token.no_progress, None);
|
||||
}
|
||||
for count in 1..=MAX_LIST_NO_PROGRESS_PAGES {
|
||||
let mut merger = ListThroughMerger::new(2, Some(&token));
|
||||
push_empty_pages(&mut merger, MergeSide::Source);
|
||||
assert!(merger.next_fetch().is_none(), "the per-request two-fetch limit stays intact");
|
||||
let outcome = merger.finish(count % 2 == 1);
|
||||
if count == MAX_LIST_NO_PROGRESS_PAGES {
|
||||
assert_eq!(outcome, Err(ListPageError::NoProgress(MergeSide::Source)));
|
||||
break;
|
||||
}
|
||||
token = outcome.expect("budget not exhausted").next_token.expect("truncated cursor");
|
||||
assert_eq!(token.no_progress, Some(count));
|
||||
let ListThroughCursor::Merged(decoded) = decode_continuation_token(&token.encode()).expect("round-trip v2") else {
|
||||
panic!("merged cursor expected");
|
||||
};
|
||||
token = *decoded;
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn objects_and_common_prefixes_reset_a_budget_at_the_boundary() {
|
||||
for entry in [ListEntryKey::object("result"), ListEntryKey::prefix("result/")] {
|
||||
for issue_tokens in [false, true] {
|
||||
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
|
||||
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||
merger
|
||||
.push_page(MergeSide::Source, vec![], true, Some("B".into()))
|
||||
.expect("empty advancing page");
|
||||
merger
|
||||
.push_page(MergeSide::Source, vec![entry.clone()], true, Some("C".into()))
|
||||
.expect("real progress");
|
||||
let outcome = merger
|
||||
.finish(issue_tokens)
|
||||
.expect("real progress does not exhaust the budget");
|
||||
assert_eq!(
|
||||
outcome.picks,
|
||||
vec![MergePick {
|
||||
side: MergeSide::Source,
|
||||
index: 0
|
||||
}]
|
||||
);
|
||||
let next = outcome.next_token.expect("source remains truncated");
|
||||
assert_eq!(next.last_key.as_deref(), Some(entry.name.as_str()));
|
||||
assert_eq!(next.v, 1);
|
||||
assert_eq!(next.no_progress, None);
|
||||
assert!(!next.encode().contains("no_progress"));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_a_new_eof_transition_resets_the_empty_page_budget() {
|
||||
for finished_side in [MergeSide::Local, MergeSide::Source] {
|
||||
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, false);
|
||||
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||
if finished_side == MergeSide::Local {
|
||||
merger
|
||||
.push_page(MergeSide::Local, vec![], false, None)
|
||||
.expect("new local EOF");
|
||||
push_empty_pages(&mut merger, MergeSide::Source);
|
||||
} else {
|
||||
push_empty_pages(&mut merger, MergeSide::Local);
|
||||
merger
|
||||
.push_page(MergeSide::Source, vec![], false, None)
|
||||
.expect("new source EOF");
|
||||
}
|
||||
let next = merger
|
||||
.finish(false)
|
||||
.expect("new EOF is progress")
|
||||
.next_token
|
||||
.expect("other side truncated");
|
||||
assert_eq!(next.no_progress, None);
|
||||
assert_eq!(next.v, 1);
|
||||
assert_eq!(next.local_done, finished_side == MergeSide::Local);
|
||||
assert_eq!(next.source_done, finished_side == MergeSide::Source);
|
||||
let mut merger = ListThroughMerger::new(2, Some(&next));
|
||||
let remaining = if finished_side == MergeSide::Local {
|
||||
MergeSide::Source
|
||||
} else {
|
||||
MergeSide::Local
|
||||
};
|
||||
push_empty_pages(&mut merger, remaining);
|
||||
let next = merger
|
||||
.finish(true)
|
||||
.expect("a new budget starts")
|
||||
.next_token
|
||||
.expect("truncated");
|
||||
assert_eq!(next.no_progress, Some(1), "an already-done side cannot reset every page");
|
||||
}
|
||||
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
|
||||
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||
merger.push_page(MergeSide::Source, vec![], false, None).expect("final EOF");
|
||||
let outcome = merger.finish(false).expect("EOF succeeds at the budget boundary");
|
||||
assert!(!outcome.is_truncated);
|
||||
assert!(outcome.next_token.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn filtered_duplicates_cannot_reset_the_no_progress_budget() {
|
||||
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
|
||||
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||
for next in ["B", "C"] {
|
||||
let entries = [ListEntryKey::object("last-key"), ListEntryKey::object("earlier")]
|
||||
.into_iter()
|
||||
.filter(|entry| merger.accepts(&entry.name))
|
||||
.collect::<Vec<_>>();
|
||||
assert!(entries.is_empty(), "both provider entries were already consumed");
|
||||
merger
|
||||
.push_page(MergeSide::Source, entries, true, Some(next.into()))
|
||||
.expect("advancing cursor");
|
||||
}
|
||||
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Source)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn no_progress_is_attributed_to_local_when_source_cannot_unblock_it() {
|
||||
for source_mode in ["disabled", "done", "empty", "data"] {
|
||||
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, source_mode == "done");
|
||||
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||
if source_mode == "disabled" {
|
||||
merger.disable_source();
|
||||
}
|
||||
push_empty_pages(&mut merger, MergeSide::Local);
|
||||
match source_mode {
|
||||
"empty" => push_empty_pages(&mut merger, MergeSide::Source),
|
||||
"data" => merger
|
||||
.push_page(MergeSide::Source, vec![ListEntryKey::object("source")], false, None)
|
||||
.expect("source data"),
|
||||
_ => {}
|
||||
}
|
||||
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Local)), "{source_mode}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn source_budget_failure_remerges_local_objects_and_prefixes_without_refetching() {
|
||||
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, false);
|
||||
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||
merger
|
||||
.push_page(MergeSide::Local, vec![ListEntryKey::object("local")], true, Some("L1".into()))
|
||||
.expect("local object");
|
||||
merger
|
||||
.push_page(MergeSide::Local, vec![ListEntryKey::prefix("prefix/")], true, Some("L2".into()))
|
||||
.expect("local prefix");
|
||||
push_empty_pages(&mut merger, MergeSide::Source);
|
||||
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Source)));
|
||||
merger.disable_source();
|
||||
assert!(merger.next_fetch().is_none(), "fallback does not perform another fetch");
|
||||
let outcome = merger.finish(false).expect("local data makes progress");
|
||||
assert_eq!(
|
||||
outcome.picks,
|
||||
vec![
|
||||
MergePick {
|
||||
side: MergeSide::Local,
|
||||
index: 0
|
||||
},
|
||||
MergePick {
|
||||
side: MergeSide::Local,
|
||||
index: 1
|
||||
}
|
||||
]
|
||||
);
|
||||
let token = outcome.next_token.expect("remaining local page");
|
||||
assert_eq!(token.local.as_deref(), Some("L2"));
|
||||
assert_eq!(token.source.as_deref(), Some("A"));
|
||||
assert_eq!(token.last_key.as_deref(), Some("prefix/"));
|
||||
assert_eq!(token.no_progress, None);
|
||||
assert_eq!(token.v, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_zero_sized_merge_preserves_an_existing_budget() {
|
||||
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
|
||||
let mut merger = ListThroughMerger::new(0, Some(&resume));
|
||||
merger
|
||||
.push_page(MergeSide::Source, vec![ListEntryKey::object("result")], true, Some("B".into()))
|
||||
.expect("source page");
|
||||
let outcome = merger.finish(false).expect("a zero-sized request cannot consume entries");
|
||||
assert!(outcome.picks.is_empty());
|
||||
assert_eq!(outcome.next_token.expect("unconsumed source").no_progress, resume.no_progress);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_plain_local_marker_stays_local() {
|
||||
assert_eq!(
|
||||
|
||||
@@ -40,9 +40,10 @@ pub use config::{
|
||||
SourceCredentials, SourceErrorPolicy, SourceTimeout, TlsConfig, ValidationContext,
|
||||
};
|
||||
pub use list_through::{
|
||||
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListThroughCursor, ListThroughMerger, ListThroughToken,
|
||||
ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MergeOutcome, MergePick, MergeSide, SOURCE_LIST_MAX_RATE_WAIT,
|
||||
SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, decode_continuation_token, source_list_plan,
|
||||
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListPageError, ListThroughCursor, ListThroughMerger,
|
||||
ListThroughToken, ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MAX_LIST_NO_PROGRESS_PAGES, MergeOutcome, MergePick,
|
||||
MergeSide, SOURCE_LIST_MAX_RATE_WAIT, SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter,
|
||||
decode_continuation_token, source_list_plan,
|
||||
};
|
||||
pub use negative_cache::{NEGATIVE_CACHE_MAX_ENTRIES, NegativeCache};
|
||||
pub use pull::{
|
||||
|
||||
@@ -3541,7 +3541,7 @@ impl TierConfigMgr {
|
||||
// Get tier configuration and create new driver
|
||||
let tier_config = self.tiers.get(tier_name).ok_or_else(|| ERR_TIER_NOT_FOUND.clone())?;
|
||||
|
||||
let driver = new_warm_backend(tier_config, false).await?;
|
||||
let driver = construct_warm_backend(tier_config).await?;
|
||||
|
||||
self.replace_driver(tier_name, driver)?;
|
||||
Ok(self
|
||||
@@ -4486,6 +4486,11 @@ impl TierConfigMgr {
|
||||
let committed_coordinator_intent =
|
||||
committed_tier_mutation_intent(coordinator_intent.as_ref(), &committed_config_etag)
|
||||
.map_err(TierConfigUpdateError::Save)?;
|
||||
// Persist Committed before notifying refresh; a Prepared disk record
|
||||
// would restore the prepared block and invalidate our publish allowance.
|
||||
let coordinator_commit =
|
||||
commit_coordinator_tier_mutation_intent(api.clone(), coordinator_intent.as_ref(), &committed_config_etag)
|
||||
.await;
|
||||
if let Some(intent) = committed_coordinator_intent.as_ref() {
|
||||
TierConfigMgr::apply_committed_mutation_intent_block(&handle, intent)
|
||||
.await
|
||||
@@ -4496,9 +4501,9 @@ impl TierConfigMgr {
|
||||
.map_err(TierConfigUpdateError::Publish)?,
|
||||
);
|
||||
}
|
||||
commit_coordinator_tier_mutation_intent(api.clone(), coordinator_intent.as_ref(), &committed_config_etag)
|
||||
.await
|
||||
.map_err(TierConfigUpdateError::Save)?;
|
||||
// Config is already saved: retain the committed fence and wake recovery
|
||||
// even when the coordinator commit failed or its outcome is unknown.
|
||||
coordinator_commit.map_err(TierConfigUpdateError::Save)?;
|
||||
if coordinated_config_update {
|
||||
drop(update.take());
|
||||
drop(config_lock.take());
|
||||
@@ -10603,6 +10608,11 @@ mod tests {
|
||||
.expect_err("coordinator committed-state CAS failure must be observable");
|
||||
assert!(matches!(err, TierConfigUpdateError::Save(_)));
|
||||
assert!(manager.read().await.tiers.contains_key("COLD-A"));
|
||||
assert!(TierConfigMgr::has_committed_mutation_block(&manager).await);
|
||||
let refresh = TierConfigMgr::mutation_refresh_notifier(&manager).await;
|
||||
tokio::time::timeout(Duration::from_secs(1), refresh.notified())
|
||||
.await
|
||||
.expect("failed coordinator commit must notify recovery after saving config");
|
||||
let blocked = match TierConfigMgr::acquire_operation_lease(&manager, "COLD-A").await {
|
||||
Ok(_) => panic!("failed coordinator commit CAS must retain the local committed fence"),
|
||||
Err(err) => err,
|
||||
@@ -14329,6 +14339,12 @@ mod tests {
|
||||
after_commit: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
struct CasCoordinatorCommitBarrier {
|
||||
arrived: tokio::sync::Notify,
|
||||
release: tokio::sync::Notify,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct CasConfigStore {
|
||||
objects: tokio::sync::Mutex<HashMap<String, (Vec<u8>, String)>>,
|
||||
@@ -14341,6 +14357,7 @@ mod tests {
|
||||
fail_delete_prefix: tokio::sync::Mutex<Option<(String, usize)>>,
|
||||
delete_log: tokio::sync::Mutex<Vec<String>>,
|
||||
list_barrier: tokio::sync::Mutex<Option<Arc<CasListBarrier>>>,
|
||||
coordinator_commit_barrier: tokio::sync::Mutex<Option<Arc<CasCoordinatorCommitBarrier>>>,
|
||||
intent_list_calls: AtomicUsize,
|
||||
fail_reference_walk: AtomicBool,
|
||||
reference_walk_send_count: AtomicUsize,
|
||||
@@ -14363,6 +14380,7 @@ mod tests {
|
||||
fail_delete_prefix: tokio::sync::Mutex::new(None),
|
||||
delete_log: tokio::sync::Mutex::new(Vec::new()),
|
||||
list_barrier: tokio::sync::Mutex::new(None),
|
||||
coordinator_commit_barrier: tokio::sync::Mutex::new(None),
|
||||
intent_list_calls: AtomicUsize::new(0),
|
||||
fail_reference_walk: AtomicBool::new(false),
|
||||
reference_walk_send_count: AtomicUsize::new(0),
|
||||
@@ -14554,6 +14572,19 @@ mod tests {
|
||||
}
|
||||
let mut payload = Vec::new();
|
||||
tokio::io::AsyncReadExt::read_to_end(&mut data.stream, &mut payload).await?;
|
||||
if object.starts_with(crate::services::tier::tier_mutation_intent::TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX)
|
||||
&& opts
|
||||
.http_preconditions
|
||||
.as_ref()
|
||||
.and_then(HTTPPreconditions::if_match_value)
|
||||
.is_some()
|
||||
{
|
||||
let barrier = self.coordinator_commit_barrier.lock().await.take();
|
||||
if let Some(barrier) = barrier {
|
||||
barrier.arrived.notify_one();
|
||||
barrier.release.notified().await;
|
||||
}
|
||||
}
|
||||
let race_rewrite = if opts
|
||||
.http_preconditions
|
||||
.as_ref()
|
||||
@@ -15651,14 +15682,7 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn force_remove_and_save_bypasses_lifecycle_only_reference() {
|
||||
// rustfs/rustfs#6832: reproduces the admin RemoveTier path (not just the lower-level
|
||||
// reference-proof function) for a tier with zero transitioned objects but a lifecycle
|
||||
// rule still pointing at it — the exact shape of
|
||||
// `test_manual_transition_async_tier_failure_reports_terminal_partial` in e2e_test,
|
||||
// which force-removes a tier a lifecycle rule still references to simulate a
|
||||
// decommissioned backend.
|
||||
async fn assert_lifecycle_only_reference_obeys_force(clear: bool, force: bool) {
|
||||
let store = Arc::new(CasConfigStore::default());
|
||||
let tier = build_rustfs_tier("COLD-A");
|
||||
let mut persisted = empty_mgr();
|
||||
@@ -15699,22 +15723,55 @@ mod tests {
|
||||
|
||||
let manager = TierConfigMgr::new();
|
||||
manager.write().await.tiers.insert("COLD-A".to_string(), tier);
|
||||
TierConfigMgr::remove_and_save_with(&manager, store.clone(), "COLD-A", true)
|
||||
.await
|
||||
.expect("force remove must bypass a lifecycle-config-only reference");
|
||||
let mutation = if clear {
|
||||
TierCandidateMutation::Clear(force)
|
||||
} else {
|
||||
TierCandidateMutation::Remove("COLD-A".to_string(), force)
|
||||
};
|
||||
let result = TIER_DRIVER_TEST_FACTORY
|
||||
.scope(
|
||||
healthy_driver_factory(),
|
||||
TierConfigMgr::update_candidate_with_config_lock(&manager, store.clone(), mutation),
|
||||
)
|
||||
.await;
|
||||
if force {
|
||||
result.expect("force mutation must bypass a lifecycle-config-only reference");
|
||||
} else {
|
||||
let err = result.expect_err("non-force mutation must reject a lifecycle-only reference");
|
||||
let TierConfigUpdateError::Publish(err) = err else {
|
||||
panic!("non-force mutation must fail during reference proof: {err:?}");
|
||||
};
|
||||
assert_eq!(err.code, ERR_TIER_BACKEND_IN_USE.code);
|
||||
assert!(err.message.contains("move-current"), "{err}");
|
||||
}
|
||||
|
||||
assert!(!manager.read().await.tiers.contains_key("COLD-A"));
|
||||
assert!(
|
||||
!load_tier_config_for_update(store)
|
||||
assert_eq!(manager.read().await.tiers.contains_key("COLD-A"), !force);
|
||||
assert_eq!(
|
||||
load_tier_config_for_update(store)
|
||||
.await
|
||||
.expect("config should still reload")
|
||||
.0
|
||||
.tiers
|
||||
.contains_key("COLD-A"),
|
||||
"force removal must persist the empty candidate"
|
||||
!force,
|
||||
"persisted state must match the force mutation result"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn remove_with_config_lock_obeys_force_for_lifecycle_only_reference() {
|
||||
for force in [false, true] {
|
||||
assert_lifecycle_only_reference_obeys_force(false, force).await;
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn clear_with_config_lock_obeys_force_for_lifecycle_only_reference() {
|
||||
for force in [false, true] {
|
||||
assert_lifecycle_only_reference_obeys_force(true, force).await;
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn zero_reference_proof_blocks_clear_before_config_save() {
|
||||
let store = Arc::new(CasConfigStore::default());
|
||||
@@ -17255,6 +17312,98 @@ mod tests {
|
||||
assert_ne!(manager_a.read().await.empty(), manager_b.read().await.empty());
|
||||
}
|
||||
|
||||
async fn assert_coordinator_commit_refresh_succeeds(mutation: TierCandidateMutation) {
|
||||
let adding = matches!(mutation, TierCandidateMutation::Add(..));
|
||||
let manager = TierConfigMgr::new();
|
||||
let store = Arc::new(CasConfigStore::default());
|
||||
if !adding {
|
||||
let mut persisted = empty_mgr();
|
||||
persisted.tiers.insert("COLD-A".to_string(), build_rustfs_tier("COLD-A"));
|
||||
persisted
|
||||
.save_tiering_config_if_current(store.clone(), None)
|
||||
.await
|
||||
.expect("existing tier fixture should persist");
|
||||
let mut guard = manager.write().await;
|
||||
install_lease_backend(&mut guard, "COLD-A", LeaseTestBackend::ready("old"));
|
||||
}
|
||||
let barrier = Arc::new(CasCoordinatorCommitBarrier::default());
|
||||
*store.coordinator_commit_barrier.lock().await = Some(barrier.clone());
|
||||
let update_manager = manager.clone();
|
||||
let update_store = store.clone();
|
||||
let update = tokio::spawn(async move {
|
||||
TIER_DRIVER_TEST_FACTORY
|
||||
.scope(
|
||||
healthy_driver_factory(),
|
||||
TIER_MUTATION_TEST_PEERS.scope(
|
||||
Vec::new(),
|
||||
TierConfigMgr::update_candidate_with_config_lock(&update_manager, update_store, mutation),
|
||||
),
|
||||
)
|
||||
.await
|
||||
});
|
||||
tokio::time::timeout(Duration::from_secs(5), barrier.arrived.notified())
|
||||
.await
|
||||
.expect("mutation should reach coordinator commit after saving config");
|
||||
assert_eq!(
|
||||
load_tier_config_for_update(store.clone())
|
||||
.await
|
||||
.expect("saved config should be readable before coordinator commit")
|
||||
.0
|
||||
.tiers
|
||||
.contains_key("COLD-A"),
|
||||
adding
|
||||
);
|
||||
assert_eq!(
|
||||
TierConfigMgr::load_coordinator_mutation_intents(store.clone())
|
||||
.await
|
||||
.expect("coordinator intent should remain readable")[0]
|
||||
.state,
|
||||
TierMutationIntentState::Prepared
|
||||
);
|
||||
|
||||
let lock_requests = lock_unpoisoned(&store.lock_requests).len();
|
||||
// Also exercise an independently scheduled refresh while the durable
|
||||
// coordinator record is still Prepared, before its commit notification.
|
||||
TierConfigMgr::request_committed_mutation_refresh(&manager).await;
|
||||
TIER_MUTATION_TEST_PEERS
|
||||
.scope(Vec::new(), async {
|
||||
let worker = TierConfigMgr::refresh_tier_config_handle_with(manager.clone(), store.clone());
|
||||
tokio::pin!(worker);
|
||||
tokio::time::timeout(Duration::from_secs(5), async {
|
||||
while lock_unpoisoned(&store.lock_requests).len() == lock_requests {
|
||||
tokio::select! {
|
||||
_ = &mut worker => panic!("refresh worker must remain available"),
|
||||
_ = tokio::task::yield_now() => {}
|
||||
}
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("refresh should reconcile the Prepared record before waiting for the config lock");
|
||||
barrier.release.notify_one();
|
||||
let result = tokio::time::timeout(Duration::from_secs(5), async {
|
||||
tokio::select! {
|
||||
_ = &mut worker => panic!("refresh worker must remain available"),
|
||||
result = update => result.expect("tier mutation task should join"),
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("tier mutation should finish with refresh running");
|
||||
result.expect("saved tier mutation must publish successfully on the first attempt");
|
||||
})
|
||||
.await;
|
||||
assert_eq!(manager.read().await.tiers.contains_key("COLD-A"), adding);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn tier_add_succeeds_with_refresh_during_coordinator_commit() {
|
||||
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true)).await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn tier_remove_succeeds_with_refresh_during_coordinator_commit() {
|
||||
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Remove("COLD-A".to_string(), true)).await;
|
||||
}
|
||||
|
||||
async fn committed_refresh_fixture(fail_cleanup: bool) -> (Arc<RwLock<TierConfigMgr>>, Arc<CasConfigStore>, uuid::Uuid) {
|
||||
let manager = TierConfigMgr::new();
|
||||
{
|
||||
|
||||
@@ -37,7 +37,7 @@ use crate::services::tier::{
|
||||
use bytes::Bytes;
|
||||
use http::StatusCode;
|
||||
use rustfs_s3_client::credentials::{Credentials, SignatureType, Static, Value};
|
||||
use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionCore};
|
||||
use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionClientTimeouts, TransitionCore};
|
||||
use rustfs_s3_client::{
|
||||
admin_handler_utils::AdminError,
|
||||
api_error_response::to_error_response,
|
||||
@@ -320,6 +320,27 @@ pub(crate) fn endpoint_authority(url: &url::Url) -> Result<String, std::io::Erro
|
||||
}
|
||||
}
|
||||
|
||||
fn transition_timeout_from_env(env_key: &str, default_secs: u64) -> Duration {
|
||||
Duration::from_secs(rustfs_utils::get_env_u64(env_key, default_secs))
|
||||
}
|
||||
|
||||
pub(crate) fn transition_client_timeouts_from_env() -> TransitionClientTimeouts {
|
||||
TransitionClientTimeouts::new(
|
||||
transition_timeout_from_env(
|
||||
rustfs_config::ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS,
|
||||
rustfs_config::DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS,
|
||||
),
|
||||
transition_timeout_from_env(
|
||||
rustfs_config::ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS,
|
||||
rustfs_config::DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS,
|
||||
),
|
||||
transition_timeout_from_env(
|
||||
rustfs_config::ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
|
||||
rustfs_config::DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
/// Build the [`WarmBackendS3`] shared by the S3-compatible warm backend providers.
|
||||
///
|
||||
/// Credential, bucket, and endpoint validation run in this order because the
|
||||
@@ -350,6 +371,7 @@ pub(crate) async fn new_s3_compatible_warm_backend(
|
||||
signer_type: SignatureType::SignatureV4,
|
||||
..Default::default()
|
||||
}));
|
||||
let timeouts = transition_client_timeouts_from_env();
|
||||
let opts = Options {
|
||||
creds,
|
||||
secure: u.scheme() == "https",
|
||||
@@ -362,7 +384,7 @@ pub(crate) async fn new_s3_compatible_warm_backend(
|
||||
// Run the SSRF guard after the host-presence check so a host-less endpoint
|
||||
// keeps this constructor's stable error text.
|
||||
(params.validate_endpoint)(&u).map_err(|err| std::io::Error::other(format!("tier endpoint is not allowed: {err}")))?;
|
||||
let client = TransitionClient::new(&endpoint, opts, params.provider_tag).await?;
|
||||
let client = TransitionClient::new_with_timeouts(&endpoint, opts, params.provider_tag, timeouts).await?;
|
||||
|
||||
let client = Arc::new(client);
|
||||
let core = TransitionCore(Arc::clone(&client));
|
||||
|
||||
@@ -26,7 +26,7 @@ use crate::services::tier::{
|
||||
tier_config::TierS3,
|
||||
warm_backend::{
|
||||
TransitionCandidateIdentity, TransitionCandidateProbe, TransitionCandidateReconciler, WarmBackend, WarmBackendGetOpts,
|
||||
build_transition_put_options, endpoint_authority,
|
||||
build_transition_put_options, endpoint_authority, transition_client_timeouts_from_env,
|
||||
},
|
||||
};
|
||||
use http::HeaderMap;
|
||||
@@ -139,6 +139,7 @@ impl WarmBackendS3 {
|
||||
} else {
|
||||
return Err(std::io::Error::other("insufficient parameters for S3 backend authentication"));
|
||||
}
|
||||
let timeouts = transition_client_timeouts_from_env();
|
||||
let opts = Options {
|
||||
creds,
|
||||
secure: u.scheme() == "https",
|
||||
@@ -147,7 +148,7 @@ impl WarmBackendS3 {
|
||||
..Default::default()
|
||||
};
|
||||
let endpoint = endpoint_authority(&u)?;
|
||||
let client = TransitionClient::new(&endpoint, opts, tier_type).await?;
|
||||
let client = TransitionClient::new_with_timeouts(&endpoint, opts, tier_type, timeouts).await?;
|
||||
|
||||
let client = Arc::new(client);
|
||||
let core = TransitionCore(Arc::clone(&client));
|
||||
|
||||
@@ -2490,9 +2490,9 @@ impl crate::storage_api_contracts::heal::HealOperations for SetDisks {
|
||||
return Ok((result, err.map(|e| e.into())));
|
||||
}
|
||||
|
||||
let disks = self.disks.read().await;
|
||||
|
||||
let disks = disks.clone();
|
||||
// The inner heal and missing-object report read the registry again;
|
||||
// release this snapshot guard before a topology writer can queue between reads.
|
||||
let disks = self.get_disks_internal().await;
|
||||
let (_, errs) = Self::read_all_fileinfo(&disks, "", bucket, object, version_id, false, false, false)
|
||||
.await
|
||||
.map_err(|e| to_object_err(e.into(), vec![bucket, object]))?;
|
||||
@@ -3419,6 +3419,366 @@ mod heal_result_report_tests {
|
||||
assert_eq!(unformatted, DiskError::UnformattedDisk);
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
enum InventoryWriterHealCase {
|
||||
Existing,
|
||||
Missing,
|
||||
MissingVersion,
|
||||
}
|
||||
|
||||
async fn assert_heal_object_inventory_writer(case: InventoryWriterHealCase) {
|
||||
use crate::set_disk::core::io_primitives::disk_call_counters;
|
||||
use std::time::Duration;
|
||||
use tokio::io::AsyncReadExt;
|
||||
|
||||
let (_temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
|
||||
let bucket = "heal-inventory-writer-bucket";
|
||||
let object = match case {
|
||||
InventoryWriterHealCase::Existing => "heal-inventory-writer-existing",
|
||||
InventoryWriterHealCase::Missing => "heal-inventory-writer-missing",
|
||||
InventoryWriterHealCase::MissingVersion => "heal-inventory-writer-missing-version",
|
||||
};
|
||||
set.make_bucket(
|
||||
bucket,
|
||||
&MakeBucketOptions {
|
||||
versioning_enabled: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("heal fixture bucket should be created");
|
||||
let body = vec![0x67; 64 * 1024];
|
||||
let stored_version = Uuid::new_v4();
|
||||
let stored_version_string = stored_version.to_string();
|
||||
let published = if matches!(case, InventoryWriterHealCase::Missing) {
|
||||
None
|
||||
} else {
|
||||
let mut reader = PutObjReader::from_vec(body.clone());
|
||||
let info = set
|
||||
.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut reader,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
versioned: true,
|
||||
version_id: Some(stored_version_string.clone()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("full-fanout PUT should seed the heal fixture");
|
||||
for disk in &disks {
|
||||
let metadata = disk
|
||||
.read_version("", bucket, object, &stored_version_string, &ReadOptions::default())
|
||||
.await
|
||||
.expect("the seeded version must be present on every disk");
|
||||
assert_eq!(metadata.version_id, Some(stored_version));
|
||||
assert_eq!(metadata.size, i64::try_from(body.len()).expect("fixture size should fit i64"));
|
||||
}
|
||||
Some(info)
|
||||
};
|
||||
let requested_version = match case {
|
||||
InventoryWriterHealCase::Existing => stored_version_string.clone(),
|
||||
InventoryWriterHealCase::Missing => String::new(),
|
||||
InventoryWriterHealCase::MissingVersion => Uuid::new_v4().to_string(),
|
||||
};
|
||||
let opts = HealOpts {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
};
|
||||
let calls = disk_call_counters::observe(object);
|
||||
let read_gate = set.disks.read().await;
|
||||
// UFCS selects the trait's outer precheck, not the same-named inherent heal.
|
||||
let heal = <SetDisks as crate::storage_api_contracts::heal::HealOperations>::heal_object(
|
||||
set.as_ref(),
|
||||
bucket,
|
||||
object,
|
||||
&requested_version,
|
||||
&opts,
|
||||
);
|
||||
tokio::pin!(heal);
|
||||
assert!(matches!(
|
||||
futures::poll!(tokio::task::unconstrained(heal.as_mut())),
|
||||
std::task::Poll::Pending
|
||||
));
|
||||
// These tests use the current-thread runtime: full-wait metadata tasks
|
||||
// have been spawned, but cannot run during the single unconstrained poll.
|
||||
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 0);
|
||||
let writer = set.disks.write();
|
||||
tokio::pin!(writer);
|
||||
assert!(matches!(
|
||||
futures::poll!(tokio::task::unconstrained(writer.as_mut())),
|
||||
std::task::Poll::Pending
|
||||
));
|
||||
assert!(set.disks.try_read().is_err(), "the writer must already block new inventory readers");
|
||||
tokio::time::timeout(Duration::from_secs(5), async {
|
||||
while calls.total(disk_call_counters::KIND_READ_VERSION) < 4 {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("the suspended trait heal must have started the real metadata fanout");
|
||||
for disk_index in 0..4 {
|
||||
assert_eq!(calls.for_disk(disk_call_counters::KIND_READ_VERSION, disk_index), 1);
|
||||
}
|
||||
drop(read_gate);
|
||||
|
||||
let (_, outcome) =
|
||||
tokio::time::timeout(Duration::from_secs(5), async { tokio::join!(async { drop(writer.await) }, heal) })
|
||||
.await
|
||||
.expect("trait heal must not deadlock its nested inventory read with the queued writer");
|
||||
let (result, error) = outcome.expect("heal should report the object's outcome");
|
||||
match case {
|
||||
InventoryWriterHealCase::Existing => assert!(error.is_none(), "existing object heal failed: {error:?}"),
|
||||
InventoryWriterHealCase::Missing => assert!(matches!(error, Some(Error::FileNotFound))),
|
||||
InventoryWriterHealCase::MissingVersion => assert!(matches!(error, Some(Error::FileVersionNotFound))),
|
||||
}
|
||||
assert_eq!(result.bucket, bucket);
|
||||
assert_eq!(result.object, object);
|
||||
assert_eq!(result.version_id, requested_version);
|
||||
assert_eq!(result.disk_count, 4);
|
||||
assert_eq!(result.before.drives.len(), 4);
|
||||
assert_eq!(result.after.drives.len(), 4);
|
||||
for disk_index in 0..4 {
|
||||
let endpoint = set.set_endpoints[disk_index].to_string();
|
||||
assert_eq!(result.before.drives[disk_index].endpoint, endpoint);
|
||||
assert_eq!(result.after.drives[disk_index].endpoint, endpoint);
|
||||
}
|
||||
if let Some(published) = published {
|
||||
tokio::time::timeout(Duration::from_secs(10), async {
|
||||
let mut reader = set
|
||||
.get_object_reader(
|
||||
bucket,
|
||||
object,
|
||||
None,
|
||||
Default::default(),
|
||||
&ObjectOptions {
|
||||
versioned: true,
|
||||
version_id: Some(stored_version_string),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("the stored version must remain readable after heal");
|
||||
assert_eq!(reader.object_info.etag, published.etag);
|
||||
assert_eq!(reader.object_info.version_id, Some(stored_version));
|
||||
let mut observed_body = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut observed_body)
|
||||
.await
|
||||
.expect("stored body should stream");
|
||||
assert_eq!(observed_body, body);
|
||||
})
|
||||
.await
|
||||
.expect("GET must finish after the inventory writer and heal");
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_object_inventory_writer_existing() {
|
||||
assert_heal_object_inventory_writer(InventoryWriterHealCase::Existing).await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_object_inventory_writer_missing() {
|
||||
assert_heal_object_inventory_writer(InventoryWriterHealCase::Missing).await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_object_inventory_writer_missing_version() {
|
||||
assert_heal_object_inventory_writer(InventoryWriterHealCase::MissingVersion).await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn heal_object_with_queued_disk_renewal() {
|
||||
use crate::layout::endpoints::SetupType;
|
||||
use crate::runtime::instance::InstanceContext;
|
||||
use crate::set_disk::core::io_primitives::disk_call_counters;
|
||||
use std::collections::HashMap;
|
||||
use std::future::Future;
|
||||
use std::task::Poll;
|
||||
use std::time::Duration;
|
||||
use tokio::io::AsyncReadExt;
|
||||
|
||||
// renew_disk still registers local disks on the ambient context. Match
|
||||
// the default serial group used by its other setup/registry fixtures,
|
||||
// and restore only this temporary endpoint, including on a failed join.
|
||||
struct RenewDiskTestState {
|
||||
ctx: Arc<InstanceContext>,
|
||||
was_dist_erasure: bool,
|
||||
map: Arc<RwLock<HashMap<String, Option<DiskStore>>>>,
|
||||
endpoint: String,
|
||||
previous_disk: Option<Option<DiskStore>>,
|
||||
}
|
||||
|
||||
impl Drop for RenewDiskTestState {
|
||||
fn drop(&mut self) {
|
||||
let ctx = self.ctx.clone();
|
||||
let was_dist_erasure = self.was_dist_erasure;
|
||||
let map = self.map.clone();
|
||||
let endpoint = self.endpoint.clone();
|
||||
let previous_disk = self.previous_disk.take();
|
||||
let handle = tokio::runtime::Handle::current();
|
||||
std::thread::spawn(move || {
|
||||
handle.block_on(async move {
|
||||
let mut map = map.write().await;
|
||||
match previous_disk {
|
||||
Some(disk) => {
|
||||
map.insert(endpoint, disk);
|
||||
}
|
||||
None => {
|
||||
map.remove(&endpoint);
|
||||
}
|
||||
}
|
||||
drop(map);
|
||||
if was_dist_erasure {
|
||||
ctx.update_erasure_type(SetupType::DistErasure).await;
|
||||
}
|
||||
});
|
||||
})
|
||||
.join()
|
||||
.expect("renew fixture state restoration should finish");
|
||||
}
|
||||
}
|
||||
|
||||
let (_temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
|
||||
let endpoint = set.set_endpoints[0].clone();
|
||||
let ctx = crate::runtime::global::current_ctx();
|
||||
let map = ctx.local_disk_map();
|
||||
let restore = RenewDiskTestState {
|
||||
ctx: ctx.clone(),
|
||||
was_dist_erasure: ctx.is_dist_erasure().await,
|
||||
map: map.clone(),
|
||||
endpoint: endpoint.to_string(),
|
||||
previous_disk: map.read().await.get(&endpoint.to_string()).cloned(),
|
||||
};
|
||||
// Only distributed erasure needs an override to avoid the ambient slot array.
|
||||
if restore.was_dist_erasure {
|
||||
ctx.update_erasure_type(SetupType::Erasure).await;
|
||||
}
|
||||
|
||||
let bucket = "heal-disk-renewal-bucket";
|
||||
let object = "heal-disk-renewal-object";
|
||||
set.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("renew fixture bucket should be created");
|
||||
let body = vec![0x73; 64 * 1024];
|
||||
let mut reader = PutObjReader::from_vec(body.clone());
|
||||
let published = set
|
||||
.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut reader,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("full-fanout PUT should seed the renewal fixture");
|
||||
for disk in &disks {
|
||||
let metadata = disk
|
||||
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
.expect("the seeded object must be present on every disk");
|
||||
assert_eq!(metadata.size, i64::try_from(body.len()).expect("fixture size should fit i64"));
|
||||
}
|
||||
|
||||
let opts = HealOpts {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
};
|
||||
let calls = disk_call_counters::observe(object);
|
||||
let read_gate = set.disks.read().await;
|
||||
let heal = <SetDisks as crate::storage_api_contracts::heal::HealOperations>::heal_object(
|
||||
set.as_ref(),
|
||||
bucket,
|
||||
object,
|
||||
"",
|
||||
&opts,
|
||||
);
|
||||
tokio::pin!(heal);
|
||||
assert!(matches!(futures::poll!(tokio::task::unconstrained(heal.as_mut())), Poll::Pending));
|
||||
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 0);
|
||||
|
||||
let renew = set.renew_disk(&endpoint);
|
||||
tokio::pin!(renew);
|
||||
tokio::time::timeout(
|
||||
Duration::from_secs(5),
|
||||
futures::future::poll_fn(|cx| {
|
||||
assert!(
|
||||
std::pin::pin!(tokio::task::unconstrained(renew.as_mut()))
|
||||
.poll(cx)
|
||||
.is_pending(),
|
||||
"renewal must reach its inventory write before returning"
|
||||
);
|
||||
if set.disks.try_read().is_err() {
|
||||
Poll::Ready(())
|
||||
} else {
|
||||
Poll::Pending
|
||||
}
|
||||
}),
|
||||
)
|
||||
.await
|
||||
.expect("real renewal must queue its topology writer behind the read gate");
|
||||
let registered = map
|
||||
.read()
|
||||
.await
|
||||
.get(&endpoint.to_string())
|
||||
.cloned()
|
||||
.flatten()
|
||||
.expect("renewal must register the connected disk before its inventory write");
|
||||
assert!(!Arc::ptr_eq(®istered, &disks[0]), "renewal must construct a new disk handle");
|
||||
tokio::time::timeout(Duration::from_secs(5), async {
|
||||
while calls.total(disk_call_counters::KIND_READ_VERSION) < 4 {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("the suspended trait heal must have started the real metadata fanout");
|
||||
for disk_index in 0..4 {
|
||||
assert_eq!(calls.for_disk(disk_call_counters::KIND_READ_VERSION, disk_index), 1);
|
||||
}
|
||||
drop(read_gate);
|
||||
|
||||
let (_, outcome) = tokio::time::timeout(Duration::from_secs(5), async { tokio::join!(renew, heal) })
|
||||
.await
|
||||
.expect("trait heal and real disk renewal must finish without a nested inventory read deadlock");
|
||||
let (report, error) = outcome.expect("heal should report the existing object");
|
||||
assert!(error.is_none(), "existing object heal failed after renewal: {error:?}");
|
||||
assert_eq!(report.bucket, bucket);
|
||||
assert_eq!(report.object, object);
|
||||
assert_eq!(report.disk_count, 4);
|
||||
let renewed = set.get_disks_internal().await[0]
|
||||
.clone()
|
||||
.expect("the renewed slot must remain online");
|
||||
assert!(Arc::ptr_eq(&renewed, ®istered), "the set must publish the newly connected handle");
|
||||
assert_eq!(renewed.endpoint(), endpoint);
|
||||
let format = load_format_erasure(&renewed, false)
|
||||
.await
|
||||
.expect("renewed disk format should remain readable");
|
||||
assert_eq!(format.erasure.this, set.format.erasure.sets[0][0]);
|
||||
tokio::time::timeout(Duration::from_secs(10), async {
|
||||
let mut reader = set
|
||||
.get_object_reader(bucket, object, None, Default::default(), &ObjectOptions::default())
|
||||
.await
|
||||
.expect("the object must remain readable after renewal and heal");
|
||||
assert_eq!(reader.object_info.etag, published.etag);
|
||||
let mut observed_body = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut observed_body)
|
||||
.await
|
||||
.expect("stored body should stream");
|
||||
assert_eq!(observed_body, body);
|
||||
})
|
||||
.await
|
||||
.expect("GET must finish after renewal and heal");
|
||||
}
|
||||
|
||||
// Regression for #955: an offline disk must contribute exactly one drive
|
||||
// record. Before the fix the offline branch fell through and pushed a second
|
||||
// (Corrupt) record for the same disk, so `before/after.drives` grew to
|
||||
|
||||
@@ -2452,10 +2452,9 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
let write_quorum = fi.write_quorum(self.default_write_quorum());
|
||||
let read_quorum = fi.read_quorum(self.default_read_quorum());
|
||||
|
||||
let disks = self.disks.read().await;
|
||||
|
||||
let disks = disks.clone();
|
||||
// let disks = Self::shuffle_disks(&disks, &fi.erasure.distribution);
|
||||
// Release the registry guard before recovery and cleanup read it again:
|
||||
// a queued topology writer would otherwise deadlock those nested reads.
|
||||
let disks = self.get_disks_internal().await;
|
||||
|
||||
let part_path = format!("{}/{}/", upload_id_path, fi.data_dir.unwrap_or(Uuid::nil()));
|
||||
self.recover_part_transactions(&part_path, read_quorum, write_quorum)
|
||||
@@ -6743,6 +6742,87 @@ mod tests {
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
#[serial]
|
||||
async fn complete_multipart_releases_disk_snapshot_before_cleanup() {
|
||||
let (temp_dirs, disk_stores, set_disks) = hermetic_set_disks(4).await;
|
||||
let bucket = "multipart-topology-lock-bucket";
|
||||
let object = "object";
|
||||
let body = vec![0x65; 4096];
|
||||
make_bucket_on_all(&disk_stores, bucket).await;
|
||||
let (upload_id, parts) =
|
||||
stage_upload_with_create_opts(&set_disks, bucket, object, &body, &ObjectOptions::default()).await;
|
||||
let upload_id_path = SetDisks::get_upload_id_dir(bucket, object, &upload_id);
|
||||
for dir in &temp_dirs {
|
||||
assert!(
|
||||
dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_id_path).exists(),
|
||||
"the test must create real upload staging on every disk"
|
||||
);
|
||||
}
|
||||
let barrier = MultipartCommitBarrier::install(bucket, object, MultipartCommitPause::AfterObjectPublication);
|
||||
let complete_store = set_disks.clone();
|
||||
let complete_upload_id = upload_id.clone();
|
||||
let complete = tokio::spawn(async move {
|
||||
complete_store
|
||||
.complete_multipart_upload(bucket, object, &complete_upload_id, parts, &ObjectOptions::default())
|
||||
.await
|
||||
});
|
||||
barrier.wait_until_paused().await;
|
||||
|
||||
// Hold a separate read gate so the real writer queues even when completion
|
||||
// correctly releases its snapshot guard. Polling Pending proves admission
|
||||
// to Tokio's write-preferring queue before the cleanup attempts another read.
|
||||
let read_gate = set_disks.disks.read().await;
|
||||
let writer = set_disks.disks.write();
|
||||
tokio::pin!(writer);
|
||||
assert!(matches!(
|
||||
futures::poll!(tokio::task::unconstrained(writer.as_mut())),
|
||||
std::task::Poll::Pending
|
||||
));
|
||||
assert!(
|
||||
set_disks.disks.try_read().is_err(),
|
||||
"the pending writer must already block new readers before the cleanup resumes"
|
||||
);
|
||||
drop(read_gate);
|
||||
barrier.release();
|
||||
|
||||
let writer_guard = tokio::time::timeout(Duration::from_secs(5), writer)
|
||||
.await
|
||||
.expect("a queued topology writer must not deadlock with multipart cleanup's disk snapshot");
|
||||
// A reconnect can publish the same handles; this test isolates admission
|
||||
// order without changing the disks that contain the committed object.
|
||||
drop(writer_guard);
|
||||
tokio::time::timeout(Duration::from_secs(10), complete)
|
||||
.await
|
||||
.expect("multipart cleanup must finish after the topology writer releases")
|
||||
.expect("completion task should not panic")
|
||||
.expect("completion should preserve the successful object commit");
|
||||
|
||||
let mut reader = tokio::time::timeout(
|
||||
Duration::from_secs(10),
|
||||
set_disks.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default()),
|
||||
)
|
||||
.await
|
||||
.expect("GET should finish after completion")
|
||||
.expect("the completed object should remain readable");
|
||||
let mut observed_body = Vec::new();
|
||||
tokio::time::timeout(Duration::from_secs(10), reader.stream.read_to_end(&mut observed_body))
|
||||
.await
|
||||
.expect("the completed object body should finish streaming")
|
||||
.expect("the completed object body should be readable");
|
||||
assert_eq!(observed_body, body);
|
||||
assert!(matches!(
|
||||
set_disks.check_upload_id_exists(bucket, object, &upload_id, false).await,
|
||||
Err(StorageError::InvalidUploadID(..))
|
||||
));
|
||||
for dir in &temp_dirs {
|
||||
assert!(
|
||||
!dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_id_path).exists(),
|
||||
"successful completion must remove its upload staging from every disk"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
#[serial]
|
||||
async fn complete_releases_object_lock_before_cleanup_and_keeps_upload_lock() {
|
||||
|
||||
@@ -45,6 +45,10 @@ use uuid::Uuid;
|
||||
|
||||
use crate::heal::task::{HealOptions, HealPriority, HealRequest, HealType};
|
||||
|
||||
/// Read-only inspection of committed MRF checkpoints. The legacy consumer
|
||||
/// remains unchanged until ownership-aware replay is deployed.
|
||||
pub mod snapshot;
|
||||
|
||||
/// Journal location inside the metadata bucket, following the resume-state
|
||||
/// layout.
|
||||
pub(crate) const MRF_JOURNAL_PATH: &str = "buckets/.heal/mrf/journal.bin";
|
||||
|
||||
@@ -0,0 +1,681 @@
|
||||
// Copyright 2026 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
//! Reader-first support for owner-local MRF checkpoints.
|
||||
//!
|
||||
//! Each of two slots has a payload and a commit manifest. The manifest binds
|
||||
//! the writer identity, persistent sequence, length and whole-payload digest.
|
||||
//! Replacing the inactive slot must leave the previous committed slot intact.
|
||||
//! Production publication and reclamation are deliberately not enabled here.
|
||||
//! An unreadable commit path cannot prove that only legacy data exists. This
|
||||
//! explicit inspection API fails closed and never mutates recovery anchors.
|
||||
//! It is not wired into the legacy consumer: that transition requires the
|
||||
//! ownership-aware replay and producer handoff before writer activation.
|
||||
//! One surviving committed replica supports process restart recovery only;
|
||||
//! this reader does not establish a replication quorum or a power-loss policy.
|
||||
|
||||
use super::{MRF_JOURNAL_PATH, MRF_SCOPED_JOURNAL_PATH, decode_journal};
|
||||
use crate::heal::RUSTFS_META_BUCKET;
|
||||
use crate::heal::storage_api::owner::{EcstoreDiskAPI, EcstoreDiskError, EcstoreDiskStore};
|
||||
use sha2::{Digest, Sha256};
|
||||
use std::collections::HashMap;
|
||||
use tokio::io::AsyncReadExt;
|
||||
use uuid::Uuid;
|
||||
|
||||
// Root-level control files avoid requiring a new directory before the first
|
||||
// atomic commit. They remain inside the storage owner's metadata volume.
|
||||
const PAYLOAD_PATHS: [&str; 2] = [".heal-mrf-snapshot.0.bin", ".heal-mrf-snapshot.1.bin"];
|
||||
const MANIFEST_PATHS: [&str; 2] = [".heal-mrf-commit.0.bin", ".heal-mrf-commit.1.bin"];
|
||||
const MAGIC: &[u8; 8] = b"RFMRFC01";
|
||||
const MANIFEST_LEN: usize = 8 + 1 + 16 + 8 + 8 + 32 + 32;
|
||||
const VERSION: u8 = 1;
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum SnapshotError {
|
||||
#[error("MRF checkpoint has an invalid or incomplete commit record")]
|
||||
Corrupt,
|
||||
#[error("MRF checkpoint format is unsupported")]
|
||||
Unsupported,
|
||||
#[error("MRF checkpoint exceeds the configured byte limit")]
|
||||
TooLarge,
|
||||
#[error("MRF checkpoint replicas disagree at the same sequence")]
|
||||
Conflict,
|
||||
#[error("MRF checkpoint storage is unavailable")]
|
||||
Disk(#[source] EcstoreDiskError),
|
||||
#[error("MRF checkpoint body could not be read")]
|
||||
Read(#[source] std::io::Error),
|
||||
}
|
||||
|
||||
#[derive(Debug, PartialEq, Eq)]
|
||||
struct Manifest {
|
||||
owner: Uuid,
|
||||
sequence: u64,
|
||||
payload_len: usize,
|
||||
payload_digest: [u8; 32],
|
||||
}
|
||||
|
||||
impl Manifest {
|
||||
fn decode(bytes: &[u8], limit: usize) -> Result<Self, SnapshotError> {
|
||||
if bytes.len() != MANIFEST_LEN || &bytes[..8] != MAGIC {
|
||||
return Err(SnapshotError::Corrupt);
|
||||
}
|
||||
if bytes[8] != VERSION {
|
||||
return Err(SnapshotError::Unsupported);
|
||||
}
|
||||
let signed = MANIFEST_LEN - 32;
|
||||
let checksum: [u8; 32] = Sha256::digest(&bytes[..signed]).into();
|
||||
if checksum != bytes[signed..] {
|
||||
return Err(SnapshotError::Corrupt);
|
||||
}
|
||||
let owner = Uuid::from_slice(&bytes[9..25]).map_err(|_| SnapshotError::Corrupt)?;
|
||||
let sequence = u64::from_le_bytes(bytes[25..33].try_into().map_err(|_| SnapshotError::Corrupt)?);
|
||||
let payload_len = u64::from_le_bytes(bytes[33..41].try_into().map_err(|_| SnapshotError::Corrupt)?);
|
||||
let payload_len = usize::try_from(payload_len).map_err(|_| SnapshotError::TooLarge)?;
|
||||
if owner.is_nil() || sequence == 0 || sequence == u64::MAX {
|
||||
return Err(SnapshotError::Corrupt);
|
||||
}
|
||||
if payload_len > limit {
|
||||
return Err(SnapshotError::TooLarge);
|
||||
}
|
||||
Ok(Self {
|
||||
owner,
|
||||
sequence,
|
||||
payload_len,
|
||||
payload_digest: bytes[41..73].try_into().map_err(|_| SnapshotError::Corrupt)?,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub struct CommittedSnapshot {
|
||||
manifest: Manifest,
|
||||
payload: Vec<u8>,
|
||||
}
|
||||
|
||||
impl CommittedSnapshot {
|
||||
/// Persistent single-writer sequence, not a process UUID ordering.
|
||||
pub fn sequence(&self) -> u64 {
|
||||
self.manifest.sequence
|
||||
}
|
||||
|
||||
/// Identity recorded by the committed checkpoint's writer.
|
||||
pub fn owner(&self) -> Uuid {
|
||||
self.manifest.owner
|
||||
}
|
||||
|
||||
/// Complete, checksum-validated record bytes. Inspection does not consume
|
||||
/// these records or acknowledge completion to any producer.
|
||||
pub fn payload(&self) -> &[u8] {
|
||||
&self.payload
|
||||
}
|
||||
|
||||
fn decode(manifest: &[u8], payload: Vec<u8>, limit: usize) -> Result<Self, SnapshotError> {
|
||||
let manifest = Manifest::decode(manifest, limit)?;
|
||||
let checksum: [u8; 32] = Sha256::digest(&payload).into();
|
||||
if payload.len() != manifest.payload_len || checksum != manifest.payload_digest {
|
||||
return Err(SnapshotError::Corrupt);
|
||||
}
|
||||
if decode_journal(&payload).1 != 0 {
|
||||
return Err(SnapshotError::Corrupt);
|
||||
}
|
||||
Ok(Self { manifest, payload })
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
pub enum RecoverySnapshot {
|
||||
/// An intact legacy snapshot, without a comparable commit sequence.
|
||||
Legacy(Vec<u8>),
|
||||
/// A committed checkpoint requiring ownership-aware replay before use.
|
||||
Committed(CommittedSnapshot),
|
||||
}
|
||||
|
||||
async fn read_bounded(disk: &EcstoreDiskStore, path: &str, limit: usize) -> Result<Option<Vec<u8>>, SnapshotError> {
|
||||
let reader = match EcstoreDiskAPI::read_file(disk.as_ref(), RUSTFS_META_BUCKET, path).await {
|
||||
Ok(reader) => reader,
|
||||
Err(EcstoreDiskError::FileNotFound | EcstoreDiskError::VolumeNotFound) => return Ok(None),
|
||||
Err(error) => return Err(SnapshotError::Disk(error)),
|
||||
};
|
||||
let maximum = limit.checked_add(1).ok_or(SnapshotError::TooLarge)?;
|
||||
let maximum = u64::try_from(maximum).map_err(|_| SnapshotError::TooLarge)?;
|
||||
let mut bytes = Vec::new();
|
||||
reader
|
||||
.take(maximum)
|
||||
.read_to_end(&mut bytes)
|
||||
.await
|
||||
.map_err(SnapshotError::Read)?;
|
||||
if bytes.len() > limit {
|
||||
return Err(SnapshotError::TooLarge);
|
||||
}
|
||||
Ok(Some(bytes))
|
||||
}
|
||||
|
||||
fn select_snapshot(selected: &mut Option<CommittedSnapshot>, candidate: CommittedSnapshot) -> Result<(), SnapshotError> {
|
||||
if let Some(current) = selected {
|
||||
if current.manifest.sequence == candidate.manifest.sequence
|
||||
&& (current.manifest != candidate.manifest || current.payload != candidate.payload)
|
||||
{
|
||||
return Err(SnapshotError::Conflict);
|
||||
}
|
||||
if current.manifest.sequence >= candidate.manifest.sequence {
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
*selected = Some(candidate);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn read_committed(disks: &[EcstoreDiskStore], limit: usize) -> Result<Option<CommittedSnapshot>, SnapshotError> {
|
||||
let mut selected = None;
|
||||
let mut damaged = None;
|
||||
let mut identities = HashMap::new();
|
||||
for disk in disks {
|
||||
for (manifest_path, payload_path) in MANIFEST_PATHS.into_iter().zip(PAYLOAD_PATHS) {
|
||||
let candidate = async {
|
||||
let Some(manifest) = read_bounded(disk, manifest_path, MANIFEST_LEN).await? else {
|
||||
return Ok(None);
|
||||
};
|
||||
let header = Manifest::decode(&manifest, limit)?;
|
||||
let payload = read_bounded(disk, payload_path, header.payload_len)
|
||||
.await?
|
||||
.ok_or(SnapshotError::Corrupt)?;
|
||||
CommittedSnapshot::decode(&manifest, payload, limit).map(Some)
|
||||
}
|
||||
.await;
|
||||
match candidate {
|
||||
Ok(Some(candidate)) => {
|
||||
let identity = (
|
||||
candidate.manifest.owner,
|
||||
candidate.manifest.payload_len,
|
||||
candidate.manifest.payload_digest,
|
||||
);
|
||||
if identities
|
||||
.insert(candidate.manifest.sequence, identity)
|
||||
.is_some_and(|previous| previous != identity)
|
||||
{
|
||||
return Err(SnapshotError::Conflict);
|
||||
}
|
||||
select_snapshot(&mut selected, candidate)?;
|
||||
}
|
||||
Ok(None) => {}
|
||||
// A future committed format may supersede all readable slots.
|
||||
Err(SnapshotError::Unsupported) => return Err(SnapshotError::Unsupported),
|
||||
Err(error) => damaged = Some(error),
|
||||
}
|
||||
}
|
||||
}
|
||||
match (selected, damaged) {
|
||||
(Some(snapshot), _) => Ok(Some(snapshot)),
|
||||
(None, Some(error)) => Err(error),
|
||||
(None, None) => Ok(None),
|
||||
}
|
||||
}
|
||||
|
||||
async fn read_legacy(disks: &[EcstoreDiskStore], path: &str, limit: usize) -> Result<Option<Vec<u8>>, SnapshotError> {
|
||||
let mut selected = None;
|
||||
let mut incomplete: Option<Vec<u8>> = None;
|
||||
for disk in disks {
|
||||
match read_bounded(disk, path, limit).await {
|
||||
Ok(Some(payload)) if decode_journal(&payload).1 == 0 => {
|
||||
if selected.as_ref().is_some_and(|current| *current != payload) {
|
||||
// Legacy snapshots have no sequence. There is no evidence
|
||||
// that the first, longest or nonempty replica is newest.
|
||||
return Err(SnapshotError::Conflict);
|
||||
}
|
||||
selected = Some(payload);
|
||||
}
|
||||
Ok(Some(payload)) => {
|
||||
if let Some(previous) = &incomplete {
|
||||
if previous.starts_with(&payload) {
|
||||
continue;
|
||||
}
|
||||
if !payload.starts_with(previous) {
|
||||
return Err(SnapshotError::Corrupt);
|
||||
}
|
||||
}
|
||||
incomplete = Some(payload);
|
||||
}
|
||||
Ok(None) => {}
|
||||
Err(error) => return Err(error),
|
||||
}
|
||||
}
|
||||
if let Some(prefix) = incomplete
|
||||
&& !selected.as_ref().is_some_and(|payload| payload.starts_with(&prefix))
|
||||
{
|
||||
// In particular, an empty O_TRUNC replica cannot supersede another
|
||||
// replica containing intact records followed by a torn tail.
|
||||
return Err(SnapshotError::Corrupt);
|
||||
}
|
||||
Ok(selected)
|
||||
}
|
||||
|
||||
/// Inspect local MRF checkpoints without replaying, acknowledging or deleting.
|
||||
///
|
||||
/// `max_bytes` bounds each payload read. Every local replica is examined and
|
||||
/// ambiguous identities, unavailable proof or unsupported formats return a
|
||||
/// typed error. This API must not authorize a writer without the separate
|
||||
/// ownership and mixed-version activation checks.
|
||||
pub async fn inspect_local_recovery_snapshot(max_bytes: usize) -> Result<Option<RecoverySnapshot>, SnapshotError> {
|
||||
read_recovery_snapshot(&super::journal_disks().await, max_bytes).await
|
||||
}
|
||||
|
||||
async fn read_recovery_snapshot(disks: &[EcstoreDiskStore], limit: usize) -> Result<Option<RecoverySnapshot>, SnapshotError> {
|
||||
if let Some(snapshot) = read_committed(disks, limit).await? {
|
||||
return Ok(Some(RecoverySnapshot::Committed(snapshot)));
|
||||
}
|
||||
// RUSTFS_COMPAT_TODO(backlog-2263): inspect retained legacy MRF journals. Remove after all supported upgrade and rollback readers understand committed snapshots and retained journals have migrated.
|
||||
if let Some(payload) = read_legacy(disks, MRF_SCOPED_JOURNAL_PATH, limit).await? {
|
||||
return Ok(Some(RecoverySnapshot::Legacy(payload)));
|
||||
}
|
||||
Ok(read_legacy(disks, MRF_JOURNAL_PATH, limit)
|
||||
.await?
|
||||
.map(RecoverySnapshot::Legacy))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::heal::mrf_queue::encode_intent;
|
||||
use crate::heal::storage_api::owner::{EcstoreConditionalFileUpdate, EcstoreDiskBytes};
|
||||
use crate::heal::{DiskOption, Endpoint, new_disk};
|
||||
use rustfs_common::mrf_channel::{MrfIntent, MrfKind, MrfScope};
|
||||
use std::sync::Arc;
|
||||
use tempfile::TempDir;
|
||||
|
||||
fn payload(object: &str) -> Vec<u8> {
|
||||
let intent = MrfIntent {
|
||||
bucket: Arc::from("bucket"),
|
||||
object: Arc::from(object),
|
||||
version_id: None,
|
||||
kind: MrfKind::PartialWrite,
|
||||
scope: None,
|
||||
lease: None,
|
||||
enqueued_at_ms: 1234,
|
||||
attempts: 0,
|
||||
};
|
||||
let mut bytes = Vec::new();
|
||||
assert!(encode_intent(&intent, &mut bytes), "fixture must encode a full record");
|
||||
bytes
|
||||
}
|
||||
|
||||
fn manifest(owner: Uuid, sequence: u64, payload: &[u8]) -> Vec<u8> {
|
||||
let mut bytes = Vec::with_capacity(MANIFEST_LEN);
|
||||
bytes.extend_from_slice(MAGIC);
|
||||
bytes.push(VERSION);
|
||||
bytes.extend_from_slice(owner.as_bytes());
|
||||
bytes.extend_from_slice(&sequence.to_le_bytes());
|
||||
bytes.extend_from_slice(&u64::try_from(payload.len()).expect("fixture length fits").to_le_bytes());
|
||||
bytes.extend_from_slice(&Sha256::digest(payload));
|
||||
bytes.extend_from_slice(&Sha256::digest(&bytes));
|
||||
bytes
|
||||
}
|
||||
|
||||
async fn disk(root: &TempDir, name: &str) -> EcstoreDiskStore {
|
||||
let path = root.path().join(name);
|
||||
std::fs::create_dir_all(&path).expect("create disk directory");
|
||||
let endpoint = Endpoint::try_from(path.to_string_lossy().as_ref()).expect("valid disk endpoint");
|
||||
let disk = new_disk(
|
||||
&endpoint,
|
||||
&DiskOption {
|
||||
cleanup: false,
|
||||
health_check: false,
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("open disk");
|
||||
let result = EcstoreDiskAPI::make_volume(disk.as_ref(), RUSTFS_META_BUCKET).await;
|
||||
assert!(
|
||||
matches!(result, Ok(()) | Err(EcstoreDiskError::VolumeExists)),
|
||||
"metadata volume: {result:?}"
|
||||
);
|
||||
disk
|
||||
}
|
||||
|
||||
// Exercise the existing storage owner's atomic CAS primitive. No production
|
||||
// caller publishes this format until ownership-aware replay is available.
|
||||
async fn install(disk: &EcstoreDiskStore, path: &str, bytes: &[u8]) {
|
||||
let expected = EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, path).await.ok();
|
||||
let result = EcstoreDiskAPI::compare_and_update_file(
|
||||
disk.as_ref(),
|
||||
RUSTFS_META_BUCKET,
|
||||
path,
|
||||
expected,
|
||||
Some(EcstoreDiskBytes::copy_from_slice(bytes)),
|
||||
)
|
||||
.await
|
||||
.expect("atomic snapshot slot write");
|
||||
assert_eq!(result, EcstoreConditionalFileUpdate::Updated);
|
||||
}
|
||||
|
||||
async fn commit(disk: &EcstoreDiskStore, slot: usize, owner: Uuid, sequence: u64, bytes: &[u8]) {
|
||||
install(disk, PAYLOAD_PATHS[slot], bytes).await;
|
||||
install(disk, MANIFEST_PATHS[slot], &manifest(owner, sequence, bytes)).await;
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn manifest_validates_identity_sequence_length_and_digest() {
|
||||
let bytes = payload("object");
|
||||
let owner = Uuid::new_v4();
|
||||
assert!(CommittedSnapshot::decode(&manifest(owner, 1, &bytes), bytes.clone(), bytes.len()).is_ok());
|
||||
for (owner, sequence) in [(Uuid::nil(), 1), (owner, 0), (owner, u64::MAX)] {
|
||||
assert!(matches!(
|
||||
Manifest::decode(&manifest(owner, sequence, &bytes), bytes.len()),
|
||||
Err(SnapshotError::Corrupt)
|
||||
));
|
||||
}
|
||||
assert!(matches!(
|
||||
Manifest::decode(&manifest(owner, 1, &bytes), bytes.len() - 1),
|
||||
Err(SnapshotError::TooLarge)
|
||||
));
|
||||
let mut corrupt = manifest(owner, 1, &bytes);
|
||||
corrupt[25] ^= 1;
|
||||
assert!(matches!(Manifest::decode(&corrupt, bytes.len()), Err(SnapshotError::Corrupt)));
|
||||
let mut unsupported = manifest(owner, 1, &bytes);
|
||||
unsupported[8] = 2;
|
||||
assert!(matches!(Manifest::decode(&unsupported, bytes.len()), Err(SnapshotError::Unsupported)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn whole_payload_integrity_is_required_even_with_a_valid_manifest() {
|
||||
let bytes = payload("object");
|
||||
let owner = Uuid::new_v4();
|
||||
let header = manifest(owner, 1, &bytes);
|
||||
assert!(matches!(
|
||||
CommittedSnapshot::decode(&header, bytes[..bytes.len() - 1].to_vec(), bytes.len()),
|
||||
Err(SnapshotError::Corrupt)
|
||||
));
|
||||
let invalid = b"not an MRF record".to_vec();
|
||||
assert!(matches!(
|
||||
CommittedSnapshot::decode(&manifest(owner, 2, &invalid), invalid, bytes.len()),
|
||||
Err(SnapshotError::Corrupt)
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn newest_complete_replica_wins_in_both_disk_orders() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let first = disk(&root, "first").await;
|
||||
let second = disk(&root, "second").await;
|
||||
let owner = Uuid::new_v4();
|
||||
commit(&first, 0, owner, 1, &payload("old")).await;
|
||||
commit(&second, 1, owner, 2, &payload("new")).await;
|
||||
for disks in [vec![first.clone(), second.clone()], vec![second.clone(), first.clone()]] {
|
||||
let recovered = read_committed(&disks, 4096)
|
||||
.await
|
||||
.expect("read replicas")
|
||||
.expect("committed snapshot");
|
||||
assert_eq!(recovered.manifest.sequence, 2);
|
||||
assert_eq!(recovered.payload, payload("new"));
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn divergent_commits_at_same_sequence_fail_closed() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let first = disk(&root, "first").await;
|
||||
let second = disk(&root, "second").await;
|
||||
let owner = Uuid::new_v4();
|
||||
commit(&first, 0, owner, 7, &payload("a")).await;
|
||||
commit(&second, 1, owner, 7, &payload("b")).await;
|
||||
assert!(matches!(read_committed(&[first, second], 4096).await, Err(SnapshotError::Conflict)));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn newer_slot_does_not_hide_a_conflicting_commit_history() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let first = disk(&root, "first").await;
|
||||
let second = disk(&root, "second").await;
|
||||
let owner = Uuid::new_v4();
|
||||
commit(&first, 0, owner, 8, &payload("newest")).await;
|
||||
commit(&first, 1, owner, 7, &payload("a")).await;
|
||||
commit(&second, 1, owner, 7, &payload("b")).await;
|
||||
assert!(matches!(read_committed(&[first, second], 4096).await, Err(SnapshotError::Conflict)));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn uncommitted_or_torn_successor_preserves_previous_slot() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let disk = disk(&root, "disk").await;
|
||||
let owner = Uuid::new_v4();
|
||||
let old = payload("old");
|
||||
let next = payload("next");
|
||||
commit(&disk, 0, owner, 1, &old).await;
|
||||
install(&disk, PAYLOAD_PATHS[1], &next).await;
|
||||
let recovered = read_committed(std::slice::from_ref(&disk), 4096)
|
||||
.await
|
||||
.expect("staged payload is not a commit")
|
||||
.expect("old snapshot");
|
||||
assert_eq!(recovered.payload, old);
|
||||
install(&disk, MANIFEST_PATHS[1], &manifest(owner, 2, &next)[..20]).await;
|
||||
let recovered = read_committed(std::slice::from_ref(&disk), 4096)
|
||||
.await
|
||||
.expect("torn manifest preserves old slot")
|
||||
.expect("old snapshot");
|
||||
assert_eq!(recovered.manifest.sequence, 1);
|
||||
install(&disk, MANIFEST_PATHS[1], &manifest(owner, 2, &next)).await;
|
||||
install(&disk, PAYLOAD_PATHS[1], b"torn").await;
|
||||
let recovered = read_committed(&[disk], 4096)
|
||||
.await
|
||||
.expect("torn payload preserves old slot")
|
||||
.expect("old snapshot");
|
||||
assert_eq!(recovered.manifest.sequence, 1);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn stale_manifest_cas_cannot_replace_committed_anchor() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let disk = disk(&root, "disk").await;
|
||||
let owner = Uuid::new_v4();
|
||||
let bytes = payload("object");
|
||||
commit(&disk, 0, owner, 1, &bytes).await;
|
||||
let result = EcstoreDiskAPI::compare_and_update_file(
|
||||
disk.as_ref(),
|
||||
RUSTFS_META_BUCKET,
|
||||
MANIFEST_PATHS[0],
|
||||
None,
|
||||
Some(manifest(owner, 2, &bytes).into()),
|
||||
)
|
||||
.await
|
||||
.expect("CAS call");
|
||||
assert_eq!(result, EcstoreConditionalFileUpdate::Mismatch);
|
||||
let recovered = read_committed(&[disk], 4096)
|
||||
.await
|
||||
.expect("read old anchor")
|
||||
.expect("snapshot");
|
||||
assert_eq!(recovered.manifest.sequence, 1);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn legacy_import_requires_complete_consistent_replicas() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let first = disk(&root, "first").await;
|
||||
let second = disk(&root, "second").await;
|
||||
let bytes = payload("object");
|
||||
for (disk, data) in [(&first, &bytes[..bytes.len() - 1]), (&second, bytes.as_slice())] {
|
||||
EcstoreDiskAPI::write_all(
|
||||
disk.as_ref(),
|
||||
RUSTFS_META_BUCKET,
|
||||
MRF_SCOPED_JOURNAL_PATH,
|
||||
EcstoreDiskBytes::copy_from_slice(data),
|
||||
)
|
||||
.await
|
||||
.expect("legacy fixture");
|
||||
}
|
||||
let disks = [first.clone(), second];
|
||||
assert!(
|
||||
matches!(read_recovery_snapshot(&disks, 4096).await.expect("intact legacy replica"), Some(RecoverySnapshot::Legacy(data)) if data == bytes)
|
||||
);
|
||||
EcstoreDiskAPI::write_all(first.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, payload("different").into())
|
||||
.await
|
||||
.expect("divergent fixture");
|
||||
assert!(matches!(read_recovery_snapshot(&disks, 4096).await, Err(SnapshotError::Conflict)));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn committed_inspection_leaves_payload_and_manifest_unchanged() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let disk = disk(&root, "disk").await;
|
||||
let owner = Uuid::new_v4();
|
||||
let bytes = payload("object");
|
||||
commit(&disk, 0, owner, 3, &bytes).await;
|
||||
assert!(matches!(
|
||||
read_recovery_snapshot(std::slice::from_ref(&disk), 4096)
|
||||
.await
|
||||
.expect("new snapshot"),
|
||||
Some(RecoverySnapshot::Committed(_))
|
||||
));
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, MANIFEST_PATHS[0])
|
||||
.await
|
||||
.expect("manifest retained")
|
||||
.as_ref(),
|
||||
manifest(owner, 3, &bytes)
|
||||
);
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, PAYLOAD_PATHS[0])
|
||||
.await
|
||||
.expect("payload retained")
|
||||
.as_ref(),
|
||||
bytes
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn legacy_inspection_rejects_complete_subsets_and_scope_ambiguity() {
|
||||
let scoped = |set_index| {
|
||||
let intent = MrfIntent {
|
||||
bucket: Arc::from("bucket"),
|
||||
object: Arc::from("a"),
|
||||
version_id: None,
|
||||
kind: MrfKind::PartialWrite,
|
||||
scope: Some(MrfScope {
|
||||
pool_index: 0,
|
||||
set_index,
|
||||
}),
|
||||
lease: None,
|
||||
enqueued_at_ms: 1234,
|
||||
attempts: 0,
|
||||
};
|
||||
let mut bytes = Vec::new();
|
||||
assert!(encode_intent(&intent, &mut bytes), "scoped fixture must encode");
|
||||
bytes
|
||||
};
|
||||
let mut superset = payload("a");
|
||||
superset.extend_from_slice(&payload("b"));
|
||||
for (case, first_bytes, second_bytes) in [
|
||||
("complete-subset", payload("a"), superset),
|
||||
("different-set", scoped(1), scoped(2)),
|
||||
("unknown-scope", payload("a"), scoped(1)),
|
||||
] {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let first = disk(&root, "first").await;
|
||||
let second = disk(&root, "second").await;
|
||||
for (disk, bytes) in [(&first, &first_bytes), (&second, &second_bytes)] {
|
||||
assert_eq!(decode_journal(bytes).1, 0, "{case}: complete fixture");
|
||||
EcstoreDiskAPI::write_all(
|
||||
disk.as_ref(),
|
||||
RUSTFS_META_BUCKET,
|
||||
MRF_SCOPED_JOURNAL_PATH,
|
||||
EcstoreDiskBytes::copy_from_slice(bytes),
|
||||
)
|
||||
.await
|
||||
.expect("write legacy replica");
|
||||
}
|
||||
for disks in [vec![first.clone(), second.clone()], vec![second.clone(), first.clone()]] {
|
||||
assert!(
|
||||
matches!(read_recovery_snapshot(&disks, 4096).await, Err(SnapshotError::Conflict)),
|
||||
"{case}: neither replica order proves a latest snapshot"
|
||||
);
|
||||
}
|
||||
for (disk, bytes) in [(&first, &first_bytes), (&second, &second_bytes)] {
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH)
|
||||
.await
|
||||
.expect("legacy evidence retained")
|
||||
.as_ref(),
|
||||
bytes.as_slice(),
|
||||
"{case}: inspection must preserve both source replicas"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn oversized_or_corrupt_scoped_snapshot_never_falls_back_to_legacy() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let disk = disk(&root, "disk").await;
|
||||
EcstoreDiskAPI::write_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, vec![0; 1025].into())
|
||||
.await
|
||||
.expect("oversized fixture");
|
||||
EcstoreDiskAPI::write_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_JOURNAL_PATH, payload("old").into())
|
||||
.await
|
||||
.expect("legacy fixture");
|
||||
assert!(matches!(
|
||||
read_recovery_snapshot(std::slice::from_ref(&disk), 1024).await,
|
||||
Err(SnapshotError::TooLarge)
|
||||
));
|
||||
assert!(matches!(read_recovery_snapshot(&[disk], 2048).await, Err(SnapshotError::Corrupt)));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn empty_legacy_replica_cannot_erase_records_in_a_torn_replica() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let first = disk(&root, "first").await;
|
||||
let second = disk(&root, "second").await;
|
||||
let mut incomplete = payload("durable-object");
|
||||
incomplete.extend_from_slice(b"torn");
|
||||
EcstoreDiskAPI::write_all(first.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, Vec::new().into())
|
||||
.await
|
||||
.expect("empty truncated replica");
|
||||
EcstoreDiskAPI::write_all(second.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, incomplete.clone().into())
|
||||
.await
|
||||
.expect("records and torn tail");
|
||||
for disks in [vec![first.clone(), second.clone()], vec![second.clone(), first.clone()]] {
|
||||
assert!(matches!(read_recovery_snapshot(&disks, 4096).await, Err(SnapshotError::Corrupt)));
|
||||
}
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(second.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH)
|
||||
.await
|
||||
.expect("recovery anchor preserved")
|
||||
.as_ref(),
|
||||
incomplete
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn unreadable_commit_record_never_implies_legacy_only() {
|
||||
let root = TempDir::new().expect("test directory");
|
||||
let disk = disk(&root, "disk").await;
|
||||
let legacy = payload("old");
|
||||
EcstoreDiskAPI::write_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_JOURNAL_PATH, legacy.clone().into())
|
||||
.await
|
||||
.expect("legacy fixture");
|
||||
// Opening a directory as a record either fails at open or at read,
|
||||
// depending on the platform. Neither outcome proves absence.
|
||||
std::fs::create_dir(root.path().join("disk").join(RUSTFS_META_BUCKET).join(MANIFEST_PATHS[0]))
|
||||
.expect("unreadable manifest fixture");
|
||||
let recovered = read_recovery_snapshot(std::slice::from_ref(&disk), 4096).await;
|
||||
assert!(
|
||||
matches!(recovered, Err(SnapshotError::Disk(_) | SnapshotError::Read(_))),
|
||||
"must preserve unavailable proof: {recovered:?}"
|
||||
);
|
||||
assert_eq!(
|
||||
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_JOURNAL_PATH)
|
||||
.await
|
||||
.expect("legacy remains")
|
||||
.as_ref(),
|
||||
legacy
|
||||
);
|
||||
}
|
||||
}
|
||||
+477
-62
@@ -43,6 +43,10 @@ const ERR_LIFECYCLE_BUCKET_LOCKED: &str =
|
||||
"ExpiredObjectAllVersions element and DelMarkerExpiration action cannot be used on an object locked bucket";
|
||||
const ERR_LIFECYCLE_TOO_MANY_RULES: &str = "Lifecycle configuration should have at most 1000 rules";
|
||||
const ERR_LIFECYCLE_INVALID_EXPIRATION_DAYS: &str = "'Days' for Expiration action must be a positive integer";
|
||||
const ERR_LIFECYCLE_EXPIRATION_DAYS_DATE_CONFLICT: &str = "Expiration cannot specify both Days and Date";
|
||||
const ERR_LIFECYCLE_MULTIPLE_TRANSITIONS: &str = "Only one Transition action per lifecycle rule is supported";
|
||||
const ERR_LIFECYCLE_MULTIPLE_NONCURRENT_TRANSITIONS: &str =
|
||||
"Only one NoncurrentVersionTransition action per lifecycle rule is supported";
|
||||
const ERR_LIFECYCLE_INVALID_NONCURRENT_EXPIRATION_DAYS: &str =
|
||||
"'NoncurrentDays' for NoncurrentVersionExpiration action must be a positive integer";
|
||||
const ERR_LIFECYCLE_INVALID_ABORT_INCOMPLETE_MPU_DAYS: &str =
|
||||
@@ -514,6 +518,12 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
{
|
||||
return Err(std::io::Error::other(ERR_LIFECYCLE_INVALID_EXPIRED_OBJECT_ALL_VERSIONS));
|
||||
}
|
||||
if expiration.days.is_some() && expiration.date.is_some() {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::InvalidInput,
|
||||
ERR_LIFECYCLE_EXPIRATION_DAYS_DATE_CONFLICT,
|
||||
));
|
||||
}
|
||||
if let Some(expiration_date) = &expiration.date {
|
||||
let date = OffsetDateTime::from(expiration_date.clone());
|
||||
if date.hour() != 0 || date.minute() != 0 || date.second() != 0 || date.nanosecond() != 0 {
|
||||
@@ -547,11 +557,20 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
}
|
||||
}
|
||||
if let Some(transitions) = &r.transitions {
|
||||
if transitions.len() > 1 {
|
||||
return Err(std::io::Error::new(std::io::ErrorKind::InvalidInput, ERR_LIFECYCLE_MULTIPLE_TRANSITIONS));
|
||||
}
|
||||
for transition in transitions {
|
||||
TransitionOps::validate(transition)?;
|
||||
}
|
||||
}
|
||||
if let Some(noncurrent_transitions) = &r.noncurrent_version_transitions {
|
||||
if noncurrent_transitions.len() > 1 {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::InvalidInput,
|
||||
ERR_LIFECYCLE_MULTIPLE_NONCURRENT_TRANSITIONS,
|
||||
));
|
||||
}
|
||||
for transition in noncurrent_transitions {
|
||||
NoncurrentVersionTransitionOps::validate(transition)?;
|
||||
}
|
||||
@@ -626,6 +645,8 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
}
|
||||
|
||||
async fn eval(&self, obj: &ObjectOpts) -> Event {
|
||||
// A single-object lookup cannot prove how many newer historical versions
|
||||
// survive. Count-dependent actions wait for the complete-group evaluator.
|
||||
self.eval_inner(obj, OffsetDateTime::now_utc(), 0).await
|
||||
}
|
||||
|
||||
@@ -689,23 +710,8 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
return Event::default();
|
||||
};
|
||||
|
||||
if let Some(restore_expires) = obj.restore_expires
|
||||
&& restore_expires.unix_timestamp() != 0
|
||||
&& now.unix_timestamp() > restore_expires.unix_timestamp()
|
||||
{
|
||||
let mut action = IlmAction::DeleteRestoredAction;
|
||||
if !obj.is_latest {
|
||||
action = IlmAction::DeleteRestoredVersionAction;
|
||||
}
|
||||
|
||||
events.push(Event {
|
||||
action,
|
||||
due: Some(now),
|
||||
rule_id: "".into(),
|
||||
noncurrent_days: 0,
|
||||
newer_noncurrent_versions: 0,
|
||||
storage_class: "".into(),
|
||||
});
|
||||
if let Some(event) = obj.restored_copy_expiry(now) {
|
||||
events.push(event);
|
||||
}
|
||||
|
||||
if let Some(ref lc_rules) = self.filter_rules(obj).await {
|
||||
@@ -781,21 +787,15 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
continue;
|
||||
}
|
||||
|
||||
if !obj.is_latest
|
||||
&& let Some(ref noncurrent_version_expiration) = rule.noncurrent_version_expiration
|
||||
&& let Some(retain_newer_noncurrent_versions) = noncurrent_version_expiration.newer_noncurrent_versions
|
||||
&& let Some(retained) = retained_noncurrent_versions(retain_newer_noncurrent_versions)
|
||||
&& newer_noncurrent_versions < retained
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if !obj.is_latest
|
||||
&& let Some(ref noncurrent_version_expiration) = rule.noncurrent_version_expiration
|
||||
&& (noncurrent_version_expiration.noncurrent_days.is_some()
|
||||
|| noncurrent_version_expiration
|
||||
.newer_noncurrent_versions
|
||||
.is_some_and(|count| count > 0))
|
||||
&& noncurrent_version_expiration
|
||||
.newer_noncurrent_versions
|
||||
.is_none_or(|retain| usize::try_from(retain).is_ok_and(|retain| newer_noncurrent_versions >= retain))
|
||||
{
|
||||
// A count-only rule (MinIO extension) has no age condition:
|
||||
// every version past the retained count is due as soon as it
|
||||
@@ -829,7 +829,11 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
&& let Some(noncurrent_version_transition) = rule
|
||||
.noncurrent_version_transitions
|
||||
.as_ref()
|
||||
.filter(|transitions| transitions.len() == 1)
|
||||
.and_then(|transitions| transitions.first())
|
||||
&& noncurrent_version_transition
|
||||
.newer_noncurrent_versions
|
||||
.is_none_or(|retain| usize::try_from(retain).is_ok_and(|retain| newer_noncurrent_versions >= retain))
|
||||
&& let Some(storage_class) = noncurrent_version_transition.storage_class.as_ref()
|
||||
&& !storage_class.as_str().is_empty()
|
||||
&& !obj.delete_marker
|
||||
@@ -913,7 +917,11 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
}
|
||||
|
||||
if obj.transition_status != TRANSITION_COMPLETE
|
||||
&& let Some(transition) = rule.transitions.as_ref().and_then(|transitions| transitions.first())
|
||||
&& let Some(transition) = rule
|
||||
.transitions
|
||||
.as_ref()
|
||||
.filter(|transitions| transitions.len() == 1)
|
||||
.and_then(|transitions| transitions.first())
|
||||
&& let Some(storage_class) = transition.storage_class.as_ref()
|
||||
&& !storage_class.as_str().is_empty()
|
||||
{
|
||||
@@ -936,18 +944,15 @@ impl Lifecycle for BucketLifecycleConfiguration {
|
||||
}
|
||||
|
||||
if !events.is_empty() {
|
||||
// Select the winning event using a strict total order (MinIO semantics):
|
||||
// the earliest `due` wins, and ties break toward delete-type actions. A
|
||||
// missing `due` is treated as UNIX_EPOCH. This replaces a hand-written
|
||||
// `sort_by` comparator that was not a strict weak ordering (it could return
|
||||
// `Ordering::Less` for both `(a, b)` and `(b, a)`), which panics on the
|
||||
// repository toolchain and did not deterministically pick the earliest event.
|
||||
// Eligible expiration takes precedence over transition, even when a
|
||||
// failed transition has an earlier deadline. Within each action class,
|
||||
// prefer the earliest deadline using a deterministic total order.
|
||||
let event = events
|
||||
.iter()
|
||||
.min_by_key(|event| {
|
||||
(
|
||||
event.due.unwrap_or(OffsetDateTime::UNIX_EPOCH).unix_timestamp(),
|
||||
ilm_action_priority_rank(&event.action),
|
||||
event.due.unwrap_or(OffsetDateTime::UNIX_EPOCH).unix_timestamp(),
|
||||
)
|
||||
})
|
||||
.cloned()
|
||||
@@ -1223,6 +1228,27 @@ impl ObjectOpts {
|
||||
pub fn expired_object_deletemarker(&self) -> bool {
|
||||
self.delete_marker && self.is_latest && self.num_versions == 1
|
||||
}
|
||||
|
||||
pub(crate) fn restored_copy_expiry(&self, now: OffsetDateTime) -> Option<Event> {
|
||||
let restore_expires = self.restore_expires?;
|
||||
// Restore metadata alone does not prove that a durable remote copy exists.
|
||||
if self.transition_status != TRANSITION_COMPLETE
|
||||
|| restore_expires.unix_timestamp() == 0
|
||||
|| now.unix_timestamp() <= restore_expires.unix_timestamp()
|
||||
{
|
||||
return None;
|
||||
}
|
||||
let action = if self.is_latest {
|
||||
IlmAction::DeleteRestoredAction
|
||||
} else {
|
||||
IlmAction::DeleteRestoredVersionAction
|
||||
};
|
||||
expiration_action_has_valid_target(action, self.version_id, self.is_latest, self.delete_marker).then(|| Event {
|
||||
action,
|
||||
due: Some(now),
|
||||
..Default::default()
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns whether an expiry action has enough identity to target the object
|
||||
@@ -1245,11 +1271,8 @@ pub fn expiration_action_has_valid_target(
|
||||
}
|
||||
}
|
||||
|
||||
/// Total-order rank for lifecycle actions used to break `due` ties.
|
||||
///
|
||||
/// Delete-type actions rank before every other action so that, when two events
|
||||
/// share the same `due`, a delete wins (MinIO semantics). The concrete numeric
|
||||
/// values only matter relative to each other.
|
||||
/// Eligible logical expiration takes precedence over transition and restore-copy
|
||||
/// cleanup. Deadlines break ties within an action class.
|
||||
fn ilm_action_priority_rank(action: &IlmAction) -> u8 {
|
||||
match action {
|
||||
IlmAction::DeleteAllVersionsAction
|
||||
@@ -4344,24 +4367,6 @@ mod tests {
|
||||
assert_eq!(event.action, IlmAction::NoneAction);
|
||||
}
|
||||
|
||||
// Property-based tests for the rule evaluator (backlog#1148 ilm-14,
|
||||
// follow-up to backlog#1030 / rustfs#4455).
|
||||
//
|
||||
// backlog#1030 found that the old hand-written winner comparator was not a
|
||||
// strict weak ordering and could panic — proof that enumerated cases do
|
||||
// not cover the space of colliding events. These properties pin, over
|
||||
// randomized rule sets and object states:
|
||||
//
|
||||
// * `eval_inner` never panics and is deterministic for a fixed input;
|
||||
// * the winning event matches an independently recomputed candidate set:
|
||||
// earliest `due` wins, ties break toward delete-class actions (the
|
||||
// `min_by_key` selection that replaced the rustfs#4455 comparator);
|
||||
// * `expected_expiry_time` is monotonically non-decreasing in `days` and
|
||||
// always lands on the processing boundary, both at production defaults
|
||||
// and under an explicit `RUSTFS_ILM_PROCESS_TIME`.
|
||||
//
|
||||
// Case counts are tuned so the whole module runs in seconds inside the
|
||||
// default CI test job.
|
||||
// ---- backlog#2201: retention-count and Filter invariants -----------------
|
||||
|
||||
fn rule_with_noncurrent_expiration(expiration: NoncurrentVersionExpiration) -> LifecycleRule {
|
||||
@@ -4939,6 +4944,410 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
mod adversarial_regressions {
|
||||
use super::*;
|
||||
use s3s::dto::NoncurrentVersionExpiration;
|
||||
|
||||
fn run(test: impl std::future::Future<Output = ()>) {
|
||||
with_default_ilm_process_time(|| {
|
||||
tokio::runtime::Builder::new_current_thread()
|
||||
.build()
|
||||
.expect("lifecycle regression runtime should build")
|
||||
.block_on(test);
|
||||
});
|
||||
}
|
||||
|
||||
fn noncurrent_object() -> ObjectOpts {
|
||||
ObjectOpts {
|
||||
name: "logs/object".to_string(),
|
||||
mod_time: Some(datetime!(2020-01-01 00:00:00 UTC)),
|
||||
successor_mod_time: Some(datetime!(2020-01-02 00:00:00 UTC)),
|
||||
version_id: Some(Uuid::from_u128(1)),
|
||||
size: 1024 * 1024,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn noncurrent_transition_retains_the_requested_newer_versions() {
|
||||
run(async {
|
||||
let mut rule = enabled_rule(None, None, Some("retain-two-hot-versions"));
|
||||
rule.filter = Some(LifecycleRuleFilter::default());
|
||||
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
|
||||
noncurrent_days: Some(1),
|
||||
newer_noncurrent_versions: Some(2),
|
||||
storage_class: Some(TransitionStorageClass::from_static("WARM")),
|
||||
}]);
|
||||
let lc = Arc::new(BucketLifecycleConfiguration {
|
||||
rules: vec![rule],
|
||||
expiry_updated_at: None,
|
||||
});
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("valid noncurrent transition policy");
|
||||
let objects = (0..4)
|
||||
.map(|index| ObjectOpts {
|
||||
mod_time: Some(datetime!(2020-01-05 00:00:00 UTC) - Duration::days(index)),
|
||||
successor_mod_time: (index > 0).then_some(datetime!(2020-01-06 00:00:00 UTC) - Duration::days(index)),
|
||||
version_id: Some(Uuid::from_u128(u128::try_from(index + 1).expect("small version index"))),
|
||||
is_latest: index == 0,
|
||||
num_versions: 4,
|
||||
..noncurrent_object()
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let actions = crate::Evaluator::new(lc)
|
||||
.eval(&objects)
|
||||
.await
|
||||
.expect("complete version chain should evaluate")
|
||||
.into_iter()
|
||||
.map(|event| event.action)
|
||||
.collect::<Vec<_>>();
|
||||
assert_eq!(
|
||||
actions,
|
||||
[
|
||||
IlmAction::NoneAction,
|
||||
IlmAction::NoneAction,
|
||||
IlmAction::NoneAction,
|
||||
IlmAction::TransitionVersionAction
|
||||
],
|
||||
"the two newest noncurrent versions must remain in their current storage class"
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn noncurrent_transition_checks_count_age_and_single_object_context() {
|
||||
run(async {
|
||||
let mut rule = enabled_rule(None, None, Some("retain-two"));
|
||||
rule.filter = Some(LifecycleRuleFilter::default());
|
||||
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
|
||||
noncurrent_days: Some(3),
|
||||
newer_noncurrent_versions: Some(2),
|
||||
storage_class: Some(TransitionStorageClass::from_static("WARM")),
|
||||
}]);
|
||||
let mut lc = BucketLifecycleConfiguration {
|
||||
rules: vec![rule],
|
||||
expiry_updated_at: None,
|
||||
};
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("valid counted transition");
|
||||
let object = noncurrent_object();
|
||||
let now = datetime!(2020-01-10 00:00:00 UTC);
|
||||
for (newer, expected) in [
|
||||
(0, IlmAction::NoneAction),
|
||||
(1, IlmAction::NoneAction),
|
||||
(2, IlmAction::TransitionVersionAction),
|
||||
(3, IlmAction::TransitionVersionAction),
|
||||
] {
|
||||
assert_eq!(lc.eval_inner(&object, now, newer).await.action, expected, "newer count: {newer}");
|
||||
}
|
||||
assert_eq!(
|
||||
lc.eval_inner(&object, datetime!(2020-01-04 00:00:00 UTC), 2).await.action,
|
||||
IlmAction::NoneAction,
|
||||
"the retention count does not replace the age condition"
|
||||
);
|
||||
assert_eq!(
|
||||
lc.eval(&object).await.action,
|
||||
IlmAction::NoneAction,
|
||||
"a single-object lookup must not assume a complete version history"
|
||||
);
|
||||
for retain in [None, Some(0), Some(-1), Some(i32::MAX)] {
|
||||
lc.rules[0]
|
||||
.noncurrent_version_transitions
|
||||
.as_mut()
|
||||
.expect("transition exists")[0]
|
||||
.newer_noncurrent_versions = retain;
|
||||
let expected = if matches!(retain, None | Some(0)) {
|
||||
IlmAction::TransitionVersionAction
|
||||
} else {
|
||||
IlmAction::NoneAction
|
||||
};
|
||||
assert_eq!(lc.eval_inner(&object, now, 2).await.action, expected, "retention: {retain:?}");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn noncurrent_expiration_and_transition_have_independent_retention_counts() {
|
||||
run(async {
|
||||
let mut rule = enabled_rule(None, None, Some("independent-counts"));
|
||||
rule.filter = Some(LifecycleRuleFilter::default());
|
||||
rule.noncurrent_version_expiration = Some(NoncurrentVersionExpiration {
|
||||
noncurrent_days: Some(90),
|
||||
newer_noncurrent_versions: Some(4),
|
||||
});
|
||||
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
|
||||
noncurrent_days: Some(30),
|
||||
newer_noncurrent_versions: Some(2),
|
||||
storage_class: Some(TransitionStorageClass::from_static("WARM")),
|
||||
}]);
|
||||
let lc = BucketLifecycleConfiguration {
|
||||
rules: vec![rule],
|
||||
expiry_updated_at: None,
|
||||
};
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("valid independent retention limits");
|
||||
let object = noncurrent_object();
|
||||
let now = datetime!(2020-05-01 00:00:00 UTC);
|
||||
for (newer, expected) in [
|
||||
(1, IlmAction::NoneAction),
|
||||
(2, IlmAction::TransitionVersionAction),
|
||||
(3, IlmAction::TransitionVersionAction),
|
||||
(4, IlmAction::DeleteVersionAction),
|
||||
] {
|
||||
assert_eq!(lc.eval_inner(&object, now, newer).await.action, expected, "newer count: {newer}");
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn expiration_retention_does_not_skip_an_independent_transition() {
|
||||
run(async {
|
||||
let mut rule = enabled_rule(None, None, Some("transition-then-expire"));
|
||||
rule.filter = Some(LifecycleRuleFilter::default());
|
||||
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
|
||||
noncurrent_days: Some(1),
|
||||
newer_noncurrent_versions: None,
|
||||
storage_class: Some(TransitionStorageClass::from_static("WARM")),
|
||||
}]);
|
||||
let mut lc = BucketLifecycleConfiguration {
|
||||
rules: vec![rule],
|
||||
expiry_updated_at: None,
|
||||
};
|
||||
let object = noncurrent_object();
|
||||
let now = datetime!(2020-01-10 00:00:00 UTC);
|
||||
let transition_only = lc.eval_inner(&object, now, 0).await;
|
||||
assert_eq!(transition_only.action, IlmAction::TransitionVersionAction);
|
||||
|
||||
lc.rules[0].noncurrent_version_expiration = Some(NoncurrentVersionExpiration {
|
||||
noncurrent_days: Some(90),
|
||||
newer_noncurrent_versions: Some(2),
|
||||
});
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("valid combined policy");
|
||||
let combined = lc.eval_inner(&object, now, 0).await;
|
||||
assert_eq!(combined.action, transition_only.action, "retention limits expiration, not transition");
|
||||
assert_eq!(combined.storage_class, transition_only.storage_class);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn current_transition_rejects_multiple_stages_in_any_order() {
|
||||
run(async {
|
||||
let mut rule = enabled_rule(None, None, Some("two-current-transitions"));
|
||||
rule.transitions = Some(vec![
|
||||
Transition {
|
||||
date: Some(datetime!(2020-03-01 00:00:00 UTC).into()),
|
||||
days: None,
|
||||
storage_class: Some(TransitionStorageClass::from_static("COLD")),
|
||||
},
|
||||
Transition {
|
||||
date: Some(datetime!(2020-01-03 00:00:00 UTC).into()),
|
||||
days: None,
|
||||
storage_class: Some(TransitionStorageClass::from_static("WARM")),
|
||||
},
|
||||
]);
|
||||
let mut lc = BucketLifecycleConfiguration {
|
||||
rules: vec![rule],
|
||||
expiry_updated_at: None,
|
||||
};
|
||||
let object = ObjectOpts {
|
||||
is_latest: true,
|
||||
..noncurrent_object()
|
||||
};
|
||||
let now = datetime!(2020-01-10 00:00:00 UTC);
|
||||
for status in [ExpirationStatus::ENABLED, ExpirationStatus::DISABLED] {
|
||||
lc.rules[0].status = ExpirationStatus::from_static(status);
|
||||
for _ in 0..2 {
|
||||
let err = lc
|
||||
.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect_err("multiple transition stages must be rejected");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
|
||||
assert_eq!(err.to_string(), ERR_LIFECYCLE_MULTIPLE_TRANSITIONS);
|
||||
assert_eq!(
|
||||
lc.eval_inner(&object, now, 0).await.action,
|
||||
IlmAction::NoneAction,
|
||||
"legacy multi-stage configurations must not silently execute their first stage"
|
||||
);
|
||||
lc.rules[0]
|
||||
.transitions
|
||||
.as_mut()
|
||||
.expect("transition array is present")
|
||||
.reverse();
|
||||
}
|
||||
}
|
||||
lc.rules[0]
|
||||
.transitions
|
||||
.as_mut()
|
||||
.expect("transition array is present")
|
||||
.remove(0);
|
||||
lc.rules[0].status = ExpirationStatus::from_static(ExpirationStatus::ENABLED);
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("one stage is supported");
|
||||
let event = lc.eval_inner(&object, now, 0).await;
|
||||
assert_eq!(event.action, IlmAction::TransitionAction);
|
||||
assert_eq!(event.storage_class, "WARM");
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn noncurrent_transition_rejects_multiple_stages_in_any_order() {
|
||||
run(async {
|
||||
let mut rule = enabled_rule(None, None, Some("two-noncurrent-transitions"));
|
||||
rule.noncurrent_version_transitions = Some(vec![
|
||||
NoncurrentVersionTransition {
|
||||
noncurrent_days: Some(30),
|
||||
newer_noncurrent_versions: None,
|
||||
storage_class: Some(TransitionStorageClass::from_static("COLD")),
|
||||
},
|
||||
NoncurrentVersionTransition {
|
||||
noncurrent_days: Some(1),
|
||||
newer_noncurrent_versions: None,
|
||||
storage_class: Some(TransitionStorageClass::from_static("WARM")),
|
||||
},
|
||||
]);
|
||||
let mut lc = BucketLifecycleConfiguration {
|
||||
rules: vec![rule],
|
||||
expiry_updated_at: None,
|
||||
};
|
||||
let object = noncurrent_object();
|
||||
let now = datetime!(2020-01-10 00:00:00 UTC);
|
||||
for status in [ExpirationStatus::ENABLED, ExpirationStatus::DISABLED] {
|
||||
lc.rules[0].status = ExpirationStatus::from_static(status);
|
||||
for _ in 0..2 {
|
||||
let err = lc
|
||||
.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect_err("multiple noncurrent transition stages must be rejected");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
|
||||
assert_eq!(err.to_string(), ERR_LIFECYCLE_MULTIPLE_NONCURRENT_TRANSITIONS);
|
||||
assert_eq!(
|
||||
lc.eval_inner(&object, now, 0).await.action,
|
||||
IlmAction::NoneAction,
|
||||
"legacy multi-stage configurations must not silently execute their first stage"
|
||||
);
|
||||
lc.rules[0]
|
||||
.noncurrent_version_transitions
|
||||
.as_mut()
|
||||
.expect("transition array is present")
|
||||
.reverse();
|
||||
}
|
||||
}
|
||||
lc.rules[0]
|
||||
.noncurrent_version_transitions
|
||||
.as_mut()
|
||||
.expect("transition array is present")
|
||||
.remove(0);
|
||||
lc.rules[0].status = ExpirationStatus::from_static(ExpirationStatus::ENABLED);
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("one stage is supported");
|
||||
let event = lc.eval_inner(&object, now, 0).await;
|
||||
assert_eq!(event.action, IlmAction::TransitionVersionAction);
|
||||
assert_eq!(event.storage_class, "WARM");
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn expiration_rejects_simultaneous_days_and_date() {
|
||||
run(async {
|
||||
let mut lc = BucketLifecycleConfiguration {
|
||||
rules: vec![enabled_rule(
|
||||
Some(LifecycleExpiration {
|
||||
days: Some(1),
|
||||
..Default::default()
|
||||
}),
|
||||
None,
|
||||
Some("ambiguous-expiry"),
|
||||
)],
|
||||
expiry_updated_at: None,
|
||||
};
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("a single Days expiration is valid");
|
||||
lc.rules[0].expiration.as_mut().expect("expiration is present").date =
|
||||
Some(datetime!(2099-01-01 00:00:00 UTC).into());
|
||||
let err = lc
|
||||
.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect_err("Days and Date are mutually exclusive; accepting both silently overrides Days");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
|
||||
assert_eq!(err.to_string(), ERR_LIFECYCLE_EXPIRATION_DAYS_DATE_CONFLICT);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn overdue_transition_does_not_starve_permanent_expiration() {
|
||||
run(async {
|
||||
let mut rule = enabled_rule(
|
||||
Some(LifecycleExpiration {
|
||||
days: Some(90),
|
||||
..Default::default()
|
||||
}),
|
||||
None,
|
||||
Some("archive-then-delete"),
|
||||
);
|
||||
rule.transitions = Some(vec![Transition {
|
||||
days: Some(30),
|
||||
date: None,
|
||||
storage_class: Some(TransitionStorageClass::from_static("WARM")),
|
||||
}]);
|
||||
let lc = BucketLifecycleConfiguration {
|
||||
rules: vec![rule],
|
||||
expiry_updated_at: None,
|
||||
};
|
||||
lc.validate(&ObjectLockConfiguration::default())
|
||||
.await
|
||||
.expect("valid transition and expiration policy");
|
||||
let object = ObjectOpts {
|
||||
is_latest: true,
|
||||
version_id: None,
|
||||
transition_status: TRANSITION_PENDING.to_string(),
|
||||
..noncurrent_object()
|
||||
};
|
||||
let before_expiration = lc.eval_inner(&object, datetime!(2020-02-15 00:00:00 UTC), 0).await;
|
||||
assert_eq!(before_expiration.action, IlmAction::TransitionAction);
|
||||
let overdue = lc.eval_inner(&object, datetime!(2020-05-01 00:00:00 UTC), 0).await;
|
||||
assert_eq!(
|
||||
overdue.action,
|
||||
IlmAction::DeleteAction,
|
||||
"an unavailable tier must not prevent permanent expiration indefinitely"
|
||||
);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
/// Property-based tests for the rule evaluator (backlog#1148 ilm-14,
|
||||
/// follow-up to backlog#1030 / rustfs#4455).
|
||||
///
|
||||
/// backlog#1030 found that the old hand-written winner comparator was not a
|
||||
/// strict weak ordering and could panic — proof that enumerated cases do
|
||||
/// not cover the space of colliding events. These properties pin, over
|
||||
/// randomized rule sets and object states:
|
||||
///
|
||||
/// * `eval_inner` never panics and is deterministic for a fixed input;
|
||||
/// * the winning event matches an independently recomputed candidate set:
|
||||
/// eligible expiration wins over transition, then earliest `due` wins (the
|
||||
/// `min_by_key` selection that replaced the rustfs#4455 comparator);
|
||||
/// * `expected_expiry_time` is monotonically non-decreasing in `days` and
|
||||
/// always lands on the processing boundary, both at production defaults
|
||||
/// and under an explicit `RUSTFS_ILM_PROCESS_TIME`.
|
||||
///
|
||||
/// Case counts are tuned so the whole module runs in seconds inside the
|
||||
/// default CI test job.
|
||||
mod proptests {
|
||||
use super::*;
|
||||
use proptest::prelude::*;
|
||||
@@ -5220,8 +5629,8 @@ mod tests {
|
||||
/// consider for a live current version under `selection`-shaped rules
|
||||
/// (expiration and first-transition only, no filters): expiration
|
||||
/// fires when `now >= due`, transition when `now > due` and the object
|
||||
/// has not already transitioned. Selection semantics under test:
|
||||
/// earliest due wins, ties prefer delete-class.
|
||||
/// has not already transitioned. Eligible expiration wins over transition;
|
||||
/// the earliest deadline wins within the selected action class.
|
||||
fn oracle_candidates(lc: &BucketLifecycleConfiguration, obj: &ObjectOpts, now: OffsetDateTime) -> Vec<Candidate> {
|
||||
let mod_time = obj.mod_time.expect("selection strategy always sets mod_time");
|
||||
let mut candidates = Vec::new();
|
||||
@@ -5310,8 +5719,8 @@ mod tests {
|
||||
/// Differential test of winner selection (the rustfs#4455 fix):
|
||||
/// for a live current version under randomized expiration and
|
||||
/// transition rules, `eval_inner`'s winner must carry the
|
||||
/// minimum `(due, rank)` of the independently recomputed
|
||||
/// candidate set — earliest due wins, ties prefer delete-class —
|
||||
/// earliest expiration from the independently recomputed candidate
|
||||
/// set, or the earliest transition when no expiration is eligible,
|
||||
/// and must be `NoneAction` exactly when that set is empty.
|
||||
#[test]
|
||||
#[serial]
|
||||
@@ -5340,7 +5749,13 @@ mod tests {
|
||||
|
||||
// Oracle and evaluator must observe the same (pinned) time env.
|
||||
let (event, expected) = with_production_time_env(|| {
|
||||
let expected = oracle_candidates(&lc, &obj, now).into_iter().min();
|
||||
let candidates = oracle_candidates(&lc, &obj, now);
|
||||
let expected = candidates
|
||||
.iter()
|
||||
.filter(|(_, rank)| *rank == 0)
|
||||
.min()
|
||||
.copied()
|
||||
.or_else(|| candidates.into_iter().min());
|
||||
let rt = tokio::runtime::Builder::new_current_thread()
|
||||
.enable_all()
|
||||
.build()
|
||||
|
||||
@@ -119,13 +119,10 @@ impl Evaluator {
|
||||
break 'top_loop;
|
||||
}
|
||||
}
|
||||
IlmAction::DeleteAction
|
||||
| IlmAction::DeleteRestoredAction
|
||||
| IlmAction::DeleteVersionAction
|
||||
| IlmAction::DeleteRestoredVersionAction
|
||||
if self.is_object_locked(obj) =>
|
||||
{
|
||||
event = Event::default();
|
||||
// Restore expiry removes only the temporary local copy; the
|
||||
// retained logical version and its remote data remain intact.
|
||||
IlmAction::DeleteAction | IlmAction::DeleteVersionAction if self.is_object_locked(obj) => {
|
||||
event = obj.restored_copy_expiry(now).unwrap_or_default();
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
@@ -220,6 +217,95 @@ mod tests {
|
||||
|
||||
use super::*;
|
||||
use rustfs_replication::{ReplicationStatusType, VersionPurgeStatusType};
|
||||
|
||||
#[tokio::test]
|
||||
async fn adversarial_restore_expiry_survives_legal_hold() {
|
||||
let mut policy = (*latest_expiration_lifecycle()).clone();
|
||||
policy.rules[0].status = ExpirationStatus::from_static(ExpirationStatus::DISABLED);
|
||||
let policy = Arc::new(policy);
|
||||
policy
|
||||
.validate(&lock_enabled_without_default_retention())
|
||||
.await
|
||||
.expect("valid disabled lifecycle rule");
|
||||
let mut objects = [true, false].map(|is_latest| ObjectOpts {
|
||||
is_latest,
|
||||
num_versions: 2,
|
||||
mod_time: Some(
|
||||
OffsetDateTime::from_unix_timestamp(if is_latest { 1_200_000 } else { 1_000_000 })
|
||||
.expect("fixed version timestamp"),
|
||||
),
|
||||
successor_mod_time: (!is_latest)
|
||||
.then(|| OffsetDateTime::from_unix_timestamp(1_200_000).expect("fixed successor timestamp")),
|
||||
transition_status: crate::TRANSITION_COMPLETE.to_string(),
|
||||
restore_expires: Some(OffsetDateTime::from_unix_timestamp(2_000_000).expect("fixed expired restore timestamp")),
|
||||
..current_object_opts(ReplicationStatusType::Completed)
|
||||
});
|
||||
let evaluator = Evaluator::new(policy).with_lock_retention(Some(lock_enabled_without_default_retention()));
|
||||
let expected = [IlmAction::DeleteRestoredAction, IlmAction::DeleteRestoredVersionAction];
|
||||
let unlocked = evaluator
|
||||
.eval(&objects)
|
||||
.await
|
||||
.expect("unlocked restored versions should evaluate");
|
||||
assert_eq!(unlocked.iter().map(|event| event.action).collect::<Vec<_>>(), expected);
|
||||
|
||||
for object in &mut objects {
|
||||
object
|
||||
.user_defined
|
||||
.insert(X_AMZ_OBJECT_LOCK_LEGAL_HOLD.as_str().to_string(), "ON".to_string());
|
||||
}
|
||||
let locked = evaluator
|
||||
.eval(&objects)
|
||||
.await
|
||||
.expect("locked restored versions should evaluate");
|
||||
assert_eq!(
|
||||
locked.iter().map(|event| event.action).collect::<Vec<_>>(),
|
||||
expected,
|
||||
"expiring a restored local copy preserves the retained logical version and remote object"
|
||||
);
|
||||
|
||||
let mut expiring_policy = (*latest_expiration_lifecycle()).clone();
|
||||
expiring_policy.rules[0].noncurrent_version_expiration = Some(NoncurrentVersionExpiration {
|
||||
noncurrent_days: Some(1),
|
||||
newer_noncurrent_versions: None,
|
||||
});
|
||||
let expiring_evaluator =
|
||||
Evaluator::new(Arc::new(expiring_policy)).with_lock_retention(Some(lock_enabled_without_default_retention()));
|
||||
let locked = expiring_evaluator
|
||||
.eval(&objects)
|
||||
.await
|
||||
.expect("locked expired versions should evaluate");
|
||||
assert_eq!(
|
||||
locked.iter().map(|event| event.action).collect::<Vec<_>>(),
|
||||
expected,
|
||||
"blocked logical expiration must still allow an eligible restore-copy cleanup"
|
||||
);
|
||||
|
||||
for status in [ReplicationStatusType::Pending, ReplicationStatusType::Failed] {
|
||||
for object in &mut objects {
|
||||
object.replication_status = status.clone();
|
||||
}
|
||||
for evaluator in [&evaluator, &expiring_evaluator] {
|
||||
let events = evaluator.eval(&objects).await.expect("pending replication should evaluate");
|
||||
assert!(events.iter().all(|event| event.action == IlmAction::NoneAction));
|
||||
}
|
||||
}
|
||||
for object in &mut objects {
|
||||
object.replication_status = ReplicationStatusType::Completed;
|
||||
}
|
||||
for transition_status in ["", crate::TRANSITION_PENDING, "unknown"] {
|
||||
for object in &mut objects {
|
||||
object.transition_status = transition_status.to_string();
|
||||
}
|
||||
for evaluator in [&evaluator, &expiring_evaluator] {
|
||||
let events = evaluator.eval(&objects).await.expect("incomplete transition should evaluate");
|
||||
assert!(
|
||||
events.iter().all(|event| event.action == IlmAction::NoneAction),
|
||||
"restore metadata cannot authorize cleanup without a completed transition"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn expired_marker_lifecycle() -> Arc<BucketLifecycleConfiguration> {
|
||||
Arc::new(BucketLifecycleConfiguration {
|
||||
expiry_updated_at: None,
|
||||
|
||||
@@ -120,18 +120,10 @@ impl TransitionClient {
|
||||
|
||||
let h = resp.headers().clone();
|
||||
|
||||
let mut body = resp.into_body();
|
||||
let body_vec = if let Some(limit) = max_response_bytes {
|
||||
collect_response_body(body, limit).await?
|
||||
self.collect_response_body(resp.into_body(), limit).await?
|
||||
} else {
|
||||
let mut body_vec = Vec::new();
|
||||
while let Some(frame) = body.frame().await {
|
||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
||||
if let Some(data) = frame.data_ref() {
|
||||
body_vec.extend_from_slice(data);
|
||||
}
|
||||
}
|
||||
body_vec
|
||||
self.collect_response_body_unbounded(resp.into_body()).await?
|
||||
};
|
||||
Ok((object_stat, h, BufReader::new(Cursor::new(body_vec))))
|
||||
}
|
||||
@@ -143,7 +135,7 @@ mod bounded_response_tests {
|
||||
use crate::{
|
||||
api_get_options::GetObjectOptions,
|
||||
credentials::{Credentials, SignatureType, Static, Value},
|
||||
transition_api::{BucketLookupType, Options, TransitionClient, collect_response_body},
|
||||
transition_api::{BucketLookupType, Options, TransitionClient, TransitionClientTimeouts, collect_response_body},
|
||||
};
|
||||
use http_body_util::Full;
|
||||
use hyper::body::Bytes;
|
||||
@@ -175,7 +167,31 @@ mod bounded_response_tests {
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||
}
|
||||
|
||||
async fn bounded_get_fixture(body: &'static [u8]) -> Option<(TransitionClient, tokio::task::JoinHandle<String>)> {
|
||||
fn test_options() -> Options {
|
||||
Options {
|
||||
creds: Credentials::new(Static(Value {
|
||||
access_key_id: "access-key".to_string(),
|
||||
secret_access_key: "secret-key".to_string(),
|
||||
signer_type: SignatureType::SignatureV4,
|
||||
..Default::default()
|
||||
})),
|
||||
region: "us-east-1".to_string(),
|
||||
bucket_lookup: BucketLookupType::BucketLookupPath,
|
||||
max_retries: 1,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
async fn client_for_endpoint(endpoint: &str, timeouts: TransitionClientTimeouts) -> TransitionClient {
|
||||
TransitionClient::new_with_timeouts(endpoint, test_options(), "", timeouts)
|
||||
.await
|
||||
.expect("fixture client should build")
|
||||
}
|
||||
|
||||
async fn bounded_get_fixture_with_timeouts(
|
||||
body: &'static [u8],
|
||||
timeouts: TransitionClientTimeouts,
|
||||
) -> Option<(TransitionClient, tokio::task::JoinHandle<String>)> {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return None,
|
||||
@@ -209,27 +225,14 @@ mod bounded_response_tests {
|
||||
stream.write_all(body).await.expect("fixture should write response body");
|
||||
request
|
||||
});
|
||||
let client = TransitionClient::new(
|
||||
&endpoint,
|
||||
Options {
|
||||
creds: Credentials::new(Static(Value {
|
||||
access_key_id: "access-key".to_string(),
|
||||
secret_access_key: "secret-key".to_string(),
|
||||
signer_type: SignatureType::SignatureV4,
|
||||
..Default::default()
|
||||
})),
|
||||
region: "us-east-1".to_string(),
|
||||
bucket_lookup: BucketLookupType::BucketLookupPath,
|
||||
max_retries: 1,
|
||||
..Default::default()
|
||||
},
|
||||
"",
|
||||
)
|
||||
.await
|
||||
.expect("fixture client should build");
|
||||
let client = client_for_endpoint(&endpoint, timeouts).await;
|
||||
Some((client, request))
|
||||
}
|
||||
|
||||
async fn bounded_get_fixture(body: &'static [u8]) -> Option<(TransitionClient, tokio::task::JoinHandle<String>)> {
|
||||
bounded_get_fixture_with_timeouts(body, TransitionClientTimeouts::default()).await
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn real_transport_accepts_the_exact_closed_range_length() {
|
||||
let Some((client, request)) = bounded_get_fixture(b"RustFS!").await else {
|
||||
@@ -292,24 +295,7 @@ mod bounded_response_tests {
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
let client = TransitionClient::new(
|
||||
&endpoint,
|
||||
Options {
|
||||
creds: Credentials::new(Static(Value {
|
||||
access_key_id: "access-key".to_string(),
|
||||
secret_access_key: "secret-key".to_string(),
|
||||
signer_type: SignatureType::SignatureV4,
|
||||
..Default::default()
|
||||
})),
|
||||
region: "us-east-1".to_string(),
|
||||
bucket_lookup: BucketLookupType::BucketLookupPath,
|
||||
max_retries: 1,
|
||||
..Default::default()
|
||||
},
|
||||
"",
|
||||
)
|
||||
.await
|
||||
.expect("fixture client should build");
|
||||
let client = client_for_endpoint(&endpoint, TransitionClientTimeouts::default()).await;
|
||||
let mut opts = GetObjectOptions::default();
|
||||
opts.headers
|
||||
.insert("range".to_string(), "bytes=0-18446744073709551615".to_string());
|
||||
@@ -326,6 +312,176 @@ mod bounded_response_tests {
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn connection_refused_returns_without_waiting_for_the_request_timeout() {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||
Err(err) => panic!("test listener should bind: {err}"),
|
||||
};
|
||||
let endpoint = listener
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
drop(listener);
|
||||
|
||||
let client = client_for_endpoint(
|
||||
&endpoint,
|
||||
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_secs(5), Duration::from_secs(1)),
|
||||
)
|
||||
.await;
|
||||
let mut opts = GetObjectOptions::default();
|
||||
opts.set_range(0, 6).expect("the probe range should be valid");
|
||||
|
||||
let result = tokio::time::timeout(Duration::from_secs(2), client.get_object_inner("bucket", "probe", &opts))
|
||||
.await
|
||||
.expect("connection refused should return before the broader request timeout");
|
||||
|
||||
assert!(result.is_err(), "connection refused must fail instead of hanging");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn response_header_stall_returns_timed_out() {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||
Err(err) => panic!("test listener should bind: {err}"),
|
||||
};
|
||||
let endpoint = listener
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
let fixture = tokio::spawn(async move {
|
||||
let (mut stream, _) = listener.accept().await.expect("fixture should accept one GET");
|
||||
let mut request = Vec::new();
|
||||
let mut buffer = [0; 1024];
|
||||
loop {
|
||||
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
|
||||
assert_ne!(read, 0, "connection closed before request headers were received");
|
||||
request.extend_from_slice(&buffer[..read]);
|
||||
if request.windows(4).any(|window| window == b"\r\n\r\n") {
|
||||
break;
|
||||
}
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||
});
|
||||
let client = client_for_endpoint(
|
||||
&endpoint,
|
||||
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_millis(50), Duration::from_secs(1)),
|
||||
)
|
||||
.await;
|
||||
let mut opts = GetObjectOptions::default();
|
||||
opts.set_range(0, 6).expect("the probe range should be valid");
|
||||
|
||||
let err = client
|
||||
.get_object_inner("bucket", "probe", &opts)
|
||||
.await
|
||||
.expect_err("response header stalls must be bounded");
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
|
||||
fixture.await.expect("fixture should join");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn response_body_idle_stall_returns_timed_out() {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||
Err(err) => panic!("test listener should bind: {err}"),
|
||||
};
|
||||
let endpoint = listener
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
let fixture = tokio::spawn(async move {
|
||||
let (mut stream, _) = listener.accept().await.expect("fixture should accept one GET");
|
||||
let mut request = Vec::new();
|
||||
let mut buffer = [0; 1024];
|
||||
loop {
|
||||
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
|
||||
assert_ne!(read, 0, "connection closed before request headers were received");
|
||||
request.extend_from_slice(&buffer[..read]);
|
||||
if request.windows(4).any(|window| window == b"\r\n\r\n") {
|
||||
break;
|
||||
}
|
||||
}
|
||||
stream
|
||||
.write_all(b"HTTP/1.1 206 Partial Content\r\nContent-Length: 7\r\nConnection: close\r\n\r\nRu")
|
||||
.await
|
||||
.expect("fixture should write the first body chunk");
|
||||
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||
});
|
||||
let client = client_for_endpoint(
|
||||
&endpoint,
|
||||
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_secs(1), Duration::from_millis(50)),
|
||||
)
|
||||
.await;
|
||||
let mut opts = GetObjectOptions::default();
|
||||
opts.set_range(0, 6).expect("the probe range should be valid");
|
||||
|
||||
let err = client
|
||||
.get_object_inner("bucket", "probe", &opts)
|
||||
.await
|
||||
.expect_err("body stalls after partial progress must be bounded");
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
|
||||
fixture.await.expect("fixture should join");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn response_body_idle_timer_resets_on_progress() {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||
Err(err) => panic!("test listener should bind: {err}"),
|
||||
};
|
||||
let endpoint = listener
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
let fixture = tokio::spawn(async move {
|
||||
let (mut stream, _) = listener.accept().await.expect("fixture should accept one GET");
|
||||
let mut request = Vec::new();
|
||||
let mut buffer = [0; 1024];
|
||||
loop {
|
||||
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
|
||||
assert_ne!(read, 0, "connection closed before request headers were received");
|
||||
request.extend_from_slice(&buffer[..read]);
|
||||
if request.windows(4).any(|window| window == b"\r\n\r\n") {
|
||||
break;
|
||||
}
|
||||
}
|
||||
stream
|
||||
.write_all(b"HTTP/1.1 206 Partial Content\r\nContent-Length: 7\r\nConnection: close\r\n\r\n")
|
||||
.await
|
||||
.expect("fixture should write response headers");
|
||||
for byte in b"RustFS!" {
|
||||
stream.write_all(&[*byte]).await.expect("fixture should write body progress");
|
||||
tokio::time::sleep(Duration::from_millis(20)).await;
|
||||
}
|
||||
});
|
||||
let client = client_for_endpoint(
|
||||
&endpoint,
|
||||
TransitionClientTimeouts::new(Duration::from_millis(10), Duration::from_secs(1), Duration::from_millis(100)),
|
||||
)
|
||||
.await;
|
||||
let mut opts = GetObjectOptions::default();
|
||||
opts.set_range(0, 6).expect("the probe range should be valid");
|
||||
|
||||
let (_, _, mut reader) = client
|
||||
.get_object_inner("bucket", "probe", &opts)
|
||||
.await
|
||||
.expect("continuous body progress must not be killed by the idle timer");
|
||||
let mut body = Vec::new();
|
||||
reader
|
||||
.read_to_end(&mut body)
|
||||
.await
|
||||
.expect("bounded response should be readable");
|
||||
|
||||
assert_eq!(body, b"RustFS!");
|
||||
fixture.await.expect("fixture should join");
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
|
||||
@@ -27,7 +27,6 @@ use crate::{
|
||||
transition_api::{ReaderImpl, RequestMetadata, TransitionClient, collect_response_body},
|
||||
};
|
||||
use http::{HeaderMap, StatusCode};
|
||||
use http_body_util::BodyExt;
|
||||
use hyper::body::Body;
|
||||
use hyper::body::Bytes;
|
||||
use rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE;
|
||||
@@ -124,14 +123,9 @@ impl TransitionClient {
|
||||
}
|
||||
|
||||
//let mut list_bucket_result = ListBucketV2Result::default();
|
||||
let mut body_vec = Vec::new();
|
||||
let mut body = resp.into_body();
|
||||
while let Some(frame) = body.frame().await {
|
||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
||||
if let Some(data) = frame.data_ref() {
|
||||
body_vec.extend_from_slice(data);
|
||||
}
|
||||
}
|
||||
let body_vec = self
|
||||
.collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE)
|
||||
.await?;
|
||||
let mut list_bucket_result = match quick_xml::de::from_str::<ListBucketV2Result>(&String::from_utf8_lossy(&body_vec)) {
|
||||
Ok(result) => result,
|
||||
Err(err) => {
|
||||
@@ -214,7 +208,9 @@ impl TransitionClient {
|
||||
|
||||
let resp_status = resp.status();
|
||||
let headers = resp.headers().clone();
|
||||
let body = collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE).await?;
|
||||
let body = self
|
||||
.collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE)
|
||||
.await?;
|
||||
if resp_status != StatusCode::OK {
|
||||
return Err(std::io::Error::other(http_resp_to_error_response(
|
||||
resp_status,
|
||||
@@ -428,6 +424,30 @@ fn decode_s3_name(name: &str, encoding_type: &str) -> Result<String, std::io::Er
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::{
|
||||
credentials::{Credentials, SignatureType, Static, Value},
|
||||
transition_api::{BucketLookupType, Options, TransitionClientTimeouts},
|
||||
};
|
||||
use std::time::Duration;
|
||||
use tokio::{
|
||||
io::{AsyncReadExt, AsyncWriteExt},
|
||||
net::TcpListener,
|
||||
};
|
||||
|
||||
fn timeout_test_options() -> Options {
|
||||
Options {
|
||||
creds: Credentials::new(Static(Value {
|
||||
access_key_id: "access-key".to_string(),
|
||||
secret_access_key: "secret-key".to_string(),
|
||||
signer_type: SignatureType::SignatureV4,
|
||||
..Default::default()
|
||||
})),
|
||||
region: "us-east-1".to_string(),
|
||||
bucket_lookup: BucketLookupType::BucketLookupPath,
|
||||
max_retries: 1,
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_versions_xml_preserves_versions_and_delete_markers() {
|
||||
@@ -525,4 +545,56 @@ mod tests {
|
||||
assert_eq!(parsed.common_prefixes.len(), 1);
|
||||
assert_eq!(parsed.common_prefixes[0].prefix, "subdir/");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn list_objects_v2_body_stall_returns_timed_out() {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||
Err(err) => panic!("test listener should bind: {err}"),
|
||||
};
|
||||
let endpoint = listener
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
let fixture = tokio::spawn(async move {
|
||||
let (mut stream, _) = listener.accept().await.expect("fixture should accept one list request");
|
||||
let mut request = Vec::new();
|
||||
let mut buffer = [0; 1024];
|
||||
loop {
|
||||
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
|
||||
assert_ne!(read, 0, "connection closed before request headers were received");
|
||||
request.extend_from_slice(&buffer[..read]);
|
||||
if request.windows(4).any(|window| window == b"\r\n\r\n") {
|
||||
break;
|
||||
}
|
||||
}
|
||||
stream
|
||||
.write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 512\r\nConnection: close\r\n\r\n<ListBucketResult><Name>warm")
|
||||
.await
|
||||
.expect("fixture should write a partial list response");
|
||||
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||
});
|
||||
let client = TransitionClient::new_with_timeouts(
|
||||
&endpoint,
|
||||
timeout_test_options(),
|
||||
"",
|
||||
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_secs(1), Duration::from_millis(50)),
|
||||
)
|
||||
.await
|
||||
.expect("fixture client should build");
|
||||
client
|
||||
.bucket_loc_cache
|
||||
.lock()
|
||||
.expect("location cache should lock")
|
||||
.set("bucket", "us-east-1");
|
||||
|
||||
let err = client
|
||||
.list_objects_v2_query("bucket", "", "", false, false, "", "", 1, HeaderMap::new())
|
||||
.await
|
||||
.expect_err("a stalled ListObjectsV2 body must be bounded");
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
|
||||
fixture.await.expect("fixture should join");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -18,7 +18,6 @@
|
||||
#![allow(clippy::all)]
|
||||
|
||||
use http::{HeaderMap, HeaderName, StatusCode};
|
||||
use http_body_util::BodyExt;
|
||||
use hyper::body::Bytes;
|
||||
use s3s::S3ErrorCode;
|
||||
use std::collections::HashMap;
|
||||
@@ -247,14 +246,9 @@ impl TransitionClient {
|
||||
// Parse the CreateMultipartUpload response for the UploadId. Returning a
|
||||
// default (empty) result here made every multipart transition fail at the
|
||||
// first UploadPart with "UploadID cannot be empty" (rustfs/rustfs#4811).
|
||||
let mut body_vec = Vec::new();
|
||||
let mut body = resp.into_body();
|
||||
while let Some(frame) = body.frame().await {
|
||||
let frame = frame.map_err(|e| std::io::Error::other(e.to_string()))?;
|
||||
if let Some(data) = frame.data_ref() {
|
||||
body_vec.extend_from_slice(data);
|
||||
}
|
||||
}
|
||||
let body_vec = self
|
||||
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE)
|
||||
.await?;
|
||||
let initiate_multipart_upload_result =
|
||||
quick_xml::de::from_str::<InitiateMultipartUploadResult>(&String::from_utf8_lossy(&body_vec))
|
||||
.map_err(|e| std::io::Error::other(format!("failed to parse CreateMultipartUpload response: {e}")))?;
|
||||
|
||||
@@ -19,7 +19,6 @@
|
||||
#![allow(clippy::all)]
|
||||
|
||||
use http::{HeaderMap, HeaderValue, Method, StatusCode};
|
||||
use http_body_util::BodyExt;
|
||||
use hyper::body::Body;
|
||||
use hyper::body::Bytes;
|
||||
use rustfs_utils::HashAlgorithm;
|
||||
@@ -351,14 +350,9 @@ impl TransitionClient {
|
||||
)
|
||||
.await?;
|
||||
|
||||
let mut body_vec = Vec::new();
|
||||
let mut body = resp.into_body();
|
||||
while let Some(frame) = body.frame().await {
|
||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
||||
if let Some(data) = frame.data_ref() {
|
||||
body_vec.extend_from_slice(data);
|
||||
}
|
||||
}
|
||||
let body_vec = self
|
||||
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE)
|
||||
.await?;
|
||||
process_remove_multi_objects_response(
|
||||
ReaderImpl::Body(Bytes::from(body_vec)),
|
||||
bucket_name,
|
||||
|
||||
@@ -19,7 +19,6 @@
|
||||
#![allow(clippy::all)]
|
||||
|
||||
use http::{HeaderMap, HeaderValue, StatusCode};
|
||||
use http_body_util::BodyExt;
|
||||
use hyper::body::Body;
|
||||
use hyper::body::Bytes;
|
||||
use rustfs_utils::EMPTY_STRING_SHA256_HASH;
|
||||
@@ -119,14 +118,9 @@ impl TransitionClient {
|
||||
let resp_status = resp.status();
|
||||
let h = resp.headers().clone();
|
||||
|
||||
let mut body_vec = Vec::new();
|
||||
let mut body = resp.into_body();
|
||||
while let Some(frame) = body.frame().await {
|
||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
||||
if let Some(data) = frame.data_ref() {
|
||||
body_vec.extend_from_slice(data);
|
||||
}
|
||||
}
|
||||
let body_vec = self
|
||||
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE)
|
||||
.await?;
|
||||
let resperr = http_resp_to_error_response(resp_status, &h, body_vec, bucket_name, "");
|
||||
|
||||
warn!("bucket exists, resperr: {:?}", resperr);
|
||||
@@ -170,11 +164,13 @@ impl TransitionClient {
|
||||
let resp_status = resp.status();
|
||||
let h = resp.headers().clone();
|
||||
|
||||
let body_vec = collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE).await?;
|
||||
let body_vec = self
|
||||
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE)
|
||||
.await?;
|
||||
parse_bucket_versioning_response(resp_status, &h, body_vec, bucket_name)
|
||||
}
|
||||
|
||||
Err(err) => Err(std::io::Error::other(err)),
|
||||
Err(err) => Err(err),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -274,8 +270,14 @@ impl TransitionClient {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::parse_bucket_versioning_response;
|
||||
use crate::{
|
||||
credentials::{Credentials, SignatureType, Static, Value},
|
||||
transition_api::{BucketLookupType, Options, TransitionClient, TransitionClientTimeouts},
|
||||
};
|
||||
use http::{HeaderMap, StatusCode};
|
||||
use s3s::dto::BucketVersioningStatus;
|
||||
use std::time::Duration;
|
||||
use tokio::{io::AsyncReadExt, net::TcpListener};
|
||||
|
||||
#[test]
|
||||
fn parses_bucket_versioning_statuses_mfa_delete_and_unversioned_state() {
|
||||
@@ -338,4 +340,63 @@ mod tests {
|
||||
assert_eq!(strict_err.kind(), std::io::ErrorKind::InvalidData);
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_bucket_versioning_preserves_request_timeout_kind() {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||
Err(err) => panic!("test listener should bind: {err}"),
|
||||
};
|
||||
let endpoint = listener
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
let fixture = tokio::spawn(async move {
|
||||
let (mut stream, _) = listener.accept().await.expect("fixture should accept one versioning request");
|
||||
let mut request = Vec::new();
|
||||
let mut buffer = [0; 1024];
|
||||
loop {
|
||||
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
|
||||
assert_ne!(read, 0, "connection closed before request headers were received");
|
||||
request.extend_from_slice(&buffer[..read]);
|
||||
if request.windows(4).any(|window| window == b"\r\n\r\n") {
|
||||
break;
|
||||
}
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||
});
|
||||
let client = TransitionClient::new_with_timeouts(
|
||||
&endpoint,
|
||||
Options {
|
||||
creds: Credentials::new(Static(Value {
|
||||
access_key_id: "access-key".to_string(),
|
||||
secret_access_key: "secret-key".to_string(),
|
||||
signer_type: SignatureType::SignatureV4,
|
||||
..Default::default()
|
||||
})),
|
||||
region: "us-east-1".to_string(),
|
||||
bucket_lookup: BucketLookupType::BucketLookupPath,
|
||||
max_retries: 1,
|
||||
..Default::default()
|
||||
},
|
||||
"",
|
||||
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_millis(50), Duration::from_secs(1)),
|
||||
)
|
||||
.await
|
||||
.expect("fixture client should build");
|
||||
client
|
||||
.bucket_loc_cache
|
||||
.lock()
|
||||
.expect("location cache should lock")
|
||||
.set("bucket", "us-east-1");
|
||||
|
||||
let err = client
|
||||
.get_bucket_versioning("bucket")
|
||||
.await
|
||||
.expect_err("a stalled versioning request must time out");
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
|
||||
fixture.await.expect("fixture should join");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -26,7 +26,6 @@ use crate::{
|
||||
transition_api::{CreateBucketConfiguration, LocationConstraint, TransitionClient},
|
||||
};
|
||||
use http::Request;
|
||||
use http_body_util::BodyExt;
|
||||
use hyper::StatusCode;
|
||||
use hyper::body::Body;
|
||||
use hyper::body::Bytes;
|
||||
@@ -86,7 +85,7 @@ impl TransitionClient {
|
||||
let req = self.get_bucket_location_request(bucket_name)?;
|
||||
|
||||
let mut resp = self.doit(req).await?;
|
||||
location = process_bucket_location_response(resp, bucket_name, &self.tier_type).await?;
|
||||
location = process_bucket_location_response(self, resp, bucket_name, &self.tier_type).await?;
|
||||
{
|
||||
if let Ok(mut bucket_loc_cache) = self.bucket_loc_cache.lock() {
|
||||
bucket_loc_cache.set(bucket_name, &location);
|
||||
@@ -198,6 +197,7 @@ impl TransitionClient {
|
||||
}
|
||||
|
||||
async fn process_bucket_location_response(
|
||||
client: &TransitionClient,
|
||||
mut resp: http::Response<Incoming>,
|
||||
bucket_name: &str,
|
||||
tier_type: &str,
|
||||
@@ -237,14 +237,9 @@ async fn process_bucket_location_response(
|
||||
}
|
||||
//}
|
||||
|
||||
let mut body_vec = Vec::new();
|
||||
let mut body = resp.into_body();
|
||||
while let Some(frame) = body.frame().await {
|
||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
||||
if let Some(data) = frame.data_ref() {
|
||||
body_vec.extend_from_slice(data);
|
||||
}
|
||||
}
|
||||
let body_vec = client
|
||||
.collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE)
|
||||
.await?;
|
||||
let mut location = "".to_string();
|
||||
if tier_type == "huaweicloud" {
|
||||
if let Ok(body_str) = String::from_utf8(body_vec) {
|
||||
|
||||
@@ -41,7 +41,7 @@ use http::{
|
||||
request::{Builder, Request},
|
||||
};
|
||||
use http_body::Body;
|
||||
use http_body_util::{BodyExt, LengthLimitError, Limited};
|
||||
use http_body_util::BodyExt;
|
||||
use hyper::body::Bytes;
|
||||
use hyper::body::Incoming;
|
||||
use hyper_rustls::{ConfigBuilderExt, HttpsConnector};
|
||||
@@ -67,10 +67,12 @@ use s3s::dto::Owner;
|
||||
use s3s::dto::ReplicationStatus;
|
||||
use serde::{Deserialize, Serialize};
|
||||
use sha2::Sha256;
|
||||
use std::error::Error as StdError;
|
||||
use std::io::Cursor;
|
||||
use std::pin::Pin;
|
||||
use std::sync::atomic::{AtomicI32, Ordering};
|
||||
use std::task::{Context, Poll};
|
||||
use std::time::Duration as StdDuration;
|
||||
use std::{
|
||||
collections::HashMap,
|
||||
sync::{Arc, Mutex},
|
||||
@@ -79,28 +81,108 @@ use time::Duration;
|
||||
use time::OffsetDateTime;
|
||||
use tokio::io::BufReader;
|
||||
use tokio::io::{AsyncRead, AsyncReadExt};
|
||||
use tracing::{debug, error, warn};
|
||||
use tracing::{debug, error, trace, warn};
|
||||
use url::{Url, form_urlencoded};
|
||||
use uuid::Uuid;
|
||||
|
||||
const C_USER_AGENT: &str = "RustFS (linux; x86)";
|
||||
pub const MAX_S3_ERROR_RESPONSE_SIZE: usize = 64 * 1024;
|
||||
const EVENT_TIER_REMOTE_TRANSPORT: &str = "tier_remote_transport";
|
||||
const LOG_COMPONENT_S3_CLIENT: &str = "s3_client";
|
||||
const LOG_SUBSYSTEM_TIER: &str = "tier";
|
||||
|
||||
const SUCCESS_STATUS: [StatusCode; 3] = [StatusCode::OK, StatusCode::NO_CONTENT, StatusCode::PARTIAL_CONTENT];
|
||||
|
||||
fn response_body_exceeds_limit_error() -> std::io::Error {
|
||||
std::io::Error::new(std::io::ErrorKind::InvalidData, "remote tier response body exceeds limit")
|
||||
}
|
||||
|
||||
fn remote_tier_timeout_error(message: &'static str) -> std::io::Error {
|
||||
std::io::Error::new(std::io::ErrorKind::TimedOut, message)
|
||||
}
|
||||
|
||||
fn source_chain_has_io_kind(error: &(dyn StdError + 'static), kind: std::io::ErrorKind) -> bool {
|
||||
let mut current = Some(error);
|
||||
while let Some(error) = current {
|
||||
if error
|
||||
.downcast_ref::<std::io::Error>()
|
||||
.is_some_and(|io_error| io_error.kind() == kind)
|
||||
{
|
||||
return true;
|
||||
}
|
||||
current = error.source();
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
fn transition_transport_error(err: hyper_util::client::legacy::Error) -> std::io::Error {
|
||||
if source_chain_has_io_kind(&err, std::io::ErrorKind::TimedOut) {
|
||||
return remote_tier_timeout_error("remote tier connection timed out");
|
||||
}
|
||||
std::io::Error::other(err)
|
||||
}
|
||||
|
||||
async fn next_response_body_data<B>(
|
||||
mut body: Pin<&mut B>,
|
||||
idle_timeout: Option<StdDuration>,
|
||||
) -> Result<Option<Bytes>, std::io::Error>
|
||||
where
|
||||
B: Body<Data = Bytes>,
|
||||
B::Error: Into<Box<dyn StdError + Send + Sync>>,
|
||||
{
|
||||
let next_nonempty_data = async {
|
||||
loop {
|
||||
let Some(frame) = std::future::poll_fn(|cx| body.as_mut().poll_frame(cx)).await else {
|
||||
return Ok(None);
|
||||
};
|
||||
let frame = frame.map_err(std::io::Error::other)?;
|
||||
let Ok(data) = frame.into_data() else {
|
||||
continue;
|
||||
};
|
||||
if !data.is_empty() {
|
||||
return Ok(Some(data));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if let Some(idle_timeout) = idle_timeout {
|
||||
tokio::time::timeout(idle_timeout, next_nonempty_data)
|
||||
.await
|
||||
.map_err(|_| remote_tier_timeout_error("remote tier response body stalled"))?
|
||||
} else {
|
||||
next_nonempty_data.await
|
||||
}
|
||||
}
|
||||
|
||||
async fn collect_response_body_inner<B>(
|
||||
body: B,
|
||||
limit: Option<usize>,
|
||||
idle_timeout: Option<StdDuration>,
|
||||
) -> Result<Vec<u8>, std::io::Error>
|
||||
where
|
||||
B: Body<Data = Bytes>,
|
||||
B::Error: Into<Box<dyn StdError + Send + Sync>>,
|
||||
{
|
||||
let mut body_vec = Vec::new();
|
||||
let mut body = std::pin::pin!(body);
|
||||
while let Some(data) = next_response_body_data(body.as_mut(), idle_timeout).await? {
|
||||
let Some(new_len) = body_vec.len().checked_add(data.len()) else {
|
||||
return Err(response_body_exceeds_limit_error());
|
||||
};
|
||||
if limit.is_some_and(|limit| new_len > limit) {
|
||||
return Err(response_body_exceeds_limit_error());
|
||||
}
|
||||
body_vec.extend_from_slice(&data);
|
||||
}
|
||||
Ok(body_vec)
|
||||
}
|
||||
|
||||
pub async fn collect_response_body<B>(body: B, limit: usize) -> Result<Vec<u8>, std::io::Error>
|
||||
where
|
||||
B: Body<Data = Bytes>,
|
||||
B::Error: Into<Box<dyn std::error::Error + Send + Sync>>,
|
||||
B::Error: Into<Box<dyn StdError + Send + Sync>>,
|
||||
{
|
||||
let body = Limited::new(body, limit).collect().await.map_err(|err| {
|
||||
if err.is::<LengthLimitError>() {
|
||||
std::io::Error::new(std::io::ErrorKind::InvalidData, "remote tier response body exceeds limit")
|
||||
} else {
|
||||
std::io::Error::other(err)
|
||||
}
|
||||
})?;
|
||||
Ok(body.to_bytes().to_vec())
|
||||
collect_response_body_inner(body, Some(limit), None).await
|
||||
}
|
||||
|
||||
const C_UNKNOWN: i32 = -1;
|
||||
@@ -196,6 +278,62 @@ pub struct TransitionClient {
|
||||
pub trailing_header_support: bool,
|
||||
pub max_retries: i64,
|
||||
pub tier_type: String,
|
||||
pub timeouts: TransitionClientTimeouts,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub struct TransitionClientTimeouts {
|
||||
pub connect_timeout: StdDuration,
|
||||
pub request_timeout: StdDuration,
|
||||
pub response_body_idle_timeout: StdDuration,
|
||||
}
|
||||
|
||||
impl TransitionClientTimeouts {
|
||||
pub const fn new(
|
||||
connect_timeout: StdDuration,
|
||||
request_timeout: StdDuration,
|
||||
response_body_idle_timeout: StdDuration,
|
||||
) -> Self {
|
||||
Self {
|
||||
connect_timeout,
|
||||
request_timeout,
|
||||
response_body_idle_timeout,
|
||||
}
|
||||
}
|
||||
|
||||
fn validate(self) -> Result<Self, std::io::Error> {
|
||||
if self.connect_timeout.is_zero() {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::InvalidInput,
|
||||
"remote tier connect timeout must be greater than zero",
|
||||
));
|
||||
}
|
||||
if self.request_timeout.is_zero() {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::InvalidInput,
|
||||
"remote tier request timeout must be greater than zero",
|
||||
));
|
||||
}
|
||||
if self.response_body_idle_timeout.is_zero() {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::InvalidInput,
|
||||
"remote tier response body idle timeout must be greater than zero",
|
||||
));
|
||||
}
|
||||
Ok(self)
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for TransitionClientTimeouts {
|
||||
fn default() -> Self {
|
||||
Self {
|
||||
connect_timeout: StdDuration::from_secs(rustfs_config::DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS),
|
||||
request_timeout: StdDuration::from_secs(rustfs_config::DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS),
|
||||
response_body_idle_timeout: StdDuration::from_secs(
|
||||
rustfs_config::DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
|
||||
),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
@@ -288,12 +426,28 @@ async fn build_tls_config() -> Result<rustls::ClientConfig, std::io::Error> {
|
||||
|
||||
impl TransitionClient {
|
||||
pub async fn new(endpoint: &str, opts: Options, tier_type: &str) -> Result<TransitionClient, std::io::Error> {
|
||||
let client = Self::private_new(endpoint, opts, tier_type).await?;
|
||||
|
||||
Ok(client)
|
||||
Self::private_new(endpoint, opts, tier_type, TransitionClientTimeouts::default()).await
|
||||
}
|
||||
|
||||
async fn private_new(endpoint: &str, opts: Options, tier_type: &str) -> Result<TransitionClient, std::io::Error> {
|
||||
/// Builds a transition client with explicit transport timeout budgets.
|
||||
///
|
||||
/// [`Self::new`] keeps the historical constructor surface and uses the
|
||||
/// production defaults from [`TransitionClientTimeouts::default`].
|
||||
pub async fn new_with_timeouts(
|
||||
endpoint: &str,
|
||||
opts: Options,
|
||||
tier_type: &str,
|
||||
timeouts: TransitionClientTimeouts,
|
||||
) -> Result<TransitionClient, std::io::Error> {
|
||||
Self::private_new(endpoint, opts, tier_type, timeouts).await
|
||||
}
|
||||
|
||||
async fn private_new(
|
||||
endpoint: &str,
|
||||
opts: Options,
|
||||
tier_type: &str,
|
||||
timeouts: TransitionClientTimeouts,
|
||||
) -> Result<TransitionClient, std::io::Error> {
|
||||
if rustls::crypto::CryptoProvider::get_default().is_none() {
|
||||
// No default provider is set yet; try to install aws-lc-rs.
|
||||
// `install_default` can only fail if another thread races us and installs a provider
|
||||
@@ -306,15 +460,19 @@ impl TransitionClient {
|
||||
}
|
||||
|
||||
let endpoint_url = get_endpoint_url(endpoint, opts.secure)?;
|
||||
let timeouts = timeouts.validate()?;
|
||||
|
||||
let tls = build_tls_config().await?;
|
||||
|
||||
let mut http = HttpConnector::new();
|
||||
http.enforce_http(false);
|
||||
http.set_connect_timeout(Some(timeouts.connect_timeout));
|
||||
let https = hyper_rustls::HttpsConnectorBuilder::new()
|
||||
.with_tls_config(tls)
|
||||
.https_or_http()
|
||||
.enable_http1()
|
||||
.enable_http2()
|
||||
.build();
|
||||
.wrap_connector(http);
|
||||
let http_client = Client::builder(TokioExecutor::new()).build(https);
|
||||
|
||||
let mut client = TransitionClient {
|
||||
@@ -337,6 +495,7 @@ impl TransitionClient {
|
||||
trailing_header_support: opts.trailing_headers,
|
||||
max_retries: opts.max_retries,
|
||||
tier_type: tier_type.to_string(),
|
||||
timeouts,
|
||||
};
|
||||
|
||||
{
|
||||
@@ -501,29 +660,43 @@ impl TransitionClient {
|
||||
}
|
||||
|
||||
pub async fn doit(&self, req: Request<s3s::Body>) -> Result<Response<Incoming>, std::io::Error> {
|
||||
let req_method;
|
||||
let req_uri;
|
||||
let resp;
|
||||
let http_client = self.http_client.clone();
|
||||
{
|
||||
req_method = req.method().clone();
|
||||
req_uri = req.uri().clone();
|
||||
|
||||
debug!("endpoint_url: {}", self.endpoint_url.as_str().to_string());
|
||||
resp = http_client.request(req);
|
||||
}
|
||||
let resp = resp.await;
|
||||
debug!("http_client url: {} {}", req_method, req_uri);
|
||||
if let Err(err) = resp {
|
||||
error!("http_client call error: {:?}", err);
|
||||
return Err(std::io::Error::other(err));
|
||||
}
|
||||
|
||||
let req_method = req.method().clone();
|
||||
let resp = tokio::time::timeout(self.timeouts.request_timeout, http_client.request(req)).await;
|
||||
let resp = match resp {
|
||||
Ok(r) => r,
|
||||
Err(_) => return Err(std::io::Error::other("Unexpected error in response")),
|
||||
Ok(Ok(resp)) => resp,
|
||||
Ok(Err(err)) => {
|
||||
let err = transition_transport_error(err);
|
||||
error!(
|
||||
event = EVENT_TIER_REMOTE_TRANSPORT,
|
||||
component = LOG_COMPONENT_S3_CLIENT,
|
||||
subsystem = LOG_SUBSYSTEM_TIER,
|
||||
method = %req_method,
|
||||
error_kind = ?err.kind(),
|
||||
"remote tier request failed"
|
||||
);
|
||||
return Err(err);
|
||||
}
|
||||
Err(_) => {
|
||||
warn!(
|
||||
event = EVENT_TIER_REMOTE_TRANSPORT,
|
||||
component = LOG_COMPONENT_S3_CLIENT,
|
||||
subsystem = LOG_SUBSYSTEM_TIER,
|
||||
method = %req_method,
|
||||
timeout_ms = self.timeouts.request_timeout.as_millis(),
|
||||
"remote tier request timed out before response headers"
|
||||
);
|
||||
return Err(remote_tier_timeout_error("remote tier request timed out before response headers"));
|
||||
}
|
||||
};
|
||||
debug!(status = %resp.status(), "remote tier response received");
|
||||
trace!(
|
||||
event = EVENT_TIER_REMOTE_TRANSPORT,
|
||||
component = LOG_COMPONENT_S3_CLIENT,
|
||||
subsystem = LOG_SUBSYSTEM_TIER,
|
||||
method = %req_method,
|
||||
status = %resp.status(),
|
||||
"remote tier response received"
|
||||
);
|
||||
|
||||
//let b = resp.body_mut().store_all_unlimited().await.unwrap().to_vec();
|
||||
//debug!("http_resp_body: {}", String::from_utf8(b).unwrap());
|
||||
@@ -537,7 +710,15 @@ impl TransitionClient {
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.unwrap_or_default()
|
||||
.to_string();
|
||||
warn!(status = %status, request_id, "remote tier request rejected");
|
||||
warn!(
|
||||
event = EVENT_TIER_REMOTE_TRANSPORT,
|
||||
component = LOG_COMPONENT_S3_CLIENT,
|
||||
subsystem = LOG_SUBSYSTEM_TIER,
|
||||
method = %req_method,
|
||||
status = %status,
|
||||
request_id,
|
||||
"remote tier request rejected"
|
||||
);
|
||||
}
|
||||
Ok(resp)
|
||||
}
|
||||
@@ -581,7 +762,9 @@ impl TransitionClient {
|
||||
let resp_status = resp.status();
|
||||
let h = resp.headers().clone();
|
||||
|
||||
let body_vec = collect_response_body(resp.into_body(), MAX_S3_ERROR_RESPONSE_SIZE).await?;
|
||||
let body_vec = self
|
||||
.collect_response_body(resp.into_body(), MAX_S3_ERROR_RESPONSE_SIZE)
|
||||
.await?;
|
||||
let parsed_error =
|
||||
http_resp_to_error_response(resp_status, &h, body_vec, &metadata.bucket_name, &metadata.object_name);
|
||||
let routing_region = parsed_error.region;
|
||||
@@ -635,6 +818,22 @@ impl TransitionClient {
|
||||
Err(std::io::Error::other("remote tier request did not produce a response"))
|
||||
}
|
||||
|
||||
pub async fn collect_response_body<B>(&self, body: B, limit: usize) -> Result<Vec<u8>, std::io::Error>
|
||||
where
|
||||
B: Body<Data = Bytes>,
|
||||
B::Error: Into<Box<dyn StdError + Send + Sync>>,
|
||||
{
|
||||
collect_response_body_inner(body, Some(limit), Some(self.timeouts.response_body_idle_timeout)).await
|
||||
}
|
||||
|
||||
pub async fn collect_response_body_unbounded<B>(&self, body: B) -> Result<Vec<u8>, std::io::Error>
|
||||
where
|
||||
B: Body<Data = Bytes>,
|
||||
B::Error: Into<Box<dyn StdError + Send + Sync>>,
|
||||
{
|
||||
collect_response_body_inner(body, None, Some(self.timeouts.response_body_idle_timeout)).await
|
||||
}
|
||||
|
||||
async fn new_request(
|
||||
&self,
|
||||
method: &http::Method,
|
||||
@@ -1504,12 +1703,17 @@ pub struct CreateBucketConfiguration {
|
||||
mod tests {
|
||||
use super::{
|
||||
MAX_S3_CLIENT_RESPONSE_SIZE, MAX_S3_ERROR_RESPONSE_SIZE, SignatureType, build_tls_config, collect_response_body,
|
||||
signer_error_to_io_error, to_object_info_for_provider, validate_header_values, with_rustls_init_guard,
|
||||
collect_response_body_inner, signer_error_to_io_error, to_object_info_for_provider, validate_header_values,
|
||||
with_rustls_init_guard,
|
||||
};
|
||||
use crate::provider_versions::{BucketVersioningState, ProviderVersionCapabilities, RemoteVersion};
|
||||
use http::{HeaderMap, HeaderValue};
|
||||
use http_body_util::Full;
|
||||
use futures::stream;
|
||||
use http::{HeaderMap, HeaderValue, Request};
|
||||
use http_body::Frame;
|
||||
use http_body_util::{Full, StreamBody};
|
||||
use hyper::body::Bytes;
|
||||
use std::time::Duration as StdDuration;
|
||||
use tokio::net::TcpListener;
|
||||
use uuid::Uuid;
|
||||
|
||||
#[tokio::test]
|
||||
@@ -1540,6 +1744,77 @@ mod tests {
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn empty_data_frames_do_not_reset_the_body_idle_timeout() {
|
||||
let frames = stream::unfold((), |_| async {
|
||||
tokio::time::sleep(StdDuration::from_millis(10)).await;
|
||||
Some((Ok::<_, std::io::Error>(Frame::data(Bytes::new())), ()))
|
||||
});
|
||||
let body = StreamBody::new(Box::pin(frames));
|
||||
|
||||
let err = tokio::time::timeout(
|
||||
StdDuration::from_millis(200),
|
||||
collect_response_body_inner(body, Some(1), Some(StdDuration::from_millis(50))),
|
||||
)
|
||||
.await
|
||||
.expect("the collector should enforce its own body idle timeout")
|
||||
.expect_err("empty frames must not count as body progress");
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn public_body_collector_accepts_non_unpin_bodies() {
|
||||
let body = StreamBody::new(stream::once(async { Ok::<_, std::io::Error>(Frame::data(Bytes::from_static(b"ok"))) }));
|
||||
|
||||
let collected = collect_response_body(body, 2)
|
||||
.await
|
||||
.expect("the public collector should pin non-Unpin bodies internally");
|
||||
|
||||
assert_eq!(collected, b"ok");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn https_endpoints_reach_the_transport_connector() {
|
||||
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||
Ok(listener) => listener,
|
||||
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||
Err(err) => panic!("test listener should bind: {err}"),
|
||||
};
|
||||
let endpoint = listener
|
||||
.local_addr()
|
||||
.expect("listener local address should be available")
|
||||
.to_string();
|
||||
let accepted = tokio::spawn(async move {
|
||||
let (stream, _) = tokio::time::timeout(StdDuration::from_secs(1), listener.accept())
|
||||
.await
|
||||
.expect("HTTPS connector should reach the TCP listener")
|
||||
.expect("fixture should accept the HTTPS connection");
|
||||
drop(stream);
|
||||
});
|
||||
let client = super::TransitionClient::new_with_timeouts(
|
||||
&endpoint,
|
||||
super::Options {
|
||||
secure: true,
|
||||
..Default::default()
|
||||
},
|
||||
"",
|
||||
super::TransitionClientTimeouts::new(StdDuration::from_secs(1), StdDuration::from_secs(1), StdDuration::from_secs(1)),
|
||||
)
|
||||
.await
|
||||
.expect("fixture client should build");
|
||||
let request = Request::builder()
|
||||
.uri(format!("https://{endpoint}/"))
|
||||
.body(s3s::Body::empty())
|
||||
.expect("fixture request should build");
|
||||
|
||||
client
|
||||
.doit(request)
|
||||
.await
|
||||
.expect_err("the fixture closes before completing the TLS handshake");
|
||||
accepted.await.expect("fixture should join");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rustls_guard_converts_panics_to_io_errors() {
|
||||
let err = with_rustls_init_guard(|| -> Result<(), std::io::Error> { panic!("missing provider") })
|
||||
@@ -1573,6 +1848,18 @@ mod tests {
|
||||
assert!(outcome.is_ok(), "provider install guard must not panic when a provider is already set");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn transition_timeouts_reject_zero_budgets() {
|
||||
for timeouts in [
|
||||
super::TransitionClientTimeouts::new(StdDuration::ZERO, StdDuration::from_secs(1), StdDuration::from_secs(1)),
|
||||
super::TransitionClientTimeouts::new(StdDuration::from_secs(1), StdDuration::ZERO, StdDuration::from_secs(1)),
|
||||
super::TransitionClientTimeouts::new(StdDuration::from_secs(1), StdDuration::from_secs(1), StdDuration::ZERO),
|
||||
] {
|
||||
let err = timeouts.validate().expect_err("zero timeout budgets must fail closed");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn validate_header_values_returns_header_name_for_non_utf8_values() {
|
||||
let mut headers = HeaderMap::new();
|
||||
|
||||
Reference in New Issue
Block a user