mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-23 04:39:04 +00:00
Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 48f742644c | |||
| f8203b43f9 |
@@ -39,10 +39,11 @@ jobs:
|
|||||||
env:
|
env:
|
||||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout main branch
|
||||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
ref: main
|
||||||
|
|
||||||
- name: Setup Rust environment
|
- name: Setup Rust environment
|
||||||
uses: ./.github/actions/setup
|
uses: ./.github/actions/setup
|
||||||
@@ -88,10 +89,11 @@ jobs:
|
|||||||
# either casing.
|
# either casing.
|
||||||
NO_PROXY: 127.0.0.1,localhost
|
NO_PROXY: 127.0.0.1,localhost
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout main branch
|
||||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
ref: main
|
||||||
|
|
||||||
- name: Setup Rust environment
|
- name: Setup Rust environment
|
||||||
uses: ./.github/actions/setup
|
uses: ./.github/actions/setup
|
||||||
@@ -176,10 +178,11 @@ jobs:
|
|||||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
||||||
NO_PROXY: 127.0.0.1,localhost
|
NO_PROXY: 127.0.0.1,localhost
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout main branch
|
||||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
ref: main
|
||||||
|
|
||||||
- name: Setup Rust environment
|
- name: Setup Rust environment
|
||||||
uses: ./.github/actions/setup
|
uses: ./.github/actions/setup
|
||||||
|
|||||||
@@ -462,16 +462,19 @@ pub(crate) fn local_decommission_queue_prefix(endpoints: &EndpointServerPools, i
|
|||||||
Ok(local)
|
Ok(local)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn resumable_decommission_queue_indices(meta: &PoolMeta) -> Vec<usize> {
|
fn first_resumable_decommission_queue_indices(meta: &PoolMeta) -> Vec<usize> {
|
||||||
let mut indices = Vec::new();
|
let mut indices = Vec::new();
|
||||||
for (idx, pool) in meta.pools.iter().enumerate() {
|
for (idx, pool) in meta.pools.iter().enumerate() {
|
||||||
if let Some(decommission) = &pool.decommission {
|
if let Some(decommission) = &pool.decommission {
|
||||||
if !decommission.has_decommission_state() {
|
if !decommission.has_decommission_state() {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if decommission.complete || decommission.failed || decommission.canceled {
|
if decommission.complete {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
if decommission.failed || decommission.canceled {
|
||||||
|
break;
|
||||||
|
}
|
||||||
indices.push(idx);
|
indices.push(idx);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -2403,10 +2406,24 @@ impl PoolMeta {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn return_resumable_pools(&self) -> Vec<PoolStatus> {
|
pub fn return_resumable_pools(&self) -> Vec<PoolStatus> {
|
||||||
resumable_decommission_queue_indices(self)
|
let mut new_pools = Vec::new();
|
||||||
.into_iter()
|
for pool in &self.pools {
|
||||||
.map(|idx| self.pools[idx].clone())
|
if let Some(decommission) = &pool.decommission {
|
||||||
.collect()
|
if !decommission.has_decommission_state() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if decommission.complete || decommission.failed || decommission.canceled {
|
||||||
|
// Recovery is not required when:
|
||||||
|
// - Decommissioning completed
|
||||||
|
// - Decommissioning failed and must be explicitly restarted or cleared
|
||||||
|
// - Decommissioning was cancelled
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// All other scenarios require recovery
|
||||||
|
new_pools.push(pool.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
new_pools
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -3315,7 +3332,7 @@ impl ECStore {
|
|||||||
let _start_guard = self.start_gate.lock().await;
|
let _start_guard = self.start_gate.lock().await;
|
||||||
let indices = {
|
let indices = {
|
||||||
let pool_meta = self.pool_meta.read().await;
|
let pool_meta = self.pool_meta.read().await;
|
||||||
resumable_decommission_queue_indices(&pool_meta)
|
first_resumable_decommission_queue_indices(&pool_meta)
|
||||||
.into_iter()
|
.into_iter()
|
||||||
.filter(|idx| indices.contains(idx))
|
.filter(|idx| indices.contains(idx))
|
||||||
.collect::<Vec<_>>()
|
.collect::<Vec<_>>()
|
||||||
@@ -3360,7 +3377,7 @@ impl ECStore {
|
|||||||
pub async fn spawn_missing_local_decommission_routines(self: &Arc<Self>) -> Result<()> {
|
pub async fn spawn_missing_local_decommission_routines(self: &Arc<Self>) -> Result<()> {
|
||||||
let indices = {
|
let indices = {
|
||||||
let pool_meta = self.pool_meta.read().await;
|
let pool_meta = self.pool_meta.read().await;
|
||||||
resumable_decommission_queue_indices(&pool_meta)
|
first_resumable_decommission_queue_indices(&pool_meta)
|
||||||
};
|
};
|
||||||
let indices = local_decommission_queue_prefix(&self.endpoints(), &indices)?;
|
let indices = local_decommission_queue_prefix(&self.endpoints(), &indices)?;
|
||||||
if indices.is_empty() {
|
if indices.is_empty() {
|
||||||
@@ -6346,12 +6363,12 @@ mod pools_tests {
|
|||||||
ensure_decommission_start_keeps_active_pool, ensure_decommission_start_local_leader,
|
ensure_decommission_start_keeps_active_pool, ensure_decommission_start_local_leader,
|
||||||
ensure_decommission_start_pool_states, ensure_decommission_start_rebalance_meta_allowed,
|
ensure_decommission_start_pool_states, ensure_decommission_start_rebalance_meta_allowed,
|
||||||
ensure_decommission_start_target_capacity, ensure_decommission_terminal_operation_supported,
|
ensure_decommission_start_target_capacity, ensure_decommission_terminal_operation_supported,
|
||||||
ensure_local_decommission_pool_leaders, ensure_valid_decommission_pool_index, get_by_index, guard_decommission_cancelers,
|
ensure_local_decommission_pool_leaders, ensure_valid_decommission_pool_index, first_resumable_decommission_queue_indices,
|
||||||
has_active_decommission_canceler, is_decommission_active, is_decommission_cancel_requested,
|
get_by_index, guard_decommission_cancelers, has_active_decommission_canceler, is_decommission_active,
|
||||||
load_decommission_entry_versions, local_decommission_queue_prefix, mark_decommission_bucket_done,
|
is_decommission_cancel_requested, load_decommission_entry_versions, local_decommission_queue_prefix,
|
||||||
merge_pool_status_refresh, missing_decommission_worker_prefix, observe_decommission_terminal_reload_result,
|
mark_decommission_bucket_done, merge_pool_status_refresh, missing_decommission_worker_prefix,
|
||||||
pool_meta_has_active_decommission, require_decommission_store, reserve_decommission_start_cancelers,
|
observe_decommission_terminal_reload_result, pool_meta_has_active_decommission, require_decommission_store,
|
||||||
resolve_decommission_bucket_done_save_result, resolve_decommission_bucket_state,
|
reserve_decommission_start_cancelers, resolve_decommission_bucket_done_save_result, resolve_decommission_bucket_state,
|
||||||
resolve_decommission_check_after_list_result, resolve_decommission_entry_cleanup_delete_result,
|
resolve_decommission_check_after_list_result, resolve_decommission_entry_cleanup_delete_result,
|
||||||
resolve_decommission_entry_exact_versions, resolve_decommission_entry_reload_result,
|
resolve_decommission_entry_exact_versions, resolve_decommission_entry_reload_result,
|
||||||
resolve_decommission_listing_worker_result, resolve_decommission_optional_bucket_config_result,
|
resolve_decommission_listing_worker_result, resolve_decommission_optional_bucket_config_result,
|
||||||
@@ -6359,9 +6376,9 @@ mod pools_tests {
|
|||||||
resolve_decommission_preflight_heal_result, resolve_decommission_progress_save_result,
|
resolve_decommission_preflight_heal_result, resolve_decommission_progress_save_result,
|
||||||
resolve_decommission_terminal_mark_after_error_result, resolve_decommission_terminal_mark_result,
|
resolve_decommission_terminal_mark_after_error_result, resolve_decommission_terminal_mark_result,
|
||||||
resolve_decommission_update_after_result, resolve_start_decommission_pool_meta_reload_result,
|
resolve_decommission_update_after_result, resolve_start_decommission_pool_meta_reload_result,
|
||||||
resumable_decommission_queue_indices, rollback_start_decommission_pool_meta, run_decommission_buckets_bounded,
|
rollback_start_decommission_pool_meta, run_decommission_buckets_bounded, run_decommission_listing_with_retry,
|
||||||
run_decommission_listing_with_retry, run_decommission_listing_with_retry_and_drain, run_decommission_side_effect,
|
run_decommission_listing_with_retry_and_drain, run_decommission_side_effect, should_cleanup_decommission_source_entry,
|
||||||
should_cleanup_decommission_source_entry, should_continue_decommission_queue, should_count_decommission_version_complete,
|
should_continue_decommission_queue, should_count_decommission_version_complete,
|
||||||
should_preserve_decommission_canceled_state, should_reject_decommission_cancel_as_terminal,
|
should_preserve_decommission_canceled_state, should_reject_decommission_cancel_as_terminal,
|
||||||
should_retry_decommission_cancel_reload, should_retry_decommission_listing, should_skip_canceled_decommission_routine,
|
should_retry_decommission_cancel_reload, should_retry_decommission_listing, should_skip_canceled_decommission_routine,
|
||||||
spawn_decommission_index_cancelers, split_decommission_buckets, take_and_cancel_decommission_canceler,
|
spawn_decommission_index_cancelers, split_decommission_buckets, take_and_cancel_decommission_canceler,
|
||||||
@@ -9370,7 +9387,7 @@ mod pools_tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_resumable_decommission_queue_indices_skip_terminal_predecessors() {
|
fn test_first_resumable_decommission_queue_indices_stops_at_failed_or_canceled_state() {
|
||||||
let meta = PoolMeta {
|
let meta = PoolMeta {
|
||||||
pools: vec![
|
pools: vec![
|
||||||
decommission_test_pool_status(
|
decommission_test_pool_status(
|
||||||
@@ -9406,11 +9423,11 @@ mod pools_tests {
|
|||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|
||||||
assert_eq!(resumable_decommission_queue_indices(&meta), vec![3]);
|
assert!(first_resumable_decommission_queue_indices(&meta).is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_resumable_decommission_queue_indices_preserve_active_predecessor_order() {
|
fn test_first_resumable_decommission_queue_indices_allows_after_completed_prefix() {
|
||||||
let meta = PoolMeta {
|
let meta = PoolMeta {
|
||||||
pools: vec![
|
pools: vec![
|
||||||
decommission_test_pool_status(
|
decommission_test_pool_status(
|
||||||
@@ -9423,7 +9440,7 @@ mod pools_tests {
|
|||||||
decommission_test_pool_status(
|
decommission_test_pool_status(
|
||||||
1,
|
1,
|
||||||
Some(PoolDecommissionInfo {
|
Some(PoolDecommissionInfo {
|
||||||
start_time: Some(OffsetDateTime::UNIX_EPOCH),
|
queued: true,
|
||||||
..Default::default()
|
..Default::default()
|
||||||
}),
|
}),
|
||||||
),
|
),
|
||||||
@@ -9438,11 +9455,11 @@ mod pools_tests {
|
|||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|
||||||
assert_eq!(resumable_decommission_queue_indices(&meta), vec![1, 2]);
|
assert_eq!(first_resumable_decommission_queue_indices(&meta), vec![1, 2]);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_runtime_and_startup_use_the_same_resumable_queue() {
|
fn test_return_resumable_pools_skips_failed_decommission() {
|
||||||
let meta = PoolMeta {
|
let meta = PoolMeta {
|
||||||
pools: vec![
|
pools: vec![
|
||||||
decommission_test_pool_status(
|
decommission_test_pool_status(
|
||||||
@@ -9454,20 +9471,6 @@ mod pools_tests {
|
|||||||
),
|
),
|
||||||
decommission_test_pool_status(
|
decommission_test_pool_status(
|
||||||
1,
|
1,
|
||||||
Some(PoolDecommissionInfo {
|
|
||||||
canceled: true,
|
|
||||||
..Default::default()
|
|
||||||
}),
|
|
||||||
),
|
|
||||||
decommission_test_pool_status(
|
|
||||||
2,
|
|
||||||
Some(PoolDecommissionInfo {
|
|
||||||
start_time: Some(OffsetDateTime::UNIX_EPOCH),
|
|
||||||
..Default::default()
|
|
||||||
}),
|
|
||||||
),
|
|
||||||
decommission_test_pool_status(
|
|
||||||
3,
|
|
||||||
Some(PoolDecommissionInfo {
|
Some(PoolDecommissionInfo {
|
||||||
queued: true,
|
queued: true,
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -9477,15 +9480,10 @@ mod pools_tests {
|
|||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|
||||||
let runtime_indices = resumable_decommission_queue_indices(&meta);
|
let resumable = meta.return_resumable_pools();
|
||||||
let startup_ids = meta
|
|
||||||
.return_resumable_pools()
|
|
||||||
.into_iter()
|
|
||||||
.map(|pool| pool.id)
|
|
||||||
.collect::<Vec<_>>();
|
|
||||||
|
|
||||||
assert_eq!(runtime_indices, vec![2, 3]);
|
assert_eq!(resumable.len(), 1);
|
||||||
assert_eq!(startup_ids, vec![2, 3]);
|
assert_eq!(resumable[0].id, 1);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -784,24 +784,6 @@ pub(crate) fn create_deferred_bitrot_reader_with_stripe_handle(
|
|||||||
///
|
///
|
||||||
/// # Returns
|
/// # Returns
|
||||||
/// A Result containing the BitrotWriterWrapper or an error
|
/// A Result containing the BitrotWriterWrapper or an error
|
||||||
/// Size hint handed to `DiskAPI::create_file` for a bitrot-wrapped shard.
|
|
||||||
///
|
|
||||||
/// A known length is grown by one checksum per shard so the on-disk file size
|
|
||||||
/// matches what the bitrot writer emits. A negative length is the
|
|
||||||
/// unknown-size sentinel (`HashReader::SIZE_PRESERVE_LAYER`, used by SSE and
|
|
||||||
/// compression) and must be preserved: `RemoteDisk::create_file` forwards it
|
|
||||||
/// in the `put_file_stream` query, and the receiver only treats `size > 0` as
|
|
||||||
/// a fixed body length when locating the authenticated trailer. Clamping it
|
|
||||||
/// to `0` would claim an empty body and misframe the stream. `0` stays `0`
|
|
||||||
/// because a genuinely empty object still means an empty body.
|
|
||||||
fn bitrot_create_file_size(length: i64, shard_size: usize, checksum_algo: &HashAlgorithm) -> i64 {
|
|
||||||
if length <= 0 {
|
|
||||||
return length;
|
|
||||||
}
|
|
||||||
let length = length as usize;
|
|
||||||
(length.div_ceil(shard_size) * checksum_algo.size() + length) as i64
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn create_bitrot_writer(
|
pub async fn create_bitrot_writer(
|
||||||
is_inline_buffer: bool,
|
is_inline_buffer: bool,
|
||||||
disk: Option<&DiskStore>,
|
disk: Option<&DiskStore>,
|
||||||
@@ -814,7 +796,12 @@ pub async fn create_bitrot_writer(
|
|||||||
let writer = if is_inline_buffer {
|
let writer = if is_inline_buffer {
|
||||||
CustomWriter::new_inline_buffer()
|
CustomWriter::new_inline_buffer()
|
||||||
} else if let Some(disk) = disk {
|
} else if let Some(disk) = disk {
|
||||||
let length = bitrot_create_file_size(length, shard_size, &checksum_algo);
|
let length = if length > 0 {
|
||||||
|
let length = length as usize;
|
||||||
|
(length.div_ceil(shard_size) * checksum_algo.size() + length) as i64
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
|
||||||
let file = disk.create_file("", volume, path, length).await?;
|
let file = disk.create_file("", volume, path, length).await?;
|
||||||
#[cfg(feature = "hotpath")]
|
#[cfg(feature = "hotpath")]
|
||||||
@@ -833,25 +820,6 @@ mod tests {
|
|||||||
use rustfs_rio::ChunkReader;
|
use rustfs_rio::ChunkReader;
|
||||||
use std::collections::VecDeque;
|
use std::collections::VecDeque;
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn bitrot_create_file_size_grows_known_length_by_checksums() {
|
|
||||||
// 10 bytes over 4-byte shards = 3 shards, each followed by a 32-byte hash.
|
|
||||||
assert_eq!(bitrot_create_file_size(10, 4, &HashAlgorithm::HighwayHash256), 10 + 3 * 32);
|
|
||||||
assert_eq!(bitrot_create_file_size(10, 4, &HashAlgorithm::None), 10);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn bitrot_create_file_size_keeps_empty_and_unknown_distinct() {
|
|
||||||
assert_eq!(bitrot_create_file_size(0, 4, &HashAlgorithm::HighwayHash256), 0);
|
|
||||||
// SSE/compression streams advertise SIZE_PRESERVE_LAYER (-1); the remote
|
|
||||||
// put_file_stream receiver relies on a non-positive size to parse the auth
|
|
||||||
// trailer from the stream tail, so the sentinel must survive untouched.
|
|
||||||
assert_eq!(
|
|
||||||
bitrot_create_file_size(rustfs_rio::HashReader::SIZE_PRESERVE_LAYER, 4, &HashAlgorithm::HighwayHash256),
|
|
||||||
rustfs_rio::HashReader::SIZE_PRESERVE_LAYER
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
struct TestChunkReader {
|
struct TestChunkReader {
|
||||||
chunks: VecDeque<Bytes>,
|
chunks: VecDeque<Bytes>,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -16,14 +16,14 @@
|
|||||||
//!
|
//!
|
||||||
//! `scripts/test/vault_ha_kms_live.sh` owns the official Vault containers and
|
//! `scripts/test/vault_ha_kms_live.sh` owns the official Vault containers and
|
||||||
//! kills the active node while this test continuously decrypts through a
|
//! kills the active node while this test continuously decrypts through a
|
||||||
//! surviving standby. KV2 and Transit must recover after the bounded circuit
|
//! surviving standby. KV2 and Transit requests must remain successful, use a
|
||||||
//! interval, use a bounded number of attempts, and leave the circuit and
|
//! bounded number of attempts, and leave the circuit and in-flight gauges at
|
||||||
//! in-flight gauges at zero after a new leader is elected.
|
//! zero after a new leader is elected.
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
use std::path::{Path, PathBuf};
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::sync::Arc;
|
||||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||||
use std::sync::{Arc, Mutex};
|
|
||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
|
|
||||||
use metrics_util::MetricKind;
|
use metrics_util::MetricKind;
|
||||||
@@ -43,11 +43,6 @@ const OPERATION_ATTEMPTS: &str = "rustfs_kms_backend_operation_attempts";
|
|||||||
const IN_FLIGHT: &str = "rustfs_kms_backend_in_flight";
|
const IN_FLIGHT: &str = "rustfs_kms_backend_in_flight";
|
||||||
const CIRCUIT_OPEN: &str = "rustfs_kms_backend_circuit_open";
|
const CIRCUIT_OPEN: &str = "rustfs_kms_backend_circuit_open";
|
||||||
const MAX_ATTEMPTS: u32 = 10;
|
const MAX_ATTEMPTS: u32 = 10;
|
||||||
const ATTEMPT_TIMEOUT: Duration = Duration::from_secs(2);
|
|
||||||
const HEALTHY_PROGRESS_TIMEOUT: Duration = Duration::from_secs(20);
|
|
||||||
// The circuit remains open for 30s after five failed attempts.
|
|
||||||
const POST_FAILOVER_PROGRESS_TIMEOUT: Duration = Duration::from_secs(35);
|
|
||||||
const FAILOVER_ERROR_POLL_INTERVAL: Duration = Duration::from_millis(100);
|
|
||||||
|
|
||||||
type MetricEntry = (
|
type MetricEntry = (
|
||||||
metrics_util::CompositeKey,
|
metrics_util::CompositeKey,
|
||||||
@@ -69,7 +64,7 @@ fn config(backend: KmsBackend, backend_config: BackendConfig) -> KmsConfig {
|
|||||||
backend,
|
backend,
|
||||||
backend_config,
|
backend_config,
|
||||||
allow_insecure_dev_defaults: true,
|
allow_insecure_dev_defaults: true,
|
||||||
timeout: ATTEMPT_TIMEOUT,
|
timeout: Duration::from_secs(2),
|
||||||
retry_attempts: MAX_ATTEMPTS,
|
retry_attempts: MAX_ATTEMPTS,
|
||||||
enable_cache: false,
|
enable_cache: false,
|
||||||
..KmsConfig::default()
|
..KmsConfig::default()
|
||||||
@@ -169,31 +164,14 @@ fn retryable_failures(snapshot: &[MetricEntry], operation: &str) -> u64 {
|
|||||||
.sum()
|
.sum()
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn wait_for_count(
|
async fn wait_for_count(counter: &AtomicU64, minimum: u64, description: &str) {
|
||||||
counter: &AtomicU64,
|
tokio::time::timeout(Duration::from_secs(20), async {
|
||||||
failure: &Mutex<Option<String>>,
|
|
||||||
minimum: u64,
|
|
||||||
description: &str,
|
|
||||||
timeout: Duration,
|
|
||||||
) {
|
|
||||||
tokio::time::timeout(timeout, async {
|
|
||||||
while counter.load(Ordering::SeqCst) < minimum {
|
while counter.load(Ordering::SeqCst) < minimum {
|
||||||
if let Some(error) = failure.lock().expect("decrypt failure lock poisoned").as_ref() {
|
|
||||||
panic!(
|
|
||||||
"{description} worker failed after {} successful decrypts: {error}",
|
|
||||||
counter.load(Ordering::SeqCst)
|
|
||||||
);
|
|
||||||
}
|
|
||||||
tokio::time::sleep(Duration::from_millis(25)).await;
|
tokio::time::sleep(Duration::from_millis(25)).await;
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
.await
|
.await
|
||||||
.unwrap_or_else(|_| {
|
.unwrap_or_else(|_| panic!("timed out waiting for {description}"));
|
||||||
panic!(
|
|
||||||
"timed out after {timeout:?} waiting for {description}: completed {}, expected {minimum}",
|
|
||||||
counter.load(Ordering::SeqCst)
|
|
||||||
)
|
|
||||||
});
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn wait_for_file(path: &Path, description: &str) {
|
async fn wait_for_file(path: &Path, description: &str) {
|
||||||
@@ -211,8 +189,7 @@ async fn decrypt_loop<B: KmsBackendTrait + Send + Sync + 'static>(
|
|||||||
request: DecryptRequest,
|
request: DecryptRequest,
|
||||||
expected: Vec<u8>,
|
expected: Vec<u8>,
|
||||||
completed: Arc<AtomicU64>,
|
completed: Arc<AtomicU64>,
|
||||||
allow_failover_errors: Arc<AtomicBool>,
|
failed: Arc<AtomicBool>,
|
||||||
failure: Arc<Mutex<Option<String>>>,
|
|
||||||
stop: CancellationToken,
|
stop: CancellationToken,
|
||||||
) {
|
) {
|
||||||
while !stop.is_cancelled() {
|
while !stop.is_cancelled() {
|
||||||
@@ -220,18 +197,8 @@ async fn decrypt_loop<B: KmsBackendTrait + Send + Sync + 'static>(
|
|||||||
Ok(response) if response.plaintext == expected => {
|
Ok(response) if response.plaintext == expected => {
|
||||||
completed.fetch_add(1, Ordering::SeqCst);
|
completed.fetch_add(1, Ordering::SeqCst);
|
||||||
}
|
}
|
||||||
Ok(_) => {
|
Ok(_) | Err(_) => {
|
||||||
*failure.lock().expect("decrypt failure lock poisoned") =
|
failed.store(true, Ordering::SeqCst);
|
||||||
Some("decrypt returned unexpected plaintext".to_string());
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
Err(rustfs_kms::KmsError::BackendError { .. } | rustfs_kms::KmsError::OperationTimedOut { .. })
|
|
||||||
if allow_failover_errors.load(Ordering::SeqCst) =>
|
|
||||||
{
|
|
||||||
tokio::time::sleep(FAILOVER_ERROR_POLL_INTERVAL).await;
|
|
||||||
}
|
|
||||||
Err(error) => {
|
|
||||||
*failure.lock().expect("decrypt failure lock poisoned") = Some(error.to_string());
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -329,9 +296,7 @@ async fn exercise_failover(snapshotter: &Snapshotter) {
|
|||||||
);
|
);
|
||||||
|
|
||||||
let stop = CancellationToken::new();
|
let stop = CancellationToken::new();
|
||||||
let allow_failover_errors = Arc::new(AtomicBool::new(false));
|
let failed = Arc::new(AtomicBool::new(false));
|
||||||
let kv2_failure = Arc::new(Mutex::new(None));
|
|
||||||
let transit_failure = Arc::new(Mutex::new(None));
|
|
||||||
let kv2_completed = Arc::new(AtomicU64::new(0));
|
let kv2_completed = Arc::new(AtomicU64::new(0));
|
||||||
let transit_completed = Arc::new(AtomicU64::new(0));
|
let transit_completed = Arc::new(AtomicU64::new(0));
|
||||||
let kv2_worker = tokio::spawn(decrypt_loop(
|
let kv2_worker = tokio::spawn(decrypt_loop(
|
||||||
@@ -339,8 +304,7 @@ async fn exercise_failover(snapshotter: &Snapshotter) {
|
|||||||
kv2_request,
|
kv2_request,
|
||||||
kv2_data_key.plaintext_key,
|
kv2_data_key.plaintext_key,
|
||||||
Arc::clone(&kv2_completed),
|
Arc::clone(&kv2_completed),
|
||||||
Arc::clone(&allow_failover_errors),
|
Arc::clone(&failed),
|
||||||
Arc::clone(&kv2_failure),
|
|
||||||
stop.clone(),
|
stop.clone(),
|
||||||
));
|
));
|
||||||
let transit_worker = tokio::spawn(decrypt_loop(
|
let transit_worker = tokio::spawn(decrypt_loop(
|
||||||
@@ -348,21 +312,12 @@ async fn exercise_failover(snapshotter: &Snapshotter) {
|
|||||||
transit_request,
|
transit_request,
|
||||||
transit_data_key.plaintext_key,
|
transit_data_key.plaintext_key,
|
||||||
Arc::clone(&transit_completed),
|
Arc::clone(&transit_completed),
|
||||||
Arc::clone(&allow_failover_errors),
|
Arc::clone(&failed),
|
||||||
Arc::clone(&transit_failure),
|
|
||||||
stop.clone(),
|
stop.clone(),
|
||||||
));
|
));
|
||||||
|
|
||||||
wait_for_count(&kv2_completed, &kv2_failure, 2, "two healthy KV2 decrypts", HEALTHY_PROGRESS_TIMEOUT).await;
|
wait_for_count(&kv2_completed, 2, "two healthy KV2 decrypts").await;
|
||||||
wait_for_count(
|
wait_for_count(&transit_completed, 2, "two healthy Transit decrypts").await;
|
||||||
&transit_completed,
|
|
||||||
&transit_failure,
|
|
||||||
2,
|
|
||||||
"two healthy Transit decrypts",
|
|
||||||
HEALTHY_PROGRESS_TIMEOUT,
|
|
||||||
)
|
|
||||||
.await;
|
|
||||||
allow_failover_errors.store(true, Ordering::SeqCst);
|
|
||||||
std::fs::write(&marker, b"ready").expect("publish failover readiness marker");
|
std::fs::write(&marker, b"ready").expect("publish failover readiness marker");
|
||||||
|
|
||||||
wait_for_file(&elected, "the replacement Vault leader").await;
|
wait_for_file(&elected, "the replacement Vault leader").await;
|
||||||
@@ -371,39 +326,18 @@ async fn exercise_failover(snapshotter: &Snapshotter) {
|
|||||||
|
|
||||||
let kv2_after_election = kv2_completed.load(Ordering::SeqCst) + 2;
|
let kv2_after_election = kv2_completed.load(Ordering::SeqCst) + 2;
|
||||||
let transit_after_election = transit_completed.load(Ordering::SeqCst) + 2;
|
let transit_after_election = transit_completed.load(Ordering::SeqCst) + 2;
|
||||||
wait_for_count(
|
wait_for_count(&kv2_completed, kv2_after_election, "post-failover KV2 decrypts").await;
|
||||||
&kv2_completed,
|
wait_for_count(&transit_completed, transit_after_election, "post-failover Transit decrypts").await;
|
||||||
&kv2_failure,
|
|
||||||
kv2_after_election,
|
|
||||||
"post-failover KV2 decrypts",
|
|
||||||
POST_FAILOVER_PROGRESS_TIMEOUT,
|
|
||||||
)
|
|
||||||
.await;
|
|
||||||
wait_for_count(
|
|
||||||
&transit_completed,
|
|
||||||
&transit_failure,
|
|
||||||
transit_after_election,
|
|
||||||
"post-failover Transit decrypts",
|
|
||||||
POST_FAILOVER_PROGRESS_TIMEOUT,
|
|
||||||
)
|
|
||||||
.await;
|
|
||||||
|
|
||||||
stop.cancel();
|
stop.cancel();
|
||||||
kv2_worker.await.expect("KV2 decrypt worker must join");
|
kv2_worker.await.expect("KV2 decrypt worker must join");
|
||||||
transit_worker.await.expect("Transit decrypt worker must join");
|
transit_worker.await.expect("Transit decrypt worker must join");
|
||||||
assert!(
|
assert!(!failed.load(Ordering::SeqCst), "no decrypt may fail or return different plaintext");
|
||||||
kv2_failure.lock().expect("KV2 failure lock poisoned").is_none(),
|
|
||||||
"no KV2 decrypt may fail or return different plaintext"
|
|
||||||
);
|
|
||||||
assert!(
|
|
||||||
transit_failure.lock().expect("Transit failure lock poisoned").is_none(),
|
|
||||||
"no Transit decrypt may fail or return different plaintext"
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[ignore = "requires a real three-node Vault Raft cluster; run scripts/test/vault_ha_kms_live.sh"]
|
#[ignore = "requires a real three-node Vault Raft cluster; run scripts/test/vault_ha_kms_live.sh"]
|
||||||
fn vault_raft_leader_failure_recovers_kv2_and_transit_decrypts() {
|
fn vault_raft_leader_failure_preserves_kv2_and_transit_decrypts() {
|
||||||
let recorder = DebuggingRecorder::new();
|
let recorder = DebuggingRecorder::new();
|
||||||
let snapshotter = recorder.snapshotter();
|
let snapshotter = recorder.snapshotter();
|
||||||
metrics::with_local_recorder(&recorder, || {
|
metrics::with_local_recorder(&recorder, || {
|
||||||
@@ -415,6 +349,11 @@ fn vault_raft_leader_failure_recovers_kv2_and_transit_decrypts() {
|
|||||||
});
|
});
|
||||||
let snapshot = snapshotter.snapshot().into_vec();
|
let snapshot = snapshotter.snapshot().into_vec();
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
counter_value(&snapshot, OPERATIONS_TOTAL, &[("outcome", "circuit_open")]),
|
||||||
|
0,
|
||||||
|
"a bounded leader election must not open the circuit"
|
||||||
|
);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
counter_value(&snapshot, OPERATIONS_TOTAL, &[("outcome", "budget_exhausted")]),
|
counter_value(&snapshot, OPERATIONS_TOTAL, &[("outcome", "budget_exhausted")]),
|
||||||
0,
|
0,
|
||||||
|
|||||||
@@ -138,6 +138,11 @@ const SITE_REPL_RESYNC_DEFAULT_PAGE_SIZE: usize = 100;
|
|||||||
const SITE_REPL_RESYNC_MAX_PAGE_SIZE: usize = 1000;
|
const SITE_REPL_RESYNC_MAX_PAGE_SIZE: usize = 1000;
|
||||||
const SITE_REPLICATION_PEER_REQUEST_TIMEOUT: Duration = Duration::from_secs(10);
|
const SITE_REPLICATION_PEER_REQUEST_TIMEOUT: Duration = Duration::from_secs(10);
|
||||||
const SITE_REPLICATION_PEER_CONNECT_TIMEOUT: Duration = Duration::from_secs(3);
|
const SITE_REPLICATION_PEER_CONNECT_TIMEOUT: Duration = Duration::from_secs(3);
|
||||||
|
/// Bound on waiting for the lifecycle lock (below). 3x the peer request
|
||||||
|
/// timeout: outlives one full peer round of a healthy concurrent lifecycle
|
||||||
|
/// operation, while converting a holder wedged on unreachable peers into a
|
||||||
|
/// retryable 503 for the waiter instead of an unbounded hang.
|
||||||
|
const SITE_REPLICATION_LIFECYCLE_LOCK_TIMEOUT: Duration = Duration::from_secs(30);
|
||||||
const SITE_REPLICATION_PEER_ERROR_DETAIL_LIMIT: usize = 256;
|
const SITE_REPLICATION_PEER_ERROR_DETAIL_LIMIT: usize = 256;
|
||||||
const SITE_REPLICATION_INITIAL_SYNC_ERROR_LIMIT: usize = 32;
|
const SITE_REPLICATION_INITIAL_SYNC_ERROR_LIMIT: usize = 32;
|
||||||
const MAX_PEER_CA_CERT_PEM_SIZE: usize = 256 * 1024;
|
const MAX_PEER_CA_CERT_PEM_SIZE: usize = 256 * 1024;
|
||||||
@@ -388,9 +393,17 @@ struct SiteReplicationLifecycleGuard {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl SiteReplicationLifecycleGuard {
|
impl SiteReplicationLifecycleGuard {
|
||||||
async fn acquire() -> Self {
|
/// Bounded acquire: a holder wedged on unreachable peers (each probe
|
||||||
Self {
|
/// costs up to [`SITE_REPLICATION_PEER_REQUEST_TIMEOUT`]) must not hang
|
||||||
_guard: SITE_REPLICATION_LIFECYCLE_LOCK.lock().await,
|
/// every other lifecycle operation indefinitely, so waiters get a
|
||||||
|
/// retryable 503 after [`SITE_REPLICATION_LIFECYCLE_LOCK_TIMEOUT`].
|
||||||
|
async fn acquire() -> S3Result<Self> {
|
||||||
|
match tokio::time::timeout(SITE_REPLICATION_LIFECYCLE_LOCK_TIMEOUT, SITE_REPLICATION_LIFECYCLE_LOCK.lock()).await {
|
||||||
|
Ok(guard) => Ok(Self { _guard: guard }),
|
||||||
|
Err(_) => Err(S3Error::with_message(
|
||||||
|
S3ErrorCode::ServiceUnavailable,
|
||||||
|
"another site replication lifecycle operation is in progress; retry later".to_string(),
|
||||||
|
)),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2128,6 +2141,27 @@ async fn remote_add_preflight_info(site: &PeerSite) -> S3Result<SiteReplicationA
|
|||||||
add_preflight_info_from_sr_info(site, info, idp_settings)
|
add_preflight_info_from_sr_info(site, info, idp_settings)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Preflight every site in an add request while the lifecycle lock is held.
|
||||||
|
/// Probes run concurrently (matching the other peer fan-outs in this file):
|
||||||
|
/// k unreachable sites cost roughly one peer request timeout, not k of them.
|
||||||
|
/// Results (and the first error, if any) are reported in request order.
|
||||||
|
async fn add_preflight_infos(
|
||||||
|
sites: &[PeerSite],
|
||||||
|
current_state: &SiteReplicationState,
|
||||||
|
local_peer: &PeerInfo,
|
||||||
|
) -> S3Result<Vec<SiteReplicationAddPreflightInfo>> {
|
||||||
|
futures::future::join_all(sites.iter().map(|site| async move {
|
||||||
|
if same_identity_endpoint(&site.endpoint, &local_peer.endpoint) {
|
||||||
|
local_add_preflight_info(current_state, local_peer, site).await
|
||||||
|
} else {
|
||||||
|
remote_add_preflight_info(site).await
|
||||||
|
}
|
||||||
|
}))
|
||||||
|
.await
|
||||||
|
.into_iter()
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
fn validate_add_preflight_topology(infos: &[SiteReplicationAddPreflightInfo], local_peer: &PeerInfo) -> S3Result<()> {
|
fn validate_add_preflight_topology(infos: &[SiteReplicationAddPreflightInfo], local_peer: &PeerInfo) -> S3Result<()> {
|
||||||
let mut deployment_ids = HashSet::new();
|
let mut deployment_ids = HashSet::new();
|
||||||
let mut local_seen = false;
|
let mut local_seen = false;
|
||||||
@@ -9872,7 +9906,7 @@ impl Operation for SiteReplicationAddHandler {
|
|||||||
let cred = validate_site_replication_admin_request(&req, AdminAction::SiteReplicationAddAction).await?;
|
let cred = validate_site_replication_admin_request(&req, AdminAction::SiteReplicationAddAction).await?;
|
||||||
reject_site_replicator_on_public_admin(&cred)?;
|
reject_site_replicator_on_public_admin(&cred)?;
|
||||||
let replicate_ilm_expiry = sr_add_replicate_ilm_expiry(&req.uri);
|
let replicate_ilm_expiry = sr_add_replicate_ilm_expiry(&req.uri);
|
||||||
let lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await;
|
let lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await?;
|
||||||
// Everything up to the commit below is preflight: peer probes, IAM
|
// Everything up to the commit below is preflight: peer probes, IAM
|
||||||
// work and the join fan-out all talk to the network, so none of it may
|
// work and the join fan-out all talk to the network, so none of it may
|
||||||
// run inside the state transaction. The snapshot read here is what the
|
// run inside the state transaction. The snapshot read here is what the
|
||||||
@@ -9887,14 +9921,7 @@ impl Operation for SiteReplicationAddHandler {
|
|||||||
// inject it so the add preflight (which requires the local deployment) succeeds. No-op for `mc`.
|
// inject it so the add preflight (which requires the local deployment) succeeds. No-op for `mc`.
|
||||||
ensure_local_site_present(&mut sites, &local_peer);
|
ensure_local_site_present(&mut sites, &local_peer);
|
||||||
validate_add_sites(&sites, &local_peer)?;
|
validate_add_sites(&sites, &local_peer)?;
|
||||||
let mut preflight_infos = Vec::with_capacity(sites.len());
|
let preflight_infos = add_preflight_infos(&sites, ¤t_state, &local_peer).await?;
|
||||||
for site in &sites {
|
|
||||||
if same_identity_endpoint(&site.endpoint, &local_peer.endpoint) {
|
|
||||||
preflight_infos.push(local_add_preflight_info(¤t_state, &local_peer, site).await?);
|
|
||||||
} else {
|
|
||||||
preflight_infos.push(remote_add_preflight_info(site).await?);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
validate_add_preflight_topology(&preflight_infos, &local_peer)?;
|
validate_add_preflight_topology(&preflight_infos, &local_peer)?;
|
||||||
let expected_updated_at = current_state.updated_at;
|
let expected_updated_at = current_state.updated_at;
|
||||||
require_add_peer_tls_capability(&sites, &local_peer).await?;
|
require_add_peer_tls_capability(&sites, &local_peer).await?;
|
||||||
@@ -10102,7 +10129,7 @@ impl Operation for SiteReplicationRemoveHandler {
|
|||||||
async fn call(&self, req: S3Request<Body>, _params: Params<'_, '_>) -> S3Result<S3Response<(StatusCode, Body)>> {
|
async fn call(&self, req: S3Request<Body>, _params: Params<'_, '_>) -> S3Result<S3Response<(StatusCode, Body)>> {
|
||||||
let cred = validate_site_replication_admin_request(&req, AdminAction::SiteReplicationRemoveAction).await?;
|
let cred = validate_site_replication_admin_request(&req, AdminAction::SiteReplicationRemoveAction).await?;
|
||||||
reject_site_replicator_on_public_admin(&cred)?;
|
reject_site_replicator_on_public_admin(&cred)?;
|
||||||
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await;
|
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await?;
|
||||||
// The request body is read before the bucket-op guard and the state
|
// The request body is read before the bucket-op guard and the state
|
||||||
// transaction: a client that stalls mid-body must hold neither the
|
// transaction: a client that stalls mid-body must hold neither the
|
||||||
// state-object lock nor the write half of the bucket-op RwLock (which
|
// state-object lock nor the write half of the bucket-op RwLock (which
|
||||||
@@ -10312,7 +10339,7 @@ where
|
|||||||
F: FnOnce(SRPeerJoinReq) -> Fut + Send + 'static,
|
F: FnOnce(SRPeerJoinReq) -> Fut + Send + 'static,
|
||||||
Fut: std::future::Future<Output = S3Result<()>> + Send + 'static,
|
Fut: std::future::Future<Output = S3Result<()>> + Send + 'static,
|
||||||
{
|
{
|
||||||
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await;
|
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await?;
|
||||||
admit_peer_join_across_nodes(local_endpoint, join_req, defer_sync_state_enable, apply_iam).await
|
admit_peer_join_across_nodes(local_endpoint, join_req, defer_sync_state_enable, apply_iam).await
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -11142,7 +11169,7 @@ impl Operation for SRPeerRemoveHandler {
|
|||||||
async fn call(&self, req: S3Request<Body>, _params: Params<'_, '_>) -> S3Result<S3Response<(StatusCode, Body)>> {
|
async fn call(&self, req: S3Request<Body>, _params: Params<'_, '_>) -> S3Result<S3Response<(StatusCode, Body)>> {
|
||||||
validate_site_replication_admin_request(&req, AdminAction::SiteReplicationRemoveAction).await?;
|
validate_site_replication_admin_request(&req, AdminAction::SiteReplicationRemoveAction).await?;
|
||||||
let remove_req: SRRemoveReq = read_site_replication_json(req, "", false).await?;
|
let remove_req: SRRemoveReq = read_site_replication_json(req, "", false).await?;
|
||||||
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await;
|
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await?;
|
||||||
let _bucket_op_guard = SITE_REPLICATION_BUCKET_OP_LOCK.write().await;
|
let _bucket_op_guard = SITE_REPLICATION_BUCKET_OP_LOCK.write().await;
|
||||||
let removed_deployment_ids = update_site_replication_state(move |state| {
|
let removed_deployment_ids = update_site_replication_state(move |state| {
|
||||||
if pending_endpoint_refresh(state).is_some() {
|
if pending_endpoint_refresh(state).is_some() {
|
||||||
@@ -11186,7 +11213,7 @@ impl Operation for SiteReplicationResyncOpHandler {
|
|||||||
let operation = query.get("operation").cloned().unwrap_or_default();
|
let operation = query.get("operation").cloned().unwrap_or_default();
|
||||||
let resolved_store = object_store_from_req(&req);
|
let resolved_store = object_store_from_req(&req);
|
||||||
let requested_peer: PeerInfo = read_site_replication_json(req, "", false).await?;
|
let requested_peer: PeerInfo = read_site_replication_json(req, "", false).await?;
|
||||||
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await;
|
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await?;
|
||||||
let (peer, existing_status) = {
|
let (peer, existing_status) = {
|
||||||
let state = load_site_replication_state().await?;
|
let state = load_site_replication_state().await?;
|
||||||
let local_peer = current_local_runtime_peer(&state);
|
let local_peer = current_local_runtime_peer(&state);
|
||||||
@@ -11478,7 +11505,7 @@ impl Operation for SRRotateServiceAccountHandler {
|
|||||||
// mid-repair and race its own IAM write against the reconciler's
|
// mid-repair and race its own IAM write against the reconciler's
|
||||||
// stale one. (The removed process mutex used to provide this
|
// stale one. (The removed process mutex used to provide this
|
||||||
// exclusion as a side effect.)
|
// exclusion as a side effect.)
|
||||||
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await;
|
let _lifecycle_guard = SiteReplicationLifecycleGuard::acquire().await?;
|
||||||
let local_endpoint = site_replication_local_endpoint(&req.uri, &req.headers);
|
let local_endpoint = site_replication_local_endpoint(&req.uri, &req.headers);
|
||||||
let rotation_parent = cred.access_key.clone();
|
let rotation_parent = cred.access_key.clone();
|
||||||
let (pending_rotation, local_peer, previous_access_key) = update_site_replication_state_when_changed(move |state| {
|
let (pending_rotation, local_peer, previous_access_key) = update_site_replication_state_when_changed(move |state| {
|
||||||
@@ -14019,7 +14046,9 @@ mod tests {
|
|||||||
async fn test_add_bootstrap_scope_only_allows_expected_bucket_setup_until_guard_drops() {
|
async fn test_add_bootstrap_scope_only_allows_expected_bucket_setup_until_guard_drops() {
|
||||||
let token;
|
let token;
|
||||||
{
|
{
|
||||||
let lifecycle = SiteReplicationLifecycleGuard::acquire().await;
|
let lifecycle = SiteReplicationLifecycleGuard::acquire()
|
||||||
|
.await
|
||||||
|
.expect("acquire lifecycle guard");
|
||||||
let guard = SiteReplicationAddInProgressGuard::start(lifecycle, HashSet::from(["legacy-bucket".to_string()]))
|
let guard = SiteReplicationAddInProgressGuard::start(lifecycle, HashSet::from(["legacy-bucket".to_string()]))
|
||||||
.expect("start site replication add guard");
|
.expect("start site replication add guard");
|
||||||
token = guard.token.to_string();
|
token = guard.token.to_string();
|
||||||
@@ -14075,14 +14104,18 @@ mod tests {
|
|||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial]
|
#[serial]
|
||||||
async fn test_add_lifecycle_allows_callback_before_remove_writer() {
|
async fn test_add_lifecycle_allows_callback_before_remove_writer() {
|
||||||
let lifecycle = SiteReplicationLifecycleGuard::acquire().await;
|
let lifecycle = SiteReplicationLifecycleGuard::acquire()
|
||||||
|
.await
|
||||||
|
.expect("acquire lifecycle guard");
|
||||||
let add_guard =
|
let add_guard =
|
||||||
SiteReplicationAddInProgressGuard::start(lifecycle, HashSet::new()).expect("start site replication add guard");
|
SiteReplicationAddInProgressGuard::start(lifecycle, HashSet::new()).expect("start site replication add guard");
|
||||||
let (started_tx, started_rx) = tokio::sync::oneshot::channel();
|
let (started_tx, started_rx) = tokio::sync::oneshot::channel();
|
||||||
let (entered_tx, mut entered_rx) = tokio::sync::oneshot::channel();
|
let (entered_tx, mut entered_rx) = tokio::sync::oneshot::channel();
|
||||||
let remove = tokio::spawn(async move {
|
let remove = tokio::spawn(async move {
|
||||||
let _ = started_tx.send(());
|
let _ = started_tx.send(());
|
||||||
let _lifecycle = SiteReplicationLifecycleGuard::acquire().await;
|
let _lifecycle = SiteReplicationLifecycleGuard::acquire()
|
||||||
|
.await
|
||||||
|
.expect("acquire lifecycle guard");
|
||||||
let _bucket_op = SITE_REPLICATION_BUCKET_OP_LOCK.write().await;
|
let _bucket_op = SITE_REPLICATION_BUCKET_OP_LOCK.write().await;
|
||||||
let _ = entered_tx.send(());
|
let _ = entered_tx.send(());
|
||||||
});
|
});
|
||||||
@@ -14102,6 +14135,111 @@ mod tests {
|
|||||||
entered_rx.await.expect("remove entered lifecycle");
|
entered_rx.await.expect("remove entered lifecycle");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Deleting either constant (or "simplifying" the client builders to
|
||||||
|
/// inline values) removes the only bound on how long a lifecycle
|
||||||
|
/// operation can be wedged per unreachable peer (#1889 C1 / #1952 C2).
|
||||||
|
#[test]
|
||||||
|
fn test_peer_timeout_constants_bound_unreachable_peer_probes() {
|
||||||
|
assert_eq!(SITE_REPLICATION_PEER_REQUEST_TIMEOUT, Duration::from_secs(10));
|
||||||
|
assert_eq!(SITE_REPLICATION_PEER_CONNECT_TIMEOUT, Duration::from_secs(3));
|
||||||
|
assert!(
|
||||||
|
SITE_REPLICATION_LIFECYCLE_LOCK_TIMEOUT >= SITE_REPLICATION_PEER_REQUEST_TIMEOUT,
|
||||||
|
"a waiter must not give up before the holder's single wedged peer probe can finish"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial]
|
||||||
|
async fn test_lifecycle_guard_acquire_times_out_with_retryable_503() {
|
||||||
|
let holder = SiteReplicationLifecycleGuard::acquire().await.expect("first acquire");
|
||||||
|
let err =
|
||||||
|
match tokio::time::timeout(SITE_REPLICATION_LIFECYCLE_LOCK_TIMEOUT * 2, SiteReplicationLifecycleGuard::acquire())
|
||||||
|
.await
|
||||||
|
.expect("bounded acquire must not hang while the lock is held")
|
||||||
|
{
|
||||||
|
Ok(_) => panic!("acquire while the lock is held should time out"),
|
||||||
|
Err(err) => err,
|
||||||
|
};
|
||||||
|
assert_eq!(err.code(), &S3ErrorCode::ServiceUnavailable);
|
||||||
|
|
||||||
|
drop(holder);
|
||||||
|
tokio::time::timeout(Duration::from_secs(1), SiteReplicationLifecycleGuard::acquire())
|
||||||
|
.await
|
||||||
|
.expect("acquire after release must not wait")
|
||||||
|
.expect("acquire after release");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
struct PreflightFanoutTestState {
|
||||||
|
metainfo_barrier: Arc<tokio::sync::Barrier>,
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn preflight_fanout_test_handler(State(state): State<PreflightFanoutTestState>, uri: Uri) -> (StatusCode, String) {
|
||||||
|
if uri.path().ends_with("/site-replication/metainfo") {
|
||||||
|
state.metainfo_barrier.wait().await;
|
||||||
|
}
|
||||||
|
(StatusCode::OK, "{}".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn test_add_preflight_probes_sites_concurrently() {
|
||||||
|
temp_env::async_with_vars(
|
||||||
|
[(ALLOW_LOOPBACK_REPLICATION_TARGET_ENV, Some("true"))],
|
||||||
|
add_preflight_probes_sites_concurrently_inner(),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn add_preflight_probes_sites_concurrently_inner() {
|
||||||
|
const REMOTE_SITES: usize = 3;
|
||||||
|
let listener = match TcpListener::bind("127.0.0.1:0").await {
|
||||||
|
Ok(listener) => listener,
|
||||||
|
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
|
||||||
|
Err(err) => panic!("bind preflight test server: {err}"),
|
||||||
|
};
|
||||||
|
let endpoint = format!("http://{}", listener.local_addr().expect("preflight test address"));
|
||||||
|
let state = PreflightFanoutTestState {
|
||||||
|
metainfo_barrier: Arc::new(tokio::sync::Barrier::new(REMOTE_SITES)),
|
||||||
|
};
|
||||||
|
let server = tokio::spawn(async move {
|
||||||
|
axum::serve(listener, Router::new().fallback(any(preflight_fanout_test_handler)).with_state(state))
|
||||||
|
.await
|
||||||
|
.expect("serve preflight test requests");
|
||||||
|
});
|
||||||
|
|
||||||
|
let sites: Vec<PeerSite> = (0..REMOTE_SITES)
|
||||||
|
.map(|index| PeerSite {
|
||||||
|
name: format!("site-{index}"),
|
||||||
|
endpoint: endpoint.clone(),
|
||||||
|
access_key: "test-access".to_string(),
|
||||||
|
secret_key: "test-secret".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let local_peer = PeerInfo {
|
||||||
|
deployment_id: "local".to_string(),
|
||||||
|
endpoint: "http://192.0.2.1:9000".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let current_state = SiteReplicationState::default();
|
||||||
|
|
||||||
|
// Each site's metainfo request parks on a barrier that only releases
|
||||||
|
// once every site's request has arrived: serial probing never sends
|
||||||
|
// the second request and dies on the peer request timeout, so
|
||||||
|
// finishing well inside that timeout proves the probes overlap —
|
||||||
|
// which is what caps k unreachable sites at one timeout, not k.
|
||||||
|
let infos = tokio::time::timeout(
|
||||||
|
SITE_REPLICATION_PEER_REQUEST_TIMEOUT / 2,
|
||||||
|
add_preflight_infos(&sites, ¤t_state, &local_peer),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("preflight probes must fan out concurrently, not serially")
|
||||||
|
.expect("preflight infos");
|
||||||
|
assert_eq!(infos.len(), REMOTE_SITES);
|
||||||
|
server.abort();
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_merge_add_sites_propagates_replicate_ilm_expiry() {
|
fn test_merge_add_sites_propagates_replicate_ilm_expiry() {
|
||||||
let state = merge_add_sites(
|
let state = merge_add_sites(
|
||||||
|
|||||||
@@ -241,7 +241,7 @@ env \
|
|||||||
RUSTFS_TEST_VAULT_FAILOVER_MARKER="$MARKER" \
|
RUSTFS_TEST_VAULT_FAILOVER_MARKER="$MARKER" \
|
||||||
RUSTFS_TEST_VAULT_OLD_LEADER="$OLD_LEADER" \
|
RUSTFS_TEST_VAULT_OLD_LEADER="$OLD_LEADER" \
|
||||||
cargo test -p rustfs-kms --test vault_ha_failover_live \
|
cargo test -p rustfs-kms --test vault_ha_failover_live \
|
||||||
vault_raft_leader_failure_recovers_kv2_and_transit_decrypts -- \
|
vault_raft_leader_failure_preserves_kv2_and_transit_decrypts -- \
|
||||||
--ignored --nocapture --test-threads=1 &
|
--ignored --nocapture --test-threads=1 &
|
||||||
TEST_PID=$!
|
TEST_PID=$!
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user