mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-05 21:07:43 +00:00
fix(ecstore): harden rebalance data movement (#3234)
* fix(ecstore): harden rebalance data movement * fix(ecstore): preserve failed rebalance status * fix(ecstore): avoid rebalance walkdir total timeout * fix(ecstore): retry rebalance listing timeouts * fix(ecstore): restrict data movement resume target * refactor(ecstore): simplify multipart movement target * fix(ecstore): restore resume target checks * perf(ecstore): speed up rebalance bucket merges * fix: keep rebalance listing alive on transient failures --------- Co-authored-by: houseme <housemecn@gmail.com>
This commit is contained in:
@@ -56,6 +56,7 @@ async fn peek_with_timeout<R: AsyncRead + Unpin>(reader: &mut MetacacheReader<R>
|
||||
pub(crate) enum TestReaderBehavior {
|
||||
Eof,
|
||||
Stall,
|
||||
IgnoreCancel,
|
||||
ProducerError(DiskError),
|
||||
PartialThenTimeout(Vec<MetaCacheEntry>),
|
||||
}
|
||||
@@ -72,6 +73,7 @@ pub struct ListPathRawOptions {
|
||||
pub min_disks: usize,
|
||||
pub report_not_found: bool,
|
||||
pub per_disk_limit: i32,
|
||||
pub skip_walkdir_total_timeout: bool,
|
||||
pub agreed: Option<AgreedFn>,
|
||||
pub partial: Option<PartialFn>,
|
||||
pub finished: Option<FinishedFn>,
|
||||
@@ -97,6 +99,7 @@ impl Clone for ListPathRawOptions {
|
||||
min_disks: self.min_disks,
|
||||
report_not_found: self.report_not_found,
|
||||
per_disk_limit: self.per_disk_limit,
|
||||
skip_walkdir_total_timeout: self.skip_walkdir_total_timeout,
|
||||
#[cfg(test)]
|
||||
test_reader_behaviors: self.test_reader_behaviors.clone(),
|
||||
#[cfg(test)]
|
||||
@@ -137,6 +140,11 @@ pub async fn list_path_raw(rx: CancellationToken, opts: ListPathRawOptions) -> d
|
||||
cancel_rx_clone.cancelled().await;
|
||||
return Ok(());
|
||||
}
|
||||
TestReaderBehavior::IgnoreCancel => {
|
||||
let _held_writer = wr;
|
||||
std::future::pending::<()>().await;
|
||||
return Ok(());
|
||||
}
|
||||
TestReaderBehavior::ProducerError(err) => {
|
||||
record_producer_error(&producer_errs_clone, disk_idx, &err);
|
||||
return Err(err);
|
||||
@@ -162,6 +170,7 @@ pub async fn list_path_raw(rx: CancellationToken, opts: ListPathRawOptions) -> d
|
||||
filter_prefix: opts_clone.filter_prefix.clone(),
|
||||
forward_to: opts_clone.forward_to.clone(),
|
||||
limit: opts_clone.per_disk_limit,
|
||||
skip_total_timeout: opts_clone.skip_walkdir_total_timeout,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
@@ -213,6 +222,7 @@ pub async fn list_path_raw(rx: CancellationToken, opts: ListPathRawOptions) -> d
|
||||
filter_prefix: opts_clone.filter_prefix.clone(),
|
||||
forward_to: opts_clone.forward_to.clone(),
|
||||
limit: opts_clone.per_disk_limit,
|
||||
skip_total_timeout: opts_clone.skip_walkdir_total_timeout,
|
||||
..Default::default()
|
||||
},
|
||||
&mut wr,
|
||||
@@ -485,6 +495,9 @@ pub async fn list_path_raw(rx: CancellationToken, opts: ListPathRawOptions) -> d
|
||||
if let Err(err) = revjob.await.map_err(std::io::Error::other)? {
|
||||
error!("list_path_raw: revjob err {:?}", err);
|
||||
cancel_rx.cancel();
|
||||
for job in jobs {
|
||||
job.abort();
|
||||
}
|
||||
|
||||
return Err(err);
|
||||
}
|
||||
@@ -492,9 +505,15 @@ pub async fn list_path_raw(rx: CancellationToken, opts: ListPathRawOptions) -> d
|
||||
// The merge consumer can finish successfully before every producer finishes
|
||||
// (for example after reaching EOF quorum while a tolerated drive is stalled,
|
||||
// or after the requested listing limit is satisfied). Cancel remaining walk
|
||||
// jobs before joining them so list calls do not wait for slow remote streams.
|
||||
// jobs before aborting them so list calls do not wait for slow remote streams.
|
||||
cancel_rx.cancel();
|
||||
|
||||
for job in jobs.iter() {
|
||||
if !job.is_finished() {
|
||||
job.abort();
|
||||
}
|
||||
}
|
||||
|
||||
let results = join_all(jobs).await;
|
||||
let mut job_errs = Vec::new();
|
||||
for result in results {
|
||||
@@ -509,6 +528,9 @@ pub async fn list_path_raw(rx: CancellationToken, opts: ListPathRawOptions) -> d
|
||||
job_errs.push(err);
|
||||
}
|
||||
Err(err) => {
|
||||
if err.is_cancelled() {
|
||||
continue;
|
||||
}
|
||||
error!("list_path_raw join err {:?}", err);
|
||||
job_errs.push(err.into());
|
||||
}
|
||||
@@ -582,6 +604,31 @@ mod tests {
|
||||
.expect("listing should complete when healthy quorum reached EOF and only a tolerated drive stalled");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn list_path_raw_aborts_unresponsive_producer_after_quorum_eof() {
|
||||
let result = timeout(
|
||||
Duration::from_millis(200),
|
||||
list_path_raw(
|
||||
CancellationToken::new(),
|
||||
ListPathRawOptions {
|
||||
disks: vec![None, None, None],
|
||||
min_disks: 2,
|
||||
test_reader_behaviors: vec![
|
||||
TestReaderBehavior::Eof,
|
||||
TestReaderBehavior::Eof,
|
||||
TestReaderBehavior::IgnoreCancel,
|
||||
],
|
||||
peek_timeout: Some(Duration::from_millis(20)),
|
||||
..Default::default()
|
||||
},
|
||||
),
|
||||
)
|
||||
.await;
|
||||
|
||||
let listing = result.expect("list_path_raw should abort unresponsive producer instead of hanging");
|
||||
assert!(listing.is_ok());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn list_path_raw_returns_timeout_when_producer_fails_after_partial_entry() {
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
|
||||
Reference in New Issue
Block a user