mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-07 20:46:11 +00:00
fix(storage): harden ODM and scanner publication (#7187)
* fix(storage): harden ODM and scanner publication * fix(app): simplify absent SSE configuration matching * test(heal): settle PUT rename tails before disk-wipe fixtures * fix(ecstore): remove duplicate local rename implementation Keep the canonical commit module after concurrent storage changes merged. The control-write and rollback changes are already present there. Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * fix(ci): satisfy new clippy lints * style(scanner): order merged test imports * fix(scanner): invalidate bucket work after namespace completion * fix(scanner): fence cached snapshots by scan execution --------- Co-authored-by: houseme <housemecn@gmail.com> Co-authored-by: heihutu <heihutu@gmail.com> Co-authored-by: zhi22915 <qiuzgang@gmail.com>
This commit is contained in:
@@ -20,6 +20,7 @@ use crate::scanner_folder::ScannerItem;
|
||||
use crate::storage_api::EcstoreScannerPeerDirtyUsageSnapshot;
|
||||
use crate::storage_api::owner::{
|
||||
EcstorePoolDecommissionInfo, EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta, EcstoreRebalanceStats,
|
||||
ecstore_hold_namespace_commit,
|
||||
};
|
||||
use crate::storage_api::scan::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions, ObjectIO as _};
|
||||
use crate::{
|
||||
@@ -343,6 +344,16 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
|
||||
.put_object(&bucket, object, &mut reader, &ScannerObjectOptions::default())
|
||||
.await
|
||||
.expect("object should be written to its selected pool");
|
||||
|
||||
// Quorum ACK can precede tail publication on the disk chosen to scan.
|
||||
let lock = store.pools[pool_index].disk_set[0]
|
||||
.new_ns_lock(&bucket, object)
|
||||
.await
|
||||
.expect("fixture namespace lock should be created");
|
||||
let _settled = lock
|
||||
.get_write_lock(Duration::from_secs(30))
|
||||
.await
|
||||
.expect("fixture rename tail should finish before the usage scan");
|
||||
}
|
||||
|
||||
let ctx = CancellationToken::new();
|
||||
@@ -362,7 +373,7 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
|
||||
.buckets_usage
|
||||
.get(&bucket)
|
||||
.expect("combined bucket usage should be present");
|
||||
assert_eq!(bucket_usage.objects_count, 2);
|
||||
assert_eq!(bucket_usage.objects_count, 2, "{usage:?}");
|
||||
assert_eq!(bucket_usage.size, 11);
|
||||
assert_eq!(usage.objects_total_count, 2);
|
||||
assert_eq!(usage.objects_total_size, 11);
|
||||
@@ -372,6 +383,102 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn pending_put_commit_keeps_scanner_walk_live_without_authoritative_usage() {
|
||||
let (_temp_dir, store) = setup_two_pool_scanner_store().await;
|
||||
let bucket = format!("scanner-pending-put-{}", Uuid::new_v4().simple());
|
||||
store
|
||||
.make_bucket(&bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created across both pools");
|
||||
for (pool_index, (object, body)) in [("pool-a", b"first".as_slice()), ("pool-b", b"second".as_slice())]
|
||||
.into_iter()
|
||||
.enumerate()
|
||||
{
|
||||
let mut reader = ScannerPutObjReader::from_vec(body.to_vec());
|
||||
store.pools[pool_index].disk_set[0]
|
||||
.put_object(
|
||||
&bucket,
|
||||
object,
|
||||
&mut reader,
|
||||
&ScannerObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("fixture objects must finish their rename fanouts before scanning");
|
||||
}
|
||||
|
||||
let mut pending = Some(ecstore_hold_namespace_commit(store.as_ref()));
|
||||
let mut previous_activity_digest = None;
|
||||
let mut structural_plan_digest = None;
|
||||
for (cycle, converged) in [(1, false), (2, true)] {
|
||||
if converged {
|
||||
drop(pending.take());
|
||||
}
|
||||
assert_eq!(store.scanner_data_usage_publication_blocked().await, !converged);
|
||||
assert!(!store.scanner_data_movement_pause_status().await.paused);
|
||||
let activity = crate::scanner::probe_scanner_activity(store.as_ref(), false)
|
||||
.await
|
||||
.expect("the fixture activity should be observable");
|
||||
let activity_digest = crate::scanner::scanner_activity_snapshot_digest(&activity);
|
||||
if let Some(previous) = previous_activity_digest.replace(activity_digest) {
|
||||
assert_ne!(previous, activity_digest, "draining a namespace commit must change the publication proof");
|
||||
}
|
||||
let ctx = CancellationToken::new();
|
||||
let budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
|
||||
let (updates, mut receiver) = mpsc::channel(1);
|
||||
let result = tokio::time::timeout(
|
||||
Duration::from_secs(30),
|
||||
ScannerIOCycle::nsscanner_with_status(
|
||||
store.as_ref(),
|
||||
ctx,
|
||||
Arc::clone(&budget),
|
||||
updates,
|
||||
cycle,
|
||||
1,
|
||||
HealScanMode::Normal,
|
||||
),
|
||||
)
|
||||
.await
|
||||
.expect("namespace scanning must finish while a PUT commit is pending")
|
||||
.expect("namespace scanning must remain available during a pending PUT commit");
|
||||
assert_eq!(result.activity_digest(), Some(activity_digest));
|
||||
if !converged {
|
||||
assert_eq!(budget.progress().0, 2, "the pending commit must not suppress actual object traversal");
|
||||
}
|
||||
assert_eq!(
|
||||
result.status,
|
||||
if converged {
|
||||
ScannerCycleStatus::Complete
|
||||
} else {
|
||||
ScannerCycleStatus::Superseded
|
||||
}
|
||||
);
|
||||
let usage = receiver
|
||||
.recv()
|
||||
.await
|
||||
.expect("the completed walk should produce a usage candidate");
|
||||
assert_eq!(usage.usage_snapshot_converged, Some(converged));
|
||||
assert_eq!(usage.scanner_cycle, Some(cycle));
|
||||
assert_eq!(usage.objects_total_count, 2);
|
||||
assert_eq!(usage.objects_total_size, 11);
|
||||
assert_eq!(usage.usage_snapshot_set_states.len(), 2);
|
||||
for state in &usage.usage_snapshot_set_states {
|
||||
let digest = state
|
||||
.scan_plan_digest
|
||||
.expect("each set must retain its structural cache identity");
|
||||
assert_eq!(*structural_plan_digest.get_or_insert(digest), digest);
|
||||
}
|
||||
let bucket_usage = usage.buckets_usage.get(&bucket).expect("the walked bucket must be present");
|
||||
assert_eq!(bucket_usage.objects_count, 2);
|
||||
assert_eq!(bucket_usage.size, 11);
|
||||
assert!(receiver.recv().await.is_none(), "each walk must emit exactly one terminal candidate");
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn multi_pool_scanner_cycle_zero_fills_bucket_absent_from_first_pool() {
|
||||
@@ -387,6 +494,16 @@ async fn multi_pool_scanner_cycle_zero_fills_bucket_absent_from_first_pool() {
|
||||
.put_object(&bucket, "pool-b", &mut reader, &ScannerObjectOptions::default())
|
||||
.await
|
||||
.expect("object should be written only to the second pool");
|
||||
{
|
||||
let lock = store.pools[1].disk_set[0]
|
||||
.new_ns_lock(&bucket, "pool-b")
|
||||
.await
|
||||
.expect("fixture namespace lock should be created");
|
||||
let _settled = lock
|
||||
.get_write_lock(Duration::from_secs(30))
|
||||
.await
|
||||
.expect("fixture rename tail should finish before the usage scan");
|
||||
}
|
||||
store.pools[0]
|
||||
.delete_bucket(&bucket, &DeleteBucketOptions::default())
|
||||
.await
|
||||
@@ -797,6 +914,124 @@ fn complete_set_usage_cache(buckets: &[(&str, usize)], scan_plan_digest: DataUsa
|
||||
cache
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn set_snapshot_reuse_requires_execution_identity_and_fences_stale_writers() {
|
||||
let (_temp_dir, store) = setup_two_pool_scanner_store().await;
|
||||
let set = Arc::clone(&store.pools[0].disk_set[0]);
|
||||
let epoch = scanner_publication_epoch(Arc::clone(&set)).await.expect("idle set admission");
|
||||
let mut legacy = complete_set_usage_cache(&[("photos", 5)], DataUsageScanPlanDigest([1; 32]));
|
||||
legacy.info.source = Some(DataUsageCacheSource::new(0, 0));
|
||||
legacy
|
||||
.save(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
|
||||
.await
|
||||
.expect("seed legacy set cache");
|
||||
let mut persisted = DataUsageCache::default();
|
||||
let initial = persisted
|
||||
.load_with_revisions(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
|
||||
.await
|
||||
.expect("capture the shared starting revision");
|
||||
let mut fresh = legacy.clone();
|
||||
fresh.info.scan_execution_digest = Some(DataUsageScanPlanDigest([2; 32]));
|
||||
fresh.replace(
|
||||
"photos",
|
||||
DATA_USAGE_ROOT,
|
||||
DataUsageEntry {
|
||||
size: 20,
|
||||
objects: 1,
|
||||
..Default::default()
|
||||
},
|
||||
);
|
||||
let cycle_floor = AtomicU64::new(fresh.info.next_cycle);
|
||||
let (tx, mut rx) = mpsc::channel(1);
|
||||
assert!(
|
||||
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh.clone(), Some(&initial), &cycle_floor, epoch)
|
||||
.await
|
||||
.is_some(),
|
||||
"a legacy cache without execution identity must be refreshed"
|
||||
);
|
||||
let published = rx.try_recv().expect("fresh snapshot should be forwarded");
|
||||
assert_eq!(published.find("photos").expect("published bucket").size, 20);
|
||||
assert_eq!(published.info.scan_execution_digest, fresh.info.scan_execution_digest);
|
||||
let current = persisted
|
||||
.load_with_revisions(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
|
||||
.await
|
||||
.expect("capture the current revision for the unidentified execution");
|
||||
|
||||
let mut stale = legacy.clone();
|
||||
stale.info.scan_execution_digest = Some(DataUsageScanPlanDigest([3; 32]));
|
||||
for (candidate, revisions) in [(stale, &initial), (legacy, ¤t)] {
|
||||
assert!(
|
||||
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, candidate, Some(revisions), &cycle_floor, epoch)
|
||||
.await
|
||||
.is_none(),
|
||||
"a stale or unidentified execution must not replace the newer snapshot"
|
||||
);
|
||||
assert!(matches!(rx.try_recv(), Err(mpsc::error::TryRecvError::Empty)));
|
||||
}
|
||||
fresh.info.scan_execution_digest = Some(DataUsageScanPlanDigest([4; 32]));
|
||||
assert!(
|
||||
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh.clone(), None, &cycle_floor, epoch)
|
||||
.await
|
||||
.is_none(),
|
||||
"an unreadable starting revision must not authorize an overwrite"
|
||||
);
|
||||
|
||||
fresh.info.scan_execution_digest = published.info.scan_execution_digest;
|
||||
fresh.replace("photos", DATA_USAGE_ROOT, DataUsageEntry::default());
|
||||
assert!(
|
||||
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh, Some(&initial), &cycle_floor, epoch)
|
||||
.await
|
||||
.is_some(),
|
||||
"an overlapping identical execution must reuse the completed snapshot"
|
||||
);
|
||||
assert_eq!(
|
||||
rx.try_recv()
|
||||
.expect("reused snapshot")
|
||||
.find("photos")
|
||||
.expect("reused bucket")
|
||||
.size,
|
||||
20
|
||||
);
|
||||
persisted
|
||||
.load(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
|
||||
.await
|
||||
.expect("read the final durable set cache");
|
||||
assert_eq!(persisted.find("photos").expect("durable bucket").size, 20);
|
||||
assert_eq!(persisted.info.scan_execution_digest, published.info.scan_execution_digest);
|
||||
|
||||
let ctx = CancellationToken::new();
|
||||
let empty_execution = DataUsageScanPlanDigest([5; 32]);
|
||||
set.nsscanner_cache(
|
||||
ctx.clone(),
|
||||
ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default()),
|
||||
ScannerBucketScanPlan {
|
||||
buckets: Vec::new(),
|
||||
all_buckets: Arc::new(Vec::new()),
|
||||
scope: ScannerBucketScanScope::default(),
|
||||
digest: DataUsageScanPlanDigest([6; 32]),
|
||||
execution_digest: empty_execution,
|
||||
leader_epoch: 11,
|
||||
tier_registry_generation: 13,
|
||||
publication_epoch: Some(epoch),
|
||||
dirty_usage_buckets: Arc::new(HashMap::new()),
|
||||
bucket_failures: ScannerBucketFailureState::default(),
|
||||
pending_maintenance_work: Arc::new(AtomicBool::new(false)),
|
||||
cache_cycle_floor: Arc::new(AtomicU64::new(8)),
|
||||
},
|
||||
tx,
|
||||
8,
|
||||
HealScanMode::Normal,
|
||||
)
|
||||
.await
|
||||
.expect("empty set scope should replace its prior nonempty cache");
|
||||
let empty = rx.try_recv().expect("empty set snapshot should be published");
|
||||
assert_eq!(empty.info.scan_execution_digest, Some(empty_execution));
|
||||
assert!(empty.info.snapshot_complete);
|
||||
let root = empty.checked_flatten(DATA_USAGE_ROOT).expect("complete empty root");
|
||||
assert_eq!((root.size, root.objects), (0, 0));
|
||||
}
|
||||
|
||||
fn complete_usage_baseline(
|
||||
source: DataUsageCacheSource,
|
||||
scan_plan_digest: DataUsageScanPlanDigest,
|
||||
|
||||
Reference in New Issue
Block a user