fix(storage): harden ODM and scanner publication (#7187)

* fix(storage): harden ODM and scanner publication

* fix(app): simplify absent SSE configuration matching

* test(heal): settle PUT rename tails before disk-wipe fixtures

* fix(ecstore): remove duplicate local rename implementation

Keep the canonical commit module after concurrent storage changes merged.
The control-write and rollback changes are already present there.

Co-Authored-By: heihutu <heihutu@gmail.com>
Co-Authored-By: zhi22915 <qiuzgang@gmail.com>

* fix(ci): satisfy new clippy lints

* style(scanner): order merged test imports

* fix(scanner): invalidate bucket work after namespace completion

* fix(scanner): fence cached snapshots by scan execution

---------

Co-authored-by: houseme <housemecn@gmail.com>
Co-authored-by: heihutu <heihutu@gmail.com>
Co-authored-by: zhi22915 <qiuzgang@gmail.com>
This commit is contained in:
Zhengchao An
2026-09-05 21:47:12 +08:00
committed by GitHub
parent 447f3c704b
commit e2a921bc16
35 changed files with 2252 additions and 274 deletions
+10 -2
View File
@@ -196,7 +196,7 @@ pub(crate) async fn read_config_revision<S: ScannerObjectIO>(store: Arc<S>, path
}
}
#[derive(Clone, Debug)]
#[derive(Clone, Debug, PartialEq, Eq)]
pub(crate) struct DataUsageCacheRevisions {
main: DataUsageCacheRevision,
backup: Option<DataUsageCacheRevision>,
@@ -503,6 +503,10 @@ pub struct DataUsageCacheInfo {
pub lkg_leader_epoch: Option<u64>,
#[serde(default)]
pub lkg_scan_plan_digest: Option<DataUsageScanPlanDigest>,
/// Activity-sensitive identity for same-cycle set snapshot reuse. The
/// structural plan remains reusable across ordinary bucket writes.
#[serde(default)]
pub scan_execution_digest: Option<DataUsageScanPlanDigest>,
}
impl Serialize for DataUsageCacheInfo {
@@ -519,7 +523,8 @@ impl Serialize for DataUsageCacheInfo {
+ usize::from(self.lkg_next_cycle.is_some())
+ usize::from(self.lkg_last_update.is_some())
+ usize::from(self.lkg_leader_epoch.is_some())
+ usize::from(self.lkg_scan_plan_digest.is_some());
+ usize::from(self.lkg_scan_plan_digest.is_some())
+ usize::from(self.scan_execution_digest.is_some());
let mut state = serializer.serialize_map(Some(field_count))?;
state.serialize_entry("name", &self.name)?;
state.serialize_entry("next_cycle", &self.next_cycle)?;
@@ -558,6 +563,9 @@ impl Serialize for DataUsageCacheInfo {
if let Some(scan_plan_digest) = self.lkg_scan_plan_digest {
state.serialize_entry("lkg_scan_plan_digest", &scan_plan_digest)?;
}
if let Some(scan_execution_digest) = self.scan_execution_digest {
state.serialize_entry("scan_execution_digest", &scan_execution_digest)?;
}
state.end()
}
}
@@ -1067,6 +1067,7 @@ fn test_data_usage_cache_info_deserialize_defaults_scan_resume_after() {
assert!(decoded.source.is_none());
assert!(!decoded.snapshot_complete);
assert!(decoded.scan_plan_digest.is_none());
assert!(decoded.scan_execution_digest.is_none());
assert_eq!(decoded.cache_key_format, 0);
}
@@ -1109,6 +1110,7 @@ fn test_data_usage_cache_info_unmarshal_old_msgpack_defaults_scan_resume_after()
assert!(decoded.source.is_none());
assert!(!decoded.snapshot_complete);
assert!(decoded.scan_plan_digest.is_none());
assert!(decoded.scan_execution_digest.is_none());
assert_eq!(decoded.cache_key_format, 0);
}
@@ -1145,6 +1147,7 @@ fn test_new_data_usage_cache_msgpack_round_trips_and_supports_old_reader() {
source: Some(DataUsageCacheSource::new(1, 2)),
snapshot_complete: true,
scan_plan_digest: Some(TEST_PLAN_DIGEST),
scan_execution_digest: Some(DataUsageScanPlanDigest([42; 32])),
cache_key_format: DATA_USAGE_CACHE_KEY_FORMAT,
..Default::default()
},
@@ -1164,6 +1167,7 @@ fn test_new_data_usage_cache_msgpack_round_trips_and_supports_old_reader() {
assert_eq!(current.info.source, Some(DataUsageCacheSource::new(1, 2)));
assert!(current.info.snapshot_complete);
assert_eq!(current.info.scan_plan_digest, Some(TEST_PLAN_DIGEST));
assert_eq!(current.info.scan_execution_digest, Some(DataUsageScanPlanDigest([42; 32])));
assert_eq!(current.info.cache_key_format, DATA_USAGE_CACHE_KEY_FORMAT);
assert_eq!(current.find("bucket").map(|entry| entry.objects), Some(3));
+31 -5
View File
@@ -1616,7 +1616,7 @@ where
// Refresh the storage-owned movement snapshot before reading background
// heal state. A missing heal object yields an in-memory default; do not
// let that default influence a cycle while publication is blocked.
if storeapi.scanner_data_usage_publication_blocked().await {
if storeapi.scanner_data_movement_pause_status().await.paused {
mark_scan_cycle_idle(cycle_info, &mut cycle_metrics_guard).await;
return ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement);
}
@@ -1816,6 +1816,19 @@ where
let publication_defer_reason = publication_defer_reason
.or(remote_lease_defer_reason)
.or(remote_lease_fence_defer_reason);
// A PUT tail can finish between the walk and lease acquisition without
// changing the movement epoch accepted by those leases. Re-prove the
// namespace baseline only after every peer has granted publication.
let post_lease_activity_defer_reason = if publication_defer_reason.is_none()
&& remote_publication_leases.is_some()
&& let Ok(result) = &scan_result
&& result.status == ScannerCycleStatus::Complete
{
scanner_post_lease_activity_defer_reason(result.activity_digest(), probe_scanner_activity(storeapi.as_ref(), true).await)
} else {
None
};
let publication_defer_reason = publication_defer_reason.or(post_lease_activity_defer_reason);
// Include reasons discovered while acquiring or validating remote leases.
let publication_deferred = publication_defer_reason.is_some();
let budget_elapsed = cycle_budget.budget_elapsed() && !ctx.is_cancelled();
@@ -3240,6 +3253,21 @@ where
}
}
fn scanner_post_lease_activity_defer_reason(
expected_digest: Option<[u8; 32]>,
activity: Result<ScannerActivitySnapshot, String>,
) -> Option<ScannerCycleDeferReason> {
match activity {
Ok(snapshot)
if scanner_activity_allows_usage_publication(&snapshot)
&& expected_digest == Some(scanner_activity_snapshot_digest(&snapshot)) =>
{
None
}
Ok(_) | Err(_) => Some(ScannerCycleDeferReason::ActivityBaselineUnavailable),
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum ScannerCyclePreCommitOutcome {
RecoverCacheCycle(u64),
@@ -3428,13 +3456,11 @@ use cycle_state::*;
use leadership::*;
use usage_store::*;
#[cfg(test)]
pub(crate) use activity::scanner_activity_snapshot_digest;
pub use activity::scanner_topology_digest;
pub(crate) use activity::{
ScannerActivitySnapshot, ScannerDirtyUsageAcknowledgement, probe_scanner_activity, scanner_activity_allows_usage_publication,
scanner_activity_dirty_usage_state_for_host, scanner_activity_publication_lease_targets, scanner_activity_structural_digest,
scanner_dirty_usage_acknowledgements,
scanner_activity_dirty_usage_state_for_host, scanner_activity_publication_lease_targets, scanner_activity_snapshot_digest,
scanner_activity_structural_digest, scanner_dirty_usage_acknowledgements,
};
pub(crate) use activity::{ScannerCycleOutcome, scanner_cycle_outcome_with_pending_maintenance};
pub use backlog::{
-1
View File
@@ -902,7 +902,6 @@ where
observation
}
#[cfg(test)]
pub(crate) fn scanner_activity_snapshot_digest(snapshot: &ScannerActivitySnapshot) -> [u8; 32] {
let mut hasher = Sha256::new();
hasher.update(u64::try_from(snapshot.len()).unwrap_or(u64::MAX).to_be_bytes());
+172 -1
View File
@@ -15,7 +15,8 @@
use super::heal_info::{classify_background_heal_read_error, decode_background_heal_info};
use super::*;
use crate::EcstoreResult;
use crate::storage_api::scan::BucketOperations as _;
use crate::storage_api::owner::ecstore_hold_namespace_commit;
use crate::storage_api::scan::{BucketOperations as _, ObjectIO as _};
use crate::{
DATA_USAGE_BLOOM_RECOVERY_PATH, DATA_USAGE_CACHE_KEY_FORMAT, DATA_USAGE_CACHE_NAME, DATA_USAGE_ROOT,
DataUsageCachePrepareOutcome, DataUsageCacheSource, DataUsageEntry, DataUsageScanPlanDigest, Endpoint, EndpointServerPools,
@@ -1165,6 +1166,116 @@ async fn run_data_scanner_cycle_publishes_activity_for_owner_lifetime() {
global_metrics().set_cycle(None).await;
}
#[tokio::test]
#[serial]
async fn coordinator_walks_during_pending_put_without_persisting_or_acknowledging_usage() {
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
let (_temp_dir, store) = setup_scanner_cycle_store().await;
let bucket = format!("scanner-coordinator-pending-{}", Uuid::new_v4().simple());
store
.make_bucket(&bucket, &crate::storage_api::scan::MakeBucketOptions::default())
.await
.expect("fixture bucket should be created");
let mut reader = PutObjReader::from_vec(b"first".to_vec());
store.pools[0].disk_set[0]
.put_object(
&bucket,
"object",
&mut reader,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("fixture object should finish its rename fanout");
crate::scanner_io::record_dirty_usage_bucket(&bucket);
let dirty_before = crate::scanner_io::dirty_usage_buckets_for_tests();
let baseline = read_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("fixture usage baseline should be readable");
let pending = ecstore_hold_namespace_commit(store.as_ref());
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
let mut cycle_info = CurrentCycle {
next: 1,
..Default::default()
};
let mut revision = DataUsageCacheRevision::Missing;
let outcome = tokio::time::timeout(
Duration::from_secs(30),
run_data_scanner_cycle_with_budget(&ctx, &store, &mut cycle_info, &mut revision, 1, Arc::clone(&budget)),
)
.await
.expect("the coordinator must finish its namespace walk while a PUT is pending");
assert_eq!(budget.progress().0, 1, "the coordinator must reach actual object traversal");
assert_eq!(outcome, ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert_eq!(cycle_info.next, 1, "a rejected publication must not advance the cycle");
assert_eq!(revision, DataUsageCacheRevision::Missing);
assert_eq!(crate::scanner_io::dirty_usage_buckets_for_tests(), dirty_before);
assert_eq!(
read_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("the prior authoritative usage must remain readable"),
baseline,
"the pending candidate must not replace the authoritative baseline"
);
let committed_body = b"committed-after-walk";
let mut reader = PutObjReader::from_vec(committed_body.to_vec());
store.pools[0].disk_set[0]
.put_object(
&bucket,
"object",
&mut reader,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("the pending tail must change the physical object before it drains");
assert_eq!(crate::scanner_io::dirty_usage_buckets_for_tests(), dirty_before);
drop(pending);
let retry_budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
let outcome = tokio::time::timeout(
Duration::from_secs(30),
run_data_scanner_cycle_with_budget(&ctx, &store, &mut cycle_info, &mut revision, 1, Arc::clone(&retry_budget)),
)
.await
.expect("the same cycle must converge after the pending PUT drains");
assert_eq!(
retry_budget.progress().0,
1,
"the same-cycle retry must not reuse the pre-tail bucket cache"
);
assert!(matches!(
outcome,
ScannerCycleOutcome::Completed | ScannerCycleOutcome::CompletedWithPendingMaintenance
));
assert_eq!(cycle_info.next, 2);
assert!(!crate::scanner_io::dirty_usage_buckets_for_tests().contains_key(&bucket));
let usage = read_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("the converged usage should be persisted");
let usage: DataUsageInfo = serde_json::from_slice(&usage).expect("the persisted usage should decode");
assert_eq!(usage.usage_snapshot_converged, Some(true));
assert_eq!(usage.scanner_cycle, Some(1));
assert_eq!(usage.objects_total_count, 1);
assert_eq!(
usage.objects_total_size,
u64::try_from(committed_body.len()).expect("fixture body length")
);
let bucket_usage = usage
.buckets_usage
.get(&bucket)
.expect("the scanned bucket should be published");
assert_eq!(bucket_usage.objects_count, 1);
assert_eq!(bucket_usage.size, u64::try_from(committed_body.len()).expect("fixture body length"));
global_metrics().set_cycle(None).await;
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test]
#[serial]
async fn test_finalize_partial_scan_cycle_advances_and_persists_counter() {
@@ -8485,6 +8596,66 @@ fn scanner_node_activity(epoch: &str, namespace_generation: u64, maintenance_gen
}
}
#[test]
fn post_lease_activity_proof_rejects_a_put_tail_that_finished_before_lease_acquisition() {
let before = BTreeMap::from([("node-2".to_string(), scanner_node_activity("epoch-a", 7, 3))]);
let expected_digest = Some(scanner_activity_snapshot_digest(&before));
assert_eq!(scanner_post_lease_activity_defer_reason(expected_digest, Ok(before.clone())), None);
let mut after = before.clone();
after
.get_mut("node-2")
.expect("writer should be present")
.namespace_generation += 1;
assert_eq!(
before["node-2"].movement_generation, after["node-2"].movement_generation,
"the existing movement-only lease remains valid after a PUT tail drains"
);
assert!(scanner_activity_allows_usage_publication(&after));
let reason = scanner_post_lease_activity_defer_reason(expected_digest, Ok(after));
assert_eq!(reason, Some(ScannerCycleDeferReason::ActivityBaselineUnavailable));
let result = ScannerCycleResult::new(ScannerCycleStatus::Complete, None).with_remote_dirty_usage_acknowledgements(vec![
ScannerDirtyUsageAcknowledgement {
host: "node-2".to_string(),
instance_id: "epoch-a".to_string(),
generation: 5,
},
]);
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(
result,
DataUsagePersistOutcome::Deferred(reason.expect("changed namespace should defer publication")),
);
assert_eq!(
outcome,
ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::ActivityBaselineUnavailable)
);
assert!(
acknowledgements.is_empty(),
"a rejected publication must not acknowledge the peer's dirty usage"
);
}
#[test]
fn post_lease_activity_proof_requires_a_complete_matching_baseline() {
let before = BTreeMap::from([("node-2".to_string(), scanner_node_activity("epoch-a", 7, 3))]);
let digest = scanner_activity_snapshot_digest(&before);
let mut blocked = before.clone();
blocked.get_mut("node-2").expect("peer should be present").publication_blocked = true;
let blocked_digest = scanner_activity_snapshot_digest(&blocked);
for (expected, observed) in [
(None, Ok(before)),
(Some(digest), Err("peer is unavailable".to_string())),
(Some(digest), Ok(BTreeMap::new())),
(Some(blocked_digest), Ok(blocked)),
] {
assert_eq!(
scanner_post_lease_activity_defer_reason(expected, observed),
Some(ScannerCycleDeferReason::ActivityBaselineUnavailable)
);
}
}
#[test]
fn scanner_activity_snapshot_digest_fences_storage_topology() {
let first = BTreeMap::from([("node-2".to_string(), scanner_node_activity("epoch-a", 7, 3))]);
+18 -2
View File
@@ -12,7 +12,7 @@
// See the License for the specific language governing permissions and
// limitations under the License.
use crate::data_usage_define::DATA_USAGE_CACHE_KEY_FORMAT;
use crate::data_usage_define::{DATA_USAGE_CACHE_KEY_FORMAT, DataUsageCacheRevisions};
use crate::scanner_budget::ScannerCycleBudget;
use crate::scanner_folder::{ScannerItem, scan_data_folder};
use crate::sleeper::SCANNER_SLEEPER;
@@ -271,6 +271,8 @@ pub struct ScannerBucketScanPlan {
all_buckets: Arc<Vec<BucketInfo>>,
scope: ScannerBucketScanScope,
digest: DataUsageScanPlanDigest,
// Cache work must invalidate on namespace completion even when its scoped baseline remains reusable.
execution_digest: DataUsageScanPlanDigest,
leader_epoch: u64,
tier_registry_generation: u64,
/// Epoch captured once for the whole scanner cycle. `None` is retained
@@ -456,9 +458,12 @@ async fn scanner_cycle_activity_status<S>(
where
S: ScannerStorage,
{
// Read the pending-commit barrier before sampling its completion generation.
// A tail that drains during this await must invalidate the earlier baseline.
let publication_blocked = store.scanner_data_usage_publication_blocked().await;
match crate::scanner::probe_scanner_activity(store, distributed).await {
Ok(after) => {
let status = if after == *before {
let status = if !publication_blocked && after == *before {
ScannerCycleActivityStatus::Unchanged
} else {
ScannerCycleActivityStatus::Changed
@@ -760,6 +765,7 @@ fn scanner_activity_preflight(
pub(crate) struct ScannerCycleResult {
pub(crate) status: ScannerCycleStatus,
publication_epoch: Option<u64>,
activity_digest: Option<[u8; 32]>,
observational_snapshot_published: bool,
dirty_usage_clear: Option<DirtyUsageBuckets>,
remote_dirty_usage_acknowledgements: Vec<crate::scanner::ScannerDirtyUsageAcknowledgement>,
@@ -774,6 +780,7 @@ impl ScannerCycleResult {
Self {
status,
publication_epoch: None,
activity_digest: None,
observational_snapshot_published: false,
dirty_usage_clear,
remote_dirty_usage_acknowledgements: Vec::new(),
@@ -793,6 +800,15 @@ impl ScannerCycleResult {
self.publication_epoch
}
fn with_activity_digest(mut self, activity_digest: [u8; 32]) -> Self {
self.activity_digest = Some(activity_digest);
self
}
pub(crate) fn activity_digest(&self) -> Option<[u8; 32]> {
self.activity_digest
}
pub(crate) fn with_observational_snapshot_published(mut self, published: bool) -> Self {
self.observational_snapshot_published = published;
self
+30 -12
View File
@@ -604,10 +604,12 @@ pub(super) async fn persist_and_publish_cache_snapshot(
store: Arc<SetDisks>,
updates: &mpsc::Sender<DataUsageCache>,
mut cache_snapshot: DataUsageCache,
initial_revisions: Option<&DataUsageCacheRevisions>,
cache_cycle_floor: &AtomicU64,
expected_publication_epoch: u64,
) -> Option<SystemTime> {
let source = cache_snapshot.info.source?;
let execution_digest = cache_snapshot.info.scan_execution_digest?;
let guard = match acquire_scanner_cache_locks(store.as_ref(), DATA_USAGE_CACHE_NAME, source).await {
Ok(guard) => guard,
Err(err) => {
@@ -672,20 +674,36 @@ pub(super) async fn persist_and_publish_cache_snapshot(
);
return None;
}
if matches!(
current_cache_root_entry_with_generation(
&persisted,
DATA_USAGE_ROOT,
source,
cache_snapshot.info.next_cycle,
cache_snapshot.info.leader_epoch,
scan_plan_digest,
cache_snapshot.info.tier_registry_generation,
),
Ok(Some(_))
) {
if persisted.info.scan_execution_digest == Some(execution_digest)
&& matches!(
current_cache_root_entry_with_generation(
&persisted,
DATA_USAGE_ROOT,
source,
cache_snapshot.info.next_cycle,
cache_snapshot.info.leader_epoch,
scan_plan_digest,
cache_snapshot.info.tier_registry_generation,
),
Ok(Some(_))
)
{
cache_snapshot = persisted;
} else {
// A later execution may have completed while this scan was walking.
// Only replace the cache revision from which this scan started.
if initial_revisions != Some(&revisions) {
warn!(
target: "rustfs::scanner::io",
event = EVENT_SCANNER_CACHE_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_IO,
state = "scan_baseline_revision_changed",
cache_name = DATA_USAGE_CACHE_NAME,
"Scanner skipped set snapshot without an unchanged baseline revision"
);
return None;
}
if guard.is_lock_lost() {
error!(
target: "rustfs::scanner::io",
+24 -15
View File
@@ -118,6 +118,7 @@ impl ScannerIOCache for SetDisks {
all_buckets,
scope,
digest: scan_plan_digest,
execution_digest,
leader_epoch,
tier_registry_generation,
publication_epoch,
@@ -137,20 +138,24 @@ impl ScannerIOCache for SetDisks {
.ok_or_else(|| StorageError::other("scanner cache publication is blocked by data movement"))?,
};
let mut old_cache = DataUsageCache::default();
if let Err(e) = old_cache.load(self.clone(), DATA_USAGE_CACHE_NAME).await {
warn!(
target: "rustfs::scanner::io",
event = EVENT_SCANNER_CACHE_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_IO,
pool = self.pool_index,
set = self.set_index,
cache_name = DATA_USAGE_CACHE_NAME,
state = "old_cache_load_failed",
error = %e,
"Scanner old data usage cache load failed; rebuilding from bucket caches"
);
}
let initial_revisions = match old_cache.load_with_revisions(self.clone(), DATA_USAGE_CACHE_NAME).await {
Ok(revisions) => Some(revisions),
Err(e) => {
warn!(
target: "rustfs::scanner::io",
event = EVENT_SCANNER_CACHE_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_IO,
pool = self.pool_index,
set = self.set_index,
cache_name = DATA_USAGE_CACHE_NAME,
state = "old_cache_load_failed",
error = %e,
"Scanner old data usage cache load failed; rebuilding from bucket caches"
);
None
}
};
let scoped_scan = prepare_scoped_set_scan(
&old_cache,
&buckets,
@@ -195,6 +200,7 @@ impl ScannerIOCache for SetDisks {
};
cache.info.last_update = Some(now);
cache.info.snapshot_complete = true;
cache.info.scan_execution_digest = Some(execution_digest);
cache.info.lkg_snapshot_complete = false;
cache.info.lkg_next_cycle = None;
cache.info.lkg_last_update = None;
@@ -208,6 +214,7 @@ impl ScannerIOCache for SetDisks {
self,
&updates,
cache,
initial_revisions.as_ref(),
cache_cycle_floor.as_ref(),
expected_publication_epoch,
)
@@ -637,7 +644,7 @@ impl ScannerIOCache for SetDisks {
let cache_name = path_join_buf(&[&bucket.name, DATA_USAGE_CACHE_NAME]);
let bucket_scan_plan_digest =
scanner_bucket_cache_digest(scan_plan_digest, dirty_usage_buckets_clone.get(&bucket.name).copied());
scanner_bucket_cache_digest(execution_digest, dirty_usage_buckets_clone.get(&bucket.name).copied());
if let Some(server_epoch) = remote_server_epoch {
let request_sequence = remote_session_sequence;
@@ -1360,6 +1367,7 @@ impl ScannerIOCache for SetDisks {
cache.info.next_cycle = want_cycle;
cache.info.last_update.get_or_insert_with(SystemTime::now);
cache.info.snapshot_complete = true;
cache.info.scan_execution_digest = Some(execution_digest);
cache.info.lkg_snapshot_complete = false;
cache.info.lkg_next_cycle = None;
cache.info.lkg_last_update = None;
@@ -1371,6 +1379,7 @@ impl ScannerIOCache for SetDisks {
self.clone(),
&updates,
cache_snapshot,
initial_revisions.as_ref(),
cache_cycle_floor.as_ref(),
expected_publication_epoch,
)
+9 -1
View File
@@ -180,7 +180,7 @@ where
// canceled decommission remains suspended after its worker exits, so
// starting a scan in that state could build a snapshot that cannot be
// routed to the authoritative metadata object.
if store.scanner_data_usage_publication_blocked().await {
if store.scanner_data_movement_pause_status().await.paused {
debug!(
target: "rustfs::scanner::io",
event = EVENT_SCANNER_SET_STATE,
@@ -260,8 +260,13 @@ where
}
}
bucket_plan_complete &= buckets_by_source.keys().copied().collect::<HashSet<_>>() == *expected_sources;
let activity_digest = crate::scanner::scanner_activity_snapshot_digest(&activity_before);
let scan_plan_digest =
scanner_bucket_plan_digest(&all_buckets, crate::scanner::scanner_activity_structural_digest(&activity_before));
let mut execution_hasher = Sha256::new();
execution_hasher.update(scan_plan_digest.0);
execution_hasher.update(activity_digest);
let execution_digest = DataUsageScanPlanDigest(execution_hasher.finalize().into());
let dirty_usage_snapshot = Arc::new(snapshot_dirty_usage_buckets(&all_buckets, dirty_generation_before_bucket_list));
let scan_scope = resolve_scanner_bucket_scan_scope(
store,
@@ -326,6 +331,7 @@ where
};
return Ok(ScannerCycleResult::new(status, dirty_usage_clear)
.with_publication_epoch(publication_epoch)
.with_activity_digest(activity_digest)
.with_observational_snapshot_published(observational_snapshot_published)
.with_remote_publication_lease_targets(remote_publication_lease_targets)
.with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements));
@@ -410,6 +416,7 @@ where
all_buckets: Arc::clone(&all_buckets),
scope: scan_scope.clone(),
digest: scan_plan_digest,
execution_digest,
leader_epoch,
tier_registry_generation,
publication_epoch,
@@ -598,6 +605,7 @@ where
};
Ok(ScannerCycleResult::new(cycle_status, dirty_usage_clear)
.with_publication_epoch(publication_epoch)
.with_activity_digest(activity_digest)
.with_observational_snapshot_published(observational_snapshot_published)
.with_remote_publication_lease_targets(remote_publication_lease_targets)
.with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements)
+236 -1
View File
@@ -20,6 +20,7 @@ use crate::scanner_folder::ScannerItem;
use crate::storage_api::EcstoreScannerPeerDirtyUsageSnapshot;
use crate::storage_api::owner::{
EcstorePoolDecommissionInfo, EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta, EcstoreRebalanceStats,
ecstore_hold_namespace_commit,
};
use crate::storage_api::scan::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions, ObjectIO as _};
use crate::{
@@ -343,6 +344,16 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
.put_object(&bucket, object, &mut reader, &ScannerObjectOptions::default())
.await
.expect("object should be written to its selected pool");
// Quorum ACK can precede tail publication on the disk chosen to scan.
let lock = store.pools[pool_index].disk_set[0]
.new_ns_lock(&bucket, object)
.await
.expect("fixture namespace lock should be created");
let _settled = lock
.get_write_lock(Duration::from_secs(30))
.await
.expect("fixture rename tail should finish before the usage scan");
}
let ctx = CancellationToken::new();
@@ -362,7 +373,7 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
.buckets_usage
.get(&bucket)
.expect("combined bucket usage should be present");
assert_eq!(bucket_usage.objects_count, 2);
assert_eq!(bucket_usage.objects_count, 2, "{usage:?}");
assert_eq!(bucket_usage.size, 11);
assert_eq!(usage.objects_total_count, 2);
assert_eq!(usage.objects_total_size, 11);
@@ -372,6 +383,102 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
);
}
#[tokio::test]
#[serial]
async fn pending_put_commit_keeps_scanner_walk_live_without_authoritative_usage() {
let (_temp_dir, store) = setup_two_pool_scanner_store().await;
let bucket = format!("scanner-pending-put-{}", Uuid::new_v4().simple());
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created across both pools");
for (pool_index, (object, body)) in [("pool-a", b"first".as_slice()), ("pool-b", b"second".as_slice())]
.into_iter()
.enumerate()
{
let mut reader = ScannerPutObjReader::from_vec(body.to_vec());
store.pools[pool_index].disk_set[0]
.put_object(
&bucket,
object,
&mut reader,
&ScannerObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("fixture objects must finish their rename fanouts before scanning");
}
let mut pending = Some(ecstore_hold_namespace_commit(store.as_ref()));
let mut previous_activity_digest = None;
let mut structural_plan_digest = None;
for (cycle, converged) in [(1, false), (2, true)] {
if converged {
drop(pending.take());
}
assert_eq!(store.scanner_data_usage_publication_blocked().await, !converged);
assert!(!store.scanner_data_movement_pause_status().await.paused);
let activity = crate::scanner::probe_scanner_activity(store.as_ref(), false)
.await
.expect("the fixture activity should be observable");
let activity_digest = crate::scanner::scanner_activity_snapshot_digest(&activity);
if let Some(previous) = previous_activity_digest.replace(activity_digest) {
assert_ne!(previous, activity_digest, "draining a namespace commit must change the publication proof");
}
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
let (updates, mut receiver) = mpsc::channel(1);
let result = tokio::time::timeout(
Duration::from_secs(30),
ScannerIOCycle::nsscanner_with_status(
store.as_ref(),
ctx,
Arc::clone(&budget),
updates,
cycle,
1,
HealScanMode::Normal,
),
)
.await
.expect("namespace scanning must finish while a PUT commit is pending")
.expect("namespace scanning must remain available during a pending PUT commit");
assert_eq!(result.activity_digest(), Some(activity_digest));
if !converged {
assert_eq!(budget.progress().0, 2, "the pending commit must not suppress actual object traversal");
}
assert_eq!(
result.status,
if converged {
ScannerCycleStatus::Complete
} else {
ScannerCycleStatus::Superseded
}
);
let usage = receiver
.recv()
.await
.expect("the completed walk should produce a usage candidate");
assert_eq!(usage.usage_snapshot_converged, Some(converged));
assert_eq!(usage.scanner_cycle, Some(cycle));
assert_eq!(usage.objects_total_count, 2);
assert_eq!(usage.objects_total_size, 11);
assert_eq!(usage.usage_snapshot_set_states.len(), 2);
for state in &usage.usage_snapshot_set_states {
let digest = state
.scan_plan_digest
.expect("each set must retain its structural cache identity");
assert_eq!(*structural_plan_digest.get_or_insert(digest), digest);
}
let bucket_usage = usage.buckets_usage.get(&bucket).expect("the walked bucket must be present");
assert_eq!(bucket_usage.objects_count, 2);
assert_eq!(bucket_usage.size, 11);
assert!(receiver.recv().await.is_none(), "each walk must emit exactly one terminal candidate");
}
}
#[tokio::test]
#[serial]
async fn multi_pool_scanner_cycle_zero_fills_bucket_absent_from_first_pool() {
@@ -387,6 +494,16 @@ async fn multi_pool_scanner_cycle_zero_fills_bucket_absent_from_first_pool() {
.put_object(&bucket, "pool-b", &mut reader, &ScannerObjectOptions::default())
.await
.expect("object should be written only to the second pool");
{
let lock = store.pools[1].disk_set[0]
.new_ns_lock(&bucket, "pool-b")
.await
.expect("fixture namespace lock should be created");
let _settled = lock
.get_write_lock(Duration::from_secs(30))
.await
.expect("fixture rename tail should finish before the usage scan");
}
store.pools[0]
.delete_bucket(&bucket, &DeleteBucketOptions::default())
.await
@@ -797,6 +914,124 @@ fn complete_set_usage_cache(buckets: &[(&str, usize)], scan_plan_digest: DataUsa
cache
}
#[tokio::test]
#[serial]
async fn set_snapshot_reuse_requires_execution_identity_and_fences_stale_writers() {
let (_temp_dir, store) = setup_two_pool_scanner_store().await;
let set = Arc::clone(&store.pools[0].disk_set[0]);
let epoch = scanner_publication_epoch(Arc::clone(&set)).await.expect("idle set admission");
let mut legacy = complete_set_usage_cache(&[("photos", 5)], DataUsageScanPlanDigest([1; 32]));
legacy.info.source = Some(DataUsageCacheSource::new(0, 0));
legacy
.save(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("seed legacy set cache");
let mut persisted = DataUsageCache::default();
let initial = persisted
.load_with_revisions(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("capture the shared starting revision");
let mut fresh = legacy.clone();
fresh.info.scan_execution_digest = Some(DataUsageScanPlanDigest([2; 32]));
fresh.replace(
"photos",
DATA_USAGE_ROOT,
DataUsageEntry {
size: 20,
objects: 1,
..Default::default()
},
);
let cycle_floor = AtomicU64::new(fresh.info.next_cycle);
let (tx, mut rx) = mpsc::channel(1);
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh.clone(), Some(&initial), &cycle_floor, epoch)
.await
.is_some(),
"a legacy cache without execution identity must be refreshed"
);
let published = rx.try_recv().expect("fresh snapshot should be forwarded");
assert_eq!(published.find("photos").expect("published bucket").size, 20);
assert_eq!(published.info.scan_execution_digest, fresh.info.scan_execution_digest);
let current = persisted
.load_with_revisions(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("capture the current revision for the unidentified execution");
let mut stale = legacy.clone();
stale.info.scan_execution_digest = Some(DataUsageScanPlanDigest([3; 32]));
for (candidate, revisions) in [(stale, &initial), (legacy, &current)] {
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, candidate, Some(revisions), &cycle_floor, epoch)
.await
.is_none(),
"a stale or unidentified execution must not replace the newer snapshot"
);
assert!(matches!(rx.try_recv(), Err(mpsc::error::TryRecvError::Empty)));
}
fresh.info.scan_execution_digest = Some(DataUsageScanPlanDigest([4; 32]));
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh.clone(), None, &cycle_floor, epoch)
.await
.is_none(),
"an unreadable starting revision must not authorize an overwrite"
);
fresh.info.scan_execution_digest = published.info.scan_execution_digest;
fresh.replace("photos", DATA_USAGE_ROOT, DataUsageEntry::default());
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh, Some(&initial), &cycle_floor, epoch)
.await
.is_some(),
"an overlapping identical execution must reuse the completed snapshot"
);
assert_eq!(
rx.try_recv()
.expect("reused snapshot")
.find("photos")
.expect("reused bucket")
.size,
20
);
persisted
.load(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("read the final durable set cache");
assert_eq!(persisted.find("photos").expect("durable bucket").size, 20);
assert_eq!(persisted.info.scan_execution_digest, published.info.scan_execution_digest);
let ctx = CancellationToken::new();
let empty_execution = DataUsageScanPlanDigest([5; 32]);
set.nsscanner_cache(
ctx.clone(),
ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default()),
ScannerBucketScanPlan {
buckets: Vec::new(),
all_buckets: Arc::new(Vec::new()),
scope: ScannerBucketScanScope::default(),
digest: DataUsageScanPlanDigest([6; 32]),
execution_digest: empty_execution,
leader_epoch: 11,
tier_registry_generation: 13,
publication_epoch: Some(epoch),
dirty_usage_buckets: Arc::new(HashMap::new()),
bucket_failures: ScannerBucketFailureState::default(),
pending_maintenance_work: Arc::new(AtomicBool::new(false)),
cache_cycle_floor: Arc::new(AtomicU64::new(8)),
},
tx,
8,
HealScanMode::Normal,
)
.await
.expect("empty set scope should replace its prior nonempty cache");
let empty = rx.try_recv().expect("empty set snapshot should be published");
assert_eq!(empty.info.scan_execution_digest, Some(empty_execution));
assert!(empty.info.snapshot_complete);
let root = empty.checked_flatten(DATA_USAGE_ROOT).expect("complete empty root");
assert_eq!((root.size, root.objects), (0, 0));
}
fn complete_usage_baseline(
source: DataUsageCacheSource,
scan_plan_digest: DataUsageScanPlanDigest,
+3
View File
@@ -127,6 +127,9 @@ pub(crate) use rustfs_lifecycle::{
use rustfs_storage_api as storage_contracts;
pub(crate) mod owner {
#[cfg(test)]
pub(crate) use rustfs_ecstore::api::set_disk::test_util::hold_namespace_commit as ecstore_hold_namespace_commit;
pub(crate) use super::storage_contracts::{
HTTPPreconditions, HTTPRangeSpec, NS_SCANNER_PROTOCOL_VERSION, ObjectIO, ObjectOperations, ObjectToDelete,
};