refactor(storage): separate metadata scheduling from quorum decisions (#8327)

This commit is contained in:
Chris
2026-10-04 01:04:37 +08:00
committed by GitHub
parent 309128f034
commit d087416ff9
9 changed files with 1827 additions and 1294 deletions
File diff suppressed because it is too large Load Diff
@@ -29,6 +29,27 @@ use crate::disk::error_reduce::OBJECT_OP_IGNORED_ERRS;
use crate::set_disk::file_info_is_valid_for_metadata;
use rustfs_filemeta::FileInfo;
/// One disk's observation. Pending is neither a successful vote nor an offline
/// disk: the scheduler may not have issued this slot or may still be waiting.
// Option<Result<...>> keeps FileInfo inline without a per-response Box.
// None is pending; Some(Ok(_)) is success; Some(Err(_)) retains a disk error.
pub(in crate::set_disk) type MetadataDiskResult = Option<crate::disk::error::Result<FileInfo>>;
#[derive(Debug)]
pub(in crate::set_disk) struct MetadataDiskObservation {
pub(in crate::set_disk) disk_index: usize,
pub(in crate::set_disk) result: MetadataDiskResult,
}
impl MetadataDiskObservation {
pub(in crate::set_disk) fn file_info(&self) -> Option<&FileInfo> {
match &self.result {
Some(Ok(metadata)) => Some(metadata),
None | Some(Err(_)) => None,
}
}
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub(in crate::set_disk) struct MetadataEarlyStopDecision {
pub(in crate::set_disk) reason: &'static str,
@@ -59,6 +80,14 @@ pub(in crate::set_disk) struct MetadataQuorumAccumulator {
}
impl MetadataQuorumAccumulator {
pub(in crate::set_disk) fn observe(&mut self, observation: &MetadataDiskObservation) {
match &observation.result {
None => {}
Some(Ok(metadata)) => self.observe_file_info_at(observation.disk_index, metadata),
Some(Err(error)) => self.observe_error(error),
}
}
pub(in crate::set_disk) fn new(total_disks: usize, default_parity_count: usize, allow_early_stop: bool) -> Self {
Self {
total_disks,
@@ -86,6 +115,7 @@ impl MetadataQuorumAccumulator {
self
}
#[cfg(test)]
pub(in crate::set_disk) fn observe_file_info(&mut self, file_info: &FileInfo) {
self.observe_file_info_with_index(None, file_info);
}
@@ -383,3 +413,186 @@ pub(in crate::set_disk) fn metadata_early_stop_candidate_matches(left: &FileInfo
pub(in crate::set_disk) fn is_metadata_fanout_ignored_error(err: &DiskError) -> bool {
OBJECT_OP_IGNORED_ERRS.iter().any(|ignored| ignored == err)
}
#[cfg(test)]
mod tests {
use super::*;
use crate::set_disk::SetDisks;
use time::OffsetDateTime;
use uuid::Uuid;
fn payload(disk_index: usize) -> FileInfo {
let mut metadata = FileInfo::new("object", 2, 2);
metadata.volume = "bucket".to_string();
metadata.name = "object".to_string();
metadata.size = 1;
metadata.mod_time = Some(OffsetDateTime::UNIX_EPOCH);
metadata.data_dir = Some(Uuid::from_u128(1));
metadata.erasure.index = metadata.erasure.distribution[disk_index];
metadata.metadata.insert("etag".to_string(), "etag".to_string());
metadata.add_object_part(1, "etag".to_string(), 1, None, 1, None, None);
assert!(file_info_is_valid_for_metadata(&metadata), "the observation corpus needs a valid payload");
metadata
}
fn observation(disk_index: usize, state: usize) -> MetadataDiskObservation {
MetadataDiskObservation {
disk_index,
result: match state {
0 => Some(Ok(payload(disk_index))),
1 => Some(Err(DiskError::FileNotFound)),
2 => Some(Err(DiskError::FileCorrupt)),
3 => Some(Err(DiskError::DiskNotFound)),
4 => None,
_ => panic!("unexpected test state"),
},
}
}
fn assert_same_reduction(typed: &MetadataQuorumAccumulator, legacy: &MetadataQuorumAccumulator) {
assert_eq!(typed.early_stop_decision(), legacy.early_stop_decision());
assert_eq!(typed.version_early_stop_decision(), legacy.version_early_stop_decision());
assert_eq!(typed.final_miss_reason(), legacy.final_miss_reason());
assert_eq!(typed.valid_responses, legacy.valid_responses);
assert_eq!(typed.not_found_responses, legacy.not_found_responses);
assert_eq!(typed.version_not_found_responses, legacy.version_not_found_responses);
assert_eq!(typed.ignored_errors, legacy.ignored_errors);
assert_eq!(typed.hard_errors, legacy.hard_errors);
assert_eq!(typed.candidate_votes, legacy.candidate_votes);
assert_eq!(typed.candidate_shard_mask, legacy.candidate_shard_mask);
assert_eq!(typed.matching_version_votes, legacy.matching_version_votes);
assert_eq!(typed.delete_marker_votes, legacy.delete_marker_votes);
assert_eq!(typed.conflicting_metadata, legacy.conflicting_metadata);
assert_eq!(
typed.candidate.as_ref().map(SetDisks::file_info_quorum_hash),
legacy.candidate.as_ref().map(SetDisks::file_info_quorum_hash)
);
}
fn observe_legacy(accumulator: &mut MetadataQuorumAccumulator, observation: &MetadataDiskObservation) {
match &observation.result {
None => {}
Some(Ok(metadata)) => accumulator.observe_file_info_at(observation.disk_index, metadata),
Some(Err(error)) => accumulator.observe_error(error),
}
}
#[test]
fn metadata_observation_all_four_slot_states_and_arrival_orders_preserve_reduction() {
let permutations = (0..4)
.flat_map(|a| (0..4).filter(move |b| *b != a).map(move |b| (a, b)))
.flat_map(|(a, b)| (0..4).filter(move |c| *c != a && *c != b).map(move |c| (a, b, c)))
.map(|(a, b, c)| [a, b, c, 6 - a - b - c])
.collect::<Vec<_>>();
assert_eq!(permutations.len(), 24);
// 5^4 states and all 4! arrivals, both with early-stop enabled and disabled.
for encoded in 0..625usize {
let states = [encoded % 5, encoded / 5 % 5, encoded / 25 % 5, encoded / 125 % 5];
let inputs = std::array::from_fn::<_, 4, _>(|index| observation(index, states[index]));
for enabled in [false, true] {
for order in &permutations {
let mut typed = MetadataQuorumAccumulator::new(4, 2, enabled);
let mut legacy = MetadataQuorumAccumulator::new(4, 2, enabled);
for &index in order {
typed.observe(&inputs[index]);
observe_legacy(&mut legacy, &inputs[index]);
assert_same_reduction(&typed, &legacy);
}
assert_eq!(typed.valid_responses, states.iter().filter(|&&state| state == 0).count());
assert_eq!(typed.not_found_responses, states.iter().filter(|&&state| state == 1).count());
assert_eq!(typed.hard_errors, states.iter().filter(|&&state| state == 2).count());
assert_eq!(typed.ignored_errors, states.iter().filter(|&&state| state == 3).count());
let expected_early_stop =
enabled && typed.valid_responses >= 3 && !states.iter().any(|&state| state == 1 || state == 2);
assert_eq!(
typed.early_stop_decision().is_some(),
expected_early_stop,
"states={states:?}, order={order:?}"
);
}
}
}
}
#[test]
fn metadata_observation_pending_newer_version_and_same_time_different_directory_force_full_wait() {
let mut accumulator = MetadataQuorumAccumulator::new(4, 2, true);
for index in 0..2 {
accumulator.observe(&observation(index, 0));
}
accumulator.observe(&observation(2, 4));
assert_eq!(accumulator.candidate_votes, 2, "an outstanding response cannot supply the deciding vote");
assert_eq!(accumulator.early_stop_decision(), None);
let mut newer = payload(2);
newer.mod_time = Some(OffsetDateTime::UNIX_EPOCH + time::Duration::seconds(1));
accumulator.observe(&MetadataDiskObservation {
disk_index: 2,
result: Some(Ok(newer)),
});
assert!(accumulator.conflicting_metadata);
assert_eq!(accumulator.early_stop_decision(), None);
let mut same_time_different_directory = payload(3);
same_time_different_directory.data_dir = Some(Uuid::from_u128(2));
accumulator.observe(&MetadataDiskObservation {
disk_index: 3,
result: Some(Ok(same_time_different_directory)),
});
assert_eq!(accumulator.early_stop_decision(), None);
}
#[test]
fn metadata_observation_duplicate_shard_cannot_supply_a_read_reserve() {
let mut accumulator = MetadataQuorumAccumulator::new(4, 2, true);
let first = payload(0);
for disk_index in 0..3 {
accumulator.observe(&MetadataDiskObservation {
disk_index,
result: Some(Ok(first.clone())),
});
}
assert_eq!(accumulator.candidate_votes, 3);
assert!(
!accumulator.candidate_has_read_reserve(),
"copied erasure indexes cannot supply independent data shards"
);
}
#[test]
fn metadata_observation_null_versions_markers_and_invalid_success_keep_their_meaning() {
let mut null_version = MetadataQuorumAccumulator::new(4, 2, true);
for index in 0..3 {
null_version.observe(&observation(index, 0));
}
assert!(null_version.early_stop_decision().is_some());
assert_eq!(null_version.matching_version_votes, 0);
let mut marker = MetadataQuorumAccumulator::new(4, 2, true);
for disk_index in 0..3 {
let metadata = FileInfo {
volume: "bucket".to_string(),
name: "object".to_string(),
deleted: true,
mod_time: Some(OffsetDateTime::UNIX_EPOCH + time::Duration::seconds(1)),
..Default::default()
};
assert!(metadata.is_canonical_delete_marker(), "the null marker fixture must reach marker voting");
marker.observe(&MetadataDiskObservation {
disk_index,
result: Some(Ok(metadata)),
});
}
assert_eq!(marker.delete_marker_votes, 3);
assert!(marker.early_stop_decision().is_some());
let mut invalid = MetadataQuorumAccumulator::new(4, 2, true);
invalid.observe(&MetadataDiskObservation {
disk_index: 0,
result: Some(Ok(FileInfo::default())),
});
assert_eq!(invalid.valid_responses, 0);
assert_eq!(invalid.hard_errors, 1);
assert_eq!(invalid.early_stop_decision(), None);
}
}
File diff suppressed because it is too large Load Diff
+2 -1
View File
@@ -18,4 +18,5 @@
//! duplicating read/write/erasure logic.
pub(crate) mod io_primitives;
mod metadata_quorum;
pub(in crate::set_disk) mod metadata_quorum;
pub(in crate::set_disk) mod metadata_read;
+20 -3
View File
@@ -5527,12 +5527,29 @@ fn collect_inline_data_shard_fileinfos_by_index_or_reason<'a>(
parts_metadata: &'a [FileInfo],
fi: &FileInfo,
data_shards: usize,
disk_is_online: impl FnMut(usize) -> bool,
) -> std::result::Result<Vec<&'a FileInfo>, &'static str> {
collect_inline_data_shard_fileinfos_from_observations(
parts_metadata
.iter()
.enumerate()
.map(|(index, metadata)| (index, Some(metadata))),
fi,
data_shards,
disk_is_online,
)
}
fn collect_inline_data_shard_fileinfos_from_observations<'a>(
observations: impl IntoIterator<Item = (usize, Option<&'a FileInfo>)>,
fi: &FileInfo,
data_shards: usize,
mut disk_is_online: impl FnMut(usize) -> bool,
) -> std::result::Result<Vec<&'a FileInfo>, &'static str> {
let distribution = &fi.erasure.distribution;
let mut data_files = vec![None; data_shards];
for (disk_index, file_info) in parts_metadata.iter().enumerate() {
for (disk_index, file_info) in observations {
if !disk_is_online(disk_index) {
continue;
}
@@ -5542,9 +5559,9 @@ fn collect_inline_data_shard_fileinfos_by_index_or_reason<'a>(
if block_index == 0 || block_index > data_shards {
continue;
}
if file_info.name.is_empty() {
let Some(file_info) = file_info.filter(|metadata| !metadata.name.is_empty()) else {
return Err(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_MISSING_SHARD);
}
};
if file_info.erasure.index != block_index {
return Err(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_IDENTITY_MISMATCH);
}
+26 -20
View File
@@ -598,17 +598,18 @@ impl SetDisks {
}
scope.check()?;
let (_, errors) =
Self::read_all_fileinfo(disks.as_slice(), "", bucket, object, &version.to_string(), false, false, false).await?;
let metadata_read =
Self::read_metadata(disks.as_slice(), "", bucket, object, &version.to_string(), false, false, false).await?;
let all_absent = metadata_read
.slots
.iter()
.all(|slot| matches!(&slot.result, Some(Err(DiskError::FileNotFound | DiskError::FileVersionNotFound))));
let selected = self.get_disks_internal().await;
let same_targets = selected.len() == disks.len() && selected.iter().zip(&disks).all(|(current, original)| {
matches!((current, original), (Some(current), Some(original)) if std::sync::Arc::ptr_eq(current, original))
});
if !same_targets
|| !errors
.iter()
.all(|error| matches!(error, Some(DiskError::FileNotFound | DiskError::FileVersionNotFound)))
{
if !same_targets || !all_absent {
return Err(StorageError::SlowDown);
}
scope.check()?;
@@ -734,7 +735,7 @@ impl SetDisks {
object: &str,
version_id: &str,
) -> disk::error::Result<ReadRepairCommitFingerprint> {
let (parts_metadata, errs, _) = Self::read_all_fileinfo_observed(
let metadata_read = Self::read_metadata_observed(
disks,
"",
bucket,
@@ -747,6 +748,7 @@ impl SetDisks {
self.default_parity_count,
)
.await?;
let (parts_metadata, errs, _) = metadata_read.into_legacy();
let (read_quorum, _) = Self::object_quorum_from_meta(&parts_metadata, &errs, self.default_parity_count)?;
let read_quorum = usize::try_from(read_quorum).map_err(|_| DiskError::ErasureReadQuorum)?;
let (_, quorum_mod_time, quorum_etag) = Self::list_online_disks(disks, &parts_metadata, &errs, read_quorum);
@@ -931,8 +933,9 @@ impl SetDisks {
}
};
let (mut parts_metadata, errs) =
Self::read_all_fileinfo(&disks, "", bucket, object, version_id, true, true, false).await?;
let (mut parts_metadata, errs, _) = Self::read_metadata(&disks, "", bucket, object, version_id, true, true, false)
.await?
.into_legacy();
trace!(
event = EVENT_SET_DISK_HEAL,
@@ -1284,8 +1287,10 @@ impl SetDisks {
// in `parts_metadata` to defaults. Re-read only before
// destructive cleanup so the guard sees every original
// identity.
let (delete_guard_metadata, delete_guard_errs) =
Self::read_all_fileinfo(&disks, "", bucket, object, version_id, true, true, false).await?;
let (delete_guard_metadata, delete_guard_errs, _) =
Self::read_metadata(&disks, "", bucket, object, version_id, true, true, false)
.await?
.into_legacy();
if self
.dangling_delete_safety(bucket, object, &delete_guard_metadata, &delete_guard_errs, &disks)
.await?
@@ -2316,16 +2321,16 @@ impl SetDisks {
return Err(DiskError::retired_marker_deferred(format!("conditional marker deletion failed: {error}")));
}
}
let (_, errors) = Self::read_all_fileinfo(disks, "", bucket, object, &version.to_string(), false, false, false).await?;
let metadata_read = Self::read_metadata(disks, "", bucket, object, &version.to_string(), false, false, false).await?;
let all_absent = metadata_read
.slots
.iter()
.all(|slot| matches!(&slot.result, Some(Err(DiskError::FileNotFound | DiskError::FileVersionNotFound))));
let selected = self.get_disks_internal().await;
let same_targets = selected.len() == disks.len() && selected.iter().zip(disks).all(|(current, original)| {
matches!((current, original), (Some(current), Some(original)) if std::sync::Arc::ptr_eq(current, original))
});
if !same_targets
|| !errors
.iter()
.all(|error| matches!(error, Some(DiskError::FileNotFound | DiskError::FileVersionNotFound)))
{
if !same_targets || !all_absent {
return Err(DiskError::retired_marker_deferred(
"complete marker absence could not be verified under the original target set",
));
@@ -3383,9 +3388,10 @@ impl SetDisks {
// The inner heal and missing-object report read the registry again;
// release this snapshot guard before a topology writer can queue between reads.
let disks = self.get_disks_internal().await;
let (_, errs) = Self::read_all_fileinfo(&disks, "", bucket, object, version_id, false, false, false)
let (_, errs, _) = Self::read_metadata(&disks, "", bucket, object, version_id, false, false, false)
.await
.map_err(|e| to_object_err(e.into(), vec![bucket, object]))?;
.map_err(|e| to_object_err(e.into(), vec![bucket, object]))?
.into_legacy();
if DiskError::is_all_not_found(&errs) {
debug!(
event = EVENT_SET_DISK_HEAL,
+31 -8
View File
@@ -113,7 +113,28 @@ use super::GetCodecStreamingObjectClass;
use super::GetCodecStreamingRollout;
#[cfg(test)]
use super::classify_get_codec_streaming_object_class;
use super::core::io_primitives::*;
use super::core::io_primitives::{
BitrotReaderSetup, BitrotReaderSetupAttribution, BitrotReaderSetupMode, DeferredReaderReopener, EVENT_SET_DISK_READ,
GetCodecStreamingReaderBuildOutcome, MetadataCacheLookup, ObjectBitrotReader, ReadRepairAdmissionSubmitter,
ReadRepairHealSubmission, SLOW_OBJECT_READ_LOG_THRESHOLD, codec_streaming_reader_setup_fallback_reason,
create_bitrot_readers_until_quorum_with_preference, create_data_block_bitrot_readers, resolved_read_repair_version_id,
send_read_repair_heal_request, shard_read_costs_for_disks, submit_read_repair_heal, submit_read_repair_heal_with_submitter,
};
#[cfg(test)]
use super::core::io_primitives::{
ENV_RUSTFS_GET_CODEC_STREAMING_DATA_BLOCKS_FIRST_READER_SETUP, ENV_RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP,
MultipartCodecStreamingReader, ReadRepairAdmissionFuture, ReadRepairAdmissionOutcome, collect_read_multiple_results,
collect_read_parts_results, create_bitrot_readers_until_quorum, create_bitrot_readers_until_quorum_all_shards,
release_read_repair_heal_reservation, reserve_read_repair_heal, resolve_read_part_from_responses, shard_read_cost_for_disk,
shard_read_cost_for_endpoint,
};
#[cfg(test)]
use super::core::metadata_quorum::{MetadataEarlyStopDecision, MetadataQuorumAccumulator};
#[cfg(test)]
use super::core::metadata_read::{
MetadataFanoutDiagnostics, MetadataFanoutObservation, metadata_early_stop_permitted, should_allow_metadata_early_stop,
};
use super::core::metadata_read::{late_materialization_candidate_is_safe, non_inline_data_read_early_stop_allowed};
#[cfg(test)]
use super::get_codec_streaming_config_cached_core;
#[cfg(test)]
@@ -607,11 +628,11 @@ impl SetDisks {
};
// Early-stop for safe metadata reads is handled inside
// read_all_fileinfo_observed (see read_all_fileinfo_early_stop in
// core/io_primitives.rs); unsafe requests and callers that opt out
// read_metadata_observed (see read_all_fileinfo_early_stop in
// core/metadata_read.rs); unsafe requests and callers that opt out
// (allow_early_stop=false) fall back to full-wait.
let (mut parts_metadata, errs, metadata_fanout_diagnostics) = if allow_read_version_coalescing {
Self::read_all_fileinfo_observed_for_get_object(
let metadata_read = if allow_read_version_coalescing {
Self::read_metadata_for_get_object(
&disks,
"",
bucket,
@@ -624,7 +645,7 @@ impl SetDisks {
)
.await?
} else {
Self::read_all_fileinfo_observed(
Self::read_metadata_observed(
&disks,
"",
bucket,
@@ -638,13 +659,14 @@ impl SetDisks {
)
.await?
};
let metadata_fanout_complete = metadata_read.is_complete();
let (mut parts_metadata, errs, metadata_fanout_diagnostics) = metadata_read.into_legacy();
let metadata_metrics_path = if crate::bucket::utils::is_meta_bucketname(bucket) {
GET_OBJECT_PATH_INTERNAL_META
} else {
GET_OBJECT_PATH_LEGACY_DUPLEX
};
metadata_fanout_diagnostics.record(metadata_metrics_path);
let metadata_fanout_complete = metadata_fanout_diagnostics.total_responses() >= disks.len();
// warn!("get_object_fileinfo parts_metadata {:?}", &parts_metadata);
// warn!("get_object_fileinfo {}/{} errs {:?}", bucket, object, &errs);
@@ -2117,7 +2139,7 @@ impl SetDisks {
expected: &LateMetadataIdentity,
metrics_path: &'static str,
) -> Result<(FileInfo, Vec<FileInfo>, Vec<Option<DiskStore>>)> {
let (mut parts_metadata, errs, diagnostics) = SetDisks::read_all_fileinfo_observed(
let metadata_read = SetDisks::read_metadata_observed(
fallback_disks,
"",
bucket,
@@ -2130,6 +2152,7 @@ impl SetDisks {
expected.parity_blocks,
)
.await?;
let (mut parts_metadata, errs, diagnostics) = metadata_read.into_legacy();
diagnostics.record(metrics_path);
let (read_quorum, write_quorum) = SetDisks::object_quorum_from_meta(&parts_metadata, &errs, expected.parity_blocks)
@@ -11,6 +11,8 @@
## Open Items
- `backlog-2253-metadata-observation-adapter` metadata observations: preserve aligned metadata/error slices for existing layout, shard and exact-version PUT, MPU and delete consumers while read/heal scheduling uses typed disk observations. Pending slots acquire empty placeholders only in this adapter. Remove after those remaining consumers use typed observations; delete the adapter and old facade in a separate cleanup.
- `connect-894` pending Connect heartbeats: replay the exact preceding producer-capability list, with or without jobs, when upgrading to the memory-service capability. Preserve request ID, sequence, and persisted body rather than inserting the new capability into a retry. Remove after upgrades from the pre-service-memory capability set are unsupported.
- `backlog-2539` administrator erasure-set scope: decode historical pending intents whose empty bucket list omitted the all-buckets marker. Normalize that representation to the explicit scope before replay or checkpoint comparison. Remove after all pre-marker administrator ErasureSet intents are retired; deployment verification must confirm that no such pending records remain on coordinator disks.
@@ -83,3 +83,39 @@ A candidate is ready for code movement only when all of these hold:
- old facade names have compatibility tests or explicit deprecation coverage;
- focused tests cover the changed owner path before any full gate is attempted;
- rollback preserves object IO, quorum, lifecycle/replication queues, scanner repair, notification/audit events, and metadata compatibility.
## Metadata read ownership and identity
`set_disk/core/metadata_read.rs` owns disk/RPC scheduling, coalescing, timeouts,
hedging and cancellation. Its result retains each disk slot as pending, a
successful `FileInfo`, or a typed `DiskError`. Pending includes both an
unscheduled disk and a response not yet observed; it contributes no vote and
does not establish absence. Task IDs retain the disk slot when a task panics.
The existing full-wait and early-stop gates, coalescer admission, cancellation
drain and fallback policy remain at this scheduling boundary.
`set_disk/core/metadata_quorum.rs` consumes those observations without IO.
`SetDisks` remains the sole owner of topology, locks and caches. Read and heal
callers use the typed result; `MetadataReadResult::into_legacy` is the explicit
adapter for existing quorum/layout and shard-reader helpers. It moves successful
metadata without cloning and creates empty placeholders only at that adapter.
Old full-wait facade callers retain the same aligned metadata/error slices.
These identities have different purposes and must not become one generic hash:
| Decision | Equal fields | Deliberately different or separately checked |
| --- | --- | --- |
| Quorum grouping (`file_info_quorum_hash`) | Size, deletion/restore/transition state, timestamps, version and data-directory UUIDs, checksums, mode/writer version, normalized metadata, target delete-marker versions and part identities. Payloads also include data/parity counts and distribution. | Object name and volume come from the request; per-disk erasure index and shard bytes are not an object identity. Canonical delete markers and zero-size objects do not add payload geometry. |
| Early-stop strict match (`metadata_early_stop_candidate_matches`) | Request name/volume, version/latest/deletion state, transition fields, size/time, mode/writer version, complete metadata/replication state/parts/checksum, version-history fields, data directory, algorithm/block size/data/parity/distribution. | Per-disk erasure index and data bytes differ. A non-inline read reserve additionally proves distinct indexes mapped to their disk slots; inline reads verify shard identity and bitrot/body content before stopping. |
| Late materialization (`LateMetadataIdentity`) | Request name/volume, algorithm, block size, legacy-checksum mode and quorum hash. | Each late shard must match the original distribution at its disk slot. A changed identity or insufficient matching shards fails read quorum; it cannot mix generations into a body. |
`metadata_observation_all_four_slot_states_and_arrival_orders_preserve_reduction`
checks all 625 four-disk success/missing/corrupt/offline/pending combinations and
24 arrival orders with early-stop on and off. The observation regressions also
cover a newer pending response, same-time different directories, duplicated
shard indexes, null versions, canonical delete markers and invalid success
payloads. Existing bounded-fanout, legacy inline bitrot, cache, late-metadata and
GET/Range regressions continue to exercise the real scheduling and reader paths;
the pure matrix does not replace those checks. Their call counters bound actual
disk reads, including canceled tails, without adding a scheduling policy or a
per-request metadata clone.