mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-23 04:39:04 +00:00
Compare commits
6 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c45933fa1b | |||
| b6ba89d9e4 | |||
| 648d5166e2 | |||
| 84eb5aebef | |||
| ce7277a334 | |||
| d5ba6b4e16 |
@@ -30,7 +30,8 @@ make build-docker BUILD_OS=ubuntu22.04
|
||||
- Crate membership: `Cargo.toml` `[workspace].members`
|
||||
- Architecture, layering, crate map: [ARCHITECTURE.md](ARCHITECTURE.md)
|
||||
- Migration guardrails & readiness contracts: [docs/architecture/](docs/architecture/README.md)
|
||||
- CI gates: `.github/workflows/ci.yml` (source of truth; never copy its steps into docs)
|
||||
- CI workflow steps: `.github/workflows/`; event, timeout, and required-status
|
||||
matrix: [docs/testing/ci-gates.md](docs/testing/ci-gates.md)
|
||||
- Test-layer taxonomy, per-layer entry commands, serial/nextest rules, flake
|
||||
policy: [docs/testing/README.md](docs/testing/README.md)
|
||||
- Tier/ILM transition debugging (xl.meta inspection, versionId tracing):
|
||||
|
||||
@@ -70,6 +70,8 @@ make pre-pr
|
||||
|
||||
> For the full test-layer taxonomy (unit / ecstore black-box / e2e / s3s-e2e / S3 compatibility / chaos / fuzz / bench), each layer's entry command, the naming conventions the migration gate depends on, and the serial/nextest rules, see [docs/testing/README.md](docs/testing/README.md).
|
||||
|
||||
> For the event, timeout, required-status, and local reproduction matrix, see [docs/testing/ci-gates.md](docs/testing/ci-gates.md).
|
||||
|
||||
### 🔒 Automated Pre-commit Hooks
|
||||
#### What `make pre-commit` and `make pre-pr` actually run
|
||||
|
||||
|
||||
Generated
+22
-27
@@ -1858,9 +1858,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "cc"
|
||||
version = "1.4.3"
|
||||
version = "1.4.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d"
|
||||
checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273"
|
||||
dependencies = [
|
||||
"find-msvc-tools",
|
||||
"jobserver",
|
||||
@@ -2522,12 +2522,6 @@ dependencies = [
|
||||
"subtle",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cty"
|
||||
version = "0.2.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b365fabc795046672053e29c954733ec3b05e4be654ab130fe8f1f94d7051f35"
|
||||
|
||||
[[package]]
|
||||
name = "curve25519-dalek"
|
||||
version = "4.1.3"
|
||||
@@ -5988,15 +5982,6 @@ version = "0.2.16"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981"
|
||||
|
||||
[[package]]
|
||||
name = "libmimalloc-sys"
|
||||
version = "0.1.49"
|
||||
source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11#6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11"
|
||||
dependencies = [
|
||||
"cc",
|
||||
"cty",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "libredox"
|
||||
version = "0.1.20"
|
||||
@@ -6397,14 +6382,6 @@ dependencies = [
|
||||
"synstructure 0.13.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "mimalloc"
|
||||
version = "0.1.52"
|
||||
source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11#6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11"
|
||||
dependencies = [
|
||||
"libmimalloc-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "mime"
|
||||
version = "0.3.17"
|
||||
@@ -9162,13 +9139,11 @@ dependencies = [
|
||||
"insta",
|
||||
"jiff",
|
||||
"libc",
|
||||
"libmimalloc-sys",
|
||||
"libsystemd",
|
||||
"matchit 0.9.2",
|
||||
"md-5 0.11.0",
|
||||
"metrics",
|
||||
"metrics-util",
|
||||
"mimalloc",
|
||||
"mime_guess",
|
||||
"opentelemetry",
|
||||
"opentelemetry_sdk",
|
||||
@@ -9204,6 +9179,8 @@ dependencies = [
|
||||
"rustfs-lock",
|
||||
"rustfs-log-analyzer",
|
||||
"rustfs-madmin",
|
||||
"rustfs-mimalloc",
|
||||
"rustfs-mimalloc-sys",
|
||||
"rustfs-notify",
|
||||
"rustfs-object-capacity",
|
||||
"rustfs-object-data-cache",
|
||||
@@ -9875,6 +9852,24 @@ dependencies = [
|
||||
"tokio",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-mimalloc"
|
||||
version = "0.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a406f4aa07084301d485beec873af6dccc8e3f8762da244743df92038b1db1a6"
|
||||
dependencies = [
|
||||
"rustfs-mimalloc-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-mimalloc-sys"
|
||||
version = "0.5.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c3051b819175f58445d4c369a72f0ab88149f3885ba8bea2aff3be01f53fe7cd"
|
||||
dependencies = [
|
||||
"cc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-notify"
|
||||
version = "1.0.0-rc.3"
|
||||
|
||||
+2
-2
@@ -350,8 +350,8 @@ russh-sftp = "2.4.0"
|
||||
dav-server = "0.11.0"
|
||||
|
||||
# Performance Analysis and Memory Profiling
|
||||
mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" }
|
||||
libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11", features = ["extended"] }
|
||||
rustfs-mimalloc = { version = "0.5.0" }
|
||||
rustfs-mimalloc-sys = { version = "0.5.0" }
|
||||
hotpath = { version = "0.23.3", default-features = false }
|
||||
# Snapshot testing for output format regression detection
|
||||
insta = { version = "1.48" }
|
||||
|
||||
@@ -241,8 +241,8 @@ pub mod cache {
|
||||
|
||||
pub mod capacity {
|
||||
pub use crate::core::pools::{
|
||||
DecommissionUnresolvedEntry, PoolDecommissionInfo, PoolStatus, get_total_usable_capacity, get_total_usable_capacity_free,
|
||||
path2_bucket_object, path2_bucket_object_with_base_path,
|
||||
PoolDecommissionInfo, PoolStatus, get_total_usable_capacity, get_total_usable_capacity_free, path2_bucket_object,
|
||||
path2_bucket_object_with_base_path,
|
||||
};
|
||||
pub use crate::store::utils::is_reserved_or_invalid_bucket;
|
||||
}
|
||||
|
||||
@@ -804,88 +804,6 @@ fn ensure_decommission_generation(meta: &PoolMeta, idx: usize, generation: Offse
|
||||
}
|
||||
}
|
||||
|
||||
fn record_decommission_unresolved_entry(
|
||||
meta: &mut PoolMeta,
|
||||
idx: usize,
|
||||
generation: OffsetDateTime,
|
||||
entry: DecommissionUnresolvedEntry,
|
||||
) -> Result<bool> {
|
||||
ensure_decommission_generation(meta, idx, generation)?;
|
||||
if entry.pool_index != idx || entry.source_generation != generation {
|
||||
return Err(Error::other("decommission unresolved entry does not match the active pool generation"));
|
||||
}
|
||||
|
||||
let pool_count = meta.pools.len();
|
||||
let Some(pool) = meta.pools.get_mut(idx) else {
|
||||
return Err(invalid_decommission_pool_index_error(pool_count, idx));
|
||||
};
|
||||
let Some(info) = pool.decommission.as_mut() else {
|
||||
return Err(decommission_metadata_not_initialized_error("record decommission unresolved entry"));
|
||||
};
|
||||
let existing = info.unresolved_entries.iter_mut().find(|existing| {
|
||||
existing.bucket == entry.bucket
|
||||
&& existing.object == entry.object
|
||||
&& existing.pool_index == entry.pool_index
|
||||
&& existing.set_index == entry.set_index
|
||||
&& existing.source_generation == entry.source_generation
|
||||
});
|
||||
let changed = match existing {
|
||||
Some(existing) if existing == &entry => false,
|
||||
Some(existing) => {
|
||||
*existing = entry;
|
||||
true
|
||||
}
|
||||
None => {
|
||||
info.unresolved_entries.push(entry);
|
||||
true
|
||||
}
|
||||
};
|
||||
if changed {
|
||||
pool.last_update = OffsetDateTime::now_utc();
|
||||
}
|
||||
Ok(changed)
|
||||
}
|
||||
|
||||
fn reconcile_decommission_unresolved_entries_for_completion(
|
||||
meta: &mut PoolMeta,
|
||||
idx: usize,
|
||||
verified_generation: Option<OffsetDateTime>,
|
||||
) -> Result<()> {
|
||||
let pool_count = meta.pools.len();
|
||||
let Some(pool) = meta.pools.get(idx) else {
|
||||
return Err(invalid_decommission_pool_index_error(pool_count, idx));
|
||||
};
|
||||
let Some(info) = pool.decommission.as_ref() else {
|
||||
return Err(decommission_metadata_not_initialized_error("reconcile decommission unresolved entries"));
|
||||
};
|
||||
if info.unresolved_entries.is_empty() {
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let Some(generation) = verified_generation else {
|
||||
return Err(Error::other(format!(
|
||||
"failed to complete decommission for pool {idx}: {} unresolved listing entries remain",
|
||||
info.unresolved_entries.len()
|
||||
)));
|
||||
};
|
||||
ensure_decommission_generation(meta, idx, generation)?;
|
||||
if info
|
||||
.unresolved_entries
|
||||
.iter()
|
||||
.any(|entry| entry.source_generation != generation)
|
||||
{
|
||||
return Err(Error::other(format!(
|
||||
"failed to complete decommission for pool {idx}: unresolved listing ledger contains a different generation"
|
||||
)));
|
||||
}
|
||||
|
||||
let Some(info) = meta.pools.get_mut(idx).and_then(|pool| pool.decommission.as_mut()) else {
|
||||
return Err(decommission_metadata_not_initialized_error("reconcile decommission unresolved entries"));
|
||||
};
|
||||
info.unresolved_entries.clear();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn run_decommission_side_effect<T, F, Fut>(
|
||||
rx: &CancellationToken,
|
||||
operation_gate: &Arc<tokio::sync::RwLock<()>>,
|
||||
@@ -1063,15 +981,18 @@ fn resolve_decommission_listing_error(listing_error: Option<Error>, entry_error:
|
||||
}
|
||||
}
|
||||
|
||||
fn decommission_unresolved_listing_error(entry: &DecommissionUnresolvedEntry) -> Error {
|
||||
fn decommission_unresolved_listing_error(
|
||||
bucket: &str,
|
||||
prefix: &str,
|
||||
candidate: Option<&str>,
|
||||
candidate_count: usize,
|
||||
disk_error_count: usize,
|
||||
pool_index: usize,
|
||||
set_index: usize,
|
||||
) -> Error {
|
||||
let location = candidate.unwrap_or(prefix);
|
||||
Error::other(format!(
|
||||
"decommission listing could not resolve metadata for {bucket}/{object} on pool {pool_index} set {set_index} ({candidate_count} candidate(s), {disk_error_count} disk error(s))",
|
||||
bucket = entry.bucket,
|
||||
object = entry.object,
|
||||
pool_index = entry.pool_index,
|
||||
set_index = entry.set_index,
|
||||
candidate_count = entry.candidate_count,
|
||||
disk_error_count = entry.disk_error_count,
|
||||
"decommission listing could not resolve metadata for {bucket}/{location} on pool {pool_index} set {set_index} ({candidate_count} candidate(s), {disk_error_count} disk error(s))"
|
||||
))
|
||||
}
|
||||
|
||||
@@ -1083,32 +1004,22 @@ fn resolve_decommission_partial_listing_entry(
|
||||
disk_error_count: usize,
|
||||
pool_index: usize,
|
||||
set_index: usize,
|
||||
source_generation: OffsetDateTime,
|
||||
) -> std::result::Result<MetaCacheEntry, DecommissionUnresolvedEntry> {
|
||||
) -> Result<MetaCacheEntry> {
|
||||
let candidate_count = entries.as_ref().iter().flatten().count();
|
||||
if let Some(entry) = entries.resolve(resolver) {
|
||||
return Ok(entry);
|
||||
}
|
||||
|
||||
let object = entries
|
||||
.as_ref()
|
||||
.iter()
|
||||
.flatten()
|
||||
.map(|entry| entry.name.as_str())
|
||||
.next()
|
||||
.unwrap_or(prefix)
|
||||
.to_string();
|
||||
Err(DecommissionUnresolvedEntry {
|
||||
bucket: bucket.to_string(),
|
||||
object,
|
||||
let candidate = entries.as_ref().iter().flatten().map(|entry| entry.name.as_str()).next();
|
||||
Err(decommission_unresolved_listing_error(
|
||||
bucket,
|
||||
prefix,
|
||||
candidate,
|
||||
candidate_count,
|
||||
disk_error_count,
|
||||
pool_index,
|
||||
set_index,
|
||||
source_generation,
|
||||
observed_at: OffsetDateTime::now_utc(),
|
||||
reason: "metadata_resolution_failed".to_string(),
|
||||
})
|
||||
))
|
||||
}
|
||||
|
||||
fn resolve_decommission_pool_meta_reload_result(result: Result<()>, stage: &str) -> Result<()> {
|
||||
@@ -1713,8 +1624,6 @@ struct PersistedPoolDecommissionInfo {
|
||||
pub terminal_reload_attempt_at: Option<OffsetDateTime>,
|
||||
#[serde(rename = "terminalReloadFailures", default)]
|
||||
pub terminal_reload_failures: Vec<String>,
|
||||
#[serde(rename = "unresolvedEntries", default)]
|
||||
pub unresolved_entries: Vec<DecommissionUnresolvedEntry>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize)]
|
||||
@@ -1846,7 +1755,6 @@ impl TryFrom<PersistedPoolDecommissionInfo> for PoolDecommissionInfo {
|
||||
bytes_failed: value.bytes_failed,
|
||||
terminal_reload_attempt_at: value.terminal_reload_attempt_at,
|
||||
terminal_reload_failures: value.terminal_reload_failures,
|
||||
unresolved_entries: value.unresolved_entries,
|
||||
progress_save_item_baseline: value.items_decommissioned.saturating_add(value.items_decommission_failed),
|
||||
progress_save_retry_after: None,
|
||||
})
|
||||
@@ -1879,7 +1787,6 @@ impl TryFrom<LegacyPoolDecommissionInfo> for PoolDecommissionInfo {
|
||||
bytes_failed: value.bytes_failed,
|
||||
terminal_reload_attempt_at: None,
|
||||
terminal_reload_failures: Vec::new(),
|
||||
unresolved_entries: Vec::new(),
|
||||
progress_save_item_baseline: value.items_decommissioned.saturating_add(value.items_decommission_failed),
|
||||
progress_save_retry_after: None,
|
||||
})
|
||||
@@ -1928,7 +1835,6 @@ impl From<&PoolDecommissionInfo> for PersistedPoolDecommissionInfo {
|
||||
bytes_failed: value.bytes_failed,
|
||||
terminal_reload_attempt_at: value.terminal_reload_attempt_at,
|
||||
terminal_reload_failures: value.terminal_reload_failures.clone(),
|
||||
unresolved_entries: value.unresolved_entries.clone(),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2529,26 +2435,6 @@ pub fn path2_bucket_object_with_base_path(base_path: &str, path: &str) -> (Strin
|
||||
path_to_bucket_object_with_base_path(base_path, path)
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub struct DecommissionUnresolvedEntry {
|
||||
pub bucket: String,
|
||||
pub object: String,
|
||||
#[serde(rename = "poolIndex")]
|
||||
pub pool_index: usize,
|
||||
#[serde(rename = "setIndex")]
|
||||
pub set_index: usize,
|
||||
#[serde(rename = "sourceGeneration", with = "time::serde::rfc3339")]
|
||||
pub source_generation: OffsetDateTime,
|
||||
#[serde(rename = "candidateCount")]
|
||||
pub candidate_count: usize,
|
||||
#[serde(rename = "diskErrorCount")]
|
||||
pub disk_error_count: usize,
|
||||
#[serde(rename = "observedAt", with = "time::serde::rfc3339")]
|
||||
pub observed_at: OffsetDateTime,
|
||||
pub reason: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||
pub struct PoolDecommissionInfo {
|
||||
#[serde(rename = "startTime", with = "time::serde::rfc3339::option")]
|
||||
@@ -2594,8 +2480,6 @@ pub struct PoolDecommissionInfo {
|
||||
#[serde(rename = "terminalReloadFailures", default)]
|
||||
pub terminal_reload_failures: Vec<String>,
|
||||
#[serde(skip)]
|
||||
pub unresolved_entries: Vec<DecommissionUnresolvedEntry>,
|
||||
#[serde(skip)]
|
||||
pub progress_save_item_baseline: usize,
|
||||
#[serde(skip)]
|
||||
pub progress_save_retry_after: Option<OffsetDateTime>,
|
||||
@@ -2629,7 +2513,6 @@ impl PoolDecommissionInfo {
|
||||
|| self.bytes_failed > 0
|
||||
|| self.terminal_reload_attempt_at.is_some()
|
||||
|| !self.terminal_reload_failures.is_empty()
|
||||
|| !self.unresolved_entries.is_empty()
|
||||
}
|
||||
|
||||
fn counted_items(&self) -> usize {
|
||||
@@ -2947,21 +2830,6 @@ impl ECStore {
|
||||
snapshot.save(self.pools.clone()).await
|
||||
}
|
||||
|
||||
async fn persist_decommission_unresolved_entry(
|
||||
&self,
|
||||
idx: usize,
|
||||
generation: OffsetDateTime,
|
||||
entry: DecommissionUnresolvedEntry,
|
||||
) -> Result<()> {
|
||||
{
|
||||
let mut pool_meta = self.pool_meta.write().await;
|
||||
record_decommission_unresolved_entry(&mut pool_meta, idx, generation, entry)?;
|
||||
}
|
||||
self.save_current_pool_meta()
|
||||
.await
|
||||
.map_err(|err| Error::other(format!("decommission unresolved entry ledger save failed: {err}")))
|
||||
}
|
||||
|
||||
async fn save_decommission_progress_checkpoint(&self, idx: usize, generation: OffsetDateTime) -> Result<bool> {
|
||||
// Lock order: save gate, then the short pool metadata read/write sections. Peer
|
||||
// reloads are intentionally performed by the caller after both locks are released.
|
||||
@@ -3810,7 +3678,6 @@ impl ECStore {
|
||||
let list_bi = bi.clone();
|
||||
let list_outstanding = outstanding.clone();
|
||||
let list_entry_error = entry_error.clone();
|
||||
let list_store = self.clone();
|
||||
let mut listing = tokio::spawn(async move {
|
||||
run_decommission_listing_with_retry_and_drain(
|
||||
list_rx.clone(),
|
||||
@@ -3824,9 +3691,8 @@ impl ECStore {
|
||||
let rx = list_rx_for_list.clone();
|
||||
let bucket = list_bi.clone();
|
||||
let entry_error = list_entry_error.clone();
|
||||
let store = list_store.clone();
|
||||
async move {
|
||||
set.list_objects_to_decommission(store, rx, bucket, callback, entry_error, idx, set_idx, generation)
|
||||
set.list_objects_to_decommission(rx, bucket, callback, entry_error, idx, set_idx)
|
||||
.await
|
||||
}
|
||||
},
|
||||
@@ -4727,7 +4593,7 @@ impl ECStore {
|
||||
state = "verifying_completion",
|
||||
"Decommission completion verification started"
|
||||
);
|
||||
if let Err(err) = self.check_after_decommission(idx, generation).await {
|
||||
if let Err(err) = self.check_after_decommission(idx).await {
|
||||
resolve_decommission_terminal_mark_result(
|
||||
self.decommission_failed_for_operation(idx, canceler).await,
|
||||
"failed",
|
||||
@@ -4754,7 +4620,7 @@ impl ECStore {
|
||||
"Decommission marking completed state"
|
||||
);
|
||||
resolve_decommission_terminal_mark_result(
|
||||
self.complete_decommission_for_operation(idx, canceler, generation).await,
|
||||
self.complete_decommission_for_operation(idx, canceler).await,
|
||||
"completed",
|
||||
&cmd_line,
|
||||
)?;
|
||||
@@ -4890,25 +4756,14 @@ impl ECStore {
|
||||
|
||||
#[tracing::instrument(skip(self))]
|
||||
pub async fn complete_decommission(&self, idx: usize) -> Result<()> {
|
||||
self.complete_decommission_with_owner(idx, None, None).await
|
||||
self.complete_decommission_with_owner(idx, None).await
|
||||
}
|
||||
|
||||
async fn complete_decommission_for_operation(
|
||||
&self,
|
||||
idx: usize,
|
||||
owner: &DecommissionCanceler,
|
||||
verified_generation: OffsetDateTime,
|
||||
) -> Result<()> {
|
||||
self.complete_decommission_with_owner(idx, Some(owner), Some(verified_generation))
|
||||
.await
|
||||
async fn complete_decommission_for_operation(&self, idx: usize, owner: &DecommissionCanceler) -> Result<()> {
|
||||
self.complete_decommission_with_owner(idx, Some(owner)).await
|
||||
}
|
||||
|
||||
async fn complete_decommission_with_owner(
|
||||
&self,
|
||||
idx: usize,
|
||||
owner: Option<&DecommissionCanceler>,
|
||||
verified_generation: Option<OffsetDateTime>,
|
||||
) -> Result<()> {
|
||||
async fn complete_decommission_with_owner(&self, idx: usize, owner: Option<&DecommissionCanceler>) -> Result<()> {
|
||||
ensure_decommission_terminal_operation_supported(self.single_pool(), "complete decommission")?;
|
||||
let _start_guard = self.start_gate.lock().await;
|
||||
|
||||
@@ -4920,13 +4775,11 @@ impl ECStore {
|
||||
let previous_pool_meta = pool_meta.clone();
|
||||
let Some(changed) =
|
||||
update_decommission_for_operation(cancelers.as_slice(), &mut pool_meta, idx, owner, |pool_meta| {
|
||||
reconcile_decommission_unresolved_entries_for_completion(pool_meta, idx, verified_generation)?;
|
||||
Ok::<bool, Error>(pool_meta.decommission_complete(idx))
|
||||
pool_meta.decommission_complete(idx)
|
||||
})
|
||||
else {
|
||||
return Ok(());
|
||||
};
|
||||
let changed = changed?;
|
||||
let terminal_canceler = if let Some(owner) = owner {
|
||||
Some(owner.clone())
|
||||
} else {
|
||||
@@ -5286,7 +5139,7 @@ impl ECStore {
|
||||
Ok(ret)
|
||||
}
|
||||
|
||||
async fn check_after_decommission(self: &Arc<Self>, idx: usize, generation: OffsetDateTime) -> Result<()> {
|
||||
async fn check_after_decommission(self: &Arc<Self>, idx: usize) -> Result<()> {
|
||||
let buckets = self.get_buckets_to_decommission().await?;
|
||||
let pool = self.pools[idx].clone();
|
||||
|
||||
@@ -5385,16 +5238,7 @@ impl ECStore {
|
||||
});
|
||||
|
||||
let list_result = set
|
||||
.list_objects_to_decommission(
|
||||
self.clone(),
|
||||
callback_rx,
|
||||
bucket_info.clone(),
|
||||
callback,
|
||||
entry_error.clone(),
|
||||
idx,
|
||||
set_index,
|
||||
generation,
|
||||
)
|
||||
.list_objects_to_decommission(callback_rx, bucket_info.clone(), callback, entry_error.clone(), idx, set_index)
|
||||
.await;
|
||||
let entry_error = entry_error.lock().await.clone();
|
||||
resolve_decommission_check_after_list_result(list_result, entry_error)?;
|
||||
@@ -5763,17 +5607,6 @@ mod tests {
|
||||
#[test]
|
||||
fn pool_meta_persists_decommission_resume_queues() {
|
||||
let start_time = OffsetDateTime::now_utc();
|
||||
let unresolved_entry = DecommissionUnresolvedEntry {
|
||||
bucket: "bucket-b".to_string(),
|
||||
object: "prefix/unresolved.txt".to_string(),
|
||||
pool_index: 0,
|
||||
set_index: 1,
|
||||
source_generation: start_time,
|
||||
candidate_count: 2,
|
||||
disk_error_count: 1,
|
||||
observed_at: start_time,
|
||||
reason: "metadata_resolution_failed".to_string(),
|
||||
};
|
||||
let pool_meta = PoolMeta {
|
||||
version: POOL_META_VERSION,
|
||||
pools: vec![PoolStatus {
|
||||
@@ -5794,7 +5627,6 @@ mod tests {
|
||||
bytes_failed: 128,
|
||||
terminal_reload_attempt_at: Some(start_time),
|
||||
terminal_reload_failures: vec!["complete_decommission: peer node-a failed".to_string()],
|
||||
unresolved_entries: vec![unresolved_entry.clone()],
|
||||
..Default::default()
|
||||
}),
|
||||
}],
|
||||
@@ -5834,7 +5666,6 @@ mod tests {
|
||||
restored_decommission.terminal_reload_failures,
|
||||
vec!["complete_decommission: peer node-a failed".to_string()]
|
||||
);
|
||||
assert_eq!(restored_decommission.unresolved_entries, vec![unresolved_entry]);
|
||||
assert!(restored_decommission.queued);
|
||||
assert_eq!(restored_decommission.items_since_last_progress_save(), 0);
|
||||
}
|
||||
@@ -5926,7 +5757,6 @@ mod tests {
|
||||
assert!(decommission.bucket.is_empty());
|
||||
assert!(decommission.prefix.is_empty());
|
||||
assert!(decommission.object.is_empty());
|
||||
assert!(decommission.unresolved_entries.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -6216,32 +6046,28 @@ async fn record_decommission_entry_error(
|
||||
entry_error: &Arc<tokio::sync::Mutex<Option<Error>>>,
|
||||
rx: &CancellationToken,
|
||||
err: Error,
|
||||
) -> bool {
|
||||
) {
|
||||
if rx.is_cancelled() {
|
||||
return false;
|
||||
return;
|
||||
}
|
||||
|
||||
let mut first_err = entry_error.lock().await;
|
||||
if first_err.is_none() && !rx.is_cancelled() {
|
||||
*first_err = Some(err);
|
||||
rx.cancel();
|
||||
return true;
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
impl SetDisks {
|
||||
#[tracing::instrument(skip(self, store, rx, cb_func, entry_error))]
|
||||
#[tracing::instrument(skip(self, rx, cb_func, entry_error))]
|
||||
async fn list_objects_to_decommission(
|
||||
self: &Arc<Self>,
|
||||
store: Arc<ECStore>,
|
||||
rx: CancellationToken,
|
||||
bucket_info: DecomBucketInfo,
|
||||
cb_func: ListCallback,
|
||||
entry_error: Arc<tokio::sync::Mutex<Option<Error>>>,
|
||||
pool_index: usize,
|
||||
set_index: usize,
|
||||
source_generation: OffsetDateTime,
|
||||
) -> Result<()> {
|
||||
let (disks, _) = self.get_online_disks_with_healing(false).await;
|
||||
ensure_decommission_listing_disks_available(!disks.is_empty(), &bucket_info.name)?;
|
||||
@@ -6262,8 +6088,6 @@ impl SetDisks {
|
||||
let unresolved_prefix = bucket_info.prefix.clone();
|
||||
let unresolved_pool_index = pool_index;
|
||||
let unresolved_set_index = set_index;
|
||||
let unresolved_generation = source_generation;
|
||||
let unresolved_store = store;
|
||||
|
||||
list_path_raw(
|
||||
rx,
|
||||
@@ -6283,10 +6107,8 @@ impl SetDisks {
|
||||
let prefix = unresolved_prefix.clone();
|
||||
let unresolved_error = unresolved_error.clone();
|
||||
let unresolved_rx = unresolved_rx.clone();
|
||||
let unresolved_store = unresolved_store.clone();
|
||||
let pool_index = unresolved_pool_index;
|
||||
let set_index = unresolved_set_index;
|
||||
let source_generation = unresolved_generation;
|
||||
let disk_error_count = errs.iter().flatten().count();
|
||||
if unresolved_rx.is_cancelled() {
|
||||
return Box::pin(async {});
|
||||
@@ -6300,7 +6122,6 @@ impl SetDisks {
|
||||
disk_error_count,
|
||||
pool_index,
|
||||
set_index,
|
||||
source_generation,
|
||||
) {
|
||||
Ok(entry) => {
|
||||
warn!("decommission_pool: list_objects_to_decommission get {}", &entry.name);
|
||||
@@ -6308,11 +6129,10 @@ impl SetDisks {
|
||||
cb_func(entry).await;
|
||||
})
|
||||
}
|
||||
Err(unresolved_entry) => Box::pin(async move {
|
||||
Err(err) => Box::pin(async move {
|
||||
if unresolved_rx.is_cancelled() {
|
||||
return;
|
||||
}
|
||||
let err = decommission_unresolved_listing_error(&unresolved_entry);
|
||||
warn!(
|
||||
event = EVENT_DECOMMISSION_BUCKET,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
@@ -6323,13 +6143,6 @@ impl SetDisks {
|
||||
error = %err,
|
||||
"Decommission listing failed closed on unresolved metadata"
|
||||
);
|
||||
let err = match unresolved_store
|
||||
.persist_decommission_unresolved_entry(pool_index, source_generation, unresolved_entry)
|
||||
.await
|
||||
{
|
||||
Ok(()) => err,
|
||||
Err(ledger_err) => Error::other(format!("{err}; {ledger_err}")),
|
||||
};
|
||||
record_decommission_entry_error(&unresolved_error, &unresolved_rx, err).await;
|
||||
}),
|
||||
}
|
||||
@@ -6538,52 +6351,47 @@ mod pools_tests {
|
||||
use super::{
|
||||
DECOMMISSION_ENTRY_CONCURRENCY_DEFAULT_CAP, DECOMMISSION_ENTRY_CONCURRENCY_HARD_CAP, DECOMMISSION_ENTRY_QUEUE_HARD_CAP,
|
||||
DECOMMISSION_PROGRESS_SAVE_INTERVAL, DECOMMISSION_PROGRESS_SAVE_ITEM_THRESHOLD, DecomBucketInfo, DecommissionCanceler,
|
||||
DecommissionEntryEnqueueResult, DecommissionStartPoolState, DecommissionTerminalState, DecommissionUnresolvedEntry,
|
||||
ListCallback, POOL_META_VERSION, PoolDecommissionInfo, PoolMeta, PoolSpaceInfo, PoolStatus, QueuedDecommissionEntry,
|
||||
apply_decommission_status_space_info, await_decommission_worker, bind_decommission_cancelers,
|
||||
bind_missing_decommission_cancelers, cancel_decommission_canceler, clamp_decommission_entry_concurrency,
|
||||
classify_decommission_terminal_state, count_decommission_item, decommission_cancel_signal_result,
|
||||
decommission_entry_queue_capacity, decommission_item_size, decommission_meta_bucket_options,
|
||||
decommission_start_pool_state, decommission_unresolved_listing_error, dedup_indices,
|
||||
default_decommission_bucket_concurrency, default_decommission_entry_concurrency, drain_decommission_entry_queue,
|
||||
enqueue_decommission_entry, ensure_decommission_cancel_allowed, ensure_decommission_clear_allowed,
|
||||
ensure_decommission_generation, ensure_decommission_listing_disks_available, ensure_decommission_not_rebalancing,
|
||||
ensure_decommission_start_allowed, ensure_decommission_start_keeps_active_pool, ensure_decommission_start_local_leader,
|
||||
DecommissionEntryEnqueueResult, DecommissionStartPoolState, DecommissionTerminalState, ListCallback,
|
||||
PoolDecommissionInfo, PoolMeta, PoolSpaceInfo, PoolStatus, QueuedDecommissionEntry, apply_decommission_status_space_info,
|
||||
await_decommission_worker, bind_decommission_cancelers, bind_missing_decommission_cancelers,
|
||||
cancel_decommission_canceler, clamp_decommission_entry_concurrency, classify_decommission_terminal_state,
|
||||
count_decommission_item, decommission_cancel_signal_result, decommission_entry_queue_capacity, decommission_item_size,
|
||||
decommission_meta_bucket_options, decommission_start_pool_state, dedup_indices, default_decommission_bucket_concurrency,
|
||||
default_decommission_entry_concurrency, drain_decommission_entry_queue, enqueue_decommission_entry,
|
||||
ensure_decommission_cancel_allowed, ensure_decommission_clear_allowed, ensure_decommission_generation,
|
||||
ensure_decommission_listing_disks_available, ensure_decommission_not_rebalancing, ensure_decommission_start_allowed,
|
||||
ensure_decommission_start_keeps_active_pool, ensure_decommission_start_local_leader,
|
||||
ensure_decommission_start_pool_states, ensure_decommission_start_rebalance_meta_allowed,
|
||||
ensure_decommission_start_target_capacity, ensure_decommission_terminal_operation_supported,
|
||||
ensure_local_decommission_pool_leaders, ensure_valid_decommission_pool_index, first_resumable_decommission_queue_indices,
|
||||
get_by_index, guard_decommission_cancelers, has_active_decommission_canceler, is_decommission_active,
|
||||
is_decommission_cancel_requested, load_decommission_entry_versions, local_decommission_queue_prefix,
|
||||
mark_decommission_bucket_done, merge_pool_status_refresh, missing_decommission_worker_prefix,
|
||||
observe_decommission_terminal_reload_result, pool_meta_has_active_decommission,
|
||||
reconcile_decommission_unresolved_entries_for_completion, record_decommission_unresolved_entry,
|
||||
require_decommission_store, reserve_decommission_start_cancelers, resolve_decommission_bucket_done_save_result,
|
||||
resolve_decommission_bucket_state, resolve_decommission_check_after_list_result,
|
||||
resolve_decommission_entry_cleanup_delete_result, resolve_decommission_entry_exact_versions,
|
||||
resolve_decommission_entry_reload_result, resolve_decommission_listing_worker_result,
|
||||
resolve_decommission_optional_bucket_config_result, resolve_decommission_partial_listing_entry,
|
||||
resolve_decommission_pool_meta_reload_result, resolve_decommission_preflight_heal_result,
|
||||
resolve_decommission_progress_save_result, resolve_decommission_terminal_mark_after_error_result,
|
||||
resolve_decommission_terminal_mark_result, resolve_decommission_update_after_result,
|
||||
resolve_start_decommission_pool_meta_reload_result, rollback_start_decommission_pool_meta,
|
||||
run_decommission_buckets_bounded, run_decommission_listing_with_retry, run_decommission_listing_with_retry_and_drain,
|
||||
run_decommission_side_effect, should_cleanup_decommission_source_entry, should_continue_decommission_queue,
|
||||
should_count_decommission_version_complete, should_preserve_decommission_canceled_state,
|
||||
should_reject_decommission_cancel_as_terminal, should_retry_decommission_cancel_reload,
|
||||
should_retry_decommission_listing, should_skip_canceled_decommission_routine, spawn_decommission_index_cancelers,
|
||||
split_decommission_buckets, take_and_cancel_decommission_canceler, take_decommission_canceler,
|
||||
track_decommission_current_object, track_decommission_current_object_stage, update_decommission_for_operation,
|
||||
validate_start_decommission_request, wait_decommission_listing_retry, wait_decommission_worker_drain,
|
||||
with_decommission_entry_context,
|
||||
observe_decommission_terminal_reload_result, pool_meta_has_active_decommission, require_decommission_store,
|
||||
reserve_decommission_start_cancelers, resolve_decommission_bucket_done_save_result, resolve_decommission_bucket_state,
|
||||
resolve_decommission_check_after_list_result, resolve_decommission_entry_cleanup_delete_result,
|
||||
resolve_decommission_entry_exact_versions, resolve_decommission_entry_reload_result,
|
||||
resolve_decommission_listing_worker_result, resolve_decommission_optional_bucket_config_result,
|
||||
resolve_decommission_partial_listing_entry, resolve_decommission_pool_meta_reload_result,
|
||||
resolve_decommission_preflight_heal_result, resolve_decommission_progress_save_result,
|
||||
resolve_decommission_terminal_mark_after_error_result, resolve_decommission_terminal_mark_result,
|
||||
resolve_decommission_update_after_result, resolve_start_decommission_pool_meta_reload_result,
|
||||
rollback_start_decommission_pool_meta, run_decommission_buckets_bounded, run_decommission_listing_with_retry,
|
||||
run_decommission_listing_with_retry_and_drain, run_decommission_side_effect, should_cleanup_decommission_source_entry,
|
||||
should_continue_decommission_queue, should_count_decommission_version_complete,
|
||||
should_preserve_decommission_canceled_state, should_reject_decommission_cancel_as_terminal,
|
||||
should_retry_decommission_cancel_reload, should_retry_decommission_listing, should_skip_canceled_decommission_routine,
|
||||
spawn_decommission_index_cancelers, split_decommission_buckets, take_and_cancel_decommission_canceler,
|
||||
take_decommission_canceler, track_decommission_current_object, track_decommission_current_object_stage,
|
||||
update_decommission_for_operation, validate_start_decommission_request, wait_decommission_listing_retry,
|
||||
wait_decommission_worker_drain, with_decommission_entry_context,
|
||||
};
|
||||
use crate::bucket::metadata_sys;
|
||||
use crate::data_movement;
|
||||
use crate::disk::{STORAGE_FORMAT_FILE, endpoint::Endpoint};
|
||||
use crate::disk::endpoint::Endpoint;
|
||||
use crate::error::{Error, StorageError};
|
||||
use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
|
||||
use crate::runtime::instance::InstanceContext;
|
||||
use crate::services::rebalance::{RebalStatus, RebalanceInfo, RebalanceMeta, RebalanceStats};
|
||||
use crate::storage_api_contracts::bucket::MakeBucketOptions;
|
||||
use crate::store::ECStore;
|
||||
use rustfs_filemeta::{FileInfo, FileInfoVersions, MetaCacheEntry, ObjectPartInfo};
|
||||
use rustfs_filemeta::{MetaCacheEntries, MetadataResolutionParams};
|
||||
@@ -7814,8 +7622,7 @@ mod pools_tests {
|
||||
|
||||
#[test]
|
||||
fn test_resolve_decommission_partial_listing_entry_rejects_unresolved_metadata() {
|
||||
let generation = OffsetDateTime::now_utc();
|
||||
let unresolved_entry = resolve_decommission_partial_listing_entry(
|
||||
let err = resolve_decommission_partial_listing_entry(
|
||||
MetaCacheEntries(vec![None]),
|
||||
MetadataResolutionParams {
|
||||
dir_quorum: 2,
|
||||
@@ -7828,160 +7635,23 @@ mod pools_tests {
|
||||
1,
|
||||
2,
|
||||
3,
|
||||
generation,
|
||||
)
|
||||
.expect_err("unresolved partial listing must fail closed");
|
||||
|
||||
assert_eq!(unresolved_entry.bucket, "bucket-a");
|
||||
assert_eq!(unresolved_entry.object, "prefix/");
|
||||
assert_eq!(unresolved_entry.pool_index, 2);
|
||||
assert_eq!(unresolved_entry.set_index, 3);
|
||||
assert_eq!(unresolved_entry.source_generation, generation);
|
||||
assert_eq!(unresolved_entry.candidate_count, 0);
|
||||
assert_eq!(unresolved_entry.disk_error_count, 1);
|
||||
assert_eq!(unresolved_entry.reason, "metadata_resolution_failed");
|
||||
|
||||
let message = decommission_unresolved_listing_error(&unresolved_entry).to_string();
|
||||
let message = err.to_string();
|
||||
assert!(message.contains("decommission listing could not resolve metadata"));
|
||||
assert!(message.contains("bucket-a/prefix/"));
|
||||
assert!(message.contains("pool 2 set 3"));
|
||||
assert!(message.contains("1 disk error(s)"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unresolved_entry_ledger_blocks_unverified_completion_and_reconciles_after_verified_sweep() {
|
||||
let generation = OffsetDateTime::now_utc();
|
||||
let mut pool_meta = PoolMeta {
|
||||
version: POOL_META_VERSION,
|
||||
pools: vec![PoolStatus {
|
||||
id: 0,
|
||||
cmd_line: "/data/pool".to_string(),
|
||||
last_update: generation,
|
||||
decommission: Some(PoolDecommissionInfo {
|
||||
start_time: Some(generation),
|
||||
..Default::default()
|
||||
}),
|
||||
}],
|
||||
dont_save: true,
|
||||
};
|
||||
let unresolved_entry = DecommissionUnresolvedEntry {
|
||||
bucket: "bucket-a".to_string(),
|
||||
object: "object-a".to_string(),
|
||||
pool_index: 0,
|
||||
set_index: 0,
|
||||
source_generation: generation,
|
||||
candidate_count: 1,
|
||||
disk_error_count: 0,
|
||||
observed_at: generation,
|
||||
reason: "metadata_resolution_failed".to_string(),
|
||||
};
|
||||
|
||||
assert!(
|
||||
record_decommission_unresolved_entry(&mut pool_meta, 0, generation, unresolved_entry)
|
||||
.expect("active generation should accept unresolved entry")
|
||||
);
|
||||
let err = reconcile_decommission_unresolved_entries_for_completion(&mut pool_meta, 0, None)
|
||||
.expect_err("unverified completion must retain unresolved entries");
|
||||
assert!(err.to_string().contains("1 unresolved listing entries remain"));
|
||||
assert_eq!(
|
||||
pool_meta.pools[0]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.expect("decommission should exist")
|
||||
.unresolved_entries
|
||||
.len(),
|
||||
1
|
||||
);
|
||||
|
||||
reconcile_decommission_unresolved_entries_for_completion(&mut pool_meta, 0, Some(generation))
|
||||
.expect("a successful same-generation final sweep should reconcile stale entries");
|
||||
assert!(
|
||||
pool_meta.pools[0]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.expect("decommission should exist")
|
||||
.unresolved_entries
|
||||
.is_empty()
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn final_sweep_persists_unresolved_entry_from_real_listing() {
|
||||
let (dirs, store) = metadata_sys::test_support::isolated_store_over_temp_disks().await;
|
||||
let bucket = "decommission-final-sweep-unresolved";
|
||||
let object = "corrupt-object";
|
||||
store
|
||||
.peer_sys
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("test bucket should be created");
|
||||
metadata_sys::init_bucket_metadata_sys(store.clone(), vec![bucket.to_string()]).await;
|
||||
|
||||
let generation = OffsetDateTime::now_utc();
|
||||
{
|
||||
let mut pool_meta = store.pool_meta.write().await;
|
||||
pool_meta.dont_save = false;
|
||||
pool_meta.pools[0].decommission = Some(PoolDecommissionInfo {
|
||||
start_time: Some(generation),
|
||||
..Default::default()
|
||||
});
|
||||
}
|
||||
|
||||
for (disk_index, dir) in dirs.iter().enumerate() {
|
||||
let object_dir = dir.path().join(bucket).join(object);
|
||||
tokio::fs::create_dir_all(&object_dir)
|
||||
.await
|
||||
.expect("corrupt object directory should be created");
|
||||
tokio::fs::write(object_dir.join(STORAGE_FORMAT_FILE), format!("corrupt-xl-meta-{disk_index}").into_bytes())
|
||||
.await
|
||||
.expect("divergent corrupt metadata should be written");
|
||||
}
|
||||
|
||||
let err = store
|
||||
.check_after_decommission(0, generation)
|
||||
.await
|
||||
.expect_err("real final sweep must fail closed on unresolved listing metadata");
|
||||
assert!(err.to_string().contains("decommission listing could not resolve metadata"));
|
||||
|
||||
let in_memory_entries = store.pool_meta.read().await.pools[0]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.expect("decommission should remain active")
|
||||
.unresolved_entries
|
||||
.clone();
|
||||
assert_eq!(in_memory_entries.len(), 1);
|
||||
let unresolved_entry = &in_memory_entries[0];
|
||||
assert_eq!(unresolved_entry.bucket, bucket);
|
||||
assert_eq!(unresolved_entry.object, object);
|
||||
assert_eq!(unresolved_entry.pool_index, 0);
|
||||
assert_eq!(unresolved_entry.set_index, 0);
|
||||
assert_eq!(unresolved_entry.source_generation, generation);
|
||||
assert_eq!(unresolved_entry.candidate_count, 4);
|
||||
assert_eq!(unresolved_entry.disk_error_count, 0);
|
||||
assert_eq!(unresolved_entry.reason, "metadata_resolution_failed");
|
||||
|
||||
let mut restored = PoolMeta::default();
|
||||
restored
|
||||
.load_for_startup(store.pools[0].clone())
|
||||
.await
|
||||
.expect("persisted pool metadata should reload after the final-sweep failure");
|
||||
assert_eq!(
|
||||
restored.pools[0]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.expect("reloaded decommission should exist")
|
||||
.unresolved_entries,
|
||||
in_memory_entries
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_record_decommission_entry_error_cancels_listing_and_preserves_first_error() {
|
||||
let entry_error = Arc::new(tokio::sync::Mutex::new(None));
|
||||
let rx = CancellationToken::new();
|
||||
|
||||
assert!(record_decommission_entry_error(&entry_error, &rx, Error::SlowDown).await);
|
||||
assert!(!record_decommission_entry_error(&entry_error, &rx, Error::OperationCanceled).await);
|
||||
record_decommission_entry_error(&entry_error, &rx, Error::SlowDown).await;
|
||||
record_decommission_entry_error(&entry_error, &rx, Error::OperationCanceled).await;
|
||||
|
||||
assert!(rx.is_cancelled());
|
||||
assert!(matches!(*entry_error.lock().await, Some(Error::SlowDown)));
|
||||
@@ -7993,7 +7663,7 @@ mod pools_tests {
|
||||
let rx = CancellationToken::new();
|
||||
rx.cancel();
|
||||
|
||||
assert!(!record_decommission_entry_error(&entry_error, &rx, Error::SlowDown).await);
|
||||
record_decommission_entry_error(&entry_error, &rx, Error::SlowDown).await;
|
||||
|
||||
assert!(entry_error.lock().await.is_none());
|
||||
}
|
||||
|
||||
@@ -2124,26 +2124,13 @@ impl SetDisks {
|
||||
|
||||
let put_object_size = known_put_object_storage_size(data.size());
|
||||
let shard_file_size_raw = erasure.shard_file_size(put_object_size);
|
||||
let is_inline_buffer =
|
||||
storage_class_config.should_inline(shard_file_size_raw, erasure.data_shards, opts.versioned);
|
||||
let is_inline_buffer = storage_class_config.should_inline(shard_file_size_raw, erasure.data_shards, opts.versioned);
|
||||
|
||||
let collect_stage_timing = rustfs_io_metrics::put_stage_metrics_enabled() || issue3031_diag_enabled();
|
||||
let shard_file_size = shard_file_size_raw;
|
||||
let shard_size = erasure.shard_size();
|
||||
let write_path = classify_put_write_path(is_inline_buffer, put_object_size, fi.erasure.block_size);
|
||||
let direct_inline_commit = matches!(write_path, SmallWritePath::Inline);
|
||||
{
|
||||
use std::io::Write;
|
||||
let msg = format!(
|
||||
"INLINE_DEBUG: bucket={} obj={} size={} shard_fs={} ds={} bs={} inline={} direct={} path={} iblock={} ver={}\n",
|
||||
bucket, object, put_object_size, shard_file_size_raw, erasure.data_shards, fi.erasure.block_size,
|
||||
is_inline_buffer, direct_inline_commit, write_path.metric_label(), storage_class_config.inline_block(), opts.versioned
|
||||
);
|
||||
if let Ok(mut f) = std::fs::OpenOptions::new().create(true).append(true).open("/tmp/rustfs_inline_debug.log") {
|
||||
let _ = f.write_all(msg.as_bytes());
|
||||
}
|
||||
let _ = std::io::stderr().write_all(msg.as_bytes());
|
||||
}
|
||||
rustfs_io_metrics::record_put_object_path(write_path.metric_label());
|
||||
let writer_setup_stage_start = collect_stage_timing.then(Instant::now);
|
||||
let (mut writers, errors) = if direct_inline_commit {
|
||||
|
||||
@@ -3194,7 +3194,7 @@ impl ECStore {
|
||||
|
||||
// Default return value
|
||||
let mut del_objects = vec![DeletedObject::default(); objects.len()];
|
||||
let mut accounting = vec![None; objects.len()];
|
||||
let accounting = vec![None; objects.len()];
|
||||
|
||||
let mut del_errs = Vec::with_capacity(objects.len());
|
||||
for _ in 0..objects.len() {
|
||||
|
||||
@@ -271,7 +271,7 @@ pub(super) fn resolve_latest_object_info_candidates(
|
||||
.filter(|candidate| latest_candidate_mod_time(candidate) == Some(latest_mod_time))
|
||||
.collect::<Vec<_>>();
|
||||
|
||||
latest_candidates.sort_by(|left, right| right.idx.cmp(&left.idx));
|
||||
latest_candidates.sort_by_key(|candidate| std::cmp::Reverse(candidate.idx));
|
||||
|
||||
let Some(winner) = latest_candidates.first() else {
|
||||
return Err(Error::ErasureReadQuorum);
|
||||
|
||||
@@ -43,9 +43,6 @@ allow-git = [
|
||||
# RustFS fork carrying presigned expiry and constant-time authentication fixes.
|
||||
# owner: rustfs-maintainers review: 2026-10
|
||||
"https://github.com/rustfs/s3s.git",
|
||||
# MiMalloc fork pinned for hotpath allocation counting support.
|
||||
# owner: houseme review: 2026-10
|
||||
"https://github.com/xonatius/mimalloc_rust.git",
|
||||
]
|
||||
|
||||
[bans]
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
# CI gate matrix
|
||||
|
||||
This file is the source of truth for which validation runs on each event, its
|
||||
configured wall-clock budget, and whether it can block a merge. Test taxonomy,
|
||||
naming, and nextest serialization rules remain in [README.md](README.md); e2e
|
||||
membership and counts remain in
|
||||
[e2e-suite-inventory.md](e2e-suite-inventory.md).
|
||||
|
||||
The distinction between **required** and **report-only** is load-bearing:
|
||||
a failing job blocks a merge only when its exact check name is present in the
|
||||
live `main` ruleset. A workflow name, a `merge_group` trigger, or a red PR check
|
||||
does not make a job required by itself.
|
||||
|
||||
## Required merge checks
|
||||
|
||||
The live `main` ruleset (`6436880`) currently requires exactly these contexts:
|
||||
|
||||
| Required context | Producer | Validation |
|
||||
|---|---|---|
|
||||
| `CLA Check` | `.github/workflows/cla.yml` | Contributor agreement |
|
||||
| `Quick Checks` | `.github/workflows/ci.yml` | Formatting and repository guard scripts |
|
||||
| `Test and Lint` | `.github/workflows/ci.yml` | Clippy, workspace nextest excluding `e2e_test`, doctests, and migration proofs |
|
||||
|
||||
For pull requests limited to the paths excluded by the main CI workflow,
|
||||
`.github/workflows/ci-docs-only.yml` reports `Quick Checks` and
|
||||
`Test and Lint` under the same names. It runs the real quick checks and the
|
||||
planning-document guard; it does not claim that Rust compilation or runtime
|
||||
tests ran. Despite the workflow name, these paths also include selected deploy,
|
||||
workflow, and lock files.
|
||||
|
||||
Verify the live rule rather than trusting this snapshot before changing merge
|
||||
policy:
|
||||
|
||||
```bash
|
||||
gh api repos/rustfs/rustfs/rulesets/6436880 \
|
||||
--jq '.rules[] | select(.type == "required_status_checks") | .parameters'
|
||||
```
|
||||
|
||||
The ruleset currently has `strict_required_status_checks_policy=false`.
|
||||
`Continuous Integration` accepts `merge_group` events and runs `e2e-full` for
|
||||
them, but `End-to-End Tests (full merge gate)` is not currently a required
|
||||
context. Therefore the repository is prepared to test a merge-queue SHA, but
|
||||
the workflow alone does not prove that every merge passed that lane.
|
||||
|
||||
## Pull request and merge matrix
|
||||
|
||||
Budgets below are job `timeout-minutes`, not typical runtimes. “Report-only”
|
||||
means the result is visible and actionable but is not in the live required
|
||||
context list.
|
||||
|
||||
| Event | Validation | Budget | Merge status | Reproduction |
|
||||
|---|---|---:|---|---|
|
||||
| PR, non-doc change | `Quick Checks` | 10 min | Required | `make pre-commit` (broader local umbrella) |
|
||||
| PR, non-doc change | `Test and Lint` | 90 min | Required | `cargo nextest run --profile ci --all --exclude e2e_test` |
|
||||
| PR, non-doc change | `Typos` | 10 min | Report-only | `typos` |
|
||||
| PR, non-doc change | `ILM Integration (serial)` | 90 min | Report-only | Use the exact command in `.github/workflows/ci.yml` |
|
||||
| PR, non-doc change | rio-v2 / swift / sftp test-and-lint variants | 90 min each | Report-only | `cargo nextest run` with the workflow's feature set |
|
||||
| PR, non-doc change | `Build RustFS Debug Binary` | 30 min | Report-only; prerequisite for black-box lanes | `cargo build -p rustfs --bins` |
|
||||
| PR, non-doc change | `io_uring Integration (real)` | 30 min | Report-only | `cargo test -p rustfs-ecstore --lib uring_ -- --test-threads=1 --nocapture` |
|
||||
| PR, non-doc change | `End-to-End Tests` (`e2e-smoke` plus `s3s-e2e`) | 30 min | Report-only | `cargo nextest run --profile e2e-smoke -p e2e_test`; then `./scripts/e2e-run.sh ./target/debug/rustfs <data-dir>` |
|
||||
| PR, non-doc change | `S3 Implemented Tests` | 60 min | Report-only | Build `rustfs`, then run `scripts/s3-tests/run.sh` with `DEPLOY_MODE=binary`, `TEST_MODE=single`, and `MAXFAIL=0` |
|
||||
| PR, non-doc change | `S3 Lifecycle Behavior Tests` | 30 min | Report-only | Use the accelerated scanner environment in `.github/workflows/ci.yml` with `scripts/s3-tests/run.sh` |
|
||||
| PR touching dependency or workflow inputs | Cargo Deny / Workflow Pin Report / Dependency Review | 20 / 5 / 30 min | Report-only | `cargo deny check`; `scripts/security/check_workflow_pins.sh` |
|
||||
| PR touching architecture rules or architecture docs | `Architecture Migration Rules` | 10 min | Report-only | `scripts/check_architecture_migration_rules.sh` |
|
||||
| PR touching Nix or workspace manifests | `Nix Build & Check` | 60 min | Report-only | `nix flake check` |
|
||||
| PR limited to main-CI-excluded paths | companion `Quick Checks` and `Test and Lint` | 10 min each | Required | `git diff --check`; `make doc-paths-check` when documentation paths changed |
|
||||
| `merge_group` | Standard CI plus `e2e-full` | 55 min for `e2e-full` | Standard required contexts only; `e2e-full` report-only | `cargo nextest run --profile e2e-full -p e2e_test` |
|
||||
| Push to `main` | Standard CI plus `e2e-full` | 55 min for `e2e-full` | Post-merge detection | Same as `merge_group` |
|
||||
| PR touching fuzz inputs or harness paths | Build plus five 60-second fuzz smoke targets | 60 min build; 30 min per target | Report-only | `MAX_TOTAL_TIME=60 ./scripts/fuzz/run.sh` |
|
||||
| PR touching selected ecstore disk/format paths | `Rename Safety` on Windows | 60 min | Report-only | Run the four `cargo test -p rustfs-ecstore --lib <filter>` commands in `windows-filesystem.yml` on Windows |
|
||||
|
||||
The authoritative e2e filters live in `.config/nextest.toml`; extend a profile
|
||||
instead of adding a second ad-hoc selector. Before a profile runs,
|
||||
`scripts/check_test_wiring.py` compares its exact membership to the committed
|
||||
digest so a silent test drop fails closed.
|
||||
|
||||
## Scheduled and manual validation
|
||||
|
||||
Scheduled lanes are independent fault domains. They do not block a pull
|
||||
request, but their workflow-local gate can fail the run and scheduled failures
|
||||
are routed to the shared failure-issue action. The scheduled-validation
|
||||
watchdog and freshness workflow separately detect incomplete runs and missing
|
||||
schedules.
|
||||
|
||||
| Cadence (UTC unless noted) | Workflow / validation | Budget | Verdict and artifacts | Reproduction |
|
||||
|---|---|---:|---|---|
|
||||
| Daily 02:17 | Fuzz: five nightly corpus targets | 60 min build; 60 min per target | Gate; corpus/crash artifacts, scheduled failure alert | `MAX_TOTAL_TIME=<seconds> ./scripts/fuzz/run.sh` |
|
||||
| Daily 03:17 | MinIO interop (EC + SSE read parity) | 40 min | Gate; scheduled failure alert | Dispatch `minio-interop.yml` or follow its pinned Docker fixture steps |
|
||||
| Daily 04:29 | Replication / cluster-fault / protocol e2e | 45 / 90 / 90 min | Three independent gates; JUnit, membership, and server logs | `cargo nextest run --profile e2e-repl-nightly -p e2e_test`; `--profile e2e-nightly`; `-j 1 --profile e2e-protocols` |
|
||||
| Daily 06:31 | Warp performance A/B | 180 min | Regression budget gate; A/B summaries and server logs | `bash scripts/run_hotpath_warp_abba.sh --help` |
|
||||
| Daily 00:07 Asia/Shanghai (16:07 UTC previous day) | Nightly GNU build and Vault lanes | 150 / 90 / 60 min | Build, live Vault, and HA failover gates | Use the commands and pinned Vault images in `nightly-gnu.yml` |
|
||||
| Daily 03:23 | Security Audit | 20 / 5 min, plus 30 min on PR dependency review | Cargo Deny and workflow-pin gates; scheduled failure alert | `cargo deny check`; `scripts/security/check_workflow_pins.sh` |
|
||||
| Daily 23:47 | Scheduled Validation Freshness | 10 min | Fails when a critical schedule was never created or is stale | Dispatch `scheduled-validation-freshness.yml` |
|
||||
| Sunday 00:11 | Full `Continuous Integration` matrix | Per-job budgets above | Weekly variant coverage, including dormant rio-v2 binary/e2e lanes | Dispatch `ci.yml` |
|
||||
| Sunday 01:13 | Seven-platform build matrix | 150 min per platform | Build/package integrity; scheduled failure alert | Dispatch `build.yml` with an exact platform set |
|
||||
| Sunday 02:19 | Ceph s3-tests full sweep: single and real four-node, four shards each | 180 min per shard | Compatibility gate; report, JUnit, exact node IDs, and server logs | `scripts/s3-tests/run.sh` against an existing single or distributed target |
|
||||
| Sunday 06:41 | Mint | 120 min | **Report-only by design**; per-suite PASS/FAIL/NA and raw `log.json` | Reproduce the pinned Docker sequence in `mint.yml` or dispatch it |
|
||||
| Sunday 07:43 | Workspace line coverage | 120 min | Report-only trend; lcov and JSON retained 90 days | `make coverage` |
|
||||
| Monthly, day 1 06:37 | Runner Hygiene | 15 min | Validates runner ephemerality; scheduled failure alert | Dispatch `runner-hygiene.yml` |
|
||||
|
||||
Manual `workflow_dispatch` exists for the scheduled workflows above. Manual
|
||||
runs are debugging evidence and intentionally do not open scheduled-failure
|
||||
issues. A manual performance run may explicitly allow a known regression; that
|
||||
override must not be treated as an ordinary passing baseline.
|
||||
|
||||
## Release validation
|
||||
|
||||
Release validation is post-merge and tag-driven; it does not substitute for a
|
||||
pull-request gate.
|
||||
|
||||
| Event | Validation | Budget | Result |
|
||||
|---|---|---:|---|
|
||||
| Push to `main` or weekly schedule | `Build and Release` platform matrix | 150 min per platform | Build artifacts for all selected targets; no release publication on a main push |
|
||||
| Valid release or preview tag | `Build and Release` plus asset checks | 150 min per platform | Draft release, checksummed assets, and publish step |
|
||||
| Successful non-preview release-tag build | Docker image build and image scan | 60 min build; 30 min scan | Multi-architecture images plus vulnerability report |
|
||||
| Successful release-tag build | DEB/RPM packaging | 30 min per architecture | Packages and checksum files uploaded to the release |
|
||||
| Successful non-preview release-tag build | Helm template test and package | 30 min build; 30 min publish | Versioned chart and repository index |
|
||||
|
||||
Use an exact preview tag for end-to-end release rehearsal. Manual dispatches
|
||||
are backfill/debug paths and do not prove the automatic `workflow_run` chain.
|
||||
|
||||
## Evidence requirements
|
||||
|
||||
A green check is useful only when it proves the intended behavior ran:
|
||||
|
||||
- Record the exact commit SHA and run URL.
|
||||
- Separate product failure from runner prerequisites, service readiness, and
|
||||
cancellation. Repair the precondition, then rerun the exact workload.
|
||||
- Preserve membership manifests, JUnit, raw compatibility logs, seeds, and
|
||||
server logs where the workflow provides them.
|
||||
- For a bug fix or a new fault checker, provide sensitivity evidence: the old
|
||||
behavior or an intentional mutation must fail the new oracle, and the fixed
|
||||
behavior must pass it.
|
||||
- Never promote a report-only lane to required from one green run. Require at
|
||||
least 14 days and 30 representative pull requests with at least 99% complete
|
||||
execution, then update the ruleset and this table together.
|
||||
|
||||
## Change checklist
|
||||
|
||||
Update this file in the same pull request when any of these change:
|
||||
|
||||
- workflow triggers, job names, timeouts, or nextest profile ownership;
|
||||
- required status contexts or strict/merge-queue policy;
|
||||
- scheduled cadence, alert routing, artifact contract, or local reproduction;
|
||||
- report-only versus gating semantics.
|
||||
|
||||
Do not copy per-module test counts here. Update
|
||||
[e2e-suite-inventory.md](e2e-suite-inventory.md) and its enforced membership
|
||||
digest instead.
|
||||
+2
-2
@@ -336,13 +336,13 @@ opentelemetry = { workspace = true }
|
||||
tracing-opentelemetry = { workspace = true }
|
||||
# Data structures
|
||||
hashbrown = { workspace = true, features = ["serde", "rayon"] }
|
||||
mimalloc = { workspace = true }
|
||||
rustfs-mimalloc = { workspace = true }
|
||||
|
||||
[target.'cfg(target_os = "linux")'.dependencies]
|
||||
libsystemd.workspace = true
|
||||
|
||||
[target.'cfg(not(target_os = "windows"))'.dependencies]
|
||||
libmimalloc-sys.workspace = true
|
||||
rustfs-mimalloc-sys.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
uuid = { workspace = true, features = ["v4", "v5", "fast-rng", "macro-diagnostics"] }
|
||||
|
||||
@@ -70,6 +70,12 @@ const SITE_REPLICATION_EDIT_ROUTE: &str = "/rustfs/admin/v3/site-replication/edi
|
||||
const SITE_REPLICATION_RESYNC_ROUTE: &str = "/rustfs/admin/v3/site-replication/resync/op";
|
||||
const SITE_REPLICATION_REPAIR_ROUTE: &str = "/rustfs/admin/v3/site-replication/repair";
|
||||
const SITE_REPLICATION_REPAIR_STATUS_ROUTE: &str = "/rustfs/admin/v3/site-replication/repair/status";
|
||||
const IAM_POLICY_ATTACH_ROUTE: &str = "/rustfs/admin/v3/idp/builtin/policy/attach";
|
||||
const IAM_POLICY_DETACH_ROUTE: &str = "/rustfs/admin/v3/idp/builtin/policy/detach";
|
||||
const IAM_POLICY_ENTITIES_ROUTE: &str = "/rustfs/admin/v3/idp/builtin/policy-entities";
|
||||
const IAM_ACCESS_KEYS_BULK_ROUTE: &str = "/rustfs/admin/v3/list-access-keys-bulk";
|
||||
const IAM_ACCESS_KEYS_BULK_LDAP_ROUTE: &str = "/rustfs/admin/v3/idp/ldap/list-access-keys-bulk";
|
||||
const IAM_ACCESS_KEYS_BULK_OPENID_ROUTE: &str = "/rustfs/admin/v3/idp/openid/list-access-keys-bulk";
|
||||
|
||||
macro_rules! log_system_request_rejected {
|
||||
($operation:expr, $reason:expr) => {
|
||||
@@ -661,9 +667,24 @@ pub struct RuntimeCapabilitiesSummary {
|
||||
pub manual_transition_jobs: CapabilityStatus,
|
||||
}
|
||||
|
||||
/// One named admin capability advertised to management clients
|
||||
/// (rustfs/backlog#1900). `name` is a cross-repo wire contract: the rc
|
||||
/// client gates commands on these exact strings (see rustfs/cli
|
||||
/// `IAM_POLICY_DETACH_CAPABILITY` etc.), so entries may be added but
|
||||
/// existing names must never be renamed or removed.
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
|
||||
pub struct AdvertisedAdminCapability {
|
||||
pub name: &'static str,
|
||||
pub status: CapabilityStatus,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize)]
|
||||
pub struct RuntimeCapabilitiesResponse {
|
||||
pub summary: RuntimeCapabilitiesSummary,
|
||||
/// Additive field: absent in responses from older servers, so clients
|
||||
/// must treat a missing list as "no dynamic advertisement" and fall
|
||||
/// back to their pinned per-version contract.
|
||||
pub advertised: Vec<AdvertisedAdminCapability>,
|
||||
pub replication: ReplicationCapabilities,
|
||||
pub manual_transition_jobs: ManualTransitionJobCapabilities,
|
||||
pub diagnostic_probes: DiagnosticProbeCapabilities,
|
||||
@@ -986,6 +1007,7 @@ pub(crate) async fn build_runtime_capabilities_response()
|
||||
|
||||
Ok(RuntimeCapabilitiesResponse {
|
||||
summary,
|
||||
advertised: advertised_admin_capabilities(),
|
||||
replication: ReplicationCapabilities::current(),
|
||||
manual_transition_jobs: ManualTransitionJobCapabilities::current(),
|
||||
diagnostic_probes: DiagnosticProbeCapabilities::current(),
|
||||
@@ -1077,6 +1099,23 @@ fn admin_route_capability(method: HttpMethod, path: &str) -> CapabilityStatus {
|
||||
admin_route_capability_from_inventory(method, path, ADMIN_ROUTE_POLICY_SPECS, DEFERRED_ADMIN_ROUTE_POLICIES)
|
||||
}
|
||||
|
||||
fn advertised_admin_capabilities() -> Vec<AdvertisedAdminCapability> {
|
||||
[
|
||||
("admin.iam.policy-attach", HttpMethod::Post, IAM_POLICY_ATTACH_ROUTE),
|
||||
("admin.iam.policy-detach", HttpMethod::Post, IAM_POLICY_DETACH_ROUTE),
|
||||
("admin.iam.policy-entities", HttpMethod::Get, IAM_POLICY_ENTITIES_ROUTE),
|
||||
("admin.iam.access-keys-bulk", HttpMethod::Get, IAM_ACCESS_KEYS_BULK_ROUTE),
|
||||
("admin.iam.access-keys-bulk.ldap", HttpMethod::Get, IAM_ACCESS_KEYS_BULK_LDAP_ROUTE),
|
||||
("admin.iam.access-keys-bulk.openid", HttpMethod::Get, IAM_ACCESS_KEYS_BULK_OPENID_ROUTE),
|
||||
]
|
||||
.into_iter()
|
||||
.map(|(name, method, route)| AdvertisedAdminCapability {
|
||||
name,
|
||||
status: admin_route_capability(method, route),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn admin_route_capability_from_inventory(
|
||||
method: HttpMethod,
|
||||
path: &str,
|
||||
@@ -1239,6 +1278,48 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
/// Wire-contract pin (rustfs/backlog#1900): the rc client keys its
|
||||
/// command gates on these exact capability names, and parses each
|
||||
/// entry as `{name, status: {state, reason?}}`. Renaming or dropping
|
||||
/// a name silently disables the corresponding rc command.
|
||||
#[tokio::test]
|
||||
async fn runtime_capabilities_response_advertises_iam_capabilities() {
|
||||
let response = build_runtime_capabilities_response()
|
||||
.await
|
||||
.expect("runtime capabilities response should build");
|
||||
|
||||
let expected_supported = [
|
||||
"admin.iam.policy-attach",
|
||||
"admin.iam.policy-detach",
|
||||
"admin.iam.policy-entities",
|
||||
"admin.iam.access-keys-bulk",
|
||||
"admin.iam.access-keys-bulk.ldap",
|
||||
"admin.iam.access-keys-bulk.openid",
|
||||
];
|
||||
for name in expected_supported {
|
||||
let entry = response
|
||||
.advertised
|
||||
.iter()
|
||||
.find(|capability| capability.name == name)
|
||||
.unwrap_or_else(|| panic!("{name} must be advertised"));
|
||||
assert_eq!(entry.status.state, CapabilityState::Supported, "{name} must be supported");
|
||||
}
|
||||
|
||||
let mut names: Vec<&str> = response.advertised.iter().map(|capability| capability.name).collect();
|
||||
let total = names.len();
|
||||
names.sort_unstable();
|
||||
names.dedup();
|
||||
assert_eq!(names.len(), total, "advertised capability names must be unique");
|
||||
|
||||
let serialized = serde_json::to_value(&response).expect("response should serialize");
|
||||
let advertised = serialized["advertised"].as_array().expect("advertised must be an array");
|
||||
let detach = advertised
|
||||
.iter()
|
||||
.find(|entry| entry["name"] == "admin.iam.policy-detach")
|
||||
.expect("serialized detach entry must exist");
|
||||
assert_eq!(detach["status"]["state"], "supported");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn runtime_capabilities_response_reports_missing_topology_before_storage_init() {
|
||||
let response = build_runtime_capabilities_response()
|
||||
|
||||
@@ -369,14 +369,8 @@ pub fn allocator_reclaim_controller_snapshot(ctx: &CancellationToken) -> Allocat
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "windows"))]
|
||||
#[allow(unsafe_code)]
|
||||
fn collect_allocator_memory(force: bool) -> Result<(), String> {
|
||||
// SAFETY: `mi_collect` is provided by the active global allocator backend
|
||||
// on this target family. It is explicitly intended to reclaim retained
|
||||
// pages/segments and does not require additional invariants from the caller.
|
||||
unsafe {
|
||||
libmimalloc_sys::mi_collect(force);
|
||||
}
|
||||
rustfs_mimalloc::MiMalloc::collect(force);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
|
||||
@@ -16,8 +16,7 @@
|
||||
|
||||
use super::storage_api::admin_usecase::admin::get_server_info;
|
||||
use super::storage_api::admin_usecase::capacity::{
|
||||
DecommissionUnresolvedEntry, PoolDecommissionInfo, PoolStatus, RebalStatus, get_total_usable_capacity,
|
||||
get_total_usable_capacity_free,
|
||||
PoolDecommissionInfo, PoolStatus, RebalStatus, get_total_usable_capacity, get_total_usable_capacity_free,
|
||||
};
|
||||
use super::storage_api::admin_usecase::contract::StorageAdminApi;
|
||||
use super::storage_api::admin_usecase::contract::bucket::{BucketOperations as _, BucketOptions};
|
||||
@@ -108,8 +107,6 @@ pub struct AdminPoolDecommissionInfo {
|
||||
pub bytes_failed: usize,
|
||||
#[serde(rename = "waitingReason")]
|
||||
pub waiting_reason: Option<String>,
|
||||
#[serde(rename = "unresolvedEntries", skip_serializing_if = "Vec::is_empty")]
|
||||
pub unresolved_entries: Vec<DecommissionUnresolvedEntry>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, serde::Serialize)]
|
||||
@@ -622,7 +619,6 @@ impl DefaultAdminUsecase {
|
||||
bytes_done: info.bytes_done,
|
||||
bytes_failed: info.bytes_failed,
|
||||
waiting_reason,
|
||||
unresolved_entries: info.unresolved_entries,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -680,7 +676,7 @@ impl DefaultAdminUsecase {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::super::storage_api::admin_usecase::capacity::{DecommissionUnresolvedEntry, PoolDecommissionInfo, PoolStatus};
|
||||
use super::super::storage_api::admin_usecase::capacity::{PoolDecommissionInfo, PoolStatus};
|
||||
use super::*;
|
||||
use time::OffsetDateTime;
|
||||
use tracing_subscriber::{Layer, Registry, layer::Context, prelude::*};
|
||||
@@ -991,17 +987,6 @@ mod tests {
|
||||
items_decommission_failed: 1,
|
||||
bytes_done: 1024,
|
||||
bytes_failed: 64,
|
||||
unresolved_entries: vec![DecommissionUnresolvedEntry {
|
||||
bucket: "bucket-a".to_string(),
|
||||
object: "prefix/unresolved.txt".to_string(),
|
||||
pool_index: 3,
|
||||
set_index: 1,
|
||||
source_generation: OffsetDateTime::UNIX_EPOCH,
|
||||
candidate_count: 2,
|
||||
disk_error_count: 1,
|
||||
observed_at: OffsetDateTime::UNIX_EPOCH,
|
||||
reason: "metadata_resolution_failed".to_string(),
|
||||
}],
|
||||
..Default::default()
|
||||
}),
|
||||
},
|
||||
@@ -1025,13 +1010,6 @@ mod tests {
|
||||
assert_eq!(value["decommissionInfo"]["objectsDecommissionedFailed"], 1);
|
||||
assert_eq!(value["decommissionInfo"]["bytesDecommissioned"], 1024);
|
||||
assert_eq!(value["decommissionInfo"]["bytesDecommissionedFailed"], 64);
|
||||
assert_eq!(value["decommissionInfo"]["unresolvedEntries"][0]["bucket"], "bucket-a");
|
||||
assert_eq!(value["decommissionInfo"]["unresolvedEntries"][0]["object"], "prefix/unresolved.txt");
|
||||
assert_eq!(
|
||||
value["decommissionInfo"]["unresolvedEntries"][0]["sourceGeneration"],
|
||||
"1970-01-01T00:00:00Z"
|
||||
);
|
||||
assert_eq!(value["decommissionInfo"]["unresolvedEntries"][0]["reason"], "metadata_resolution_failed");
|
||||
assert_eq!(value["decommissionInfo"]["waitingReason"], "queued");
|
||||
}
|
||||
|
||||
|
||||
@@ -31,7 +31,6 @@ pub(crate) mod admin {
|
||||
}
|
||||
|
||||
pub(crate) mod capacity {
|
||||
pub(crate) type DecommissionUnresolvedEntry = crate::storage::storage_api::ecstore_capacity::DecommissionUnresolvedEntry;
|
||||
pub(crate) type PoolDecommissionInfo = crate::storage::storage_api::ecstore_capacity::PoolDecommissionInfo;
|
||||
pub(crate) type PoolStatus = crate::storage::storage_api::ecstore_capacity::PoolStatus;
|
||||
pub(crate) type RebalStatus = crate::storage::storage_api::ecstore_rebalance::RebalStatus;
|
||||
|
||||
+10
-8
@@ -26,22 +26,22 @@ struct MiMallocAllocator;
|
||||
unsafe impl GlobalAlloc for MiMallocAllocator {
|
||||
unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
|
||||
// SAFETY: the caller upholds GlobalAlloc's contract for layout.
|
||||
unsafe { mimalloc::MiMalloc.alloc(layout) }
|
||||
unsafe { rustfs_mimalloc::MiMalloc.alloc(layout) }
|
||||
}
|
||||
|
||||
unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 {
|
||||
// SAFETY: the caller upholds GlobalAlloc's contract for layout.
|
||||
unsafe { mimalloc::MiMalloc.alloc_zeroed(layout) }
|
||||
unsafe { rustfs_mimalloc::MiMalloc.alloc_zeroed(layout) }
|
||||
}
|
||||
|
||||
unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) {
|
||||
// SAFETY: ptr and layout came from this allocator and are forwarded unchanged.
|
||||
unsafe { mimalloc::MiMalloc.dealloc(ptr, layout) }
|
||||
unsafe { rustfs_mimalloc::MiMalloc.dealloc(ptr, layout) }
|
||||
}
|
||||
|
||||
unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 {
|
||||
// SAFETY: ptr and layout came from this allocator and are forwarded unchanged.
|
||||
unsafe { mimalloc::MiMalloc.realloc(ptr, layout, new_size) }
|
||||
unsafe { rustfs_mimalloc::MiMalloc.realloc(ptr, layout, new_size) }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -51,7 +51,7 @@ static GLOBAL: hotpath::CountingAllocator<MiMallocAllocator> = hotpath::Counting
|
||||
|
||||
#[cfg(not(all(feature = "hotpath", feature = "hotpath-alloc")))]
|
||||
#[global_allocator]
|
||||
static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc;
|
||||
static GLOBAL: rustfs_mimalloc::MiMalloc = rustfs_mimalloc::MiMalloc;
|
||||
|
||||
fn main() {
|
||||
let _hotpath_guard = hotpath::HotpathGuardBuilder::new("main").build();
|
||||
@@ -71,8 +71,9 @@ mod tests {
|
||||
allocation.extend_from_slice(&[7_u8; 64]);
|
||||
|
||||
assert_eq!(allocation.len(), 64);
|
||||
let heap = rustfs_mimalloc::heap::Heap::main();
|
||||
// SAFETY: the live Vec pointer is valid to inspect for heap ownership.
|
||||
assert!(unsafe { libmimalloc_sys::mi_is_in_heap_region(allocation.as_ptr().cast()) });
|
||||
assert!(unsafe { heap.contains(allocation.as_ptr()) });
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -85,12 +86,13 @@ mod tests {
|
||||
let layout = Layout::from_size_align(32, 8).expect("valid test allocation layout");
|
||||
let grown_layout = Layout::from_size_align(64, 8).expect("valid grown test allocation layout");
|
||||
let allocator = super::MiMallocAllocator;
|
||||
let heap = rustfs_mimalloc::heap::Heap::main();
|
||||
|
||||
// SAFETY: The pointer is checked for null before use and later released
|
||||
// through the same allocator with the corresponding layout.
|
||||
let ptr = unsafe { allocator.alloc_zeroed(layout) };
|
||||
assert!(!ptr.is_null());
|
||||
assert!(unsafe { libmimalloc_sys::mi_is_in_heap_region(ptr.cast()) });
|
||||
assert!(unsafe { heap.contains(ptr) });
|
||||
assert!(unsafe { std::slice::from_raw_parts(ptr, 32).iter().all(|byte| *byte == 0) });
|
||||
|
||||
// SAFETY: `ptr` was allocated by `allocator` with `layout`; on failure
|
||||
@@ -102,7 +104,7 @@ mod tests {
|
||||
panic!("mimalloc realloc failed in allocator smoke test");
|
||||
}
|
||||
|
||||
assert!(unsafe { libmimalloc_sys::mi_is_in_heap_region(grown_ptr.cast()) });
|
||||
assert!(unsafe { heap.contains(grown_ptr) });
|
||||
// SAFETY: `grown_ptr` was reallocated by `allocator` and is released
|
||||
// with the matching grown layout.
|
||||
unsafe { allocator.dealloc(grown_ptr, grown_layout) };
|
||||
|
||||
@@ -17,10 +17,7 @@ use rustfs_io_metrics::{
|
||||
record_cpu_usage, record_memory_usage, record_process_memory_split,
|
||||
};
|
||||
use serde::Serialize;
|
||||
#[cfg(any(test, not(target_os = "windows")))]
|
||||
use serde_json::Value;
|
||||
#[cfg(not(target_os = "windows"))]
|
||||
use std::ffi::CStr;
|
||||
use std::path::Path;
|
||||
use std::sync::{Arc, Mutex, OnceLock};
|
||||
use std::time::Duration;
|
||||
@@ -231,7 +228,18 @@ fn read_cgroup_memory_snapshot() -> Option<CgroupMemorySnapshot> {
|
||||
read_cgroup_v2().or_else(read_cgroup_v1)
|
||||
}
|
||||
|
||||
#[cfg(any(test, not(target_os = "windows")))]
|
||||
fn read_allocator_memory_snapshot() -> Option<AllocatorMemorySnapshot> {
|
||||
let json = rustfs_mimalloc::MiMalloc::stats_json();
|
||||
if json.is_empty() {
|
||||
return None;
|
||||
}
|
||||
let observation = parse_mimalloc_stats_json(&json)?;
|
||||
Some(AllocatorMemorySnapshot {
|
||||
backend: crate::allocator_reclaim::allocator_backend(),
|
||||
observation,
|
||||
})
|
||||
}
|
||||
|
||||
fn numeric_json_value(value: &Value) -> Option<u64> {
|
||||
match value {
|
||||
Value::Number(number) => number
|
||||
@@ -242,7 +250,6 @@ fn numeric_json_value(value: &Value) -> Option<u64> {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(test, not(target_os = "windows")))]
|
||||
fn numeric_json_field(value: &Value, field: &str) -> Option<u64> {
|
||||
match value {
|
||||
Value::Object(fields) => fields
|
||||
@@ -254,7 +261,6 @@ fn numeric_json_field(value: &Value, field: &str) -> Option<u64> {
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(test, not(target_os = "windows")))]
|
||||
fn mimalloc_stat_field(value: &Value, metric: &str, field: &str) -> Option<u64> {
|
||||
match value {
|
||||
Value::Object(fields) => {
|
||||
@@ -271,12 +277,10 @@ fn mimalloc_stat_field(value: &Value, metric: &str, field: &str) -> Option<u64>
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(test, not(target_os = "windows")))]
|
||||
fn mimalloc_stat_current(value: &Value, metric: &str) -> Option<u64> {
|
||||
mimalloc_stat_field(value, metric, "current")
|
||||
}
|
||||
|
||||
#[cfg(any(test, not(target_os = "windows")))]
|
||||
fn mimalloc_stat_sum(value: &Value, metrics: &[&str], field: &str) -> Option<u64> {
|
||||
metrics
|
||||
.iter()
|
||||
@@ -285,7 +289,6 @@ fn mimalloc_stat_sum(value: &Value, metrics: &[&str], field: &str) -> Option<u64
|
||||
.filter(|value| *value > 0)
|
||||
}
|
||||
|
||||
#[cfg(any(test, not(target_os = "windows")))]
|
||||
fn parse_mimalloc_stats_json(stats_json: &str) -> Option<AllocatorMemoryObservation> {
|
||||
let value = serde_json::from_str::<Value>(stats_json).ok()?;
|
||||
let malloc_metrics = ["malloc_normal", "malloc_huge"];
|
||||
@@ -312,33 +315,6 @@ fn parse_mimalloc_stats_json(stats_json: &str) -> Option<AllocatorMemoryObservat
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(target_os = "windows"))]
|
||||
#[allow(unsafe_code)]
|
||||
fn read_allocator_memory_snapshot() -> Option<AllocatorMemorySnapshot> {
|
||||
// SAFETY: `mi_stats_get_json` returns a null-terminated JSON buffer owned by
|
||||
// mimalloc when called with a null input buffer. The mimalloc API requires
|
||||
// freeing that buffer with `mi_free`; parsing finishes before the buffer is freed.
|
||||
let observation = unsafe {
|
||||
let stats_ptr = libmimalloc_sys::mi_stats_get_json(0, std::ptr::null_mut());
|
||||
if stats_ptr.is_null() {
|
||||
return None;
|
||||
}
|
||||
|
||||
let observation = CStr::from_ptr(stats_ptr).to_str().ok().and_then(parse_mimalloc_stats_json);
|
||||
libmimalloc_sys::mi_free(stats_ptr.cast());
|
||||
observation?
|
||||
};
|
||||
Some(AllocatorMemorySnapshot {
|
||||
backend: crate::allocator_reclaim::allocator_backend(),
|
||||
observation,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(target_os = "windows")]
|
||||
fn read_allocator_memory_snapshot() -> Option<AllocatorMemorySnapshot> {
|
||||
None
|
||||
}
|
||||
|
||||
fn configured_memory_observability_interval_secs() -> u64 {
|
||||
rustfs_utils::get_env_u64(ENV_MEMORY_OBSERVABILITY_INTERVAL_SECS, DEFAULT_MEMORY_OBSERVABILITY_INTERVAL_SECS).max(1)
|
||||
}
|
||||
@@ -566,6 +542,13 @@ mod tests {
|
||||
assert_eq!(parse_mimalloc_stats_json(r#"{ "allocator": "unknown" }"#), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_allocator_memory_snapshot_uses_mimalloc_stats_json() {
|
||||
let snapshot = super::read_allocator_memory_snapshot();
|
||||
#[cfg(not(target_os = "windows"))]
|
||||
assert!(snapshot.is_some(), "allocator snapshot should be available on non-Windows");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn memory_observability_snapshot_reports_disabled_when_metrics_are_disabled() {
|
||||
let snapshot = build_memory_observability_status_snapshot(false, 15, false);
|
||||
|
||||
@@ -398,7 +398,7 @@ pub(crate) mod ecstore_bucket {
|
||||
|
||||
pub(crate) mod ecstore_capacity {
|
||||
pub(crate) use rustfs_ecstore::api::capacity::{
|
||||
DecommissionUnresolvedEntry, PoolDecommissionInfo, PoolStatus, get_total_usable_capacity, get_total_usable_capacity_free,
|
||||
PoolDecommissionInfo, PoolStatus, get_total_usable_capacity, get_total_usable_capacity_free,
|
||||
is_reserved_or_invalid_bucket,
|
||||
};
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user