Compare commits

..

5 Commits

Author SHA1 Message Date
houseme 37d79c9740 Merge branch 'main' into overtrue/activate-vault-configured-e2e 2026-08-23 12:08:38 +08:00
houseme 648d5166e2 feat(allocator): replace mimalloc/libmimalloc-sys with rustfs-mimalloc/rustfs-mimalloc-sys (#6404)
Replace the upstream xonatius/mimalloc_rust.git fork (mimalloc + libmimalloc-sys)
with the published rustfs-mimalloc (v0.5.0) and rustfs-mimalloc-sys (v0.5.0) crates
from crates.io.

The new crates are based on mimalloc V3 (v3.5.0) and provide:
- MiMalloc global allocator with safe API (collect, stats_json, process_info)
- Heap management and arena operations (heap module)
- Full FFI bindings to mimalloc V3

Changes:
- Workspace deps: mimalloc + libmimalloc-sys (git) → rustfs-mimalloc + rustfs-mimalloc-sys (crates.io)
- allocator_reclaim.rs: libmimalloc_sys::mi_collect → rustfs_mimalloc::MiMalloc::collect
- memory_observability.rs: raw FFI mi_stats_get_json → MiMalloc::stats_json()
- main.rs: heap ownership tests use Heap::contains() (V3 API)
- deny.toml: remove xonatius/mimalloc_rust.git from allow-git

Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-23 12:07:25 +08:00
houseme 84eb5aebef fix(ecstore): remove inline write debug noise (#6408)
* fix(ecstore): remove inline write debug noise

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(ecstore): satisfy warning-as-error lints

Co-Authored-By: heihutu <heihutu@gmail.com>

---------

Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-23 12:07:20 +08:00
overtrue 641f4d580b test(e2e): bind Vault selection to Linux listing 2026-08-23 08:52:46 +08:00
overtrue 7747582d83 test(e2e): activate configured Vault roundtrip 2026-08-23 06:33:04 +08:00
17 changed files with 180 additions and 1355 deletions
+2 -2
View File
@@ -1,2 +1,2 @@
sha256-darwin=9f767b37ed8b1c82da62ea441462d75487785c8086e56f08fb6f6cd89c6e2e52
sha256-linux=fbdaf42b220958d4b1e8880e0f8b5a7992d38e21051bb60596dd4538424757d6
sha256-darwin=52a05fdfae8bcf6f5828cc2b1e91b2a268139d3f7e1fc47d7b875b55fffb3995
sha256-linux=a22d8af72e250595ac4445e8c880f3f9706202e09ed196e5b7baac632dead8d8
+4 -3
View File
@@ -107,9 +107,10 @@ filter = 'package(e2e_test) & test(/^inline_fast_path_cluster_test::/)'
test-group = 'e2e-inline-boundaries'
# Vault KMS tests share the fixed dev-server port 8200. serial_test's #[serial]
# does not cross nextest process boundaries, so keep these tests in one group.
# does not cross nextest process boundaries, so keep every Vault-backed test in
# one group.
[[profile.default.overrides]]
filter = 'package(e2e_test) & test(/^kms::kms_vault_test::/)'
filter = 'package(e2e_test) & (test(/^kms::kms_vault_test::/) | test(/^kms::configured_roundtrip_test::test_configured_vault_kms_admin_and_versioned_cleanup$/))'
test-group = 'e2e-vault'
# ---------------------------------------------------------------------------
@@ -443,5 +444,5 @@ filter = 'package(e2e_test) & test(/^inline_fast_path_cluster_test::/)'
test-group = 'e2e-inline-boundaries'
[[profile.e2e-full.overrides]]
filter = 'package(e2e_test) & test(/^kms::kms_vault_test::/)'
filter = 'package(e2e_test) & (test(/^kms::kms_vault_test::/) | test(/^kms::configured_roundtrip_test::test_configured_vault_kms_admin_and_versioned_cleanup$/))'
test-group = 'e2e-vault'
Generated
+22 -27
View File
@@ -1858,9 +1858,9 @@ dependencies = [
[[package]]
name = "cc"
version = "1.4.3"
version = "1.4.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d"
checksum = "0ad534f4357a5264cce5019c989cf66a4f0dc4e0d1b1d15f8aacec0ff7360273"
dependencies = [
"find-msvc-tools",
"jobserver",
@@ -2522,12 +2522,6 @@ dependencies = [
"subtle",
]
[[package]]
name = "cty"
version = "0.2.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b365fabc795046672053e29c954733ec3b05e4be654ab130fe8f1f94d7051f35"
[[package]]
name = "curve25519-dalek"
version = "4.1.3"
@@ -5988,15 +5982,6 @@ version = "0.2.16"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b6d2cec3eae94f9f509c767b45932f1ada8350c4bdb85af2fcab4a3c14807981"
[[package]]
name = "libmimalloc-sys"
version = "0.1.49"
source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11#6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11"
dependencies = [
"cc",
"cty",
]
[[package]]
name = "libredox"
version = "0.1.20"
@@ -6397,14 +6382,6 @@ dependencies = [
"synstructure 0.13.2",
]
[[package]]
name = "mimalloc"
version = "0.1.52"
source = "git+https://github.com/xonatius/mimalloc_rust.git?rev=6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11#6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11"
dependencies = [
"libmimalloc-sys",
]
[[package]]
name = "mime"
version = "0.3.17"
@@ -9162,13 +9139,11 @@ dependencies = [
"insta",
"jiff",
"libc",
"libmimalloc-sys",
"libsystemd",
"matchit 0.9.2",
"md-5 0.11.0",
"metrics",
"metrics-util",
"mimalloc",
"mime_guess",
"opentelemetry",
"opentelemetry_sdk",
@@ -9204,6 +9179,8 @@ dependencies = [
"rustfs-lock",
"rustfs-log-analyzer",
"rustfs-madmin",
"rustfs-mimalloc",
"rustfs-mimalloc-sys",
"rustfs-notify",
"rustfs-object-capacity",
"rustfs-object-data-cache",
@@ -9875,6 +9852,24 @@ dependencies = [
"tokio",
]
[[package]]
name = "rustfs-mimalloc"
version = "0.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a406f4aa07084301d485beec873af6dccc8e3f8762da244743df92038b1db1a6"
dependencies = [
"rustfs-mimalloc-sys",
]
[[package]]
name = "rustfs-mimalloc-sys"
version = "0.5.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c3051b819175f58445d4c369a72f0ab88149f3885ba8bea2aff3be01f53fe7cd"
dependencies = [
"cc",
]
[[package]]
name = "rustfs-notify"
version = "1.0.0-rc.3"
+2 -2
View File
@@ -350,8 +350,8 @@ russh-sftp = "2.4.0"
dav-server = "0.11.0"
# Performance Analysis and Memory Profiling
mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" }
libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11", features = ["extended"] }
rustfs-mimalloc = { version = "0.5.0" }
rustfs-mimalloc-sys = { version = "0.5.0" }
hotpath = { version = "0.23.3", default-features = false }
# Snapshot testing for output format regression detection
insta = { version = "1.48" }
@@ -432,7 +432,6 @@ async fn test_configured_local_kms_admin_and_versioned_cleanup() -> TestResult {
}
#[tokio::test]
#[ignore = "requires a Vault binary"]
async fn test_configured_vault_kms_admin_and_versioned_cleanup() -> TestResult {
let mut env = VaultTestEnvironment::new().await?;
env.start_vault().await?;
File diff suppressed because it is too large Load Diff
+1 -9
View File
@@ -4736,15 +4736,7 @@ impl SetDisks {
achieved: 0,
});
}
// Rebuilt tiered metadata starts with index zero, but shuffling validates
// each source slot before assigning the shuffled index below.
let parts_metadata: Vec<FileInfo> = (0..disks.len())
.map(|disk_index| {
let mut part = fi.clone();
part.erasure.index = fi.erasure.distribution[disk_index];
part
})
.collect();
let parts_metadata = vec![fi.clone(); disks.len()];
let (shuffle_disks, parts_metadata) = Self::shuffle_disks_and_parts_metadata(&disks, &parts_metadata, &fi);
let mut errs = Vec::with_capacity(shuffle_disks.len());
+1 -14
View File
@@ -2124,26 +2124,13 @@ impl SetDisks {
let put_object_size = known_put_object_storage_size(data.size());
let shard_file_size_raw = erasure.shard_file_size(put_object_size);
let is_inline_buffer =
storage_class_config.should_inline(shard_file_size_raw, erasure.data_shards, opts.versioned);
let is_inline_buffer = storage_class_config.should_inline(shard_file_size_raw, erasure.data_shards, opts.versioned);
let collect_stage_timing = rustfs_io_metrics::put_stage_metrics_enabled() || issue3031_diag_enabled();
let shard_file_size = shard_file_size_raw;
let shard_size = erasure.shard_size();
let write_path = classify_put_write_path(is_inline_buffer, put_object_size, fi.erasure.block_size);
let direct_inline_commit = matches!(write_path, SmallWritePath::Inline);
{
use std::io::Write;
let msg = format!(
"INLINE_DEBUG: bucket={} obj={} size={} shard_fs={} ds={} bs={} inline={} direct={} path={} iblock={} ver={}\n",
bucket, object, put_object_size, shard_file_size_raw, erasure.data_shards, fi.erasure.block_size,
is_inline_buffer, direct_inline_commit, write_path.metric_label(), storage_class_config.inline_block(), opts.versioned
);
if let Ok(mut f) = std::fs::OpenOptions::new().create(true).append(true).open("/tmp/rustfs_inline_debug.log") {
let _ = f.write_all(msg.as_bytes());
}
let _ = std::io::stderr().write_all(msg.as_bytes());
}
rustfs_io_metrics::record_put_object_path(write_path.metric_label());
let writer_setup_stage_start = collect_stage_timing.then(Instant::now);
let (mut writers, errors) = if direct_inline_commit {
+1 -611
View File
@@ -626,7 +626,7 @@ mod tests {
io::Cursor,
sync::{
Arc,
atomic::{AtomicBool, AtomicUsize, Ordering},
atomic::{AtomicBool, Ordering},
},
time::Duration,
};
@@ -1307,85 +1307,6 @@ mod tests {
});
}
const DECOMMISSION_TEST_FAULT_STAGE_DELETE_MARKER: &str = "delete_marker_copy";
const DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT: &str = "migrate_object";
#[cfg(feature = "test-util")]
const DECOMMISSION_TEST_FAULT_STAGE_TIERED: &str = "decommission_tiered_object";
async fn seed_decommission_source(
store: &Arc<crate::store::ECStore>,
bucket: &str,
object: &str,
body: Vec<u8>,
opts: &ObjectOptions,
) {
let mut reader = PutObjReader::from_vec(body);
store.pools[0]
.put_object(bucket, object, &mut reader, opts)
.await
.expect("seed decommission source object");
}
async fn run_decommission_entry_retry_test(
store: &Arc<crate::store::ECStore>,
rx: CancellationToken,
bucket: &str,
object: &str,
expected_bucket_incarnation_id: Option<uuid::Uuid>,
source_changed_exhaustions: Arc<AtomicUsize>,
) -> crate::error::Result<()> {
let source_set = store.pools[0].get_disks_by_key(object);
store
.decommission_entry_with_retry_state_for_test(
rx,
0,
MetaCacheEntry {
name: object.to_string(),
..Default::default()
},
bucket.to_string(),
source_set,
expected_bucket_incarnation_id,
source_changed_exhaustions,
)
.await
}
async fn read_decommission_target_body(
store: &Arc<crate::store::ECStore>,
bucket: &str,
object: &str,
opts: &ObjectOptions,
) -> Vec<u8> {
let mut reader = store.pools[1]
.get_object_reader(bucket, object, None, HeaderMap::new(), opts)
.await
.expect("read decommission target object");
let mut body = Vec::new();
reader
.stream
.read_to_end(&mut body)
.await
.expect("drain decommission target body");
body
}
async fn assert_decommission_source_absent(
store: &Arc<crate::store::ECStore>,
bucket: &str,
object: &str,
opts: &ObjectOptions,
) {
let err = store.pools[0]
.get_object_info(bucket, object, opts)
.await
.expect_err("decommission source must be retained until target commit, then removed");
assert!(
matches!(err, StorageError::ObjectNotFound(_, _) | StorageError::VersionNotFound(_, _, _)),
"unexpected decommission source result: {err:?}"
);
}
async fn write_decommission_test_multipart_source(
store: &Arc<crate::store::ECStore>,
pool_idx: usize,
@@ -3181,537 +3102,6 @@ mod tests {
shutdown.cancel();
}
#[test]
#[serial_test::serial(storage_class_env)]
fn decommission_entry_retries_source_changed_without_canceling_other_bucket() {
let handle = std::thread::Builder::new()
.name("decommission_entry_retries_source_changed_without_canceling_other_bucket".to_string())
.stack_size(32 * 1024 * 1024)
.spawn(|| {
let runtime = tokio::runtime::Builder::new_multi_thread()
.enable_all()
.worker_threads(2)
.build()
.expect("test runtime should build");
runtime.block_on(async {
let temp_dir = tempfile::tempdir().expect("create decommission retry store dir");
let (_ctx, store, shutdown) = without_storage_class_env(build_isolated_test_store(
temp_dir.path(),
"decommission-entry-retry",
&[4, 4],
))
.await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let changed_bucket = format!("decom-retry-a-{}", uuid::Uuid::new_v4());
let other_bucket = format!("decom-retry-b-{}", uuid::Uuid::new_v4());
for bucket in [&changed_bucket, &other_bucket] {
store
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("create decommission retry bucket");
}
let changed_object = "changed.bin";
let first_version = uuid::Uuid::new_v4();
let second_version = uuid::Uuid::new_v4();
let base_time = OffsetDateTime::now_utc();
seed_decommission_source(
&store,
&changed_bucket,
changed_object,
b"first generation".to_vec(),
&ObjectOptions {
versioned: true,
version_id: Some(first_version.to_string()),
mod_time: Some(base_time),
..Default::default()
},
)
.await;
let other_object = "other.bin";
seed_decommission_source(
&store,
&other_bucket,
other_object,
b"other bucket generation".to_vec(),
&ObjectOptions::default(),
)
.await;
mark_test_pool_decommissioning(&store, 0).await;
let mutation_calls = Arc::new(AtomicUsize::new(0));
let mutation_calls_for_hook = Arc::clone(&mutation_calls);
let mutation_store = Arc::clone(&store);
let mutation_bucket = changed_bucket.clone();
let _mutation_guard = crate::core::pools::DecommissionCleanupMutationGuard::install(Arc::new(
move |bucket, object, attempt| {
let is_target = bucket == mutation_bucket.as_str() && object == changed_object;
let calls = Arc::clone(&mutation_calls_for_hook);
let store = Arc::clone(&mutation_store);
let bucket = mutation_bucket.clone();
Box::pin(async move {
if !is_target {
return;
}
calls.fetch_add(1, Ordering::SeqCst);
if attempt == 1 {
seed_decommission_source(
&store,
&bucket,
changed_object,
b"second generation".to_vec(),
&ObjectOptions {
versioned: true,
version_id: Some(second_version.to_string()),
mod_time: Some(base_time + time::Duration::seconds(1)),
..Default::default()
},
)
.await;
}
})
},
));
let ordinary_faults = Arc::new(AtomicUsize::new(0));
let ordinary_faults_for_hook = Arc::clone(&ordinary_faults);
let fault_bucket = other_bucket.clone();
let _fault_guard = crate::core::pools::DecommissionTestFaultGuard::install(Arc::new(
move |stage, bucket, object, attempt| {
let injected = stage == DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT
&& bucket == fault_bucket.as_str()
&& object == other_object
&& attempt < crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS;
if injected {
ordinary_faults_for_hook.fetch_add(1, Ordering::SeqCst);
}
injected
},
));
let rx = CancellationToken::new();
let source_changed_exhaustions = Arc::new(AtomicUsize::new(0));
let changed_incarnation = Some(
store
.bucket_incarnation_id(&changed_bucket)
.await
.expect("changed bucket incarnation"),
);
let other_incarnation = Some(
store
.bucket_incarnation_id(&other_bucket)
.await
.expect("other bucket incarnation"),
);
let (changed_result, other_result) = tokio::join!(
run_decommission_entry_retry_test(
&store,
rx.clone(),
&changed_bucket,
changed_object,
changed_incarnation,
Arc::clone(&source_changed_exhaustions),
),
run_decommission_entry_retry_test(
&store,
rx.clone(),
&other_bucket,
other_object,
other_incarnation,
Arc::clone(&source_changed_exhaustions),
)
);
changed_result.expect("SourceChanged entry retry must converge");
other_result.expect("other bucket entry must continue through ordinary copy retries");
assert!(!rx.is_cancelled(), "entry-level SourceChanged must not cancel the shared worker token");
assert_eq!(mutation_calls.load(Ordering::SeqCst), 2, "entry must be re-listed after SourceChanged");
assert_eq!(ordinary_faults.load(Ordering::SeqCst), 2, "ordinary copy must consume the retry budget");
assert_eq!(source_changed_exhaustions.load(Ordering::SeqCst), 0);
for (version_id, expected_body) in [
(first_version, b"first generation".as_slice()),
(second_version, b"second generation".as_slice()),
] {
let opts = ObjectOptions {
versioned: true,
version_id: Some(version_id.to_string()),
..Default::default()
};
assert_decommission_source_absent(&store, &changed_bucket, changed_object, &opts).await;
assert_eq!(
read_decommission_target_body(&store, &changed_bucket, changed_object, &opts).await,
expected_body
);
}
assert_decommission_source_absent(&store, &other_bucket, other_object, &ObjectOptions::default()).await;
assert_eq!(
read_decommission_target_body(&store, &other_bucket, other_object, &ObjectOptions::default()).await,
b"other bucket generation"
);
shutdown.cancel();
});
})
.expect("spawn decommission retry test thread");
if let Err(payload) = handle.join() {
std::panic::resume_unwind(payload);
}
}
#[test]
#[serial_test::serial(storage_class_env)]
fn decommission_entry_exhausted_source_changed_retains_source_and_records_failure() {
let handle = std::thread::Builder::new()
.name("decommission_entry_exhausted_source_changed_retains_source_and_records_failure".to_string())
.stack_size(32 * 1024 * 1024)
.spawn(|| {
let runtime = tokio::runtime::Builder::new_multi_thread()
.enable_all()
.worker_threads(2)
.build()
.expect("test runtime should build");
runtime.block_on(async {
let temp_dir = tempfile::tempdir().expect("create decommission exhaustion store dir");
let (_ctx, store, shutdown) = without_storage_class_env(build_isolated_test_store(
temp_dir.path(),
"decommission-entry-exhaustion",
&[4, 4],
))
.await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let bucket = format!("decom-exhausted-{}", uuid::Uuid::new_v4());
let object = "exhausted.bin";
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("create decommission exhaustion bucket");
let original_version = uuid::Uuid::new_v4();
let base_time = OffsetDateTime::now_utc();
seed_decommission_source(
&store,
&bucket,
object,
b"original generation".to_vec(),
&ObjectOptions {
versioned: true,
version_id: Some(original_version.to_string()),
mod_time: Some(base_time),
..Default::default()
},
)
.await;
mark_test_pool_decommissioning(&store, 0).await;
let mutation_calls = Arc::new(AtomicUsize::new(0));
let mutation_calls_for_hook = Arc::clone(&mutation_calls);
let mutation_store = Arc::clone(&store);
let mutation_bucket = bucket.clone();
let _mutation_guard = crate::core::pools::DecommissionCleanupMutationGuard::install(Arc::new(
move |called_bucket, called_object, _attempt| {
let is_target = called_bucket == mutation_bucket.as_str() && called_object == object;
let calls = Arc::clone(&mutation_calls_for_hook);
let store = Arc::clone(&mutation_store);
let bucket = mutation_bucket.clone();
Box::pin(async move {
if !is_target {
return;
}
let call = calls.fetch_add(1, Ordering::SeqCst) + 1;
let offset = i64::try_from(call).expect("entry retry count should fit i64");
seed_decommission_source(
&store,
&bucket,
object,
format!("concurrent generation {call}").into_bytes(),
&ObjectOptions {
versioned: true,
version_id: Some(uuid::Uuid::new_v4().to_string()),
mod_time: Some(base_time + time::Duration::seconds(offset)),
..Default::default()
},
)
.await;
})
},
));
let rx = CancellationToken::new();
let source_changed_exhaustions = Arc::new(AtomicUsize::new(0));
let incarnation = Some(store.bucket_incarnation_id(&bucket).await.expect("bucket incarnation"));
run_decommission_entry_retry_test(
&store,
rx.clone(),
&bucket,
object,
incarnation,
Arc::clone(&source_changed_exhaustions),
)
.await
.expect("entry-level exhaustion must stay local below the pool threshold");
assert!(!rx.is_cancelled(), "one exhausted entry must not cancel other bucket workers");
assert_eq!(mutation_calls.load(Ordering::SeqCst), crate::core::pools::DECOMMISSION_ENTRY_MAX_ATTEMPTS);
assert_eq!(source_changed_exhaustions.load(Ordering::SeqCst), 1);
store.pools[0]
.get_object_info(
&bucket,
object,
&ObjectOptions {
versioned: true,
version_id: Some(original_version.to_string()),
..Default::default()
},
)
.await
.expect("retry exhaustion must retain the original source version");
let pool_meta = store.pool_meta.read().await;
let info = pool_meta.pools[0]
.decommission
.as_ref()
.expect("decommission progress must be initialized");
assert_eq!(info.items_decommission_failed, 1, "exhausted entry must be visible as failed");
drop(pool_meta);
shutdown.cancel();
});
})
.expect("spawn decommission retry test thread");
if let Err(payload) = handle.join() {
std::panic::resume_unwind(payload);
}
}
#[test]
#[serial_test::serial(storage_class_env)]
fn decommission_entry_delete_marker_copy_retries_real_path() {
let handle = std::thread::Builder::new()
.name("decommission_entry_delete_marker_copy_retries_real_path".to_string())
.stack_size(32 * 1024 * 1024)
.spawn(|| {
let runtime = tokio::runtime::Builder::new_multi_thread()
.enable_all()
.worker_threads(2)
.build()
.expect("test runtime should build");
runtime.block_on(async {
let temp_dir = tempfile::tempdir().expect("create delete marker retry store dir");
let (_ctx, store, shutdown) = without_storage_class_env(build_isolated_test_store(
temp_dir.path(),
"decommission-delete-marker-retry",
&[4, 4],
))
.await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let bucket = format!("decom-marker-{}", uuid::Uuid::new_v4());
let object = "marker.bin";
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("create delete marker retry bucket");
let data_version = uuid::Uuid::new_v4();
let marker_version = uuid::Uuid::new_v4();
let base_time = OffsetDateTime::now_utc();
seed_decommission_source(
&store,
&bucket,
object,
b"delete marker data".to_vec(),
&ObjectOptions {
versioned: true,
version_id: Some(data_version.to_string()),
mod_time: Some(base_time),
..Default::default()
},
)
.await;
store.pools[0]
.delete_object(
&bucket,
object,
ObjectOptions {
versioned: true,
version_id: Some(marker_version.to_string()),
delete_marker: true,
mod_time: Some(base_time + time::Duration::seconds(1)),
..Default::default()
},
)
.await
.expect("seed source delete marker");
mark_test_pool_decommissioning(&store, 0).await;
let fault_calls = Arc::new(AtomicUsize::new(0));
let fault_calls_for_hook = Arc::clone(&fault_calls);
let fault_bucket = bucket.clone();
let _fault_guard = crate::core::pools::DecommissionTestFaultGuard::install(Arc::new(
move |stage, called_bucket, called_object, attempt| {
let injected = stage == DECOMMISSION_TEST_FAULT_STAGE_DELETE_MARKER
&& called_bucket == fault_bucket.as_str()
&& called_object == object
&& attempt < crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS;
if injected {
fault_calls_for_hook.fetch_add(1, Ordering::SeqCst);
}
injected
},
));
let incarnation = Some(store.bucket_incarnation_id(&bucket).await.expect("bucket incarnation"));
run_decommission_entry_retry_test(
&store,
CancellationToken::new(),
&bucket,
object,
incarnation,
Arc::new(AtomicUsize::new(0)),
)
.await
.expect("delete marker copy retries must converge");
assert_eq!(
fault_calls.load(Ordering::SeqCst),
crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS - 1
);
let marker_opts = ObjectOptions {
versioned: true,
version_id: Some(marker_version.to_string()),
..Default::default()
};
let target_marker = store.pools[1]
.get_object_info(&bucket, object, &marker_opts)
.await
.expect("target delete marker must exist");
assert!(target_marker.delete_marker);
assert_decommission_source_absent(&store, &bucket, object, &marker_opts).await;
let data_opts = ObjectOptions {
versioned: true,
version_id: Some(data_version.to_string()),
..Default::default()
};
assert_eq!(
read_decommission_target_body(&store, &bucket, object, &data_opts).await,
b"delete marker data"
);
assert_decommission_source_absent(&store, &bucket, object, &data_opts).await;
shutdown.cancel();
});
})
.expect("spawn decommission retry test thread");
if let Err(payload) = handle.join() {
std::panic::resume_unwind(payload);
}
}
#[cfg(feature = "test-util")]
#[test]
#[serial_test::serial(storage_class_env)]
fn decommission_entry_tiered_copy_retries_real_path() {
let handle = std::thread::Builder::new()
.name("decommission_entry_tiered_copy_retries_real_path".to_string())
.stack_size(32 * 1024 * 1024)
.spawn(|| {
let runtime = tokio::runtime::Builder::new_multi_thread()
.enable_all()
.worker_threads(2)
.build()
.expect("test runtime should build");
runtime.block_on(async {
let temp_dir = tempfile::tempdir().expect("create tiered retry store dir");
let (ctx, store, shutdown) = without_storage_class_env(build_isolated_test_store(
temp_dir.path(),
"decommission-tiered-retry",
&[4, 4],
))
.await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let bucket = format!("decom-tiered-{}", uuid::Uuid::new_v4());
let object = "tiered.bin";
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("create tiered retry bucket");
let mut reader = PutObjReader::from_vec(b"tiered generation".to_vec());
let original = store.pools[0]
.put_object(&bucket, object, &mut reader, &ObjectOptions::default())
.await
.expect("seed tiered source object");
let tier_name = format!("DECOM{}", &uuid::Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
register_mock_tier(&ctx.tier_config_mgr(), &tier_name).await;
store.pools[0]
.transition_object(
&bucket,
object,
&ObjectOptions {
transition: TransitionOptions {
status: TRANSITION_PENDING.to_string(),
tier: tier_name,
etag: original.etag.clone().expect("tiered source ETag"),
..Default::default()
},
version_id: original.version_id.map(|version_id| version_id.to_string()),
mod_time: original.mod_time,
..Default::default()
},
)
.await
.expect("transition source object to mock tier");
mark_test_pool_decommissioning(&store, 0).await;
let fault_calls = Arc::new(AtomicUsize::new(0));
let fault_calls_for_hook = Arc::clone(&fault_calls);
let fault_bucket = bucket.clone();
let _fault_guard = crate::core::pools::DecommissionTestFaultGuard::install(Arc::new(
move |stage, called_bucket, called_object, attempt| {
let injected = stage == DECOMMISSION_TEST_FAULT_STAGE_TIERED
&& called_bucket == fault_bucket.as_str()
&& called_object == object
&& attempt < crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS;
if injected {
fault_calls_for_hook.fetch_add(1, Ordering::SeqCst);
}
injected
},
));
let incarnation = Some(store.bucket_incarnation_id(&bucket).await.expect("bucket incarnation"));
run_decommission_entry_retry_test(
&store,
CancellationToken::new(),
&bucket,
object,
incarnation,
Arc::new(AtomicUsize::new(0)),
)
.await
.expect("tiered copy retries must converge");
assert_eq!(
fault_calls.load(Ordering::SeqCst),
crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS - 1
);
let target = store.pools[1]
.get_object_info(&bucket, object, &ObjectOptions::default())
.await
.expect("tiered target metadata must exist");
assert_eq!(target.transitioned_object.status, rustfs_filemeta::TRANSITION_COMPLETE);
assert_decommission_source_absent(&store, &bucket, object, &ObjectOptions::default()).await;
shutdown.cancel();
});
})
.expect("spawn decommission retry test thread");
if let Err(payload) = handle.join() {
std::panic::resume_unwind(payload);
}
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
#[serial_test::serial(storage_class_env)]
async fn decommission_outer_fence_loss_blocks_target_put_commit() {
+1 -1
View File
@@ -3194,7 +3194,7 @@ impl ECStore {
// Default return value
let mut del_objects = vec![DeletedObject::default(); objects.len()];
let mut accounting = vec![None; objects.len()];
let accounting = vec![None; objects.len()];
let mut del_errs = Vec::with_capacity(objects.len());
for _ in 0..objects.len() {
@@ -271,7 +271,7 @@ pub(super) fn resolve_latest_object_info_candidates(
.filter(|candidate| latest_candidate_mod_time(candidate) == Some(latest_mod_time))
.collect::<Vec<_>>();
latest_candidates.sort_by(|left, right| right.idx.cmp(&left.idx));
latest_candidates.sort_by_key(|candidate| std::cmp::Reverse(candidate.idx));
let Some(winner) = latest_candidates.first() else {
return Err(Error::ErasureReadQuorum);
-3
View File
@@ -43,9 +43,6 @@ allow-git = [
# RustFS fork carrying presigned expiry and constant-time authentication fixes.
# owner: rustfs-maintainers review: 2026-10
"https://github.com/rustfs/s3s.git",
# MiMalloc fork pinned for hotpath allocation counting support.
# owner: houseme review: 2026-10
"https://github.com/xonatius/mimalloc_rust.git",
]
[bans]
+2 -2
View File
@@ -58,7 +58,7 @@
| heal_erasure_disk_rebuild_test | 4 | 🌙 |
| inline_fast_path_cluster_test | 16 | |
| internode_rpc_signature_e2e_test | 5 | |
| kms | 46 | |
| kms | 47 | |
| leading_slash_key_test | 2 | ✅ |
| lifecycle_regression_test | 4 | |
| list_buckets_auth_test | 1 | ✅ |
@@ -99,4 +99,4 @@
| tls_hot_reload_test | 1 | ✅ |
| version_id_regression_test | 10 | ✅ |
**Total listed: 575 tests across 82 modules · PR smoke: 163 tests / 36 modules · merge/main full: 453 tests / 73 modules · nightly replication: 55 tests · nightly cluster faults: 28 tests / 7 modules · nightly protocols: 16 tests** · updated 2026-08-23.
**Total listed: 576 tests across 82 modules · PR smoke: 163 tests / 36 modules · merge/main full: 454 tests / 73 modules · nightly replication: 55 tests · nightly cluster faults: 28 tests / 7 modules · nightly protocols: 16 tests** · updated 2026-08-23.
+2 -2
View File
@@ -336,13 +336,13 @@ opentelemetry = { workspace = true }
tracing-opentelemetry = { workspace = true }
# Data structures
hashbrown = { workspace = true, features = ["serde", "rayon"] }
mimalloc = { workspace = true }
rustfs-mimalloc = { workspace = true }
[target.'cfg(target_os = "linux")'.dependencies]
libsystemd.workspace = true
[target.'cfg(not(target_os = "windows"))'.dependencies]
libmimalloc-sys.workspace = true
rustfs-mimalloc-sys.workspace = true
[dev-dependencies]
uuid = { workspace = true, features = ["v4", "v5", "fast-rng", "macro-diagnostics"] }
+1 -7
View File
@@ -369,14 +369,8 @@ pub fn allocator_reclaim_controller_snapshot(ctx: &CancellationToken) -> Allocat
}
#[cfg(not(target_os = "windows"))]
#[allow(unsafe_code)]
fn collect_allocator_memory(force: bool) -> Result<(), String> {
// SAFETY: `mi_collect` is provided by the active global allocator backend
// on this target family. It is explicitly intended to reclaim retained
// pages/segments and does not require additional invariants from the caller.
unsafe {
libmimalloc_sys::mi_collect(force);
}
rustfs_mimalloc::MiMalloc::collect(force);
Ok(())
}
+10 -8
View File
@@ -26,22 +26,22 @@ struct MiMallocAllocator;
unsafe impl GlobalAlloc for MiMallocAllocator {
unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
// SAFETY: the caller upholds GlobalAlloc's contract for layout.
unsafe { mimalloc::MiMalloc.alloc(layout) }
unsafe { rustfs_mimalloc::MiMalloc.alloc(layout) }
}
unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 {
// SAFETY: the caller upholds GlobalAlloc's contract for layout.
unsafe { mimalloc::MiMalloc.alloc_zeroed(layout) }
unsafe { rustfs_mimalloc::MiMalloc.alloc_zeroed(layout) }
}
unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) {
// SAFETY: ptr and layout came from this allocator and are forwarded unchanged.
unsafe { mimalloc::MiMalloc.dealloc(ptr, layout) }
unsafe { rustfs_mimalloc::MiMalloc.dealloc(ptr, layout) }
}
unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 {
// SAFETY: ptr and layout came from this allocator and are forwarded unchanged.
unsafe { mimalloc::MiMalloc.realloc(ptr, layout, new_size) }
unsafe { rustfs_mimalloc::MiMalloc.realloc(ptr, layout, new_size) }
}
}
@@ -51,7 +51,7 @@ static GLOBAL: hotpath::CountingAllocator<MiMallocAllocator> = hotpath::Counting
#[cfg(not(all(feature = "hotpath", feature = "hotpath-alloc")))]
#[global_allocator]
static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc;
static GLOBAL: rustfs_mimalloc::MiMalloc = rustfs_mimalloc::MiMalloc;
fn main() {
let _hotpath_guard = hotpath::HotpathGuardBuilder::new("main").build();
@@ -71,8 +71,9 @@ mod tests {
allocation.extend_from_slice(&[7_u8; 64]);
assert_eq!(allocation.len(), 64);
let heap = rustfs_mimalloc::heap::Heap::main();
// SAFETY: the live Vec pointer is valid to inspect for heap ownership.
assert!(unsafe { libmimalloc_sys::mi_is_in_heap_region(allocation.as_ptr().cast()) });
assert!(unsafe { heap.contains(allocation.as_ptr()) });
}
#[test]
@@ -85,12 +86,13 @@ mod tests {
let layout = Layout::from_size_align(32, 8).expect("valid test allocation layout");
let grown_layout = Layout::from_size_align(64, 8).expect("valid grown test allocation layout");
let allocator = super::MiMallocAllocator;
let heap = rustfs_mimalloc::heap::Heap::main();
// SAFETY: The pointer is checked for null before use and later released
// through the same allocator with the corresponding layout.
let ptr = unsafe { allocator.alloc_zeroed(layout) };
assert!(!ptr.is_null());
assert!(unsafe { libmimalloc_sys::mi_is_in_heap_region(ptr.cast()) });
assert!(unsafe { heap.contains(ptr) });
assert!(unsafe { std::slice::from_raw_parts(ptr, 32).iter().all(|byte| *byte == 0) });
// SAFETY: `ptr` was allocated by `allocator` with `layout`; on failure
@@ -102,7 +104,7 @@ mod tests {
panic!("mimalloc realloc failed in allocator smoke test");
}
assert!(unsafe { libmimalloc_sys::mi_is_in_heap_region(grown_ptr.cast()) });
assert!(unsafe { heap.contains(grown_ptr) });
// SAFETY: `grown_ptr` was reallocated by `allocator` and is released
// with the matching grown layout.
unsafe { allocator.dealloc(grown_ptr, grown_layout) };
+19 -36
View File
@@ -17,10 +17,7 @@ use rustfs_io_metrics::{
record_cpu_usage, record_memory_usage, record_process_memory_split,
};
use serde::Serialize;
#[cfg(any(test, not(target_os = "windows")))]
use serde_json::Value;
#[cfg(not(target_os = "windows"))]
use std::ffi::CStr;
use std::path::Path;
use std::sync::{Arc, Mutex, OnceLock};
use std::time::Duration;
@@ -231,7 +228,18 @@ fn read_cgroup_memory_snapshot() -> Option<CgroupMemorySnapshot> {
read_cgroup_v2().or_else(read_cgroup_v1)
}
#[cfg(any(test, not(target_os = "windows")))]
fn read_allocator_memory_snapshot() -> Option<AllocatorMemorySnapshot> {
let json = rustfs_mimalloc::MiMalloc::stats_json();
if json.is_empty() {
return None;
}
let observation = parse_mimalloc_stats_json(&json)?;
Some(AllocatorMemorySnapshot {
backend: crate::allocator_reclaim::allocator_backend(),
observation,
})
}
fn numeric_json_value(value: &Value) -> Option<u64> {
match value {
Value::Number(number) => number
@@ -242,7 +250,6 @@ fn numeric_json_value(value: &Value) -> Option<u64> {
}
}
#[cfg(any(test, not(target_os = "windows")))]
fn numeric_json_field(value: &Value, field: &str) -> Option<u64> {
match value {
Value::Object(fields) => fields
@@ -254,7 +261,6 @@ fn numeric_json_field(value: &Value, field: &str) -> Option<u64> {
}
}
#[cfg(any(test, not(target_os = "windows")))]
fn mimalloc_stat_field(value: &Value, metric: &str, field: &str) -> Option<u64> {
match value {
Value::Object(fields) => {
@@ -271,12 +277,10 @@ fn mimalloc_stat_field(value: &Value, metric: &str, field: &str) -> Option<u64>
}
}
#[cfg(any(test, not(target_os = "windows")))]
fn mimalloc_stat_current(value: &Value, metric: &str) -> Option<u64> {
mimalloc_stat_field(value, metric, "current")
}
#[cfg(any(test, not(target_os = "windows")))]
fn mimalloc_stat_sum(value: &Value, metrics: &[&str], field: &str) -> Option<u64> {
metrics
.iter()
@@ -285,7 +289,6 @@ fn mimalloc_stat_sum(value: &Value, metrics: &[&str], field: &str) -> Option<u64
.filter(|value| *value > 0)
}
#[cfg(any(test, not(target_os = "windows")))]
fn parse_mimalloc_stats_json(stats_json: &str) -> Option<AllocatorMemoryObservation> {
let value = serde_json::from_str::<Value>(stats_json).ok()?;
let malloc_metrics = ["malloc_normal", "malloc_huge"];
@@ -312,33 +315,6 @@ fn parse_mimalloc_stats_json(stats_json: &str) -> Option<AllocatorMemoryObservat
}
}
#[cfg(not(target_os = "windows"))]
#[allow(unsafe_code)]
fn read_allocator_memory_snapshot() -> Option<AllocatorMemorySnapshot> {
// SAFETY: `mi_stats_get_json` returns a null-terminated JSON buffer owned by
// mimalloc when called with a null input buffer. The mimalloc API requires
// freeing that buffer with `mi_free`; parsing finishes before the buffer is freed.
let observation = unsafe {
let stats_ptr = libmimalloc_sys::mi_stats_get_json(0, std::ptr::null_mut());
if stats_ptr.is_null() {
return None;
}
let observation = CStr::from_ptr(stats_ptr).to_str().ok().and_then(parse_mimalloc_stats_json);
libmimalloc_sys::mi_free(stats_ptr.cast());
observation?
};
Some(AllocatorMemorySnapshot {
backend: crate::allocator_reclaim::allocator_backend(),
observation,
})
}
#[cfg(target_os = "windows")]
fn read_allocator_memory_snapshot() -> Option<AllocatorMemorySnapshot> {
None
}
fn configured_memory_observability_interval_secs() -> u64 {
rustfs_utils::get_env_u64(ENV_MEMORY_OBSERVABILITY_INTERVAL_SECS, DEFAULT_MEMORY_OBSERVABILITY_INTERVAL_SECS).max(1)
}
@@ -566,6 +542,13 @@ mod tests {
assert_eq!(parse_mimalloc_stats_json(r#"{ "allocator": "unknown" }"#), None);
}
#[test]
fn read_allocator_memory_snapshot_uses_mimalloc_stats_json() {
let snapshot = super::read_allocator_memory_snapshot();
#[cfg(not(target_os = "windows"))]
assert!(snapshot.is_some(), "allocator snapshot should be available on non-Windows");
}
#[test]
fn memory_observability_snapshot_reports_disabled_when_metrics_are_disabled() {
let snapshot = build_memory_observability_status_snapshot(false, 15, false);