mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-31 09:18:28 +00:00
fix(ecstore): make inline-rollback reclamation file-precise to keep #5703's child-key safety (#5732)
#5724 reclaimed the synthetic inline-rollback dir after a committed rename with delete_data_dir(recursive: true), which has no notion of object metadata: for unversioned objects the synthetic UUID is a fixed, publicly-known constant, so object/<rollback-dir> can simultaneously be a legitimate child key's directory, and recursively deleting it reopens the authorization bypass #5703 closed (PutObject on K destroying K/<uuid> without DeleteObject permission). Replace the recursive pass with a file-precise one: after quorum commit, delete exactly object/<rollback>/xl.meta.bkp with a non-recursive delete on every disk whose rollback dir is not also the cleanup dir. The parent-rmdir walk removes the dir only when the backup was its sole content, so the BucketNotEmpty leak fix is preserved (#5724's regression test passes unchanged) while a child key at the same path keeps its metadata. The undo path's restore_metadata_backup now also reclaims the emptied synthetic dir, mirroring restore_delete_rollback.
This commit is contained in:
@@ -119,7 +119,7 @@ fn read_all_data_std(path: &Path) -> core::result::Result<(Vec<u8>, Option<Offse
|
|||||||
Ok((bytes, modtime))
|
Ok((bytes, modtime))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn inline_metadata_rollback_dir(version_id: Uuid, meta: &FileMeta) -> Uuid {
|
pub(crate) fn inline_metadata_rollback_dir(version_id: Uuid, meta: &FileMeta) -> Uuid {
|
||||||
let used_data_dirs: HashSet<Uuid> = meta.get_data_dirs().unwrap_or_default().into_iter().flatten().collect();
|
let used_data_dirs: HashSet<Uuid> = meta.get_data_dirs().unwrap_or_default().into_iter().flatten().collect();
|
||||||
let base = version_id.as_u128() ^ INLINE_METADATA_ROLLBACK_DIR_XOR;
|
let base = version_id.as_u128() ^ INLINE_METADATA_ROLLBACK_DIR_XOR;
|
||||||
let mut salt = 0u128;
|
let mut salt = 0u128;
|
||||||
@@ -240,8 +240,15 @@ async fn write_metadata_rollback_backup(object_dir: &Path, rollback_dir: Uuid, d
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn restore_metadata_backup(object_dir: &Path, xl_path: &Path, rollback_dir: Uuid) -> Result<()> {
|
async fn restore_metadata_backup(object_dir: &Path, xl_path: &Path, rollback_dir: Uuid) -> Result<()> {
|
||||||
let backup_path = object_dir.join(rollback_dir.to_string()).join(STORAGE_FORMAT_FILE_BACKUP);
|
let rollback_path = object_dir.join(rollback_dir.to_string());
|
||||||
rename_all(&backup_path, xl_path, object_dir).await
|
let backup_path = rollback_path.join(STORAGE_FORMAT_FILE_BACKUP);
|
||||||
|
rename_all(&backup_path, xl_path, object_dir).await?;
|
||||||
|
// A synthetic inline rollback dir held only the backup the rename above
|
||||||
|
// just consumed; reclaim it so the object dir can empty out. A real data
|
||||||
|
// dir still holds its parts, so the non-recursive remove is a benign
|
||||||
|
// no-op there (mirrors restore_delete_rollback).
|
||||||
|
let _ = fs::remove_dir(&rollback_path).await;
|
||||||
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn restore_delete_rollback(object_dir: &Path, xl_path: &Path, rollback_dir: Uuid) -> Result<()> {
|
async fn restore_delete_rollback(object_dir: &Path, xl_path: &Path, rollback_dir: Uuid) -> Result<()> {
|
||||||
@@ -12228,6 +12235,54 @@ mod test {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// The undo_write restore consumes `<rollback>/xl.meta.bkp` by rename; a
|
||||||
|
// synthetic rollback dir is then empty and must be reclaimed so the object
|
||||||
|
// dir can empty out (BucketNotEmpty leak). A real data dir still holds its
|
||||||
|
// parts and must survive the non-recursive remove.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn restore_metadata_backup_reclaims_empty_rollback_dir_only() {
|
||||||
|
use tempfile::tempdir;
|
||||||
|
|
||||||
|
let dir = tempdir().expect("temp dir should be created");
|
||||||
|
let object_dir = dir.path().join("bucket").join("obj");
|
||||||
|
let xl_path = object_dir.join(STORAGE_FORMAT_FILE);
|
||||||
|
let rollback_dir = Uuid::new_v4();
|
||||||
|
let rollback_path = object_dir.join(rollback_dir.to_string());
|
||||||
|
fs::create_dir_all(&rollback_path)
|
||||||
|
.await
|
||||||
|
.expect("rollback dir should be created");
|
||||||
|
fs::write(rollback_path.join(STORAGE_FORMAT_FILE_BACKUP), b"old-meta")
|
||||||
|
.await
|
||||||
|
.expect("backup should be written");
|
||||||
|
|
||||||
|
restore_metadata_backup(&object_dir, &xl_path, rollback_dir)
|
||||||
|
.await
|
||||||
|
.expect("restore should succeed");
|
||||||
|
assert_eq!(
|
||||||
|
fs::read(&xl_path).await.expect("xl.meta should be restored"),
|
||||||
|
b"old-meta",
|
||||||
|
"restore must move the backup back onto xl.meta"
|
||||||
|
);
|
||||||
|
assert!(!rollback_path.exists(), "an emptied synthetic rollback dir must be reclaimed");
|
||||||
|
|
||||||
|
// Real data dir: parts remain, the dir must survive.
|
||||||
|
let real_dir = Uuid::new_v4();
|
||||||
|
let real_path = object_dir.join(real_dir.to_string());
|
||||||
|
fs::create_dir_all(&real_path).await.expect("real data dir should be created");
|
||||||
|
fs::write(real_path.join(STORAGE_FORMAT_FILE_BACKUP), b"older-meta")
|
||||||
|
.await
|
||||||
|
.expect("backup should be written");
|
||||||
|
fs::write(real_path.join("part.1"), b"data")
|
||||||
|
.await
|
||||||
|
.expect("part should be written");
|
||||||
|
|
||||||
|
restore_metadata_backup(&object_dir, &xl_path, real_dir)
|
||||||
|
.await
|
||||||
|
.expect("restore should succeed");
|
||||||
|
assert!(real_path.join("part.1").exists(), "a real data dir must keep its parts");
|
||||||
|
assert!(real_path.exists(), "a non-empty data dir must not be removed");
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn rename_commit_failure_cleans_local_rollback_backup() {
|
async fn rename_commit_failure_cleans_local_rollback_backup() {
|
||||||
use tempfile::tempdir;
|
use tempfile::tempdir;
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ use crate::diagnostics::get::{
|
|||||||
use crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX;
|
use crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX;
|
||||||
use crate::disk::{
|
use crate::disk::{
|
||||||
DataDirDeleteStatus, OldCurrentSize, PART_TRANSACTION_NEW_META, PART_TRANSACTION_OLD_META, PART_TRANSACTION_ROLLBACK,
|
DataDirDeleteStatus, OldCurrentSize, PART_TRANSACTION_NEW_META, PART_TRANSACTION_OLD_META, PART_TRANSACTION_ROLLBACK,
|
||||||
PartTransactionAction, part_transaction_path,
|
PartTransactionAction, STORAGE_FORMAT_FILE_BACKUP, part_transaction_path,
|
||||||
};
|
};
|
||||||
use crate::erasure::coding::BitrotReader;
|
use crate::erasure::coding::BitrotReader;
|
||||||
use crate::io_support::bitrot::ShardReader;
|
use crate::io_support::bitrot::ShardReader;
|
||||||
@@ -2950,63 +2950,63 @@ impl SetDisks {
|
|||||||
return Err(ret_err);
|
return Err(ret_err);
|
||||||
}
|
}
|
||||||
|
|
||||||
// A synthetic inline-rollback dir (rollback_data_dir set, cleanup_data_dir
|
// The write is authoritatively committed, so the per-disk rollback
|
||||||
// unset) holds only the xl.meta.bkp consumed by the quorum-failure undo
|
// backup (`object/<rollback_dir>/xl.meta.bkp`) is dead weight now.
|
||||||
// above. Once the commit holds quorum that window is closed, and the
|
// When the rollback dir doubles as the real dereferenced data dir it
|
||||||
// commit_rename_data_dir pass reclaims real old data dirs only, so the
|
// is reclaimed wholesale by `commit_rename_data_dir`; a rollback dir
|
||||||
// synthetic dir must be reclaimed here — otherwise every overwrite of an
|
// reported separately (an overwrite of an inline version, whose dir is
|
||||||
// inline version by a non-inline one leaves <object>/<rollback-dir>/
|
// synthetic) is excluded from that recursive reclamation for safety
|
||||||
// behind, and DeleteBucket keeps failing with BucketNotEmpty after the
|
// (#5703) and must be reclaimed here instead — otherwise every inline
|
||||||
// object is deleted. Best-effort: residue must not fail the durable write.
|
// overwrite strands a backup file that keeps the object dir non-empty
|
||||||
let rollback_only_futures: Vec<_> = disks
|
// and makes a later DeleteBucket fail with BucketNotEmpty forever.
|
||||||
.iter()
|
// Delete exactly the backup file, never the directory tree: the
|
||||||
.enumerate()
|
// synthetic UUID is a fixed, publicly-known constant for unversioned
|
||||||
.filter_map(|(idx, disk)| {
|
// objects, so `object/<rollback_dir>` can simultaneously be a
|
||||||
let disk = disk.clone()?;
|
// legitimate child key's directory — recursively deleting it would
|
||||||
if errs[idx].is_some() || cleanup_data_dirs[idx].is_some() {
|
// reopen the authorization bypass #5703 closed. The non-recursive
|
||||||
return None;
|
// delete removes the directory only when the backup was its sole
|
||||||
}
|
// content. Best-effort space reclamation — like
|
||||||
let rollback_dir = data_dirs[idx]?;
|
// `commit_rename_data_dir`, this must never negate the already-durable
|
||||||
// Anti-misdelete guard, same posture as commit_rename_data_dir:
|
// ACK.
|
||||||
// never touch the data dir the commit just published.
|
let mut backup_reclaims = Vec::new();
|
||||||
if file_infos[idx].data_dir == Some(rollback_dir) {
|
for (idx, disk) in disks.iter().enumerate() {
|
||||||
return None;
|
if errs[idx].is_some() {
|
||||||
}
|
continue;
|
||||||
let dst_bucket = dst_bucket.clone();
|
}
|
||||||
let path = format!("{dst_object}/{rollback_dir}");
|
let Some(rollback_dir) = data_dirs[idx] else {
|
||||||
Some(tokio::spawn(async move {
|
continue;
|
||||||
disk.delete_data_dir(
|
};
|
||||||
&dst_bucket,
|
if cleanup_data_dirs[idx] == Some(rollback_dir) {
|
||||||
&path,
|
continue;
|
||||||
DeleteOptions {
|
}
|
||||||
recursive: true,
|
let Some(disk) = disk.clone() else {
|
||||||
..Default::default()
|
continue;
|
||||||
},
|
};
|
||||||
)
|
let dst_bucket = dst_bucket.clone();
|
||||||
.await
|
let dst_object = dst_object.clone();
|
||||||
}))
|
backup_reclaims.push(tokio::spawn(async move {
|
||||||
})
|
let backup_path = format!("{dst_object}/{rollback_dir}/{STORAGE_FORMAT_FILE_BACKUP}");
|
||||||
.collect();
|
disk.delete(&dst_bucket, &backup_path, DeleteOptions::default()).await
|
||||||
for result in join_all(rollback_only_futures).await {
|
}));
|
||||||
|
}
|
||||||
|
for result in join_all(backup_reclaims).await {
|
||||||
match result {
|
match result {
|
||||||
Ok(Ok(_)) => {}
|
Ok(Ok(())) => {}
|
||||||
Ok(Err(err)) if err == DiskError::FileNotFound || err == DiskError::VolumeNotFound => {}
|
Ok(Err(DiskError::FileNotFound | DiskError::VolumeNotFound)) => {}
|
||||||
Ok(Err(err)) => {
|
Ok(Err(err)) => {
|
||||||
warn!(
|
warn!(
|
||||||
target: "rustfs_ecstore::set_disk",
|
|
||||||
dst_bucket = %dst_bucket,
|
dst_bucket = %dst_bucket,
|
||||||
dst_object = %dst_object,
|
dst_object = %dst_object,
|
||||||
error = %err,
|
error = %err,
|
||||||
"failed to reclaim synthetic inline-rollback dir after commit"
|
"rollback backup reclamation failed after committed rename"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
Err(err) => {
|
Err(join_err) => {
|
||||||
warn!(
|
warn!(
|
||||||
target: "rustfs_ecstore::set_disk",
|
|
||||||
dst_bucket = %dst_bucket,
|
dst_bucket = %dst_bucket,
|
||||||
dst_object = %dst_object,
|
dst_object = %dst_object,
|
||||||
error = %err,
|
error = %join_err,
|
||||||
"synthetic inline-rollback reclaim task failed"
|
"rollback backup reclamation task failed after committed rename"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6309,6 +6309,136 @@ mod tests {
|
|||||||
assert_eq!(read_back.size, 9, "HEAD must observe the new version, not stale metadata");
|
assert_eq!(read_back.size, 9, "HEAD must observe the new version, not stale metadata");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Regression for the inline-overwrite rollback backup leak: #5703 stopped
|
||||||
|
// reporting the synthetic rollback dir for recursive post-commit cleanup,
|
||||||
|
// which stranded `object/<synthetic>/xl.meta.bkp` after every inline
|
||||||
|
// overwrite. The object dir then never emptied, so the s3-tests teardown
|
||||||
|
// sequence (delete object, delete bucket) failed with BucketNotEmpty
|
||||||
|
// forever. After a committed overwrite the backup must be reclaimed and a
|
||||||
|
// subsequent delete must leave nothing behind.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn inline_overwrite_reclaims_synthetic_rollback_backup() {
|
||||||
|
let set_disks = make_local_bucket_test_set_disks().await;
|
||||||
|
let bucket = "bucket-inline-rollback-leak";
|
||||||
|
let object = "obj";
|
||||||
|
|
||||||
|
set_disks
|
||||||
|
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("bucket should be created");
|
||||||
|
|
||||||
|
for body in [b"hello".to_vec(), b"goodbye".to_vec()] {
|
||||||
|
let mut reader = PutObjReader::from_vec(body);
|
||||||
|
set_disks
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
no_lock: true,
|
||||||
|
..ObjectOptions::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("inline write should succeed");
|
||||||
|
}
|
||||||
|
|
||||||
|
// The committed overwrite must leave only xl.meta in the object dir on
|
||||||
|
// every disk; a stranded rollback dir keeps the bucket undeletable.
|
||||||
|
for endpoint in &set_disks.set_endpoints {
|
||||||
|
let object_dir = std::path::PathBuf::from(endpoint.get_file_path()).join(bucket).join(object);
|
||||||
|
let mut entries: Vec<String> = std::fs::read_dir(&object_dir)
|
||||||
|
.expect("object dir should exist")
|
||||||
|
.map(|entry| entry.expect("entry should read").file_name().to_string_lossy().into_owned())
|
||||||
|
.collect();
|
||||||
|
entries.sort();
|
||||||
|
assert_eq!(
|
||||||
|
entries,
|
||||||
|
vec![STORAGE_FORMAT_FILE.to_string()],
|
||||||
|
"only xl.meta may remain after an inline overwrite in {object_dir:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// With only xl.meta left, the s3-tests teardown (delete object, delete
|
||||||
|
// bucket) empties the dir; the delete paths themselves are covered by
|
||||||
|
// their own tests. This harness has no bucket metadata sys, so the
|
||||||
|
// full delete_object flow cannot run here.
|
||||||
|
}
|
||||||
|
|
||||||
|
// #5703's security property must survive the backup reclamation: the
|
||||||
|
// synthetic rollback dir of key K maps to the directory `K/<uuid>`, which
|
||||||
|
// can simultaneously be a legitimate child key. Reclaiming the backup must
|
||||||
|
// remove exactly the backup file — never the child key's metadata.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn inline_overwrite_backup_reclaim_spares_child_key_dir() {
|
||||||
|
let set_disks = make_local_bucket_test_set_disks().await;
|
||||||
|
let bucket = "bucket-inline-rollback-child";
|
||||||
|
let object = "obj";
|
||||||
|
let synthetic = crate::disk::local::inline_metadata_rollback_dir(Uuid::nil(), &FileMeta::new());
|
||||||
|
let child_object = format!("{object}/{synthetic}");
|
||||||
|
|
||||||
|
set_disks
|
||||||
|
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("bucket should be created");
|
||||||
|
|
||||||
|
let mut reader = PutObjReader::from_vec(b"child".to_vec());
|
||||||
|
set_disks
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
&child_object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
no_lock: true,
|
||||||
|
..ObjectOptions::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("child write should succeed");
|
||||||
|
|
||||||
|
// Create then overwrite the parent key: the overwrite writes its
|
||||||
|
// rollback backup into the child's directory and must afterwards
|
||||||
|
// reclaim only that file.
|
||||||
|
for body in [b"first".to_vec(), b"second".to_vec()] {
|
||||||
|
let mut reader = PutObjReader::from_vec(body);
|
||||||
|
set_disks
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
no_lock: true,
|
||||||
|
..ObjectOptions::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("parent write should succeed");
|
||||||
|
}
|
||||||
|
|
||||||
|
let child_info = set_disks
|
||||||
|
.get_object_info(bucket, &child_object, &ObjectOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("child key must survive the parent's rollback backup reclamation");
|
||||||
|
assert_eq!(child_info.size, 5, "child key content must be untouched");
|
||||||
|
|
||||||
|
for endpoint in &set_disks.set_endpoints {
|
||||||
|
let child_dir = std::path::PathBuf::from(endpoint.get_file_path())
|
||||||
|
.join(bucket)
|
||||||
|
.join(object)
|
||||||
|
.join(synthetic.to_string());
|
||||||
|
let mut entries: Vec<String> = std::fs::read_dir(&child_dir)
|
||||||
|
.expect("child object dir should exist")
|
||||||
|
.map(|entry| entry.expect("entry should read").file_name().to_string_lossy().into_owned())
|
||||||
|
.collect();
|
||||||
|
entries.sort();
|
||||||
|
assert_eq!(
|
||||||
|
entries,
|
||||||
|
vec![STORAGE_FORMAT_FILE.to_string()],
|
||||||
|
"the child dir must keep its xl.meta and lose only the stray backup in {child_dir:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test(flavor = "multi_thread")]
|
#[tokio::test(flavor = "multi_thread")]
|
||||||
#[serial]
|
#[serial]
|
||||||
async fn test_acquire_dist_delete_object_locks_batch_succeeds_with_two_healthy_lockers() {
|
async fn test_acquire_dist_delete_object_locks_batch_succeeds_with_two_healthy_lockers() {
|
||||||
|
|||||||
Reference in New Issue
Block a user