mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-18 18:46:17 +00:00
Merge branch 'main' into hs01-mrf-wiring
This commit is contained in:
@@ -380,7 +380,7 @@ pub mod erasure {
|
||||
|
||||
pub mod event {
|
||||
pub use crate::event::name::EventName;
|
||||
pub use crate::services::event_notification::{EventArgs, register_event_dispatch_hook};
|
||||
pub use crate::services::event_notification::{EventArgs, register_event_dispatch_hook, send_event};
|
||||
}
|
||||
|
||||
pub mod global {
|
||||
@@ -483,6 +483,7 @@ pub mod store_list {
|
||||
}
|
||||
|
||||
pub mod storage {
|
||||
pub use crate::core::pools::HealLifecycleExpiryContext;
|
||||
pub use crate::store::HealWalkVersion;
|
||||
pub use crate::store::{
|
||||
ECStore, all_local_disk, all_local_disk_path, find_local_disk_by_ref, init_local_disks,
|
||||
|
||||
@@ -1549,8 +1549,8 @@ impl Default for PutObjectOptions {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl PutObjectOptions {
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn set_match_etag(&mut self, etag: &str) {
|
||||
if etag == "*" {
|
||||
self.custom_header
|
||||
@@ -1561,6 +1561,7 @@ impl PutObjectOptions {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn set_match_etag_except(&mut self, etag: &str) {
|
||||
if etag == "*" {
|
||||
self.custom_header
|
||||
@@ -1696,6 +1697,7 @@ impl PutObjectOptions {
|
||||
header
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn validate(&self, _c: Arc<TargetClient>) -> Result<(), std::io::Error> {
|
||||
//if self.checksum.is_set() {
|
||||
/*if !self.trailing_header_support {
|
||||
|
||||
@@ -456,16 +456,23 @@ impl<'a> LifecycleExpiryTrace<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl ExpiryStats {
|
||||
pub fn missed_tasks(&self) -> i64 {
|
||||
self.missed_expiry_tasks.load(Ordering::SeqCst)
|
||||
}
|
||||
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "asserted by this file's tests; the lib target cannot see test-only consumers (backlog#1823)"
|
||||
)]
|
||||
fn missed_free_vers_tasks(&self) -> i64 {
|
||||
self.missed_freevers_tasks.load(Ordering::SeqCst)
|
||||
}
|
||||
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "asserted by this file's tests; the lib target cannot see test-only consumers (backlog#1823)"
|
||||
)]
|
||||
fn missed_tier_journal_tasks(&self) -> i64 {
|
||||
self.missed_tier_journal_tasks.load(Ordering::SeqCst)
|
||||
}
|
||||
|
||||
@@ -19,7 +19,7 @@ pub mod core;
|
||||
pub mod evaluator;
|
||||
pub mod manual_transition_job;
|
||||
mod metadata_boundary;
|
||||
pub(crate) use metadata_boundary::get_expiry_configs;
|
||||
pub(crate) use metadata_boundary::{LifecycleExpiryConfigs, get_expiry_configs};
|
||||
mod object_lock_boundary;
|
||||
pub use self::core as lifecycle;
|
||||
mod replication_sink;
|
||||
|
||||
@@ -80,7 +80,10 @@ impl LastDayTierStats {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "asserted by this file's tests; the lib target cannot see test-only consumers (backlog#1823)"
|
||||
)]
|
||||
fn merge(&self, m: LastDayTierStats) -> LastDayTierStats {
|
||||
let mut cl = self.clone();
|
||||
let mut cm = m;
|
||||
|
||||
@@ -177,9 +177,10 @@ fn should_record_remote_delete_failure(err: &std::io::Error) -> bool {
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
#[allow(dead_code)]
|
||||
struct ObjSweeper {
|
||||
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||
object: String,
|
||||
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||
bucket: String,
|
||||
version_id: Option<Uuid>,
|
||||
versioned: bool,
|
||||
@@ -191,9 +192,9 @@ struct ObjSweeper {
|
||||
remote_object: String,
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl ObjSweeper {
|
||||
#[allow(clippy::new_ret_no_self)]
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
pub async fn new(bucket: &str, object: &str) -> Result<Self, std::io::Error> {
|
||||
Ok(Self {
|
||||
object: object.into(),
|
||||
@@ -202,17 +203,20 @@ impl ObjSweeper {
|
||||
})
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
pub fn with_version(&mut self, vid: Option<Uuid>) -> &Self {
|
||||
self.version_id = vid.clone();
|
||||
self
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
pub fn with_versioning(&mut self, versioned: bool, suspended: bool) -> &Self {
|
||||
self.versioned = versioned;
|
||||
self.suspended = suspended;
|
||||
self
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
pub fn get_opts(&self) -> lifecycle::ObjectOpts {
|
||||
let mut opts = ObjectOpts {
|
||||
version_id: self.version_id.clone(),
|
||||
@@ -226,6 +230,7 @@ impl ObjSweeper {
|
||||
opts
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
pub fn set_transition_state(&mut self, info: TransitionedObject) {
|
||||
self.transition_tier = info.tier;
|
||||
self.transition_status = info.status;
|
||||
@@ -266,6 +271,7 @@ impl ObjSweeper {
|
||||
None
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
pub async fn sweep(&self, api: Arc<ECStore>) {
|
||||
let Some(je) = self.should_remove_remote_object() else {
|
||||
return;
|
||||
|
||||
@@ -312,9 +312,7 @@ mod tests {
|
||||
}
|
||||
#[derive(Deserialize)]
|
||||
struct LegacyBucketQuota {
|
||||
#[allow(dead_code)]
|
||||
quota: Option<u64>,
|
||||
#[allow(dead_code)]
|
||||
quota_type: LegacyQuotaType,
|
||||
}
|
||||
let legacy = serde_json::from_slice::<LegacyBucketQuota>(&json)
|
||||
|
||||
@@ -95,7 +95,6 @@ impl TransitionClient {
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
#[allow(dead_code)]
|
||||
pub struct GetRequest {
|
||||
pub buffer: Vec<u8>,
|
||||
pub offset: i64,
|
||||
@@ -107,11 +106,12 @@ pub struct GetRequest {
|
||||
pub setting_object_info: bool,
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub struct GetResponse {
|
||||
pub size: i64,
|
||||
//pub error: error,
|
||||
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||
pub did_read: bool,
|
||||
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||
pub object_info: ObjectInfo,
|
||||
}
|
||||
|
||||
|
||||
@@ -27,7 +27,6 @@ use tracing::warn;
|
||||
use crate::client::api_error_response::err_invalid_argument;
|
||||
|
||||
#[derive(Default)]
|
||||
#[allow(dead_code)]
|
||||
pub struct AdvancedGetOptions {
|
||||
pub replication_delete_marker: bool,
|
||||
pub is_replication_ready_for_delete_marker: bool,
|
||||
|
||||
@@ -360,7 +360,6 @@ impl TransitionClient {
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
#[allow(dead_code)]
|
||||
pub struct ListObjectsOptions {
|
||||
reverse_versions: bool,
|
||||
with_versions: bool,
|
||||
|
||||
@@ -137,8 +137,8 @@ impl Default for PutObjectOptions {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl PutObjectOptions {
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn set_match_etag(&mut self, etag: &str) {
|
||||
if etag == "*" {
|
||||
self.custom_header.insert("If-Match", HeaderValue::from_static("*"));
|
||||
@@ -149,6 +149,7 @@ impl PutObjectOptions {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn set_match_etag_except(&mut self, etag: &str) {
|
||||
if etag == "*" {
|
||||
self.custom_header.insert("If-None-Match", HeaderValue::from_static("*"));
|
||||
@@ -259,6 +260,7 @@ impl PutObjectOptions {
|
||||
header
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn validate(&self, c: TransitionClient) -> Result<(), std::io::Error> {
|
||||
//if self.checksum.is_set() {
|
||||
/*if !self.trailing_header_support {
|
||||
|
||||
@@ -55,7 +55,6 @@ pub struct RemoveBucketOptions {
|
||||
const DELETE_RESPONSE_PREVIEW_LEN: usize = 1024;
|
||||
|
||||
#[derive(Debug)]
|
||||
#[allow(dead_code)]
|
||||
pub struct AdvancedRemoveOptions {
|
||||
pub replication_delete_marker: bool,
|
||||
pub replication_status: ReplicationStatus,
|
||||
@@ -465,10 +464,10 @@ impl TransitionClient {
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
#[allow(dead_code)]
|
||||
pub struct RemoveObjectError {
|
||||
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||
object_name: String,
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||
version_id: String,
|
||||
err: Option<std::io::Error>,
|
||||
}
|
||||
|
||||
@@ -372,8 +372,8 @@ pub struct Checksum {
|
||||
computed: bool,
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl Checksum {
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn new(t: ChecksumMode, b: &[u8]) -> Checksum {
|
||||
if t.is_set() && b.len() == t.raw_byte_len() {
|
||||
return Checksum {
|
||||
@@ -385,7 +385,7 @@ impl Checksum {
|
||||
Checksum::default()
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn new_checksum_string(t: ChecksumMode, s: &str) -> Result<Checksum, std::io::Error> {
|
||||
let b = match base64_decode(s.as_bytes()) {
|
||||
Ok(b) => b,
|
||||
@@ -412,7 +412,7 @@ impl Checksum {
|
||||
base64_encode(&self.r)
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn raw(&self) -> Option<Vec<u8>> {
|
||||
if !self.is_set() {
|
||||
return None;
|
||||
|
||||
@@ -37,16 +37,17 @@ pub struct PutObjReader {
|
||||
//pub sealMD5Fn: SealMD5CurrFn,
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl PutObjReader {
|
||||
pub fn new(reader: HashReader) -> Self {
|
||||
Self { reader }
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn md5_current_hex_string(&self) -> String {
|
||||
self.reader.checksum().map(|v| v.encoded).unwrap_or_default()
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn with_encryption(&mut self, enc_reader: HashReader) -> Result<(), std::io::Error> {
|
||||
self.reader = enc_reader;
|
||||
|
||||
|
||||
@@ -214,6 +214,19 @@ fn pool_write_quorum(participant_count: usize) -> usize {
|
||||
(participant_count / 2) + 1
|
||||
}
|
||||
|
||||
/// Error for a peer that reported `success = false` without an error payload.
|
||||
///
|
||||
/// The message must stay identical across the peers of one operation: `reduce_errs`
|
||||
/// buckets `Error::Io` by kind plus rendered message, so any per-peer detail (address,
|
||||
/// timing) would split one shared failure into single-count buckets and downgrade a real
|
||||
/// dominant error into `ErasureWriteQuorum`.
|
||||
fn peer_failure_without_details(op: &str, bucket: Option<&str>) -> Error {
|
||||
match bucket {
|
||||
Some(bucket) => Error::other(format!("{op}({bucket}): peer returned failure without error details")),
|
||||
None => Error::other(format!("{op}: peer returned failure without error details")),
|
||||
}
|
||||
}
|
||||
|
||||
fn reduce_pool_write_quorum_errs(per_pool_errs: &[Option<Error>]) -> Option<Error> {
|
||||
if per_pool_errs.is_empty() {
|
||||
return Some(Error::ErasureWriteQuorum);
|
||||
@@ -1078,7 +1091,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
||||
return if let Some(err) = response.error {
|
||||
Err(err.into())
|
||||
} else {
|
||||
Err(Error::other(""))
|
||||
Err(peer_failure_without_details("heal_bucket", Some(bucket)))
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1105,7 +1118,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
||||
return if let Some(err) = response.error {
|
||||
Err(err.into())
|
||||
} else {
|
||||
Err(Error::other(""))
|
||||
Err(peer_failure_without_details("list_bucket", None))
|
||||
};
|
||||
}
|
||||
let bucket_infos = response
|
||||
@@ -1136,9 +1149,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
||||
return if let Some(err) = response.error {
|
||||
Err(err.into())
|
||||
} else {
|
||||
Err(Error::other(format!(
|
||||
"make_bucket({bucket}): peer returned failure without error details"
|
||||
)))
|
||||
Err(peer_failure_without_details("make_bucket", Some(bucket)))
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1162,7 +1173,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
||||
return if let Some(err) = response.error {
|
||||
Err(err.into())
|
||||
} else {
|
||||
Err(Error::other(""))
|
||||
Err(peer_failure_without_details("get_bucket_info", Some(bucket)))
|
||||
};
|
||||
}
|
||||
let bucket_info = serde_json::from_str::<BucketInfo>(&response.bucket_info)?;
|
||||
@@ -1190,7 +1201,7 @@ impl PeerS3Client for RemotePeerS3Client {
|
||||
return if let Some(err) = response.error {
|
||||
Err(err.into())
|
||||
} else {
|
||||
Err(Error::other(""))
|
||||
Err(peer_failure_without_details("delete_bucket", Some(bucket)))
|
||||
};
|
||||
}
|
||||
|
||||
@@ -2314,4 +2325,37 @@ mod tests {
|
||||
.collect::<Vec<_>>();
|
||||
assert_eq!(calls, vec![1, 1, 0, 0, 0, 0, 0, 0]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn peer_failure_without_details_names_operation_and_bucket() {
|
||||
for op in ["heal_bucket", "make_bucket", "get_bucket_info", "delete_bucket"] {
|
||||
let message = peer_failure_without_details(op, Some("ops-bucket")).to_string();
|
||||
assert!(message.contains(op), "{op} message must name the operation: {message}");
|
||||
assert!(message.contains("ops-bucket"), "{op} message must name the bucket: {message}");
|
||||
}
|
||||
|
||||
let message = peer_failure_without_details("list_bucket", None).to_string();
|
||||
assert!(message.contains("list_bucket"), "cluster-wide message must name the operation");
|
||||
assert!(!message.trim().is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn peer_failure_without_details_keeps_one_reduce_errs_bucket_per_operation() {
|
||||
// reduce_errs groups Io errors by kind plus rendered message: peers failing the
|
||||
// same operation on the same bucket must still reach quorum as one dominant error.
|
||||
let per_pool_errs = vec![
|
||||
Some(peer_failure_without_details("delete_bucket", Some("shared"))),
|
||||
Some(peer_failure_without_details("delete_bucket", Some("shared"))),
|
||||
Some(peer_failure_without_details("delete_bucket", Some("shared"))),
|
||||
];
|
||||
assert_eq!(
|
||||
reduce_pool_write_quorum_errs(&per_pool_errs),
|
||||
Some(peer_failure_without_details("delete_bucket", Some("shared")))
|
||||
);
|
||||
|
||||
assert_ne!(
|
||||
peer_failure_without_details("delete_bucket", Some("shared")),
|
||||
peer_failure_without_details("get_bucket_info", Some("shared"))
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,7 +39,6 @@ use rustfs_config::{
|
||||
};
|
||||
use std::sync::LazyLock;
|
||||
|
||||
#[allow(dead_code)]
|
||||
#[allow(clippy::declare_interior_mutable_const)]
|
||||
/// Default KVS for audit webhook settings.
|
||||
pub static DEFAULT_AUDIT_WEBHOOK_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||
@@ -117,7 +116,6 @@ pub static DEFAULT_AUDIT_WEBHOOK_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||
])
|
||||
});
|
||||
|
||||
#[allow(dead_code)]
|
||||
#[allow(clippy::declare_interior_mutable_const)]
|
||||
/// Default KVS for audit MQTT settings.
|
||||
pub static DEFAULT_AUDIT_MQTT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||
@@ -375,7 +373,6 @@ pub static DEFAULT_AUDIT_NATS_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||
])
|
||||
});
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub static DEFAULT_AUDIT_PULSAR_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||
KVS(vec![
|
||||
KV {
|
||||
|
||||
@@ -12,12 +12,9 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use crate::error::{Error, Result};
|
||||
use rustfs_config::server_config::{KV, KVS};
|
||||
use rustfs_config::{DEFAULT_HEAL_BITROT_CYCLE_SECS, HEAL_BITROT_CYCLE};
|
||||
use rustfs_utils::string::parse_bool;
|
||||
use std::sync::LazyLock;
|
||||
use std::time::Duration;
|
||||
|
||||
pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||
KVS(vec![KV {
|
||||
@@ -26,59 +23,3 @@ pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||
hidden_if_empty: false,
|
||||
}])
|
||||
});
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
pub struct Config {
|
||||
pub bitrot: String,
|
||||
pub sleep: Duration,
|
||||
pub io_count: usize,
|
||||
pub drive_workers: usize,
|
||||
pub cache: Duration,
|
||||
}
|
||||
|
||||
impl Config {
|
||||
pub fn bitrot_scan_cycle(&self) -> Duration {
|
||||
self.cache
|
||||
}
|
||||
|
||||
pub fn get_workers(&self) -> usize {
|
||||
self.drive_workers
|
||||
}
|
||||
|
||||
pub fn update(&mut self, nopts: &Config) {
|
||||
self.bitrot = nopts.bitrot.clone();
|
||||
self.io_count = nopts.io_count;
|
||||
self.sleep = nopts.sleep;
|
||||
self.drive_workers = nopts.drive_workers;
|
||||
}
|
||||
}
|
||||
|
||||
const RUSTFS_BITROT_CYCLE_IN_MONTHS: u64 = 1;
|
||||
|
||||
fn parse_bitrot_config(s: &str) -> Result<Duration> {
|
||||
match parse_bool(s) {
|
||||
Ok(enabled) => {
|
||||
if enabled {
|
||||
Ok(Duration::from_secs_f64(0.0))
|
||||
} else {
|
||||
Ok(Duration::from_secs_f64(-1.0))
|
||||
}
|
||||
}
|
||||
Err(_) => {
|
||||
if !s.ends_with("m") {
|
||||
return Err(Error::other("unknown format"));
|
||||
}
|
||||
|
||||
match s.trim_end_matches('m').parse::<u64>() {
|
||||
Ok(months) => {
|
||||
if months < RUSTFS_BITROT_CYCLE_IN_MONTHS {
|
||||
return Err(Error::other(format!("minimum bitrot cycle is {RUSTFS_BITROT_CYCLE_IN_MONTHS} month(s)")));
|
||||
}
|
||||
|
||||
Ok(Duration::from_secs(months * 30 * 24 * 60))
|
||||
}
|
||||
Err(err) => Err(Error::other(err)),
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,7 +16,6 @@
|
||||
|
||||
mod audit;
|
||||
pub mod com;
|
||||
#[allow(dead_code)]
|
||||
pub mod heal;
|
||||
mod notify;
|
||||
mod oidc;
|
||||
|
||||
@@ -16,6 +16,7 @@ use crate::bucket::replication::replication_state_from_filemeta;
|
||||
use crate::bucket::versioning_sys::BucketVersioningSys;
|
||||
use crate::bucket::{
|
||||
lifecycle::{
|
||||
LifecycleExpiryConfigs,
|
||||
bucket_lifecycle_audit::LcEventSrc,
|
||||
bucket_lifecycle_ops::{
|
||||
LifecycleOps, apply_expiry_on_transitioned_object, apply_expiry_rule_in, eval_action_from_lifecycle,
|
||||
@@ -1996,11 +1997,11 @@ impl PoolMeta {
|
||||
Ok(false)
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
pub fn validate(&self, pools: Vec<Arc<Sets>>) -> Result<bool> {
|
||||
struct PoolInfo {
|
||||
position: usize,
|
||||
completed: bool,
|
||||
#[allow(dead_code, reason = "written but never read back (backlog#1823)")]
|
||||
decom_started: bool,
|
||||
}
|
||||
|
||||
@@ -2335,6 +2336,10 @@ fn lifecycle_action_removes_data_movement_version(action: IlmAction) -> bool {
|
||||
)
|
||||
}
|
||||
|
||||
fn lifecycle_action_skips_heal_version(action: IlmAction) -> bool {
|
||||
action.delete()
|
||||
}
|
||||
|
||||
fn resolve_data_movement_lifecycle_expiry_result(action: IlmAction, apply_actions: bool, applied: bool) -> Result<bool> {
|
||||
if !apply_actions || applied {
|
||||
return Ok(true);
|
||||
@@ -2385,7 +2390,80 @@ pub(crate) async fn should_skip_lifecycle_for_data_movement(
|
||||
}
|
||||
}
|
||||
|
||||
pub struct HealLifecycleExpiryContext {
|
||||
configs: LifecycleExpiryConfigs,
|
||||
}
|
||||
|
||||
impl ECStore {
|
||||
pub async fn load_heal_lifecycle_expiry_context(&self, bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
|
||||
if bucket == RUSTFS_META_BUCKET {
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
let configs = get_expiry_configs(self, bucket).await?;
|
||||
if configs.lifecycle.is_none() {
|
||||
return Ok(None);
|
||||
}
|
||||
|
||||
Ok(Some(HealLifecycleExpiryContext { configs }))
|
||||
}
|
||||
|
||||
pub async fn enqueue_heal_lifecycle_expiry(
|
||||
self: &Arc<Self>,
|
||||
context: &HealLifecycleExpiryContext,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
version_id: Option<&str>,
|
||||
object_info: Option<&crate::object_api::ObjectInfo>,
|
||||
) -> Result<bool> {
|
||||
let Some(lifecycle_config) = context.configs.lifecycle.as_ref() else {
|
||||
return Ok(false);
|
||||
};
|
||||
|
||||
let object_info = if let Some(object_info) = object_info {
|
||||
if object_info.bucket != bucket || object_info.name != object {
|
||||
return Ok(false);
|
||||
}
|
||||
let snapshot_version_id = object_info
|
||||
.version_id
|
||||
.filter(|version_id| !version_id.is_nil())
|
||||
.map(|version_id| version_id.to_string());
|
||||
if snapshot_version_id.as_deref() != version_id {
|
||||
return Ok(false);
|
||||
}
|
||||
object_info.clone()
|
||||
} else {
|
||||
match self
|
||||
.get_object_info(
|
||||
bucket,
|
||||
object,
|
||||
&ObjectOptions {
|
||||
version_id: version_id.map(str::to_string),
|
||||
versioned: version_id.is_some(),
|
||||
expected_bucket_incarnation_id: Some(context.configs.bucket_incarnation_id),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(object_info) => object_info,
|
||||
Err(err) if is_err_object_not_found(&err) || is_err_version_not_found(&err) => return Ok(false),
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
};
|
||||
|
||||
let event = eval_action_from_lifecycle(lifecycle_config, context.configs.object_lock.as_deref(), &object_info).await;
|
||||
if !lifecycle_action_skips_heal_version(event.action) {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
if lifecycle_delete_all_versions_blocked_by_replication(self.clone(), bucket, &object_info.name, event.action).await? {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
Ok(apply_expiry_rule_in(self.clone(), &event, &LcEventSrc::Scanner, &object_info).await)
|
||||
}
|
||||
|
||||
async fn save_current_pool_meta(&self) -> Result<()> {
|
||||
let _save_guard = self.pool_meta_save_gate.lock().await;
|
||||
let snapshot = {
|
||||
@@ -4287,6 +4365,19 @@ mod tests {
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn lifecycle_action_skips_heal_version_for_every_delete_action() {
|
||||
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteAction));
|
||||
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteVersionAction));
|
||||
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteRestoredAction));
|
||||
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteRestoredVersionAction));
|
||||
assert!(lifecycle_action_skips_heal_version(IlmAction::DeleteAllVersionsAction));
|
||||
assert!(lifecycle_action_skips_heal_version(IlmAction::DelMarkerDeleteAllVersionsAction));
|
||||
assert!(!lifecycle_action_skips_heal_version(IlmAction::TransitionAction));
|
||||
assert!(!lifecycle_action_skips_heal_version(IlmAction::TransitionVersionAction));
|
||||
assert!(!lifecycle_action_skips_heal_version(IlmAction::NoneAction));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolve_data_movement_lifecycle_expiry_result_allows_dry_run_skip() {
|
||||
let skip = resolve_data_movement_lifecycle_expiry_result(IlmAction::DeleteVersionAction, false, false)
|
||||
@@ -4958,13 +5049,19 @@ fn is_disk_online_state(state: &str) -> bool {
|
||||
}
|
||||
|
||||
#[deprecated(since = "0.1.0", note = "Use fallback_total_capacity_dedup instead")]
|
||||
#[allow(dead_code)]
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "superseded by the replacement named in the comment at pools.rs:5071 (backlog#1823)"
|
||||
)]
|
||||
fn fallback_total_capacity(disks: &[rustfs_madmin::Disk]) -> usize {
|
||||
fallback_total_capacity_dedup(disks)
|
||||
}
|
||||
|
||||
#[deprecated(since = "0.1.0", note = "Use fallback_free_capacity_dedup instead")]
|
||||
#[allow(dead_code)]
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "superseded by the replacement named in the comment at pools.rs:5071 (backlog#1823)"
|
||||
)]
|
||||
fn fallback_free_capacity(disks: &[rustfs_madmin::Disk]) -> usize {
|
||||
fallback_free_capacity_dedup(disks)
|
||||
}
|
||||
|
||||
@@ -1140,11 +1140,11 @@ impl crate::storage_api_contracts::heal::HealOperations for Sets {
|
||||
|
||||
Err(Error::DiskNotFound)
|
||||
}
|
||||
#[tracing::instrument(skip(self))]
|
||||
async fn check_abandoned_parts(&self, _bucket: &str, _object: &str, _opts: &HealOpts) -> Result<()> {
|
||||
// Multipart orphan reconciliation is intentionally retained above the pool/set layers
|
||||
// until there is a concrete caller and a stable lower-level contract to implement.
|
||||
Err(StorageError::NotImplemented)
|
||||
#[tracing::instrument(level = "debug", skip(self, opts), fields(bucket = %bucket, object = %object, dry_run = opts.dry_run))]
|
||||
async fn check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
|
||||
self.get_disks_for_heal_object(object, opts)?
|
||||
.check_abandoned_parts(bucket, object, opts)
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1996,7 +1996,7 @@ mod tests {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn sets_check_abandoned_parts_returns_typed_not_implemented_error() {
|
||||
async fn sets_check_abandoned_parts_rejects_invalid_set_scope() {
|
||||
let format = FormatV3::new(1, 1);
|
||||
let sets = Sets {
|
||||
id: format.id,
|
||||
@@ -2021,10 +2021,21 @@ mod tests {
|
||||
};
|
||||
|
||||
let err = sets
|
||||
.check_abandoned_parts("bucket", "object", &HealOpts::default())
|
||||
.check_abandoned_parts(
|
||||
"bucket",
|
||||
"object",
|
||||
&HealOpts {
|
||||
set: Some(1),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect_err("abandoned-parts ownership should stay above the pool/set storage layers");
|
||||
assert!(matches!(err, StorageError::NotImplemented));
|
||||
.expect_err("out-of-range abandoned-parts set scope must fail closed");
|
||||
assert!(
|
||||
matches!(err, StorageError::InvalidArgument(_, ref field, ref reason)
|
||||
if field == "set" && reason.contains("invalid heal set index 1")),
|
||||
"unexpected invalid set error: {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
// Builds a single-set `Sets` over `SET_DRIVE_COUNT` local temp-dir disks,
|
||||
|
||||
@@ -6562,7 +6562,7 @@ impl LocalDisk {
|
||||
Ok(f)
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn get_metrics(&self) -> DiskMetrics {
|
||||
DiskMetrics::default()
|
||||
}
|
||||
|
||||
@@ -132,7 +132,6 @@ impl RebalanceStopPropagationRecord {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct DiskStat {
|
||||
pub total_space: u64,
|
||||
|
||||
@@ -16,8 +16,16 @@ use serde::{Deserialize, Serialize};
|
||||
use std::{fmt::Display, io};
|
||||
use tracing::info;
|
||||
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "tier config wire version stamped by the parity constructors below (backlog#1823)"
|
||||
)]
|
||||
const C_TIER_CONFIG_VER: &str = "v1";
|
||||
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "tier-name validation message reached only from the parity constructors below (backlog#1823)"
|
||||
)]
|
||||
const ERR_TIER_NAME_EMPTY: &str = "remote tier name empty";
|
||||
const WASABI_US_EAST_ENDPOINT: &str = "https://s3.wasabisys.com";
|
||||
const WASABI_ALTERNATIVE_ENDPOINTS: &[(&str, &str)] = &[
|
||||
@@ -264,7 +272,6 @@ impl Clone for TierConfig {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl TierConfig {
|
||||
pub(crate) fn clone_with_credentials(&self) -> Self {
|
||||
Self {
|
||||
@@ -284,6 +291,7 @@ impl TierConfig {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn endpoint(&self) -> String {
|
||||
match self.tier_type {
|
||||
TierType::S3 => self.s3.as_ref().map(|s| s.endpoint.clone()).unwrap_or_default(),
|
||||
@@ -303,6 +311,7 @@ impl TierConfig {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn bucket(&self) -> String {
|
||||
match self.tier_type {
|
||||
TierType::S3 => self.s3.as_ref().map(|s| s.bucket.clone()).unwrap_or_default(),
|
||||
@@ -322,6 +331,7 @@ impl TierConfig {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn prefix(&self) -> String {
|
||||
match self.tier_type {
|
||||
TierType::S3 => self.s3.as_ref().map(|s| s.prefix.clone()).unwrap_or_default(),
|
||||
@@ -341,6 +351,7 @@ impl TierConfig {
|
||||
}
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn region(&self) -> String {
|
||||
match self.tier_type {
|
||||
TierType::S3 => self.s3.as_ref().map(|s| s.region.clone()).unwrap_or_default(),
|
||||
@@ -457,7 +468,7 @@ impl TierWasabi {
|
||||
}
|
||||
|
||||
impl TierS3 {
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn create<F>(
|
||||
name: &str,
|
||||
access_key: &str,
|
||||
@@ -528,7 +539,7 @@ pub struct TierMinIO {
|
||||
}
|
||||
|
||||
impl TierMinIO {
|
||||
#[allow(dead_code)]
|
||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||
fn create<F>(
|
||||
name: &str,
|
||||
endpoint: &str,
|
||||
|
||||
@@ -14,7 +14,6 @@
|
||||
|
||||
use crate::services::tier::tier::TierConfigMgr;
|
||||
|
||||
#[allow(dead_code)]
|
||||
impl TierConfigMgr {
|
||||
pub fn msg_size(&self) -> usize {
|
||||
100
|
||||
|
||||
@@ -4860,6 +4860,14 @@ impl SetDisks {
|
||||
/// is best-effort maintenance: individual delete failures are logged and
|
||||
/// skipped rather than propagated.
|
||||
pub(crate) async fn reclaim_orphan_data_dirs(&self, bucket: &str, object: &str) -> disk::error::Result<usize> {
|
||||
self.reclaim_orphan_data_dirs_inner(bucket, object, false).await
|
||||
}
|
||||
|
||||
pub(crate) async fn dry_run_reclaim_orphan_data_dirs(&self, bucket: &str, object: &str) -> disk::error::Result<usize> {
|
||||
self.reclaim_orphan_data_dirs_inner(bucket, object, true).await
|
||||
}
|
||||
|
||||
async fn reclaim_orphan_data_dirs_inner(&self, bucket: &str, object: &str, dry_run: bool) -> disk::error::Result<usize> {
|
||||
let disks = self.get_disks_internal().await;
|
||||
|
||||
// Phase 1 (read-only): build the referenced-data-dir union and record the
|
||||
@@ -4967,6 +4975,20 @@ impl SetDisks {
|
||||
continue;
|
||||
}
|
||||
let stray = format!("{object}/{dir}");
|
||||
if dry_run {
|
||||
removed += 1;
|
||||
debug!(
|
||||
target: "rustfs_ecstore::set_disk",
|
||||
event = "heal_abandoned_parts",
|
||||
component = "ecstore",
|
||||
subsystem = "heal",
|
||||
state = "dry_run_matched",
|
||||
result = "matched",
|
||||
bucket, object, data_dir = %dir,
|
||||
"Heal abandoned parts dry-run matched orphaned data directory"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
match disk
|
||||
.delete(
|
||||
bucket,
|
||||
|
||||
@@ -6998,6 +6998,100 @@ mod tests {
|
||||
assert!(object_dir.join(STORAGE_FORMAT_FILE).exists(), "metadata must be preserved");
|
||||
}
|
||||
|
||||
async fn recv_abandoned_parts_trace(
|
||||
trace: &mut rustfs_common::trace_bus::TraceSubscription,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
state: &str,
|
||||
) -> rustfs_common::trace_bus::TraceEvent {
|
||||
for _ in 0..32 {
|
||||
let event = tokio::time::timeout(std::time::Duration::from_secs(1), trace.recv())
|
||||
.await
|
||||
.expect("abandoned-parts trace event should arrive")
|
||||
.expect("trace bus should stay open");
|
||||
if event.kind == rustfs_common::trace_bus::TraceKind::Heal
|
||||
&& event.func == rustfs_common::trace_bus::TraceFunc::HealCheckAbandonedParts
|
||||
&& event.bucket.as_deref() == Some(bucket)
|
||||
&& event.object.as_deref() == Some(object)
|
||||
&& trace_attr_string(&event, "state").as_deref() == Some(state)
|
||||
{
|
||||
return (*event).clone();
|
||||
}
|
||||
}
|
||||
|
||||
panic!("expected abandoned-parts trace state {state} for {bucket}/{object}");
|
||||
}
|
||||
|
||||
fn trace_attr_string(event: &rustfs_common::trace_bus::TraceEvent, key: &str) -> Option<String> {
|
||||
event.attrs.iter().find_map(|attr| {
|
||||
if attr.key != key {
|
||||
return None;
|
||||
}
|
||||
Some(match &attr.value {
|
||||
rustfs_common::trace_bus::TraceVal::Bool(value) => value.to_string(),
|
||||
rustfs_common::trace_bus::TraceVal::U64(value) => value.to_string(),
|
||||
rustfs_common::trace_bus::TraceVal::I64(value) => value.to_string(),
|
||||
rustfs_common::trace_bus::TraceVal::Str(value) => value.to_string(),
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn check_abandoned_parts_dry_run_counts_without_deleting() {
|
||||
let mut trace = rustfs_common::trace_bus::subscribe_trace_events();
|
||||
let (dir, disk) = make_single_local_disk().await;
|
||||
let live = Uuid::new_v4();
|
||||
let orphan = Uuid::new_v4();
|
||||
|
||||
let object_dir = dir.path().join("bucket").join("obj");
|
||||
write_object_meta_with_data_dirs(&object_dir, "bucket", "obj", &[live]).await;
|
||||
fs::create_dir_all(object_dir.join(live.to_string()))
|
||||
.await
|
||||
.expect("live data dir should be created");
|
||||
fs::create_dir_all(object_dir.join(orphan.to_string()))
|
||||
.await
|
||||
.expect("orphan data dir should be created");
|
||||
|
||||
let set = make_set_disks_with(vec![Some(disk)]).await;
|
||||
set.check_abandoned_parts(
|
||||
"bucket",
|
||||
"obj",
|
||||
&HealOpts {
|
||||
dry_run: true,
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("dry-run abandoned-parts check should succeed");
|
||||
let dry_run_trace = recv_abandoned_parts_trace(&mut trace, "bucket", "obj", "dry_run_matched").await;
|
||||
assert_eq!(trace_attr_string(&dry_run_trace, "dry_run").as_deref(), Some("true"));
|
||||
assert_eq!(trace_attr_string(&dry_run_trace, "data_dirs").as_deref(), Some("1"));
|
||||
|
||||
assert!(object_dir.join(live.to_string()).exists(), "referenced data dir must be preserved");
|
||||
assert!(object_dir.join(orphan.to_string()).exists(), "dry-run must not remove orphaned data dir");
|
||||
|
||||
set.check_abandoned_parts(
|
||||
"bucket",
|
||||
"obj",
|
||||
&HealOpts {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("abandoned-parts check should reclaim stale data dir");
|
||||
let reclaim_trace = recv_abandoned_parts_trace(&mut trace, "bucket", "obj", "reclaimed").await;
|
||||
assert_eq!(trace_attr_string(&reclaim_trace, "dry_run").as_deref(), Some("false"));
|
||||
assert_eq!(trace_attr_string(&reclaim_trace, "data_dirs").as_deref(), Some("1"));
|
||||
|
||||
assert!(
|
||||
object_dir.join(live.to_string()).exists(),
|
||||
"referenced data dir must remain after reclaim"
|
||||
);
|
||||
assert!(!object_dir.join(orphan.to_string()).exists(), "orphaned data dir must be removed");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reclaim_orphan_data_dirs_recovers_deferred_cleanup_after_restart() {
|
||||
let (dir, disk) = make_single_local_disk().await;
|
||||
@@ -12233,11 +12327,18 @@ mod tests {
|
||||
.expect_err("unsupported copy_object_part should return a typed error");
|
||||
assert!(matches!(copy_part_err, StorageError::NotImplemented));
|
||||
|
||||
let abandoned_err = set_disks
|
||||
.check_abandoned_parts("bucket", "object", &HealOpts::default())
|
||||
set_disks
|
||||
.check_abandoned_parts(
|
||||
"bucket",
|
||||
"object",
|
||||
&HealOpts {
|
||||
dry_run: true,
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect_err("abandoned-parts check should stay in the upper reconciliation layer");
|
||||
assert!(matches!(abandoned_err, StorageError::NotImplemented));
|
||||
.expect("abandoned-parts check should be callable on empty disk sets");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
|
||||
@@ -16,6 +16,7 @@ use super::super::*;
|
||||
use crate::disk::disk_store::DiskStoreRenameDataExt;
|
||||
use crate::io_support::bitrot::object_mmap_read_enabled;
|
||||
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
|
||||
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, trace_emit};
|
||||
use tracing::trace;
|
||||
|
||||
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
||||
@@ -2057,11 +2058,61 @@ impl crate::storage_api_contracts::heal::HealOperations for SetDisks {
|
||||
Err(Error::DiskNotFound)
|
||||
}
|
||||
|
||||
#[tracing::instrument(skip(self))]
|
||||
async fn check_abandoned_parts(&self, _bucket: &str, _object: &str, _opts: &HealOpts) -> Result<()> {
|
||||
// Multipart orphan reconciliation is intentionally retained above the set layer
|
||||
// until there is a concrete caller and a stable lower-level contract to implement.
|
||||
Err(StorageError::NotImplemented)
|
||||
#[tracing::instrument(level = "debug", skip(self, opts), fields(bucket = %bucket, object = %object, dry_run = opts.dry_run))]
|
||||
async fn check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
|
||||
let started_at = std::time::Instant::now();
|
||||
let _write_lock_guard = if !opts.no_lock {
|
||||
let ns_lock = self.new_ns_lock(bucket, object).await?;
|
||||
Some(
|
||||
ns_lock
|
||||
.get_write_lock(get_lock_acquire_timeout())
|
||||
.await
|
||||
.map_err(|e| self.map_namespace_lock_error(bucket, object, "write", e))?,
|
||||
)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let removed = if opts.dry_run {
|
||||
self.dry_run_reclaim_orphan_data_dirs(bucket, object).await?
|
||||
} else {
|
||||
self.reclaim_orphan_data_dirs(bucket, object).await?
|
||||
};
|
||||
let state = if opts.dry_run && removed > 0 {
|
||||
"dry_run_matched"
|
||||
} else if removed > 0 {
|
||||
"reclaimed"
|
||||
} else {
|
||||
"checked"
|
||||
};
|
||||
let data_dirs = u64::try_from(removed).unwrap_or(u64::MAX);
|
||||
|
||||
trace_emit(|| {
|
||||
TraceEvent::new(TraceKind::Heal, TraceFunc::HealCheckAbandonedParts)
|
||||
.with_bucket(bucket)
|
||||
.with_object(object)
|
||||
.with_duration(started_at.elapsed())
|
||||
.with_attr("state", state)
|
||||
.with_attr("dry_run", opts.dry_run)
|
||||
.with_attr("data_dirs", data_dirs)
|
||||
});
|
||||
|
||||
if removed > 0 {
|
||||
trace!(
|
||||
event = "heal_abandoned_parts",
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_HEAL,
|
||||
state = if opts.dry_run { "dry_run_matched" } else { "reclaimed" },
|
||||
result = "ok",
|
||||
bucket,
|
||||
object,
|
||||
dry_run = opts.dry_run,
|
||||
data_dirs = removed,
|
||||
"Heal abandoned parts checked object data directories"
|
||||
);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3246,4 +3297,223 @@ mod heal_result_report_tests {
|
||||
assert!(result.detail.contains("part 1"));
|
||||
assert!(result.detail.contains("bitrot_failure=true"));
|
||||
}
|
||||
|
||||
// HS-12 (backlog#1874): a versioned DELETE racing an object heal must never
|
||||
// resurrect the deleted version. The heal has real reconstruction work (a
|
||||
// shard of the doomed version is removed), so both sides touch the same
|
||||
// (bucket, object, data_dir); whichever order the ns write lock serializes
|
||||
// them in, the committed delete must win.
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn heal_racing_version_delete_never_resurrects_the_deleted_version() {
|
||||
let (temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
|
||||
let bucket = "heal-race-delete-no-resurrect";
|
||||
let object = "object.bin";
|
||||
set.make_bucket(
|
||||
bucket,
|
||||
&MakeBucketOptions {
|
||||
versioning_enabled: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("versioned bucket should be created");
|
||||
|
||||
let mut first_reader = PutObjReader::from_vec(vec![0x11; 1024 * 1024]);
|
||||
let first_info = set
|
||||
.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut first_reader,
|
||||
&ObjectOptions {
|
||||
versioned: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("first version should be written");
|
||||
let first_version = first_info
|
||||
.version_id
|
||||
.expect("versioned put should return the first version id")
|
||||
.to_string();
|
||||
|
||||
let mut second_reader = PutObjReader::from_vec(vec![0x22; 1024 * 1024]);
|
||||
let second_info = set
|
||||
.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut second_reader,
|
||||
&ObjectOptions {
|
||||
versioned: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("second version should be written");
|
||||
let second_version = second_info
|
||||
.version_id
|
||||
.expect("versioned put should return the second version id")
|
||||
.to_string();
|
||||
|
||||
// Damage one shard of the doomed version so the racing heal performs an
|
||||
// actual reconstruction over its data dir instead of an early exit.
|
||||
let doomed_source = disks[0]
|
||||
.read_version("", bucket, object, &first_version, &ReadOptions::default())
|
||||
.await
|
||||
.expect("doomed version metadata should be readable");
|
||||
let doomed_data_dir = doomed_source
|
||||
.data_dir
|
||||
.expect("non-inline version should have a data directory");
|
||||
tokio::fs::remove_file(
|
||||
temp_dirs[1]
|
||||
.path()
|
||||
.join(bucket)
|
||||
.join(object)
|
||||
.join(doomed_data_dir.to_string())
|
||||
.join("part.1"),
|
||||
)
|
||||
.await
|
||||
.expect("shard damage should be injected before the race");
|
||||
|
||||
let delete_set = set.clone();
|
||||
let (delete_res, heal_res) = tokio::join!(
|
||||
async {
|
||||
delete_set
|
||||
.delete_object(
|
||||
bucket,
|
||||
object,
|
||||
ObjectOptions {
|
||||
versioned: true,
|
||||
version_id: Some(first_version.clone()),
|
||||
object_lock_config_snapshot: Some(Arc::new(crate::set_disk::ObjectLockConfigSnapshot::new(
|
||||
crate::bucket::metadata_sys::ObjectLockConfigState::ConfirmedAbsent,
|
||||
))),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
},
|
||||
async {
|
||||
set.heal_object(
|
||||
bucket,
|
||||
object,
|
||||
"",
|
||||
&HealOpts {
|
||||
scan_mode: HealScanMode::Deep,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
},
|
||||
);
|
||||
delete_res.expect("version delete must succeed under lock serialization");
|
||||
// The heal may legitimately report a transient failure when the version
|
||||
// it was rebuilding disappears mid-flight; only the end state matters.
|
||||
drop(heal_res);
|
||||
|
||||
let resurrected = set
|
||||
.get_object_info(
|
||||
bucket,
|
||||
object,
|
||||
&ObjectOptions {
|
||||
versioned: true,
|
||||
version_id: Some(first_version.clone()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await;
|
||||
assert!(
|
||||
matches!(&resurrected, Err(Error::FileVersionNotFound) | Err(Error::ObjectNotFound(..))),
|
||||
"a racing heal must not resurrect the deleted version: {resurrected:?}"
|
||||
);
|
||||
|
||||
let survivor = set
|
||||
.get_object_info(
|
||||
bucket,
|
||||
object,
|
||||
&ObjectOptions {
|
||||
versioned: true,
|
||||
version_id: Some(second_version.clone()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("surviving version must remain readable after the race");
|
||||
assert_eq!(survivor.size, 1024 * 1024, "survivor size must be intact");
|
||||
}
|
||||
|
||||
// HS-12 (backlog#1874): unversioned overwrite commits race a Deep heal on
|
||||
// the same object. The overwrite's post-commit tail deletes the replaced
|
||||
// data dir without the ns lock (object.rs commit tail), which is exactly
|
||||
// the intersection the audit flagged: the heal must tolerate the tail race
|
||||
// (retryable outcome) and every committed overwrite must survive — the
|
||||
// final current version is exactly the last payload written.
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn heal_racing_unversioned_overwrites_preserves_the_last_commit() {
|
||||
let (temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
|
||||
let bucket = "heal-race-put-overwrite";
|
||||
let object = "object.bin";
|
||||
set.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created");
|
||||
|
||||
const ROUNDS: usize = 8;
|
||||
const PAYLOAD_SIZE: usize = 256 * 1024;
|
||||
let mut last_etag = String::new();
|
||||
for round in 0..ROUNDS {
|
||||
// Give the heal something to rebuild on alternating rounds: remove a
|
||||
// shard of the current data dir right before the race.
|
||||
if round % 2 == 1 {
|
||||
let current = disks[2]
|
||||
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
.expect("current metadata should be readable");
|
||||
if let Some(data_dir) = current.data_dir {
|
||||
let shard = temp_dirs[3]
|
||||
.path()
|
||||
.join(bucket)
|
||||
.join(object)
|
||||
.join(data_dir.to_string())
|
||||
.join("part.1");
|
||||
if shard.exists() {
|
||||
tokio::fs::remove_file(&shard)
|
||||
.await
|
||||
.expect("shard damage should be injectable mid-race");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let payload = vec![round as u8; PAYLOAD_SIZE];
|
||||
let mut put_reader = PutObjReader::from_vec(payload);
|
||||
let put_opts = ObjectOptions::default();
|
||||
let heal_opts = HealOpts {
|
||||
scan_mode: HealScanMode::Deep,
|
||||
..Default::default()
|
||||
};
|
||||
let (put_res, heal_res) = tokio::join!(
|
||||
set.put_object(bucket, object, &mut put_reader, &put_opts),
|
||||
set.heal_object(bucket, object, "", &heal_opts),
|
||||
);
|
||||
let put_info = put_res.expect("overwrite must succeed under lock serialization");
|
||||
last_etag = put_info.etag.clone().unwrap_or_default();
|
||||
// Heal outcome is unconstrained (may hit the tail race and report a
|
||||
// retryable error); the invariant is checked on the end state.
|
||||
drop(heal_res);
|
||||
}
|
||||
|
||||
let final_info = set
|
||||
.get_object_info(bucket, object, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("object must remain readable after the race loop");
|
||||
assert_eq!(
|
||||
final_info.size, PAYLOAD_SIZE as i64,
|
||||
"final current version must be the last committed overwrite"
|
||||
);
|
||||
assert_eq!(
|
||||
final_info.etag.unwrap_or_default(),
|
||||
last_etag,
|
||||
"the racing heal loop must never leave a stale or resurrected current version"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -23,6 +23,7 @@
|
||||
//! per-version `SetDisks::heal_object`.
|
||||
|
||||
use super::super::*;
|
||||
use crate::object_api::ObjectInfo;
|
||||
use std::collections::HashSet;
|
||||
use std::sync::Mutex;
|
||||
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
|
||||
@@ -39,12 +40,16 @@ const BACKGROUND_WALKDIR_STALL_TIMEOUT: Duration = Duration::from_secs(60);
|
||||
/// it must not gate healing logic — the delete-marker vs data path is chosen
|
||||
/// inside `ops/heal.rs` from the resolved latest metadata. `version_id` is
|
||||
/// normalized (nil/absent UUID => `None`).
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct HealWalkVersion {
|
||||
/// object key
|
||||
pub name: String,
|
||||
/// normalized version id (`None` when the version is nil/absent)
|
||||
pub version_id: Option<String>,
|
||||
/// version modification time as Unix nanoseconds
|
||||
pub mod_time_unix_nanos: Option<i128>,
|
||||
/// object snapshot for lifecycle evaluation
|
||||
pub lifecycle_object_info: Option<ObjectInfo>,
|
||||
/// whether this version is a delete marker (observability only)
|
||||
pub is_delete_marker: bool,
|
||||
}
|
||||
@@ -63,6 +68,7 @@ struct HealWalkCollector {
|
||||
bucket: String,
|
||||
batch_objects: usize,
|
||||
version_budget: usize,
|
||||
include_lifecycle_object_info: bool,
|
||||
objects: Mutex<Vec<HealWalkObject>>,
|
||||
decode_error: Mutex<Option<DiskError>>,
|
||||
version_total: AtomicUsize,
|
||||
@@ -116,10 +122,25 @@ impl HealWalkCollector {
|
||||
|
||||
let mut versions = Vec::with_capacity(fiv.versions.len() + fiv.free_versions.len());
|
||||
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
|
||||
let version_uuid = fi.version_id.filter(|version_id| !version_id.is_nil());
|
||||
let lifecycle_object_info = if self.include_lifecycle_object_info {
|
||||
let mut lifecycle_fi = fi.clone();
|
||||
lifecycle_fi.version_id = version_uuid;
|
||||
Some(ObjectInfo::from_file_info(
|
||||
&lifecycle_fi,
|
||||
&self.bucket,
|
||||
&entry.name,
|
||||
version_uuid.is_some(),
|
||||
))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
versions.push(HealWalkVersion {
|
||||
name: entry.name.clone(),
|
||||
// Normalize: nil/absent version id => None.
|
||||
version_id: fi.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string()),
|
||||
version_id: version_uuid.map(|u| u.to_string()),
|
||||
mod_time_unix_nanos: fi.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos()),
|
||||
lifecycle_object_info,
|
||||
is_delete_marker: fi.deleted,
|
||||
});
|
||||
}
|
||||
@@ -173,11 +194,26 @@ impl HealWalkCollector {
|
||||
}
|
||||
};
|
||||
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
|
||||
let vid = fi.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string());
|
||||
let version_uuid = fi.version_id.filter(|version_id| !version_id.is_nil());
|
||||
let vid = version_uuid.map(|u| u.to_string());
|
||||
if seen.insert(vid.clone()) {
|
||||
let lifecycle_object_info = if self.include_lifecycle_object_info {
|
||||
let mut lifecycle_fi = fi.clone();
|
||||
lifecycle_fi.version_id = version_uuid;
|
||||
Some(ObjectInfo::from_file_info(
|
||||
&lifecycle_fi,
|
||||
&self.bucket,
|
||||
&entry.name,
|
||||
version_uuid.is_some(),
|
||||
))
|
||||
} else {
|
||||
None
|
||||
};
|
||||
versions.push(HealWalkVersion {
|
||||
name: entry.name.clone(),
|
||||
version_id: vid,
|
||||
mod_time_unix_nanos: fi.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos()),
|
||||
lifecycle_object_info,
|
||||
is_delete_marker: fi.deleted,
|
||||
});
|
||||
}
|
||||
@@ -255,6 +291,7 @@ impl SetDisks {
|
||||
forward_to: Option<&str>,
|
||||
batch_objects: usize,
|
||||
version_budget: usize,
|
||||
include_lifecycle_object_info: bool,
|
||||
) -> disk::error::Result<(Vec<HealWalkVersion>, Option<String>, bool)> {
|
||||
assert!(batch_objects >= 2, "heal_walk_versions_page requires batch_objects >= 2");
|
||||
|
||||
@@ -264,6 +301,7 @@ impl SetDisks {
|
||||
bucket: bucket.to_string(),
|
||||
batch_objects,
|
||||
version_budget: version_budget.max(1),
|
||||
include_lifecycle_object_info,
|
||||
objects: Mutex::new(Vec::new()),
|
||||
decode_error: Mutex::new(None),
|
||||
version_total: AtomicUsize::new(0),
|
||||
@@ -347,6 +385,7 @@ mod tests {
|
||||
bucket: "bucket".to_string(),
|
||||
batch_objects: 2,
|
||||
version_budget: 2,
|
||||
include_lifecycle_object_info: false,
|
||||
objects: Mutex::new(Vec::new()),
|
||||
decode_error: Mutex::new(None),
|
||||
version_total: AtomicUsize::new(0),
|
||||
@@ -388,6 +427,8 @@ mod tests {
|
||||
HealWalkVersion {
|
||||
name: name.to_string(),
|
||||
version_id: Some(id.to_string()),
|
||||
mod_time_unix_nanos: None,
|
||||
lifecycle_object_info: None,
|
||||
is_delete_marker: dm,
|
||||
}
|
||||
}
|
||||
@@ -491,6 +532,7 @@ mod tests {
|
||||
bucket: "bucket".to_string(),
|
||||
batch_objects: 1000,
|
||||
version_budget: 10_000,
|
||||
include_lifecycle_object_info: false,
|
||||
objects: Mutex::new(Vec::new()),
|
||||
version_total: AtomicUsize::new(0),
|
||||
decode_error: Mutex::new(None),
|
||||
@@ -567,7 +609,7 @@ mod tests {
|
||||
.expect("corrupt test metadata should be written");
|
||||
|
||||
let error = set_disks
|
||||
.heal_walk_versions_page(bucket, "", None, 2, 2)
|
||||
.heal_walk_versions_page(bucket, "", None, 2, 2, false)
|
||||
.await
|
||||
.expect_err("semantic metadata corruption must fail the heal disk walk");
|
||||
|
||||
|
||||
@@ -18,6 +18,7 @@ use tracing::trace;
|
||||
|
||||
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
||||
const LOG_SUBSYSTEM_HEAL: &str = "heal";
|
||||
const EVENT_HEAL_ABANDONED_PARTS: &str = "heal_abandoned_parts";
|
||||
const EVENT_HEAL_FORMAT_COMPLETED: &str = "heal_format_completed";
|
||||
const EVENT_HEAL_OBJECT_STARTED: &str = "heal_object_started";
|
||||
|
||||
@@ -256,13 +257,40 @@ impl ECStore {
|
||||
|
||||
#[instrument(skip(self))]
|
||||
pub(super) async fn handle_check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
|
||||
let _ = (bucket, object, opts);
|
||||
// Stale multipart reconciliation is already owned by the lifecycle-driven
|
||||
// background cleanup path in `bucket_lifecycle_ops.rs`. There is currently
|
||||
// no stable object-heal contract that should fan this request out through
|
||||
// pool/set storage layers, so keep the placeholder explicit at the ECStore
|
||||
// boundary instead of dispatching into lower layers.
|
||||
Err(StorageError::NotImplemented)
|
||||
let object = encode_dir_object(object);
|
||||
let pools = self.get_pools_for_heal_object(opts)?;
|
||||
|
||||
let mut futures = Vec::with_capacity(pools.len());
|
||||
for pool in pools.iter() {
|
||||
futures.push(pool.check_abandoned_parts(bucket, &object, opts));
|
||||
}
|
||||
|
||||
let mut first_error = None;
|
||||
for result in join_all(futures).await {
|
||||
if let Err(err) = result
|
||||
&& first_error.is_none()
|
||||
{
|
||||
first_error = Some(err);
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(err) = first_error {
|
||||
return Err(err);
|
||||
}
|
||||
|
||||
trace!(
|
||||
event = EVENT_HEAL_ABANDONED_PARTS,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_HEAL,
|
||||
state = "completed",
|
||||
result = "ok",
|
||||
bucket,
|
||||
object,
|
||||
dry_run = opts.dry_run,
|
||||
"Heal abandoned parts completed"
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -34,6 +34,7 @@ impl ECStore {
|
||||
forward_to: Option<&str>,
|
||||
batch_objects: usize,
|
||||
version_budget: usize,
|
||||
include_lifecycle_object_info: bool,
|
||||
) -> Result<(Vec<HealWalkVersion>, Option<String>, bool)> {
|
||||
if pool_idx >= self.pools.len() || set_idx >= self.pools[pool_idx].disk_set.len() {
|
||||
return Err(Error::other(format!(
|
||||
@@ -43,7 +44,7 @@ impl ECStore {
|
||||
}
|
||||
|
||||
self.pools[pool_idx].disk_set[set_idx]
|
||||
.heal_walk_versions_page(bucket, prefix, forward_to, batch_objects, version_budget)
|
||||
.heal_walk_versions_page(bucket, prefix, forward_to, batch_objects, version_budget, include_lifecycle_object_info)
|
||||
.await
|
||||
.map_err(Error::from)
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user