mirror of
https://github.com/rustfs/rustfs.git
synced 2026-07-27 16:48:58 +00:00
eebd16d8a4
* feat(cache): add object data cache engine * feat(cache): wire app-layer object cache flow * refactor(cache): streamline app-layer cache flow * refactor(cache): tighten cache flow internals * refactor: address final clippy cleanup * chore(deps): update quick-xml to 0.41.0 * feat(cache): wire object data cache env config * fix(cache): gate materialize fill by cache plan * chore(cache): add object data cache benchmark gate * fix(cache): guard object cache fill size mismatches * refactor(cache): streamline object cache body planning * fix(cache): align object cache rollout config * test(cache): cover buffered object cache benchmark * test(cache): isolate object cache benchmark metrics * test(cache): mark materialize rollout experimental * test(cache): tighten object cache benchmark gate * fix(cache): address review findings for object data cache - singleflight: clean up leader entry on cancellation (Drop impl) so a dropped GET future can no longer wedge all subsequent fills for the same key; switch the fill map to a std Mutex and add a regression test - adapter: honor RUSTFS_OBJECT_DATA_CACHE_ENABLE=true by defaulting to hit_only when no explicit mode is set (explicit mode still wins) - planner: treat nil version UUIDs as "no value" per repo convention so unversioned objects key under the canonical "null" instead of fragmenting the key space - multipart: invalidate the object cache on the quota-exceeded rollback delete after complete-multipart, closing a stale-cache window - layering: move the disabled-cache fallback into app::context and drop the new infra->app layer-dependency baseline entry * fix(cache): close invalidation races and drop full-cache scan on writes - index: make identity-index insert/remove/prune atomic via starshard compute_if_present/compute_if_absent so concurrent fills can no longer drop each other's keys (lost keys made entries unreachable to invalidation until TTL); add a concurrency regression test - fill: register the key in the identity index before the entry becomes visible in the cache and re-check the index afterwards, undoing the fill when an invalidation raced in between (new skipped_invalidation_race fill result) - invalidate: with the index now authoritative, remove the full-cache iter() fallback that made every PUT/DELETE of a never-cached object O(total cache entries) (two scans per PUT, 2N per batch delete) - materialize-fill: fail the GET instead of falling back to the partially consumed stream after a mid-read error (the fallback would send a body missing its prefix under a full-length Content-Length), and log the same size-mismatch warning as the sibling buffering paths Co-Authored-By: heihutu <heihutu@gmail.com> * test(storage): fix media-dependent buffer clamp expectation test_concurrency_manager_multi_factor_strategy_buffer_clamp asserted media_cap.min(MI_B), but the implementation's final safety clamp is [32KiB, media_cap.max(MI_B)] — deliberately so a media cap above 1MiB (NVMe's 2MiB default) stays effective. The test only passed on machines detected as SSD/Unknown (cap == 1MiB) and failed on NVMe-backed CI runners with 2MiB != 1MiB. Assert the media cap itself, which is what the strategy actually guarantees on every environment. Co-Authored-By: heihutu <heihutu@gmail.com> * test(storage): format buffer clamp assertion * chore(logging): update tier guardrail path --------- Signed-off-by: houseme <housemecn@gmail.com> Co-authored-by: cxymds <cxymds@gmail.com> Co-authored-by: overtrue <anzhengchao@gmail.com> Co-authored-by: heihutu <heihutu@gmail.com>
199 lines
7.4 KiB
Rust
199 lines
7.4 KiB
Rust
// Copyright 2024 RustFS Team
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
use crate::cache::ObjectDataCacheFillResult;
|
|
use crate::key::ObjectDataCacheKey;
|
|
use crate::metrics::{record_singleflight_join, set_inflight_fills};
|
|
use crate::stats::ObjectDataCacheStats;
|
|
use std::collections::HashMap;
|
|
use std::sync::{Arc, Mutex};
|
|
use tokio::sync::watch;
|
|
|
|
type FillMap = Mutex<HashMap<ObjectDataCacheKey, watch::Sender<Option<ObjectDataCacheFillResult>>>>;
|
|
|
|
fn lock_fills(
|
|
fills: &FillMap,
|
|
) -> std::sync::MutexGuard<'_, HashMap<ObjectDataCacheKey, watch::Sender<Option<ObjectDataCacheFillResult>>>> {
|
|
fills.lock().unwrap_or_else(|poisoned| poisoned.into_inner())
|
|
}
|
|
|
|
/// Shared singleflight controller for cache fill operations.
|
|
#[derive(Debug)]
|
|
pub struct ObjectDataCacheSingleflight {
|
|
fills: FillMap,
|
|
stats: Arc<ObjectDataCacheStats>,
|
|
}
|
|
|
|
impl ObjectDataCacheSingleflight {
|
|
/// Creates a new singleflight controller.
|
|
pub fn new(stats: Arc<ObjectDataCacheStats>) -> Self {
|
|
Self {
|
|
fills: Mutex::new(HashMap::new()),
|
|
stats,
|
|
}
|
|
}
|
|
|
|
/// Acquires a leader-or-waiter role for the supplied cache key.
|
|
pub async fn acquire(&self, key: ObjectDataCacheKey) -> ObjectDataCacheSingleflightAcquire<'_> {
|
|
let mut fills = lock_fills(&self.fills);
|
|
if let Some(sender) = fills.get(&key) {
|
|
record_singleflight_join(&self.stats);
|
|
return ObjectDataCacheSingleflightAcquire::Waiter(ObjectDataCacheSingleflightWaiter { rx: sender.subscribe() });
|
|
}
|
|
|
|
let (tx, _rx) = watch::channel(None);
|
|
fills.insert(key.clone(), tx.clone());
|
|
set_inflight_fills(&self.stats, "moka", fills.len());
|
|
drop(fills);
|
|
|
|
ObjectDataCacheSingleflightAcquire::Leader(ObjectDataCacheSingleflightLeader {
|
|
key,
|
|
tx,
|
|
fills: &self.fills,
|
|
stats: Arc::clone(&self.stats),
|
|
finished: false,
|
|
})
|
|
}
|
|
}
|
|
|
|
/// Leader or waiter result from the singleflight map.
|
|
pub enum ObjectDataCacheSingleflightAcquire<'a> {
|
|
/// Caller is responsible for performing the fill operation.
|
|
Leader(ObjectDataCacheSingleflightLeader<'a>),
|
|
/// Caller must wait for the leader's result.
|
|
Waiter(ObjectDataCacheSingleflightWaiter),
|
|
}
|
|
|
|
/// Leader handle for a singleflight fill operation.
|
|
pub struct ObjectDataCacheSingleflightLeader<'a> {
|
|
key: ObjectDataCacheKey,
|
|
tx: watch::Sender<Option<ObjectDataCacheFillResult>>,
|
|
fills: &'a FillMap,
|
|
stats: Arc<ObjectDataCacheStats>,
|
|
finished: bool,
|
|
}
|
|
|
|
impl<'a> ObjectDataCacheSingleflightLeader<'a> {
|
|
/// Completes the leader operation and publishes the shared result.
|
|
pub async fn finish(mut self, result: ObjectDataCacheFillResult) -> ObjectDataCacheFillResult {
|
|
self.remove_entry();
|
|
self.finished = true;
|
|
let _ = self.tx.send(Some(result.clone()));
|
|
result
|
|
}
|
|
|
|
fn remove_entry(&self) {
|
|
let mut fills = lock_fills(self.fills);
|
|
fills.remove(&self.key);
|
|
set_inflight_fills(&self.stats, "moka", fills.len());
|
|
}
|
|
}
|
|
|
|
impl<'a> Drop for ObjectDataCacheSingleflightLeader<'a> {
|
|
fn drop(&mut self) {
|
|
// A leader dropped without finish() was cancelled mid-fill (e.g. the
|
|
// GET request future was aborted). Remove the map entry so its sender
|
|
// clone is released and waiters observe the closed channel instead of
|
|
// blocking forever, and so a later fill can become the new leader.
|
|
if !self.finished {
|
|
self.remove_entry();
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Waiter handle for a singleflight fill operation.
|
|
pub struct ObjectDataCacheSingleflightWaiter {
|
|
rx: watch::Receiver<Option<ObjectDataCacheFillResult>>,
|
|
}
|
|
|
|
impl ObjectDataCacheSingleflightWaiter {
|
|
/// Waits for the shared fill result.
|
|
pub async fn wait(mut self) -> ObjectDataCacheFillResult {
|
|
if self.rx.wait_for(|value| value.is_some()).await.is_err() {
|
|
return ObjectDataCacheFillResult::SkippedSingleflightClosed;
|
|
}
|
|
|
|
self.rx
|
|
.borrow()
|
|
.as_ref()
|
|
.cloned()
|
|
.unwrap_or(ObjectDataCacheFillResult::SkippedSingleflightClosed)
|
|
}
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::{ObjectDataCacheSingleflight, ObjectDataCacheSingleflightAcquire};
|
|
use crate::cache::ObjectDataCacheFillResult;
|
|
use crate::key::{ObjectDataCacheBodyVariant, ObjectDataCacheKey};
|
|
use crate::stats::ObjectDataCacheStats;
|
|
use std::sync::Arc;
|
|
|
|
fn key() -> ObjectDataCacheKey {
|
|
ObjectDataCacheKey::new("bucket", "object", None, "etag", 1, ObjectDataCacheBodyVariant::FullObjectPlainV1)
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn singleflight_returns_waiter_for_second_caller() {
|
|
let stats = Arc::new(ObjectDataCacheStats::default());
|
|
let singleflight = ObjectDataCacheSingleflight::new(Arc::clone(&stats));
|
|
|
|
let first = singleflight.acquire(key()).await;
|
|
let second = singleflight.acquire(key()).await;
|
|
|
|
let leader = match first {
|
|
ObjectDataCacheSingleflightAcquire::Leader(leader) => leader,
|
|
ObjectDataCacheSingleflightAcquire::Waiter(_) => panic!("first caller must become leader"),
|
|
};
|
|
let waiter = match second {
|
|
ObjectDataCacheSingleflightAcquire::Leader(_) => panic!("second caller must become waiter"),
|
|
ObjectDataCacheSingleflightAcquire::Waiter(waiter) => waiter,
|
|
};
|
|
|
|
let waiter_task = tokio::spawn(async move { waiter.wait().await });
|
|
let leader_result = leader.finish(ObjectDataCacheFillResult::Inserted).await;
|
|
let waiter_result = waiter_task.await.expect("waiter task should complete");
|
|
|
|
assert_eq!(leader_result, ObjectDataCacheFillResult::Inserted);
|
|
assert_eq!(waiter_result, ObjectDataCacheFillResult::Inserted);
|
|
assert_eq!(stats.snapshot().singleflight_joins, 1);
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn cancelled_leader_releases_key_and_unblocks_waiters() {
|
|
let stats = Arc::new(ObjectDataCacheStats::default());
|
|
let singleflight = ObjectDataCacheSingleflight::new(Arc::clone(&stats));
|
|
|
|
let leader = match singleflight.acquire(key()).await {
|
|
ObjectDataCacheSingleflightAcquire::Leader(leader) => leader,
|
|
ObjectDataCacheSingleflightAcquire::Waiter(_) => panic!("first caller must become leader"),
|
|
};
|
|
let waiter = match singleflight.acquire(key()).await {
|
|
ObjectDataCacheSingleflightAcquire::Leader(_) => panic!("second caller must become waiter"),
|
|
ObjectDataCacheSingleflightAcquire::Waiter(waiter) => waiter,
|
|
};
|
|
|
|
// Dropping without finish() simulates the leader future being cancelled.
|
|
drop(leader);
|
|
|
|
let waiter_result = waiter.wait().await;
|
|
assert_eq!(waiter_result, ObjectDataCacheFillResult::SkippedSingleflightClosed);
|
|
|
|
assert!(
|
|
matches!(singleflight.acquire(key()).await, ObjectDataCacheSingleflightAcquire::Leader(_)),
|
|
"key must be released after a cancelled leader"
|
|
);
|
|
}
|
|
}
|