Files
rustfs/crates/object-data-cache/src/moka_backend.rs
T
houseme eebd16d8a4 feat(cache): add object data cache engine and app flow (#4187)
* feat(cache): add object data cache engine

* feat(cache): wire app-layer object cache flow

* refactor(cache): streamline app-layer cache flow

* refactor(cache): tighten cache flow internals

* refactor: address final clippy cleanup

* chore(deps): update quick-xml to 0.41.0

* feat(cache): wire object data cache env config

* fix(cache): gate materialize fill by cache plan

* chore(cache): add object data cache benchmark gate

* fix(cache): guard object cache fill size mismatches

* refactor(cache): streamline object cache body planning

* fix(cache): align object cache rollout config

* test(cache): cover buffered object cache benchmark

* test(cache): isolate object cache benchmark metrics

* test(cache): mark materialize rollout experimental

* test(cache): tighten object cache benchmark gate

* fix(cache): address review findings for object data cache

- singleflight: clean up leader entry on cancellation (Drop impl) so a dropped GET future can no longer wedge all subsequent fills for the same key; switch the fill map to a std Mutex and add a regression test

- adapter: honor RUSTFS_OBJECT_DATA_CACHE_ENABLE=true by defaulting to hit_only when no explicit mode is set (explicit mode still wins)

- planner: treat nil version UUIDs as "no value" per repo convention so unversioned objects key under the canonical "null" instead of fragmenting the key space

- multipart: invalidate the object cache on the quota-exceeded rollback delete after complete-multipart, closing a stale-cache window

- layering: move the disabled-cache fallback into app::context and drop the new infra->app layer-dependency baseline entry

* fix(cache): close invalidation races and drop full-cache scan on writes

- index: make identity-index insert/remove/prune atomic via starshard compute_if_present/compute_if_absent so concurrent fills can no longer drop each other's keys (lost keys made entries unreachable to invalidation until TTL); add a concurrency regression test

- fill: register the key in the identity index before the entry becomes visible in the cache and re-check the index afterwards, undoing the fill when an invalidation raced in between (new skipped_invalidation_race fill result)

- invalidate: with the index now authoritative, remove the full-cache iter() fallback that made every PUT/DELETE of a never-cached object O(total cache entries) (two scans per PUT, 2N per batch delete)

- materialize-fill: fail the GET instead of falling back to the partially consumed stream after a mid-read error (the fallback would send a body missing its prefix under a full-length Content-Length), and log the same size-mismatch warning as the sibling buffering paths

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(storage): fix media-dependent buffer clamp expectation

test_concurrency_manager_multi_factor_strategy_buffer_clamp asserted media_cap.min(MI_B), but the implementation's final safety clamp is [32KiB, media_cap.max(MI_B)] — deliberately so a media cap above 1MiB (NVMe's 2MiB default) stays effective. The test only passed on machines detected as SSD/Unknown (cap == 1MiB) and failed on NVMe-backed CI runners with 2MiB != 1MiB. Assert the media cap itself, which is what the strategy actually guarantees on every environment.

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(storage): format buffer clamp assertion

* chore(logging): update tier guardrail path

---------

Signed-off-by: houseme <housemecn@gmail.com>
Co-authored-by: cxymds <cxymds@gmail.com>
Co-authored-by: overtrue <anzhengchao@gmail.com>
Co-authored-by: heihutu <heihutu@gmail.com>
2026-07-03 18:11:14 +08:00

289 lines
12 KiB
Rust

// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use crate::cache::{ObjectDataCacheFillResult, ObjectDataCacheGetPlan, ObjectDataCacheInvalidationResult, ObjectDataCacheLookup};
use crate::config::ObjectDataCacheConfig;
use crate::entry::ObjectDataCacheEntry;
use crate::index::{ObjectDataCacheIdentityIndex, ObjectDataCacheIndexInsertResult};
use crate::key::ObjectDataCacheIdentity;
use crate::memory::ObjectDataCacheMemoryGate;
use crate::singleflight::{ObjectDataCacheSingleflight, ObjectDataCacheSingleflightAcquire};
use crate::stats::ObjectDataCacheStats;
use bytes::Bytes;
use moka::future::Cache;
use std::sync::Arc;
/// Weighted Moka backend for reusable object bodies.
#[derive(Debug)]
pub struct MokaBackend {
cache: Cache<crate::key::ObjectDataCacheKey, Arc<ObjectDataCacheEntry>>,
index: ObjectDataCacheIdentityIndex,
singleflight: ObjectDataCacheSingleflight,
memory_gate: ObjectDataCacheMemoryGate,
}
impl MokaBackend {
/// Creates a new backend from the validated configuration.
pub fn new(
config: &ObjectDataCacheConfig,
stats: Arc<ObjectDataCacheStats>,
) -> Result<Self, crate::error::ObjectDataCacheConfigError> {
let max_capacity = config.resolved_max_bytes()?;
let ttl = config.ttl;
let time_to_idle = config.time_to_idle;
let cache = Cache::builder()
.max_capacity(max_capacity)
.weigher(|key, value: &Arc<ObjectDataCacheEntry>| value.estimated_weight(key))
.time_to_live(ttl)
.time_to_idle(time_to_idle)
.build();
Ok(Self {
cache,
index: ObjectDataCacheIdentityIndex::new(usize::from(config.identity_keys_max)),
singleflight: ObjectDataCacheSingleflight::new(Arc::clone(&stats)),
memory_gate: ObjectDataCacheMemoryGate::new(config, stats),
})
}
/// Returns the current cache entry count.
pub fn entry_count(&self) -> u64 {
self.cache.entry_count()
}
/// Returns the approximate weighted size of cached entries.
pub fn weighted_size(&self) -> u64 {
self.cache.weighted_size()
}
/// Looks up a cached body for the supplied plan.
pub async fn lookup_body(&self, plan: &ObjectDataCacheGetPlan) -> ObjectDataCacheLookup {
let ObjectDataCacheGetPlan::Cacheable { key } = plan else {
return ObjectDataCacheLookup::SkipNotCacheable;
};
match self.cache.get(key).await {
Some(entry) => ObjectDataCacheLookup::Hit(entry.bytes()),
None => ObjectDataCacheLookup::Miss,
}
}
/// Inserts a cached body for the supplied plan.
pub async fn fill_body(&self, plan: &ObjectDataCacheGetPlan, bytes: Bytes) -> ObjectDataCacheFillResult {
let ObjectDataCacheGetPlan::Cacheable { key } = plan else {
return ObjectDataCacheFillResult::SkippedNotCacheable;
};
let fill_state = self.singleflight.acquire(key.clone()).await;
let ObjectDataCacheSingleflightAcquire::Leader(leader) = fill_state else {
let ObjectDataCacheSingleflightAcquire::Waiter(waiter) = fill_state else {
unreachable!();
};
return waiter.wait().await;
};
if !self.memory_gate.allows_fill(u64::try_from(bytes.len()).unwrap_or(u64::MAX)) {
return leader.finish(ObjectDataCacheFillResult::SkippedMemoryPressure).await;
}
let identity = ObjectDataCacheIdentity::new(Arc::clone(&key.bucket), Arc::clone(&key.object));
self.index
.prune_missing(&identity, |candidate| self.cache.contains_key(candidate))
.await;
// Register the key in the identity index BEFORE the entry becomes
// visible in the cache, so a concurrent invalidation always finds it.
let result = match self.index.insert(identity.clone(), key.clone()).await {
ObjectDataCacheIndexInsertResult::Inserted | ObjectDataCacheIndexInsertResult::Duplicate => {
let entry = Arc::new(ObjectDataCacheEntry::new(bytes, key.size, Arc::clone(&key.etag)));
self.cache.insert(key.clone(), entry).await;
// An invalidation may have raced between the index and cache
// inserts; re-check the index and undo the fill so the stale
// body cannot outlive the invalidation.
if self.index.contains_key(&identity, key).await {
ObjectDataCacheFillResult::Inserted
} else {
self.cache.remove(key).await;
ObjectDataCacheFillResult::SkippedInvalidationRace
}
}
ObjectDataCacheIndexInsertResult::Overflow { cleared_keys } => {
for stale_key in cleared_keys {
self.cache.remove(&stale_key).await;
}
ObjectDataCacheFillResult::SkippedIdentityOverflow
}
};
leader.finish(result).await
}
/// Conservatively invalidates all cached keys matching the object identity.
///
/// The identity index is authoritative: fills register the key in the
/// index before the entry becomes visible in the cache (and undo the fill
/// if an invalidation raced in between), so no full-cache scan fallback is
/// needed when the index has no entry for the identity.
pub async fn invalidate_object(&self, identity: &ObjectDataCacheIdentity) -> ObjectDataCacheInvalidationResult {
let keys_to_remove = self.index.remove_identity(identity).await;
for key in keys_to_remove {
self.cache.remove(&key).await;
}
ObjectDataCacheInvalidationResult::Success
}
}
#[cfg(test)]
mod tests {
use super::MokaBackend;
use crate::cache::{
ObjectDataCacheFillResult, ObjectDataCacheGetPlan, ObjectDataCacheInvalidationResult, ObjectDataCacheLookup,
};
use crate::config::{ObjectDataCacheConfig, ObjectDataCacheMode};
use crate::key::{ObjectDataCacheBodyVariant, ObjectDataCacheIdentity, ObjectDataCacheKey};
use crate::stats::ObjectDataCacheStats;
use bytes::Bytes;
use std::sync::Arc;
use std::time::Duration;
fn enabled_config() -> ObjectDataCacheConfig {
ObjectDataCacheConfig {
mode: ObjectDataCacheMode::FillMaterializeEnabled,
max_bytes: 8_388_608,
max_memory_percent: 5,
max_entry_bytes: 1_048_576,
ttl: Duration::from_millis(100),
time_to_idle: Duration::from_millis(100),
min_free_memory_percent: 20,
fill_concurrency_per_cpu: 1,
fill_concurrency_max: 32,
identity_keys_max: 16,
}
}
fn cacheable_plan(object: &str, etag: &str) -> ObjectDataCacheGetPlan {
ObjectDataCacheGetPlan::Cacheable {
key: ObjectDataCacheKey::new("bucket", object, None, etag, 5, ObjectDataCacheBodyVariant::FullObjectPlainV1),
}
}
#[tokio::test]
async fn moka_backend_round_trips_cached_body() {
let backend =
MokaBackend::new(&enabled_config(), Arc::new(ObjectDataCacheStats::default())).expect("moka backend should build");
let plan = cacheable_plan("object", "etag-a");
let fill = backend.fill_body(&plan, Bytes::from_static(b"hello")).await;
let lookup = backend.lookup_body(&plan).await;
assert!(matches!(fill, ObjectDataCacheFillResult::Inserted));
assert!(matches!(lookup, ObjectDataCacheLookup::Hit(ref bytes) if bytes.as_ref() == b"hello"));
}
#[tokio::test]
async fn moka_backend_invalidates_matching_identity() {
let backend =
MokaBackend::new(&enabled_config(), Arc::new(ObjectDataCacheStats::default())).expect("moka backend should build");
let plan_a = cacheable_plan("object-a", "etag-a");
let plan_b = cacheable_plan("object-b", "etag-b");
let _ = backend.fill_body(&plan_a, Bytes::from_static(b"aaaaa")).await;
let _ = backend.fill_body(&plan_b, Bytes::from_static(b"bbbbb")).await;
let result = backend
.invalidate_object(&ObjectDataCacheIdentity::new("bucket", "object-a"))
.await;
let lookup_a = backend.lookup_body(&plan_a).await;
let lookup_b = backend.lookup_body(&plan_b).await;
assert!(matches!(result, ObjectDataCacheInvalidationResult::Success));
assert!(matches!(lookup_a, ObjectDataCacheLookup::Miss));
assert!(matches!(lookup_b, ObjectDataCacheLookup::Hit(_)));
}
#[tokio::test]
async fn moka_backend_expires_entries_by_ttl() {
let backend =
MokaBackend::new(&enabled_config(), Arc::new(ObjectDataCacheStats::default())).expect("moka backend should build");
let plan = cacheable_plan("object", "etag-a");
let _ = backend.fill_body(&plan, Bytes::from_static(b"hello")).await;
tokio::time::sleep(Duration::from_millis(150)).await;
let lookup = backend.lookup_body(&plan).await;
assert!(matches!(lookup, ObjectDataCacheLookup::Miss));
}
#[tokio::test]
async fn moka_backend_expires_entries_by_tti() {
let mut config = enabled_config();
config.ttl = Duration::from_secs(30);
config.time_to_idle = Duration::from_millis(100);
let backend = MokaBackend::new(&config, Arc::new(ObjectDataCacheStats::default())).expect("moka backend should build");
let plan = cacheable_plan("object", "etag-a");
let _ = backend.fill_body(&plan, Bytes::from_static(b"hello")).await;
tokio::time::sleep(Duration::from_millis(150)).await;
let lookup = backend.lookup_body(&plan).await;
assert!(matches!(lookup, ObjectDataCacheLookup::Miss));
}
#[tokio::test]
async fn moka_backend_skips_fill_under_memory_pressure() {
let stats = Arc::new(ObjectDataCacheStats::default());
let backend = MokaBackend::new(&enabled_config(), Arc::clone(&stats)).expect("moka backend should build");
backend
.memory_gate
.set_test_snapshot(Some(crate::memory::ObjectDataCacheMemorySnapshot {
total_bytes: 1_000,
available_bytes: 100,
}));
let plan = cacheable_plan("object", "etag-a");
let result = backend.fill_body(&plan, Bytes::from_static(b"hello")).await;
let lookup = backend.lookup_body(&plan).await;
assert_eq!(result, ObjectDataCacheFillResult::SkippedMemoryPressure);
assert!(matches!(lookup, ObjectDataCacheLookup::Miss));
assert_eq!(stats.snapshot().memory_pressure_events, 1);
}
#[tokio::test]
async fn moka_backend_singleflight_waiter_observes_leader_result() {
let stats = Arc::new(ObjectDataCacheStats::default());
let backend = Arc::new(MokaBackend::new(&enabled_config(), Arc::clone(&stats)).expect("moka backend should build"));
let plan = cacheable_plan("object", "etag-a");
let leader_plan = plan.clone();
let plan_clone = plan.clone();
let first = Arc::clone(&backend);
let second = Arc::clone(&backend);
let leader = tokio::spawn(async move { first.fill_body(&leader_plan, Bytes::from_static(b"hello")).await });
let waiter = tokio::spawn(async move { second.fill_body(&plan_clone, Bytes::from_static(b"hello")).await });
let leader_result = leader.await.expect("leader task should complete");
let waiter_result = waiter.await.expect("waiter task should complete");
let lookup = backend.lookup_body(&plan).await;
assert_eq!(leader_result, ObjectDataCacheFillResult::Inserted);
assert_eq!(waiter_result, ObjectDataCacheFillResult::Inserted);
assert!(matches!(lookup, ObjectDataCacheLookup::Hit(ref bytes) if bytes.as_ref() == b"hello"));
}
}