mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-25 13:36:50 +00:00
feat(cache): add object data cache engine and app flow (#4187)
* feat(cache): add object data cache engine * feat(cache): wire app-layer object cache flow * refactor(cache): streamline app-layer cache flow * refactor(cache): tighten cache flow internals * refactor: address final clippy cleanup * chore(deps): update quick-xml to 0.41.0 * feat(cache): wire object data cache env config * fix(cache): gate materialize fill by cache plan * chore(cache): add object data cache benchmark gate * fix(cache): guard object cache fill size mismatches * refactor(cache): streamline object cache body planning * fix(cache): align object cache rollout config * test(cache): cover buffered object cache benchmark * test(cache): isolate object cache benchmark metrics * test(cache): mark materialize rollout experimental * test(cache): tighten object cache benchmark gate * fix(cache): address review findings for object data cache - singleflight: clean up leader entry on cancellation (Drop impl) so a dropped GET future can no longer wedge all subsequent fills for the same key; switch the fill map to a std Mutex and add a regression test - adapter: honor RUSTFS_OBJECT_DATA_CACHE_ENABLE=true by defaulting to hit_only when no explicit mode is set (explicit mode still wins) - planner: treat nil version UUIDs as "no value" per repo convention so unversioned objects key under the canonical "null" instead of fragmenting the key space - multipart: invalidate the object cache on the quota-exceeded rollback delete after complete-multipart, closing a stale-cache window - layering: move the disabled-cache fallback into app::context and drop the new infra->app layer-dependency baseline entry * fix(cache): close invalidation races and drop full-cache scan on writes - index: make identity-index insert/remove/prune atomic via starshard compute_if_present/compute_if_absent so concurrent fills can no longer drop each other's keys (lost keys made entries unreachable to invalidation until TTL); add a concurrency regression test - fill: register the key in the identity index before the entry becomes visible in the cache and re-check the index afterwards, undoing the fill when an invalidation raced in between (new skipped_invalidation_race fill result) - invalidate: with the index now authoritative, remove the full-cache iter() fallback that made every PUT/DELETE of a never-cached object O(total cache entries) (two scans per PUT, 2N per batch delete) - materialize-fill: fail the GET instead of falling back to the partially consumed stream after a mid-read error (the fallback would send a body missing its prefix under a full-length Content-Length), and log the same size-mismatch warning as the sibling buffering paths Co-Authored-By: heihutu <heihutu@gmail.com> * test(storage): fix media-dependent buffer clamp expectation test_concurrency_manager_multi_factor_strategy_buffer_clamp asserted media_cap.min(MI_B), but the implementation's final safety clamp is [32KiB, media_cap.max(MI_B)] — deliberately so a media cap above 1MiB (NVMe's 2MiB default) stays effective. The test only passed on machines detected as SSD/Unknown (cap == 1MiB) and failed on NVMe-backed CI runners with 2MiB != 1MiB. Assert the media cap itself, which is what the strategy actually guarantees on every environment. Co-Authored-By: heihutu <heihutu@gmail.com> * test(storage): format buffer clamp assertion * chore(logging): update tier guardrail path --------- Signed-off-by: houseme <housemecn@gmail.com> Co-authored-by: cxymds <cxymds@gmail.com> Co-authored-by: overtrue <anzhengchao@gmail.com> Co-authored-by: heihutu <heihutu@gmail.com>
This commit is contained in:
@@ -0,0 +1,215 @@
|
||||
// Copyright 2024 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use crate::index::{ObjectDataCacheIndexInsertResult, ObjectDataCacheKeySet};
|
||||
use crate::key::{ObjectDataCacheIdentity, ObjectDataCacheKey};
|
||||
use starshard::{AsyncShardedHashMap, DEFAULT_SHARDS, SnapshotMode};
|
||||
use std::collections::hash_map::RandomState;
|
||||
use std::sync::Arc;
|
||||
|
||||
/// Async starshard-backed identity -> keys index.
|
||||
#[derive(Clone)]
|
||||
pub struct StarshardIdentityIndex {
|
||||
by_object: Arc<AsyncShardedHashMap<ObjectDataCacheIdentity, ObjectDataCacheKeySet, RandomState>>,
|
||||
max_keys_per_identity: usize,
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for StarshardIdentityIndex {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("StarshardIdentityIndex")
|
||||
.field("max_keys_per_identity", &self.max_keys_per_identity)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl StarshardIdentityIndex {
|
||||
/// Creates a new identity index.
|
||||
pub fn new(max_keys_per_identity: usize) -> Self {
|
||||
Self {
|
||||
by_object: Arc::new(AsyncShardedHashMap::with_shards_and_hasher_and_snapshot_mode(
|
||||
DEFAULT_SHARDS,
|
||||
RandomState::new(),
|
||||
SnapshotMode::Cached,
|
||||
)),
|
||||
max_keys_per_identity,
|
||||
}
|
||||
}
|
||||
|
||||
/// Inserts a key into the identity index.
|
||||
///
|
||||
/// Runs the read-modify-write under the shard write lock so concurrent
|
||||
/// fills for the same identity cannot drop each other's keys.
|
||||
pub async fn insert(&self, identity: ObjectDataCacheIdentity, key: ObjectDataCacheKey) -> ObjectDataCacheIndexInsertResult {
|
||||
let max_keys = self.max_keys_per_identity;
|
||||
loop {
|
||||
let mut outcome = None;
|
||||
{
|
||||
let outcome = &mut outcome;
|
||||
let key = key.clone();
|
||||
let _ = self
|
||||
.by_object
|
||||
.compute_if_present(&identity, move |mut key_set| {
|
||||
let result = key_set.insert(key, max_keys);
|
||||
let keep = !matches!(result, ObjectDataCacheIndexInsertResult::Overflow { .. }) && !key_set.is_empty();
|
||||
*outcome = Some(result);
|
||||
keep.then_some(key_set)
|
||||
})
|
||||
.await;
|
||||
}
|
||||
if let Some(result) = outcome {
|
||||
return result;
|
||||
}
|
||||
|
||||
// Identity not tracked yet: publish a fresh single-key set.
|
||||
let mut fresh = ObjectDataCacheKeySet::default();
|
||||
let result = fresh.insert(key.clone(), max_keys);
|
||||
if !matches!(result, ObjectDataCacheIndexInsertResult::Inserted) {
|
||||
return result;
|
||||
}
|
||||
let final_set = self.by_object.compute_if_absent(identity.clone(), move || fresh).await;
|
||||
if final_set.contains(&key) {
|
||||
return ObjectDataCacheIndexInsertResult::Inserted;
|
||||
}
|
||||
// Lost the race to a concurrent insert; retry against the now
|
||||
// present entry.
|
||||
}
|
||||
}
|
||||
|
||||
/// Removes all keys tracked for an identity.
|
||||
pub async fn remove_identity(&self, identity: &ObjectDataCacheIdentity) -> Vec<ObjectDataCacheKey> {
|
||||
self.by_object
|
||||
.remove(identity)
|
||||
.await
|
||||
.map_or_else(Vec::new, |set| set.cloned())
|
||||
}
|
||||
|
||||
/// Removes a single key tracked under an identity.
|
||||
pub async fn remove_key(&self, identity: &ObjectDataCacheIdentity, key: &ObjectDataCacheKey) -> bool {
|
||||
let mut removed = false;
|
||||
{
|
||||
let removed = &mut removed;
|
||||
let _ = self
|
||||
.by_object
|
||||
.compute_if_present(identity, move |mut key_set| {
|
||||
*removed = key_set.remove_key(key);
|
||||
(!key_set.is_empty()).then_some(key_set)
|
||||
})
|
||||
.await;
|
||||
}
|
||||
removed
|
||||
}
|
||||
|
||||
/// Returns whether the identity currently tracks the supplied key.
|
||||
pub async fn contains_key(&self, identity: &ObjectDataCacheIdentity, key: &ObjectDataCacheKey) -> bool {
|
||||
self.by_object
|
||||
.get(identity)
|
||||
.await
|
||||
.is_some_and(|key_set| key_set.contains(key))
|
||||
}
|
||||
|
||||
/// Removes index keys that no longer exist in the cache.
|
||||
pub async fn prune_missing<F>(&self, identity: &ObjectDataCacheIdentity, mut key_exists: F)
|
||||
where
|
||||
F: FnMut(&ObjectDataCacheKey) -> bool,
|
||||
{
|
||||
let _ = self
|
||||
.by_object
|
||||
.compute_if_present(identity, move |mut key_set| {
|
||||
key_set.retain(|key| key_exists(key));
|
||||
(!key_set.is_empty()).then_some(key_set)
|
||||
})
|
||||
.await;
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::StarshardIdentityIndex;
|
||||
use crate::index::ObjectDataCacheIndexInsertResult;
|
||||
use crate::key::{ObjectDataCacheBodyVariant, ObjectDataCacheIdentity, ObjectDataCacheKey};
|
||||
|
||||
fn identity() -> ObjectDataCacheIdentity {
|
||||
ObjectDataCacheIdentity::new("bucket", "object")
|
||||
}
|
||||
|
||||
fn key(id: &str) -> ObjectDataCacheKey {
|
||||
ObjectDataCacheKey::new("bucket", "object", Some(id), "etag", 1, ObjectDataCacheBodyVariant::FullObjectPlainV1)
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn identity_index_removes_all_keys_for_identity() {
|
||||
let index = StarshardIdentityIndex::new(4);
|
||||
let identity = identity();
|
||||
let key_a = key("v1");
|
||||
let key_b = key("v2");
|
||||
|
||||
let _ = index.insert(identity.clone(), key_a.clone()).await;
|
||||
let _ = index.insert(identity.clone(), key_b.clone()).await;
|
||||
let removed = index.remove_identity(&identity).await;
|
||||
|
||||
assert_eq!(removed, vec![key_a, key_b]);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn identity_index_prunes_stale_keys() {
|
||||
let index = StarshardIdentityIndex::new(4);
|
||||
let identity = identity();
|
||||
let key_a = key("v1");
|
||||
let key_b = key("v2");
|
||||
|
||||
let _ = index.insert(identity.clone(), key_a.clone()).await;
|
||||
let _ = index.insert(identity.clone(), key_b.clone()).await;
|
||||
index.prune_missing(&identity, |candidate| candidate == &key_b).await;
|
||||
let removed = index.remove_identity(&identity).await;
|
||||
|
||||
assert_eq!(removed, vec![key_b]);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn identity_index_concurrent_inserts_keep_all_keys() {
|
||||
let index = StarshardIdentityIndex::new(64);
|
||||
let identity = identity();
|
||||
|
||||
let mut handles = Vec::new();
|
||||
for i in 0..32 {
|
||||
let index = index.clone();
|
||||
let identity = identity.clone();
|
||||
handles.push(tokio::spawn(async move { index.insert(identity, key(&format!("v{i}"))).await }));
|
||||
}
|
||||
for handle in handles {
|
||||
handle.await.expect("insert task should complete");
|
||||
}
|
||||
|
||||
let removed = index.remove_identity(&identity).await;
|
||||
assert_eq!(removed.len(), 32, "no concurrent insert may drop another fill's key");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn identity_index_overflow_clears_identity() {
|
||||
let index = StarshardIdentityIndex::new(1);
|
||||
let identity = identity();
|
||||
let key_a = key("v1");
|
||||
let key_b = key("v2");
|
||||
|
||||
let _ = index.insert(identity.clone(), key_a.clone()).await;
|
||||
let result = index.insert(identity.clone(), key_b).await;
|
||||
let removed = index.remove_identity(&identity).await;
|
||||
|
||||
assert!(matches!(
|
||||
result,
|
||||
ObjectDataCacheIndexInsertResult::Overflow { cleared_keys } if cleared_keys == vec![key_a]
|
||||
));
|
||||
assert!(removed.is_empty());
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user