feat(storage): harden internode data-path controls (#4224)

* fix(rio): propagate http writer shutdown errors

* fix(ecstore): unify remote lock rpc deadlines

* fix(storage): reject corrupt read multiple payloads

* feat(rio): add internode http tuning profiles

* feat(metrics): add internode baseline signals

* feat(ecstore): observe shard locality topology

* feat(ecstore): gate shard locality scheduling

* feat(ecstore): gate batch read version rpc

* feat(ecstore): observe batch processor adaptation

* feat(ecstore): gate batch processor observation

* docs: add get benchmark regression analysis

* docs: add issue 797 execution plan status

* fix(ecstore): require explicit batch rpc support

* fix(ecstore): honor documented batch read gate

* fix(ecstore): keep batch read gate stable per call

* chore: update workspace dependencies

* feat(ecstore): log batch read gate decisions

* feat(ecstore): count batch read gate decisions

* test(issue-797): add local internode A/B runner

* test(rio): fix tuning profile spelling fixture

* fix(protocols): adapt sftp channel open callbacks

* fix(metrics): wrap batch processor observation args

* chore(docs): keep issue notes local only

* fix(storage): address internode review feedback

* fix(storage): address internode data-path review findings

- Run the BatchReadVersion auto-mode unary fallback outside the batch
  RPC deadline so each read_version keeps its own per-op timeout and
  health accounting instead of racing the whole batch against one
  drive timeout.
- Cap adaptive batch-processor concurrency growth at a hard multiple
  of the configured baseline so sustained fast batches cannot ratchet
  past the configured limit.
- Parse RUSTFS_INTERNODE_HTTP_* tuning, RUSTFS_BATCH_PROCESSOR_ADAPTIVE,
  and RUSTFS_METADATA_BATCH_READ once per process instead of re-reading
  the environment on hot paths.
- Skip shard read-cost collection in observe mode when stage metrics
  are disabled, and cache the local endpoint host list instead of
  rebuilding it on every read.
- Allow --warp-extra-args values starting with -- and drop the unused
  warp_hosts_csv helper in the issue-797 A/B runner.

* fix(storage): address internode data-path review findings

- Run the BatchReadVersion auto-mode unary fallback outside the batch
  RPC deadline so each read_version keeps its own per-op timeout and
  health accounting instead of racing the whole batch against one
  drive timeout.
- Cap adaptive batch-processor concurrency growth at a hard multiple
  of the configured baseline so sustained fast batches cannot ratchet
  past the configured limit.
- Parse RUSTFS_INTERNODE_HTTP_* tuning, RUSTFS_BATCH_PROCESSOR_ADAPTIVE,
  and RUSTFS_METADATA_BATCH_READ once per process instead of re-reading
  the environment on hot paths.
- Skip shard read-cost collection in observe mode when stage metrics
  are disabled, and cache the local endpoint host list instead of
  rebuilding it on every read.
- Allow --warp-extra-args values starting with -- and drop the unused
  warp_hosts_csv helper in the issue-797 A/B runner.

Co-Authored-By: heihutu<heihutu@gmail.com>

* fix(storage): align buffer clamp test with media cap

* fix(ecstore): release optimized read locks before streaming

---------

Co-authored-by: Zhengchao An <anzhengchao@gmail.com>
This commit is contained in:
houseme
2026-07-03 17:08:15 +08:00
committed by GitHub
parent a9ac5f578d
commit 25d80d7c60
26 changed files with 3627 additions and 483 deletions
+34 -8
View File
@@ -270,10 +270,19 @@ impl AsyncRead for SetDiskLockGuardedReader {
}
}
fn attach_set_disk_read_lock_guard(mut reader: GetObjectReader, read_lock_guard: Option<ObjectLockDiagGuard>) -> GetObjectReader {
if let Some(guard) = read_lock_guard
&& reader.buffered_body.is_none()
{
fn finish_set_disk_read_lock(
mut reader: GetObjectReader,
read_lock_guard: Option<ObjectLockDiagGuard>,
lock_optimization_enabled: bool,
bucket: &str,
object: &str,
) -> GetObjectReader {
if lock_optimization_enabled || reader.buffered_body.is_some() {
release_materialized_read_lock(bucket, object, read_lock_guard);
return reader;
}
if let Some(guard) = read_lock_guard {
reader.stream = Box::new(SetDiskLockGuardedReader {
inner: reader.stream,
guard: Some(guard),
@@ -2314,7 +2323,13 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
opts.part_number = Some(1);
}
let gr = get_transitioned_object_reader(bucket, object, &range, &h, &object_info, &opts).await?;
return Ok(attach_set_disk_read_lock_guard(gr, read_lock_guard.take()));
return Ok(finish_set_disk_read_lock(
gr,
read_lock_guard.take(),
lock_optimization_enabled,
bucket,
object,
));
}
if is_get_small_object_direct_memory_eligible(&range, &object_info, &fi, opts) {
@@ -2418,7 +2433,13 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
);
record_get_object_reader_path_observation(GET_OBJECT_PATH_CODEC_STREAMING, object_class, size_bucket);
let (reader, _offset, _length) = GetObjectReader::new(stream, range, &object_info, opts, &h).await?;
return Ok(attach_set_disk_read_lock_guard(reader, read_lock_guard.take()));
return Ok(finish_set_disk_read_lock(
reader,
read_lock_guard.take(),
lock_optimization_enabled,
bucket,
object,
));
}
read::GetCodecStreamingReaderBuildOutcome::Fallback(reason) => {
record_get_codec_streaming_gate_decision(
@@ -2454,8 +2475,13 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
let set_index = self.set_index;
let pool_index = self.pool_index;
let skip_verify = opts.skip_verify_bitrot;
// Move the read-lock guard into the task so it lives for the duration of the read.
// Fully materialized paths release it before returning; streaming paths keep it.
if lock_optimization_enabled {
release_materialized_read_lock(&bucket, &object, read_lock_guard.take());
debug!(bucket, object, "Lock optimization: released read lock before streaming read");
}
// When lock optimization is disabled, keep the read-lock guard in the
// task so it lives for the duration of the streaming read.
tokio::spawn(async move {
let _guard = read_lock_guard;
let mut writer = wd;
+65 -14
View File
@@ -595,14 +595,64 @@ fn resolve_read_part_from_responses(
Err(DiskError::ErasureReadQuorum)
}
fn shard_read_cost_for_disk(disk: Option<&DiskStore>) -> ShardReadCost {
fn shard_read_costs_for_disks(disks: &[Option<DiskStore>]) -> Vec<ShardReadCost> {
let local_endpoint_hosts = local_endpoint_hosts_for_shard_costs();
disks
.iter()
.map(|disk| shard_read_cost_for_disk(disk.as_ref(), local_endpoint_hosts))
.collect()
}
fn shard_read_cost_for_disk(disk: Option<&DiskStore>, local_endpoint_hosts: &[String]) -> ShardReadCost {
match disk {
Some(disk) if disk.is_local() => ShardReadCost::Local,
Some(_) => ShardReadCost::Remote,
Some(disk) => shard_read_cost_for_endpoint(false, &disk.host_name(), local_endpoint_hosts),
None => ShardReadCost::Unknown,
}
}
fn shard_read_cost_for_endpoint(is_local: bool, host_name: &str, local_endpoint_hosts: &[String]) -> ShardReadCost {
if is_local {
return ShardReadCost::Local;
}
if !host_name.is_empty() && local_endpoint_hosts.iter().any(|host| host == host_name) {
return ShardReadCost::SameNode;
}
ShardReadCost::Remote
}
fn local_endpoint_hosts_for_shard_costs() -> &'static [String] {
// Endpoint pools are immutable after startup, so build the host list once
// instead of walking every pool on each read. Do not cache the empty
// pre-startup answer: only memoize once the pools are published.
static LOCAL_ENDPOINT_HOSTS: std::sync::OnceLock<Vec<String>> = std::sync::OnceLock::new();
if let Some(hosts) = LOCAL_ENDPOINT_HOSTS.get() {
return hosts;
}
let Some(endpoint_pools) = runtime_sources::endpoint_pools() else {
return &[];
};
let mut hosts = Vec::new();
for pool in endpoint_pools.as_ref() {
for endpoint in pool.endpoints.as_ref() {
if !endpoint.is_local {
continue;
}
let host = endpoint.host_port();
if !host.is_empty() && !hosts.contains(&host) {
hosts.push(host);
}
}
}
LOCAL_ENDPOINT_HOSTS.get_or_init(|| hosts)
}
#[derive(Clone, Debug, Eq, Hash, PartialEq)]
struct ReadRepairHealCacheKey {
bucket: String,
@@ -2813,12 +2863,7 @@ impl SetDisks {
let use_mmap_read = object_mmap_read_enabled();
let reader_setup_stage_start = Instant::now();
let read_costs = coding::decode::should_collect_shard_read_costs().then(|| {
disks
.iter()
.map(|disk| shard_read_cost_for_disk(disk.as_ref()))
.collect::<Vec<_>>()
});
let read_costs = coding::decode::should_collect_shard_read_costs().then(|| shard_read_costs_for_disks(&disks));
let reader_setup = create_bitrot_readers_until_quorum_with_preference(
&files,
&disks,
@@ -3244,12 +3289,7 @@ impl SetDisks {
bitrot_reader_init_stage: GET_STAGE_READER_TASK_BITROT_READER_INIT,
});
let reader_setup_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
let read_costs = coding::decode::should_collect_shard_read_costs().then(|| {
disks
.iter()
.map(|disk| shard_read_cost_for_disk(disk.as_ref()))
.collect::<Vec<_>>()
});
let read_costs = coding::decode::should_collect_shard_read_costs().then(|| shard_read_costs_for_disks(disks));
let reader_setup = create_bitrot_readers_until_quorum_with_preference(
files,
disks,
@@ -3835,6 +3875,17 @@ mod tests {
const CODEC_STREAMING_TEST_BUCKET: &str = "bucket";
const CODEC_STREAMING_TEST_OBJECT: &str = "object";
#[test]
fn shard_read_cost_for_endpoint_maps_topology_classes() {
let local_hosts = vec!["node-a:9000".to_string()];
assert_eq!(shard_read_cost_for_endpoint(true, "node-a:9000", &local_hosts), ShardReadCost::Local);
assert_eq!(shard_read_cost_for_endpoint(false, "node-a:9000", &local_hosts), ShardReadCost::SameNode);
assert_eq!(shard_read_cost_for_endpoint(false, "node-b:9000", &local_hosts), ShardReadCost::Remote);
assert_eq!(shard_read_cost_for_endpoint(false, "", &local_hosts), ShardReadCost::Remote);
assert_eq!(shard_read_cost_for_disk(None, &local_hosts), ShardReadCost::Unknown);
}
fn metadata_fanout_test_fileinfo(object: &str) -> FileInfo {
let mut fi = FileInfo::new(object, 2, 2);
fi.volume = "bucket".to_string();