Merge branch 'main' into cxymds/fix-5943-hard-quota-reservations

This commit is contained in:
cxymds
2026-08-13 15:24:25 +08:00
committed by GitHub
8 changed files with 1183 additions and 30 deletions
@@ -31,7 +31,7 @@ use rustfs_config::{
DEFAULT_INTERNODE_DATA_TRANSPORT, ENV_RUSTFS_INTERNODE_DATA_TRANSPORT, INTERNODE_DATA_TRANSPORT_TCP,
KNOWN_INTERNODE_DATA_TRANSPORT_BACKENDS,
};
use rustfs_rio::{HttpReader, HttpWriter};
use rustfs_rio::{ChunkReaderBox, HttpChunkReader, HttpReader, HttpWriter};
use sha2::{Digest, Sha256};
use std::collections::HashMap;
use std::future::Future;
@@ -221,6 +221,11 @@ pub struct NsScannerCapabilityRequest {
#[async_trait]
pub trait InternodeDataTransport: Send + Sync + std::fmt::Debug {
async fn open_read(&self, request: ReadStreamRequest) -> Result<FileReader>;
/// Opens an owned-chunk stream when this transport can retain receive-buffer
/// ownership. `None` preserves the established `open_read` fallback.
async fn open_read_chunks(&self, _request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
Ok(None)
}
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter>;
async fn open_walk_dir(&self, request: WalkDirStreamRequest) -> Result<FileReader>;
async fn open_ns_scanner(&self, _request: NsScannerStreamRequest) -> Result<FileReader> {
@@ -247,6 +252,15 @@ impl InternodeDataTransport for TcpHttpInternodeDataTransport {
))
}
async fn open_read_chunks(&self, request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
let url = build_read_file_stream_url(&request);
let mut headers = json_headers();
build_auth_headers(&url, &Method::GET, &mut headers)?;
Ok(Some(Box::new(
HttpChunkReader::new_with_stall_timeout(url, Method::GET, headers, None, request.stall_timeout).await?,
)))
}
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter> {
let server_epoch = self.put_file_auth_capability(&request.endpoint).await?;
let nonce = server_epoch.map(|_| Uuid::new_v4());
@@ -522,6 +522,33 @@ impl RemoteDisk {
}
}
async fn open_read_chunks_with_retry(&self, request: ReadStreamRequest) -> Result<Option<rustfs_rio::ChunkReaderBox>> {
let mut attempt = 1;
let mut last_retry_classification = None;
loop {
match self.data_transport.open_read_chunks(request.clone()).await {
Ok(reader) => {
if attempt > 1
&& let Some(classification) = last_retry_classification
{
crate::cluster::rpc::runtime_sources::record_remote_disk_open_read_retry_success(classification);
}
return Ok(reader);
}
Err(err) if attempt < REMOTE_DISK_OPEN_READ_MAX_ATTEMPTS && Self::is_retryable_open_read_error(&err) => {
if let Some(classification) = err.internode_http_error_kind() {
let classification = classification.metric_label();
crate::cluster::rpc::runtime_sources::record_remote_disk_open_read_retry(classification);
last_retry_classification = Some(classification);
}
tokio::time::sleep(REMOTE_DISK_OPEN_READ_RETRY_BACKOFF).await;
attempt += 1;
}
Err(err) => return Err(err),
}
}
}
pub fn record_capacity_probe(&self, total: u64, used: u64, free: u64) {
self.health.record_capacity_probe(total, used, free);
}
@@ -2454,6 +2481,30 @@ impl DiskAPI for RemoteDisk {
.await
}
async fn read_file_stream_chunks(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
) -> Result<Option<rustfs_rio::ChunkReaderBox>> {
if self.health.is_faulty() {
return Err(DiskError::FaultyDisk);
}
let disk = self.disk_ref().await;
let stall_timeout = get_object_disk_read_timeout();
self.open_read_chunks_with_retry(ReadStreamRequest {
endpoint: self.endpoint.grid_host(),
disk,
volume: volume.to_string(),
path: path.to_string(),
offset,
length,
stall_timeout: (!stall_timeout.is_zero()).then_some(stall_timeout),
})
.await
}
/// Buffered read for remote disks.
/// The transport stream is collected into owned Bytes for caller sharing.
#[tracing::instrument(level = "trace", skip_all)]
+15
View File
@@ -2022,6 +2022,21 @@ impl DiskAPI for LocalDiskWrapper {
.await
}
async fn read_file_stream_chunks(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
) -> Result<Option<rustfs_rio::ChunkReaderBox>> {
self.track_disk_health_with_op(
"read_file_stream_chunks",
|| async { self.disk.read_file_stream_chunks(volume, path, offset, length).await },
get_max_timeout_duration(),
)
.await
}
async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<bytes::Bytes> {
self.track_disk_health_with_op(
"read_file_mmap_copy",
+26
View File
@@ -65,6 +65,7 @@ use error::{Error, Result};
use local::LocalDisk;
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_madmin::info_commands::DiskMetrics;
use rustfs_rio::ChunkReaderBox;
use serde::{Deserialize, Serialize};
use std::{fmt::Debug, path::PathBuf, sync::Arc, time::Duration};
use time::OffsetDateTime;
@@ -463,6 +464,19 @@ impl DiskAPI for Disk {
}
}
async fn read_file_stream_chunks(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
) -> Result<Option<ChunkReaderBox>> {
match self {
Disk::Local(_) => Ok(None),
Disk::Remote(remote_disk) => remote_disk.read_file_stream_chunks(volume, path, offset, length).await,
}
}
#[tracing::instrument(level = "trace", skip_all)]
async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes> {
match self {
@@ -901,6 +915,18 @@ pub trait DiskAPI: Debug + Send + Sync + 'static {
async fn read_file(&self, volume: &str, path: &str) -> Result<FileReader>;
async fn read_file_stream(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<FileReader>;
/// Returns an owned-chunk stream when the backing transport can preserve
/// receive-buffer ownership. `None` retains the ordinary reader path.
async fn read_file_stream_chunks(
&self,
_volume: &str,
_path: &str,
_offset: usize,
_length: usize,
) -> Result<Option<ChunkReaderBox>> {
Ok(None)
}
/// File read using mmap-then-copy on Unix or an efficient read on non-Unix.
async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes>;
+521 -9
View File
@@ -14,7 +14,10 @@
use pin_project_lite::pin_project;
use rustfs_utils::HashAlgorithm;
use std::future::poll_fn;
use std::io::IoSlice;
use std::pin::Pin;
use std::task::{Context, Poll};
use std::time::Duration;
use tokio::io::{AsyncRead, AsyncReadExt, AsyncWrite, AsyncWriteExt};
use tracing::error;
@@ -23,6 +26,18 @@ const LOG_COMPONENT_ECSTORE: &str = "ecstore";
const LOG_SUBSYSTEM_ERASURE: &str = "erasure";
const EVENT_BITROT_SHORT_SHARD_READ: &str = "bitrot_short_shard_read";
const EVENT_BITROT_HASH_MISMATCH: &str = "bitrot_hash_mismatch";
const MAX_RETAINED_CHUNKS_PER_BLOCK: usize = 64;
const MAX_CHUNK_POLLS_PER_YIELD: usize = MAX_RETAINED_CHUNKS_PER_BLOCK + 1;
/// Result of polling an optional owned-chunk handoff.
pub enum ShardChunkRead {
/// The source does not support owned-chunk handoff and remains untouched.
Unsupported,
/// The source reached EOF.
Eof,
/// A non-empty chunk containing at most the requested number of bytes.
Chunk(bytes::Bytes),
}
/// A shard source that may already hold its bytes in memory.
///
@@ -42,6 +57,12 @@ pub trait ShardSource: AsyncRead + Send + Sync + Unpin {
fn try_take_block(&mut self, _n: usize) -> Option<bytes::Bytes> {
None
}
/// Polls one owned chunk when the source supports chunk handoff.
/// `Unsupported` must leave the source untouched.
fn poll_read_chunk(self: Pin<&mut Self>, _cx: &mut Context<'_>, _max: usize) -> Poll<std::io::Result<ShardChunkRead>> {
Poll::Ready(Ok(ShardChunkRead::Unsupported))
}
}
/// Borrowed and owned byte slices are ordinary streaming sources: they carry no
@@ -75,6 +96,9 @@ pin_project! {
// contiguous on-disk `[hash][data]` block so both are pulled in a single
// pass; grown lazily and never shrunk.
buf: Vec<u8>,
// Reused owned chunk vector for the remote HTTP fast path. Keeping the
// allocation with the reader avoids allocating once per bitrot block.
chunks: Vec<bytes::Bytes>,
skip_verify: bool,
last_verify_duration: Duration,
}
@@ -91,6 +115,7 @@ where
hash_algo: algo,
shard_size,
buf: Vec::new(),
chunks: Vec::new(),
skip_verify,
last_verify_duration: Duration::ZERO,
}
@@ -260,11 +285,6 @@ where
let need = hash_size + want;
// In-memory fast path: the block is already resident, so slice it instead
// of copying it into the scratch buffer first (rustfs/backlog#1159). One
// copy (`extend_from_slice`) instead of two. A source that cannot serve
// `need` bytes returns `None` and falls through to the scratch path,
// keeping the short-read contract.
if let Some(block) = self.inner.try_take_block(need) {
let (data, verify) = split_and_verify(&self.hash_algo, self.skip_verify, &block)?;
out.extend_from_slice(data);
@@ -272,6 +292,126 @@ where
return Ok(want);
}
self.chunks.clear();
let handed_off = {
let inner = &mut self.inner;
let chunks = &mut self.chunks;
let tail_buf = &mut self.buf;
let mut received = 0usize;
poll_fn(|cx| {
for _ in 0..MAX_CHUNK_POLLS_PER_YIELD {
let next = match Pin::new(&mut *inner).poll_read_chunk(cx, need - received) {
Poll::Ready(Ok(next)) => next,
Poll::Ready(Err(err)) => return Poll::Ready(Err(err)),
Poll::Pending => return Poll::Pending,
};
let chunk = match next {
ShardChunkRead::Unsupported if received == 0 => return Poll::Ready(Ok(false)),
ShardChunkRead::Unsupported => {
return Poll::Ready(Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"chunk handoff became unavailable after transferring data",
)));
}
ShardChunkRead::Eof => {
return Poll::Ready(Err(short_shard_read(received.saturating_sub(hash_size), want)));
}
ShardChunkRead::Chunk(chunk) => chunk,
};
if received == 0 {
tail_buf.clear();
}
if chunk.is_empty() {
return Poll::Ready(Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"chunk handoff returned an empty chunk",
)));
}
let remaining = need - received;
if chunk.len() > remaining {
return Poll::Ready(Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"chunk handoff exceeded its requested boundary",
)));
}
received += chunk.len();
if chunks.len() == MAX_RETAINED_CHUNKS_PER_BLOCK {
if tail_buf.is_empty() {
tail_buf.reserve_exact(need - (received - chunk.len()));
}
tail_buf.extend_from_slice(&chunk);
} else {
chunks.push(chunk);
}
if received == need {
return Poll::Ready(Ok(true));
}
}
cx.waker().wake_by_ref();
Poll::Pending
})
.await?
};
if handed_off {
if self.chunks.len() == 1 && self.buf.is_empty() {
let block = &self.chunks[0];
let (data, verify) = split_and_verify(&self.hash_algo, self.skip_verify, block)?;
out.extend_from_slice(data);
self.last_verify_duration = verify;
return Ok(want);
}
let block_chunks = || {
self.chunks
.iter()
.map(|chunk| chunk.as_ref())
.chain((!self.buf.is_empty()).then_some(self.buf.as_slice()))
};
if !self.skip_verify {
let verify_start = std::time::Instant::now();
let actual_hash = self
.hash_algo
.hash_encode_slices(block_chunks().scan(hash_size, |skip, chunk| {
let start = (*skip).min(chunk.len());
*skip -= start;
Some(&chunk[start..])
}));
let verify = verify_start.elapsed();
let mut hash_offset = 0;
let mut remaining = hash_size;
for chunk in block_chunks() {
let take = remaining.min(chunk.len());
if actual_hash.as_ref()[hash_offset..hash_offset + take] != chunk[..take] {
error!(
event = EVENT_BITROT_HASH_MISMATCH,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_ERASURE,
state = "failed",
data_len = want,
"bitrot hash mismatch"
);
return Err(std::io::Error::new(std::io::ErrorKind::InvalidData, "bitrot hash mismatch"));
}
hash_offset += take;
remaining -= take;
if remaining == 0 {
break;
}
}
self.last_verify_duration = verify;
}
let mut skip = hash_size;
for chunk in block_chunks() {
let start = skip.min(chunk.len());
skip -= start;
out.extend_from_slice(&chunk[start..]);
}
return Ok(want);
}
// Streaming path: same single pass and same verification as `read`; only
// the sink differs (`extend_from_slice` into `out` instead of
// `copy_from_slice` into a pre-zeroed buffer).
@@ -677,18 +817,167 @@ impl BitrotWriterWrapper {
#[cfg(test)]
mod tests {
use super::ShardSource;
use super::{
BitrotReader, BitrotWriter, BitrotWriterWrapper, CustomWriter, bitrot_shard_file_size, bitrot_verify, write_all_vectored,
};
use super::{MAX_RETAINED_CHUNKS_PER_BLOCK, ShardChunkRead, ShardSource};
use bytes::Bytes;
use rustfs_utils::HashAlgorithm;
use std::io::{Cursor, IoSlice};
use std::collections::VecDeque;
use std::io::{self, Cursor, IoSlice};
use std::pin::Pin;
use std::sync::{
Arc,
atomic::{AtomicUsize, Ordering},
};
use std::task::{Context, Poll};
use tokio::io::{AsyncWrite, AsyncWriteExt};
use std::time::Duration;
use tokio::io::{AsyncRead, AsyncWrite, AsyncWriteExt, ReadBuf};
struct FragmentedSource {
chunks: VecDeque<Bytes>,
}
impl FragmentedSource {
fn new(bytes: Vec<u8>, fragment_sizes: &[usize]) -> Self {
let mut chunks = VecDeque::new();
let mut offset = 0;
for &size in fragment_sizes {
let end = (offset + size).min(bytes.len());
if offset < end {
chunks.push_back(Bytes::copy_from_slice(&bytes[offset..end]));
}
offset = end;
}
if offset < bytes.len() {
chunks.push_back(Bytes::copy_from_slice(&bytes[offset..]));
}
Self { chunks }
}
}
impl AsyncRead for FragmentedSource {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("fragmented source must use chunk handoff")))
}
}
impl ShardSource for FragmentedSource {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
let Some(mut chunk) = self.chunks.pop_front() else {
return Poll::Ready(Ok(ShardChunkRead::Eof));
};
if chunk.len() > max {
self.chunks.push_front(chunk.split_off(max));
chunk.truncate(max);
}
Poll::Ready(Ok(ShardChunkRead::Chunk(chunk)))
}
}
struct GeneratedChunkSource {
bytes: Bytes,
offset: usize,
fragment_size: usize,
fail_at: Option<usize>,
}
impl GeneratedChunkSource {
fn new(bytes: Vec<u8>, fragment_size: usize) -> Self {
assert!(fragment_size > 0);
Self {
bytes: Bytes::from(bytes),
offset: 0,
fragment_size,
fail_at: None,
}
}
fn failing(bytes: Vec<u8>, fragment_size: usize, fail_at: usize) -> Self {
Self {
fail_at: Some(fail_at),
..Self::new(bytes, fragment_size)
}
}
}
impl AsyncRead for GeneratedChunkSource {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("generated source must use chunk handoff")))
}
}
impl ShardSource for GeneratedChunkSource {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
if self.fail_at == Some(self.offset) {
return Poll::Ready(Err(rustfs_rio::new_test_internode_http_io_error(
rustfs_rio::InternodeHttpErrorKind::BodyStreamAborted,
)));
}
if self.offset == self.bytes.len() {
return Poll::Ready(Ok(ShardChunkRead::Eof));
}
let error_limit = self.fail_at.unwrap_or(self.bytes.len());
let take = self
.fragment_size
.min(max)
.min(error_limit - self.offset)
.min(self.bytes.len() - self.offset);
let start = self.offset;
self.offset += take;
Poll::Ready(Ok(ShardChunkRead::Chunk(self.bytes.slice(start..start + take))))
}
}
struct InvalidChunkSource {
mode: InvalidChunkMode,
}
#[derive(Clone, Copy)]
enum InvalidChunkMode {
Empty,
Oversized,
UnsupportedAfterChunk,
Unsupported,
}
impl AsyncRead for InvalidChunkSource {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("invalid source must use chunk handoff")))
}
}
impl ShardSource for InvalidChunkSource {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
match self.mode {
InvalidChunkMode::Empty => Poll::Ready(Ok(ShardChunkRead::Chunk(Bytes::new()))),
InvalidChunkMode::Oversized => Poll::Ready(Ok(ShardChunkRead::Chunk(Bytes::from(vec![0; max + 1])))),
InvalidChunkMode::UnsupportedAfterChunk => {
self.mode = InvalidChunkMode::Unsupported;
Poll::Ready(Ok(ShardChunkRead::Chunk(Bytes::from_static(b"x"))))
}
InvalidChunkMode::Unsupported => Poll::Ready(Ok(ShardChunkRead::Unsupported)),
}
}
}
struct ScratchReuseSource {
block: Option<Bytes>,
saw_reused_scratch: bool,
}
impl AsyncRead for ScratchReuseSource {
fn poll_read(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
let Some(block) = self.block.take() else {
return Poll::Ready(Ok(()));
};
self.saw_reused_scratch = buf.initialize_unfilled()[..block.len()].iter().all(|byte| *byte == 0xa5);
buf.put_slice(&block);
Poll::Ready(Ok(()))
}
}
impl ShardSource for ScratchReuseSource {}
#[derive(Default)]
struct VectoredCountingWriter {
@@ -1446,6 +1735,70 @@ mod tests {
assert!(out.is_empty(), "corrupt bytes must never reach the caller's buffer");
}
#[tokio::test]
async fn chunked_handoff_verifies_data_split_across_hash_boundaries() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let data: Vec<u8> = (0..SHARD).map(|index| (index % 251) as u8).collect();
let mut encoded = Vec::new();
BitrotWriter::new(&mut encoded, SHARD, algo.clone())
.write(&data)
.await
.expect("write shard");
let mut output = Vec::with_capacity(SHARD);
BitrotReader::new(FragmentedSource::new(encoded, &[3, 11, 19, 37, 128]), SHARD, algo, false)
.read_appending(&mut output, SHARD)
.await
.expect("fragmented shard must verify");
assert_eq!(output, data);
}
#[tokio::test]
async fn chunked_handoff_never_appends_a_corrupt_shard() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let mut encoded = Vec::new();
BitrotWriter::new(&mut encoded, SHARD, algo.clone())
.write(&vec![9u8; SHARD])
.await
.expect("write shard");
let last = encoded.len() - 1;
encoded[last] ^= 0xff;
let mut output = Vec::with_capacity(SHARD);
let err = BitrotReader::new(FragmentedSource::new(encoded, &[7, 17, 31]), SHARD, algo, false)
.read_appending(&mut output, SHARD)
.await
.expect_err("corrupt fragmented shard must fail");
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
assert!(output.is_empty());
}
#[tokio::test]
async fn chunked_handoff_does_not_hash_when_verification_is_skipped() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let mut encoded = Vec::new();
BitrotWriter::new(&mut encoded, SHARD, algo.clone())
.write(&vec![9u8; SHARD])
.await
.expect("write shard");
encoded[0] ^= 0xff;
let mut output = Vec::with_capacity(SHARD);
let mut reader = BitrotReader::new(FragmentedSource::new(encoded, &[7, 17, 31]), SHARD, algo, true);
reader
.read_appending(&mut output, SHARD)
.await
.expect("skipped verification must accept fragmented shard bytes");
assert_eq!(reader.last_verify_duration(), Duration::ZERO);
assert_eq!(output, vec![9u8; SHARD]);
}
#[tokio::test]
async fn read_appending_rejects_a_want_larger_than_the_shard() {
let algo = HashAlgorithm::HighwayHash256;
@@ -1497,10 +1850,21 @@ mod tests {
// Equivalence: same bytes out of both paths.
let mut via_mem: Vec<u8> = Vec::with_capacity(SHARD);
BitrotReader::new(Cursor::new(Bytes::from(encoded.clone())), SHARD, algo.clone(), false)
let mut memory_reader = BitrotReader::new(Cursor::new(Bytes::from(encoded.clone())), SHARD, algo.clone(), false);
memory_reader
.read_appending(&mut via_mem, SHARD)
.await
.expect("in-memory read");
assert_eq!(
memory_reader.chunks.capacity(),
0,
"the synchronous fast path must not allocate chunk storage"
);
assert_eq!(
memory_reader.buf.capacity(),
0,
"the synchronous fast path must not allocate scratch storage"
);
let mut via_stream: Vec<u8> = Vec::with_capacity(SHARD);
BitrotReader::new(Cursor::new(encoded), SHARD, algo, false)
@@ -1537,4 +1901,152 @@ mod tests {
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert!(out.is_empty(), "corrupt bytes must never reach the caller's buffer");
}
#[tokio::test]
async fn streaming_fallback_reuses_initialized_scratch() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![7u8; SHARD];
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let source = ScratchReuseSource {
block: Some(Bytes::copy_from_slice(&encoded)),
saw_reused_scratch: false,
};
let mut reader = BitrotReader::new(source, SHARD, algo, false);
reader.buf = vec![0xa5; encoded.len()];
let mut output = Vec::new();
reader
.read_appending(&mut output, SHARD)
.await
.expect("streaming fallback should verify");
assert!(reader.inner.saw_reused_scratch, "capability probing must not clear reusable scratch");
assert_eq!(output, data);
}
#[tokio::test]
async fn chunked_handoff_bounds_production_sized_one_byte_fragments() {
const SHARD: usize = 1024 * 1024 / 4;
let algo = HashAlgorithm::HighwayHash256S;
let data: Vec<u8> = (0..SHARD).map(|index| (index % 251) as u8).collect();
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let encoded_len = encoded.len();
let mut reader = BitrotReader::new(GeneratedChunkSource::new(encoded, 1), SHARD, algo, false);
let mut output = Vec::with_capacity(SHARD);
reader
.read_appending(&mut output, SHARD)
.await
.expect("one-byte fragments should verify with bounded retained state");
assert_eq!(output, data);
assert_eq!(reader.chunks.len(), MAX_RETAINED_CHUNKS_PER_BLOCK);
assert!(reader.chunks.capacity() <= MAX_RETAINED_CHUNKS_PER_BLOCK);
assert_eq!(reader.buf.len(), encoded_len - MAX_RETAINED_CHUNKS_PER_BLOCK);
}
#[tokio::test]
async fn chunked_handoff_keeps_sixty_four_frames_zero_copy_and_respects_poll_budget() {
const SHARD: usize = 1024 * 1024;
const FRAME: usize = 16 * 1024;
let algo = HashAlgorithm::HighwayHash256S;
let small_data = vec![3u8; 4096];
let small_encoded = encode_one_block(&small_data, 4096, algo.clone()).await;
let mut exact_reader =
BitrotReader::new(FragmentedSource::new(small_encoded.clone(), &[1; 63]), 4096, algo.clone(), false);
let mut exact_output = Vec::new();
exact_reader
.read_appending(&mut exact_output, 4096)
.await
.expect("exactly sixty-four frames should verify");
assert_eq!(exact_output, small_data);
assert_eq!(exact_reader.chunks.len(), MAX_RETAINED_CHUNKS_PER_BLOCK);
assert!(exact_reader.buf.is_empty(), "the threshold itself must remain zero-copy");
let mut yielded_reader = BitrotReader::new(FragmentedSource::new(small_encoded, &[1; 65]), 4096, algo.clone(), false);
let mut yielded_output = Vec::new();
let mut yielded_read = Box::pin(yielded_reader.read_appending(&mut yielded_output, 4096));
let mut cx = Context::from_waker(std::task::Waker::noop());
assert!(std::future::Future::poll(yielded_read.as_mut(), &mut cx).is_pending());
assert!(matches!(std::future::Future::poll(yielded_read.as_mut(), &mut cx), Poll::Ready(Ok(4096))));
drop(yielded_read);
assert_eq!(yielded_output, small_data);
let data = vec![7u8; SHARD];
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let mut reader = BitrotReader::new(FragmentedSource::new(encoded, &[FRAME; 64]), SHARD, algo, false);
let mut output = Vec::with_capacity(SHARD);
let mut read = Box::pin(reader.read_appending(&mut output, SHARD));
assert!(
matches!(std::future::Future::poll(read.as_mut(), &mut cx), Poll::Ready(Ok(SHARD))),
"sixty-five normal HTTP frames should complete without a cooperative yield"
);
drop(read);
assert_eq!(output, data);
assert_eq!(reader.chunks.len(), MAX_RETAINED_CHUNKS_PER_BLOCK);
assert_eq!(reader.buf.len(), HashAlgorithm::HighwayHash256S.size());
}
#[tokio::test]
async fn chunked_tail_failures_preserve_errors_and_output() {
const SHARD: usize = 4096;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![7u8; SHARD];
let encoded = encode_one_block(&data, SHARD, algo.clone()).await;
let sentinel = vec![1u8, 2, 3];
let mut short_output = sentinel.clone();
let short_err = BitrotReader::new(GeneratedChunkSource::new(encoded[..100].to_vec(), 1), SHARD, algo.clone(), false)
.read_appending(&mut short_output, SHARD)
.await
.expect_err("EOF after the retention threshold must stay a short read");
assert_eq!(short_err.kind(), io::ErrorKind::UnexpectedEof);
assert_eq!(short_output, sentinel);
let mut corrupt = encoded.clone();
let last = corrupt.len() - 1;
corrupt[last] ^= 0xff;
let mut corrupt_output = sentinel.clone();
let corrupt_err = BitrotReader::new(GeneratedChunkSource::new(corrupt, 1), SHARD, algo.clone(), false)
.read_appending(&mut corrupt_output, SHARD)
.await
.expect_err("corrupt coalesced tail must fail verification");
assert_eq!(corrupt_err.kind(), io::ErrorKind::InvalidData);
assert_eq!(corrupt_output, sentinel);
let mut failed_output = sentinel.clone();
let body_err = BitrotReader::new(GeneratedChunkSource::failing(encoded, 1, 65), SHARD, algo, false)
.read_appending(&mut failed_output, SHARD)
.await
.expect_err("a terminal body error must not become EOF");
let source = body_err
.get_ref()
.and_then(|source| source.downcast_ref::<rustfs_rio::InternodeHttpError>())
.expect("body error should retain internode classification");
assert_eq!(source.kind(), rustfs_rio::InternodeHttpErrorKind::BodyStreamAborted);
assert_eq!(failed_output, sentinel);
}
#[tokio::test]
async fn chunked_handoff_rejects_invalid_source_contracts() {
const SHARD: usize = 64;
for mode in [
InvalidChunkMode::Empty,
InvalidChunkMode::Oversized,
InvalidChunkMode::UnsupportedAfterChunk,
] {
let source = InvalidChunkSource { mode };
let mut output = vec![9u8];
let err = BitrotReader::new(source, SHARD, HashAlgorithm::HighwayHash256S, false)
.read_appending(&mut output, SHARD)
.await
.expect_err("invalid chunk contracts must fail closed");
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
assert_eq!(output, vec![9u8]);
}
}
}
+117 -2
View File
@@ -22,12 +22,13 @@ use crate::diagnostics::get::{
#[cfg(feature = "hotpath")]
use crate::disk::FileWriter;
use crate::disk::{self, DiskAPI as _, DiskStore, FileReader, MmapCopyStageMetrics, error::DiskError};
use crate::erasure::coding::{BitrotReader, BitrotWriterWrapper, CustomWriter};
use crate::erasure::coding::{BitrotReader, BitrotWriterWrapper, CustomWriter, ShardChunkRead};
use bytes::Bytes;
use rustfs_config::{
DEFAULT_OBJECT_MMAP_READ_ENABLE, DEFAULT_OBJECT_MMAP_READ_MAX_LENGTH, ENV_OBJECT_MMAP_READ_ENABLE,
ENV_OBJECT_MMAP_READ_MAX_LENGTH, ENV_OBJECT_ZERO_COPY_ENABLE,
};
use rustfs_rio::ChunkReaderBox;
use rustfs_utils::HashAlgorithm;
use std::future::Future;
use std::io::{self, Cursor};
@@ -51,6 +52,7 @@ tokio::task_local! {
/// (rustfs/backlog#1159). Everything else is a stream and keeps the old path.
pub enum ShardReader {
InMemory(Cursor<Bytes>),
Chunked(ChunkReaderBox),
Stream(Box<dyn AsyncRead + Send + Sync + Unpin>),
}
@@ -58,6 +60,7 @@ impl AsyncRead for ShardReader {
fn poll_read(self: Pin<&mut Self>, cx: &mut Context<'_>, buf: &mut tokio::io::ReadBuf<'_>) -> Poll<std::io::Result<()>> {
match self.get_mut() {
Self::InMemory(cursor) => Pin::new(cursor).poll_read(cx, buf),
Self::Chunked(reader) => Pin::new(&mut **reader).poll_read(cx, buf),
Self::Stream(reader) => Pin::new(reader).poll_read(cx, buf),
}
}
@@ -67,7 +70,19 @@ impl crate::erasure::coding::ShardSource for ShardReader {
fn try_take_block(&mut self, n: usize) -> Option<Bytes> {
match self {
Self::InMemory(cursor) => cursor.try_take_block(n),
Self::Stream(_) => None,
Self::Chunked(_) | Self::Stream(_) => None,
}
}
fn poll_read_chunk(self: Pin<&mut Self>, cx: &mut Context<'_>, max: usize) -> Poll<io::Result<ShardChunkRead>> {
let Self::Chunked(reader) = self.get_mut() else {
return Poll::Ready(Ok(ShardChunkRead::Unsupported));
};
match Pin::new(&mut **reader).poll_read_chunk(cx, max) {
Poll::Ready(Ok(Some(chunk))) => Poll::Ready(Ok(ShardChunkRead::Chunk(chunk))),
Poll::Ready(Ok(None)) => Poll::Ready(Ok(ShardChunkRead::Eof)),
Poll::Ready(Err(err)) => Poll::Ready(Err(err)),
Poll::Pending => Poll::Pending,
}
}
}
@@ -345,6 +360,17 @@ async fn open_disk_reader(
let metrics_path = metrics_path.filter(|_| rustfs_io_metrics::get_stage_metrics_enabled());
let stage_metrics_enabled = metrics_path.is_some();
// Preserve HTTP body ownership only on healthy remote reads. Instrumented
// and local paths retain their existing AsyncRead wrappers.
if use_mmap_read
&& !disk.is_local()
&& !stage_metrics_enabled
&& !cfg!(feature = "hotpath")
&& let Some(reader) = disk.read_file_stream_chunks(bucket, path, offset, length).await?
{
return Ok(ShardReader::Chunked(reader));
}
// Mmap-copy materializes the whole `offset..offset+length` range as one
// owned allocation before any byte is served, and GET/heal shard reads
// request the entire part span in one call. Over-cap reads (e.g. a huge
@@ -780,6 +806,50 @@ pub async fn create_bitrot_writer(
#[cfg(test)]
mod tests {
use super::*;
use rustfs_rio::ChunkReader;
use std::collections::VecDeque;
struct TestChunkReader {
chunks: VecDeque<Bytes>,
}
impl TestChunkReader {
fn new(bytes: Bytes, fragment_sizes: &[usize]) -> Self {
let mut chunks = VecDeque::new();
let mut offset = 0;
for &size in fragment_sizes {
let end = (offset + size).min(bytes.len());
if offset < end {
chunks.push_back(bytes.slice(offset..end));
}
offset = end;
}
if offset < bytes.len() {
chunks.push_back(bytes.slice(offset..));
}
Self { chunks }
}
}
impl AsyncRead for TestChunkReader {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Ready(Err(io::Error::other("test chunk reader must use chunk handoff")))
}
}
impl ChunkReader for TestChunkReader {
fn poll_read_chunk(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, max: usize) -> Poll<io::Result<Option<Bytes>>> {
let Some(mut chunk) = self.chunks.pop_front() else {
return Poll::Ready(Ok(None));
};
let take = chunk.len().min(max);
if take < chunk.len() {
self.chunks.push_front(chunk.split_off(take));
}
chunk.truncate(take);
Poll::Ready(Ok(Some(chunk)))
}
}
#[cfg(feature = "hotpath")]
use crate::cluster::rpc::RemoteDisk;
@@ -1669,4 +1739,49 @@ mod tests {
println!("error: {error:?}");
assert_eq!(error, DiskError::DiskNotFound);
}
#[tokio::test]
async fn shard_reader_chunked_path_verifies_fragmented_remote_block() {
const SHARD_SIZE: usize = 1024;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![42u8; SHARD_SIZE];
let mut encoded = Vec::new();
crate::erasure::coding::BitrotWriter::new(&mut encoded, SHARD_SIZE, algo.clone())
.write(&data)
.await
.expect("test shard should encode");
let source = TestChunkReader::new(Bytes::from(encoded), &[3, 7, 17, 31]);
let mut reader = BitrotReader::new(ShardReader::Chunked(Box::new(source)), SHARD_SIZE, algo, false);
let mut output = Vec::with_capacity(SHARD_SIZE);
reader
.read_appending(&mut output, SHARD_SIZE)
.await
.expect("fragmented remote shard should verify");
assert_eq!(output, data);
}
#[tokio::test]
async fn shard_reader_chunked_path_handles_more_than_one_poll_budget() {
const SHARD_SIZE: usize = 1024;
let algo = HashAlgorithm::HighwayHash256S;
let data = vec![42u8; SHARD_SIZE];
let mut encoded = Vec::new();
crate::erasure::coding::BitrotWriter::new(&mut encoded, SHARD_SIZE, algo.clone())
.write(&data)
.await
.expect("test shard should encode");
let fragment_sizes = vec![1; encoded.len()];
let source = TestChunkReader::new(Bytes::from(encoded), &fragment_sizes);
let mut reader = BitrotReader::new(ShardReader::Chunked(Box::new(source)), SHARD_SIZE, algo, false);
let mut output = Vec::with_capacity(SHARD_SIZE);
reader
.read_appending(&mut output, SHARD_SIZE)
.await
.expect("fragmented remote shard should verify after multiple polls");
assert_eq!(output, data);
}
}