Compare commits

...

52 Commits

Author SHA1 Message Date
weisd c0d3f53f7a test(filemeta): flush xl meta test fixture before reopen (#2557) 2026-04-16 00:40:46 +00:00
Tunglies 49366ee200 chore(lint): clippy rules redundant_clone (#2554) 2026-04-15 13:54:07 +00:00
Strangerxxx 73e6542eea feat(helm): add generic service and ingress annotation support (#2541)
Co-authored-by: cxymds <Cxymds@qq.com>
Co-authored-by: houseme <housemecn@gmail.com>
2026-04-15 10:16:10 +00:00
Tunglies f57dd5a3c7 chore(lint): clippy rules needless_collect (#2522) 2026-04-15 10:00:03 +00:00
安正超 1bda5ed636 test: cover stored-size range handling (#2544)
Co-authored-by: cxymds <Cxymds@qq.com>
2026-04-15 09:37:01 +00:00
weisd 27971f5c03 fix(scanner): preserve totals for compacted folder scans (#2530) 2026-04-15 07:32:13 +00:00
安正超 6506ae8c7b fix(ci): report CLA check for GitHub Merge Queue (#2551) 2026-04-15 17:20:16 +08:00
安正超 8223cda5ff fix(ci): enable merge_group trigger for GitHub Merge Queue (#2549) 2026-04-15 16:27:53 +08:00
majinghe 8152c8e084 fix: add different annotations for different pvc (#2547) 2026-04-15 15:21:38 +08:00
weisd da9dfb51e7 fix(ecstore): cleanup stale put object temp data (#2529) 2026-04-15 14:40:58 +08:00
cxymds c11efce1e4 fix(admin): replicate empty site groups (#2528)
Co-authored-by: loverustfs <hello@rustfs.com>
Co-authored-by: houseme <housemecn@gmail.com>
Co-authored-by: GatewayJ <835269233@qq.com>
2026-04-15 14:21:40 +08:00
houseme fa793237fb build(deps): bump the dependencies group with 7 updates (#2546)
Co-authored-by: heihutu <heihutu@gmail.com>
Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com>
Co-authored-by: houseme <4829346+houseme@users.noreply.github.com>
2026-04-15 13:23:57 +08:00
安正超 df109680ae cleanup: remove orphaned chunk fast-path submodule files (#2545)
Co-authored-by: cxymds <Cxymds@qq.com>
2026-04-15 11:50:32 +08:00
cxymds 1eef75b0ca fix(admin): allow clearing group policy (#2527)
Co-authored-by: houseme <housemecn@gmail.com>
Co-authored-by: GatewayJ <835269233@qq.com>
2026-04-15 11:35:19 +08:00
安正超 642d83f0e4 revert: remove #2351 chunk I/O and object-io crate (phase 7) (#2543)
Co-authored-by: houseme <housemecn@gmail.com>
2026-04-15 10:54:41 +08:00
GatewayJ 16b9189e9b feat(oidc): add roles_claim and jwt:roles policy support (#2509)
Co-authored-by: GatewayJ <8352692332qq.com>
Co-authored-by: houseme <housemecn@gmail.com>
Co-authored-by: loverustfs <hello@rustfs.com>
2026-04-15 09:30:24 +08:00
安正超 1979fc7fb1 fix: restore bitrot stream and checksum metadata (#2542) 2026-04-15 08:36:55 +08:00
安正超 68d3dba9fc fix: revert standalone #2351 artifacts (phase 5) (#2535) 2026-04-14 22:35:05 +08:00
安正超 db8ef63674 fix: revert #2351 readers.rs behavioral changes (phase 6) (#2536) 2026-04-14 22:08:38 +08:00
安正超 6d476ae9a1 revert: remove BlockReadable trait and chunk-based I/O from rio, ecstore, disk layers (#2533) 2026-04-14 19:09:02 +08:00
安正超 2a3d127a86 revert: restore PutObjReader and remove PUT zero-copy chunk path (#2351 phase 2) (#2532) 2026-04-14 16:35:03 +08:00
安正超 8abf683178 fix(io-metrics): remove dead GET chunk fast path metrics (#2531) 2026-04-14 15:47:57 +08:00
安正超 5048ff8c69 test(credentials): cover URL-safe secret keys (#2524) 2026-04-14 13:57:23 +08:00
Liam Beckman 711aab73c3 docs: Minor formatting clean up in README.md (#2523)
Signed-off-by: Liam Beckman <lbeckman314@gmail.com>
2026-04-14 08:29:33 +08:00
安正超 315e8134eb test(get): cover reader-backed output context (#2516)
Signed-off-by: 安正超 <anzhengchao@gmail.com>
Co-authored-by: cxymds <Cxymds@qq.com>
Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
2026-04-13 21:59:13 +08:00
houseme 979626c370 refactor(utils): decouple config deps and move sys helpers (#2520)
Co-authored-by: houseme <4829346+houseme@users.noreply.github.com>
2026-04-13 21:05:03 +08:00
Cocoon-Break 505a566c7c fix: remove dead replace() call in gen_secret_key (credentials.rs) (#2515)
Signed-off-by: cocoon <54054995+kuishou68@users.noreply.github.com>
Signed-off-by: Cocoon-Break <54054995+kuishou68@users.noreply.github.com>
Co-authored-by: loverustfs <hello@rustfs.com>
Co-authored-by: houseme <housemecn@gmail.com>
2026-04-13 18:42:50 +08:00
安正超 8efa359a1a docs: add ARCHITECTURE.md for project-wide code navigation (#2519) 2026-04-13 18:26:58 +08:00
dependabot[bot] 522c1675d8 build(deps): bump the dependencies group with 2 updates (#2513)
Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
2026-04-13 10:02:27 +08:00
houseme e160633ac4 fix(admin): enforce event and audit target auth (#2508) 2026-04-12 23:59:42 +08:00
安正超 458086dc80 fix(get): remove GET chunk fast path (#2507) 2026-04-12 23:12:23 +08:00
安正超 0e697fa7e1 test(get): cover fast path pre-commit errors (#2502)
Co-authored-by: houseme <housemecn@gmail.com>
2026-04-12 20:13:15 +08:00
安正超 30cc481731 refactor(storage): inline acl entrypoints (#2501)
Co-authored-by: houseme <housemecn@gmail.com>
2026-04-12 20:12:44 +08:00
houseme b9f2369095 build(deps): bump the dependencies group with 3 updates (#2503) 2026-04-12 12:11:40 +08:00
houseme 4ce374a181 chore(deps): update flake.lock (#2498) 2026-04-12 08:50:40 +08:00
安正超 ff7dc77b5e fix(deploy): auto-create rustfs log directory (#2500) 2026-04-12 08:50:24 +08:00
安正超 b387f6c58a refactor(storage): inline object lock config entrypoints (#2494) 2026-04-12 08:49:51 +08:00
houseme 64508ae770 fix(storage): align archive content-encoding with S3 semantics (#2496) 2026-04-12 02:14:52 +08:00
John 6a2f98b468 fix(obs): respect error log level for http warnings (#2492)
Signed-off-by: 80347547 <jianglong@oppo.com>
Co-authored-by: 80347547 <jianglong@oppo.com>
Co-authored-by: houseme <housemecn@gmail.com>
2026-04-12 01:32:54 +08:00
houseme eb0dc24921 fix(get): validate fast path body before response commit (#2495)
Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com>
Co-authored-by: houseme <4829346+houseme@users.noreply.github.com>
2026-04-12 01:04:22 +08:00
Mohamed Zenadi c8c71e34b6 fix(storage): fix critical bug in range calculation (#2493)
Signed-off-by: Mohamed Zenadi <zeapo@users.noreply.github.com>
Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
2026-04-12 00:20:15 +08:00
John 11552eb722 fix(deploy): use standard rustfs log directory (#2491)
Signed-off-by: 80347547 <jianglong@oppo.com>
Co-authored-by: 80347547 <jianglong@oppo.com>
2026-04-11 21:48:05 +08:00
安正超 b0b7f56281 refactor(storage): inline object lock metadata writes (#2489) 2026-04-11 20:05:03 +08:00
安正超 120f1021b5 refactor(storage): inline object tagging entrypoints (#2486) 2026-04-11 19:01:24 +08:00
安正超 20e07519a1 refactor(storage): inline object metadata read entrypoints (#2485)
Signed-off-by: 安正超 <anzhengchao@gmail.com>
Co-authored-by: houseme <housemecn@gmail.com>
Co-authored-by: Copilot <175728472+Copilot@users.noreply.github.com>
2026-04-11 14:59:38 +08:00
安正超 5c4dadc0b7 test(bucket): cover inline list response params (#2483) 2026-04-11 14:59:17 +08:00
Henry Guo 0f7e7f35e9 test(utils): make retry timer assertions deterministic (#2468) 2026-04-11 12:02:38 +08:00
安正超 5e97377aa0 refactor(storage): drop ecfs test-only helper (#2484) 2026-04-11 11:49:10 +08:00
安正超 b49570d87a refactor(storage): inline extract put-object branch (#2482) 2026-04-11 11:00:11 +08:00
安正超 d70ca1990e refactor(storage): inline list response params (#2476) 2026-04-11 09:25:40 +08:00
Henry Guo 70fdf79297 fix(server): warn on default credentials with console enabled (#2448)
Co-authored-by: 安正超 <anzhengchao@gmail.com>
Co-authored-by: houseme <housemecn@gmail.com>
2026-04-11 00:03:06 +08:00
安正超 4472324125 refactor(storage): trim thin s3 forwarding layers (#2474) 2026-04-10 23:17:18 +08:00
185 changed files with 7009 additions and 17460 deletions
+6
View File
@@ -25,6 +25,12 @@ Current `test-and-lint` gate includes:
- `cargo nextest run --all --exclude e2e_test`
- `cargo test --all --doc`
- `cargo test -p rustfs get_object_chunk_fast_path`
- `cargo test -p rustfs materialize_chunk_stream_before_commit`
- `touch rustfs/build.rs`
- `cargo build -p rustfs --bins --jobs 2`
- `cargo test -p e2e_test archive_multipart_roundtrip_preserves_bytes`
- `cargo test -p e2e_test presigned_get_and_reverse_proxy_preserve_multipart_bytes_with_fast_path`
- `cargo fmt --all --check`
- `cargo clippy --all-targets --all-features -- -D warnings`
- `./scripts/check_layer_dependencies.sh`
+2
View File
@@ -53,6 +53,8 @@ on:
- ".github/workflows/docker.yml"
- ".github/workflows/audit.yml"
- ".github/workflows/performance.yml"
merge_group:
types: [checks_requested]
schedule:
- cron: "0 0 * * 0" # Weekly on Sunday at midnight UTC
workflow_dispatch:
+22
View File
@@ -17,6 +17,8 @@ name: CLA Check
on:
pull_request_target:
types: [opened, synchronize, reopened]
merge_group:
types: [checks_requested]
issue_comment:
types: [created, edited]
@@ -31,7 +33,26 @@ jobs:
if: ${{ github.event_name != 'issue_comment' || github.event.issue.pull_request }}
runs-on: ubuntu-latest
steps:
- name: Report CLA result for merge queue
if: github.event_name == 'merge_group'
uses: actions/github-script@v8
with:
script: |
await github.rest.checks.create({
owner: context.repo.owner,
repo: context.repo.repo,
name: 'CLA Check',
head_sha: context.sha,
status: 'completed',
conclusion: 'success',
output: {
title: 'CLA requirements satisfied for merge queue',
summary: 'Queued pull requests must satisfy the required CLA check before they enter the merge queue. This reports the existing CLA result on the merge-group SHA.'
}
});
- name: Create token for rustfs/cla
if: github.event_name != 'merge_group'
id: registry-token
uses: actions/create-github-app-token@v3
with:
@@ -42,6 +63,7 @@ jobs:
permission-contents: write
- name: Run CLA Bot
if: github.event_name != 'merge_group'
uses: overtrue/cla-bot@v0.0.9
with:
github-token: ${{ github.token }}
+377
View File
@@ -0,0 +1,377 @@
# ARCHITECTURE.md
> Last updated: 2026-04-13 · Revision: 1 (draft)
>
> This document describes the high-level architecture of RustFS.
> If you want to familiarize yourself with the code base, you are in the right place!
>
> See also [CONTRIBUTING.md](CONTRIBUTING.md) for development workflow.
## Bird's Eye View
RustFS is a high-performance, S3-compatible distributed object storage system written
in Rust. It uses erasure coding for data durability, supports multi-tenancy through
IAM/STS, and provides a web-based admin console.
A running RustFS node exposes:
- **S3 API** (port 9000) — the primary data path for object CRUD
- **Admin API** (port 9000, `/minio/` prefix) — cluster management, IAM, metrics
- **Console** (port 9001) — web UI backed by the Admin API
- **Inter-node RPC** (gRPC/tonic) — cluster communication for distributed mode
The core data flow for a PUT request looks like:
```
HTTP request
→ server (TLS, auth, routing, compression)
→ app/object_usecase (validation, policy, lifecycle)
→ storage/ecfs (erasure coding, encryption, checksums)
→ ecstore (disk pool selection, data distribution)
→ rio (reader pipeline: encrypt → compress → hash → write)
→ io-core (zero-copy I/O, buffer pool, direct I/O)
→ local disk / remote disk via RPC
```
## Code Map
The repository is a Cargo workspace with a flat `crates/` layout:
```
rustfs/ # Workspace root (virtual manifest)
├── rustfs/ # Main binary + library crate (75K lines)
│ └── src/
│ ├── main.rs # Entry point, startup sequence
│ ├── lib.rs # Module tree root
│ ├── server/ # HTTP server, TLS, routing, middleware
│ ├── admin/ # Admin API handlers and console
│ ├── app/ # Use-case layer (object, bucket, multipart)
│ ├── storage/ # Storage engine interface and implementation
│ ├── auth.rs # S3 request authentication
│ ├── config/ # CLI args, config parsing, workload profiles
│ └── ...
├── crates/ # 39 library crates
│ ├── ecstore/ # Erasure-coded storage engine (⚠️ 87K lines)
│ ├── rio/ # Reader I/O pipeline (encrypt, compress, hash)
│ ├── io-core/ # Zero-copy I/O, scheduling, buffer pool
│ ├── io-metrics/ # I/O metrics collection
│ ├── common/ # Shared runtime state, globals, data usage types
│ ├── config/ # Configuration types and parsing
│ ├── utils/ # Pure utility functions
│ ├── ... # (see "Crate Reference" below)
│ └── e2e_test/ # End-to-end integration tests
└── docs/ # Design documents and analysis
```
### Main Crate Layers (`rustfs/src/`)
The main crate is organized in layers, top to bottom:
| Layer | Directory | Responsibility |
|-------|-----------|----------------|
| **Server** | `server/` | HTTP listener, TLS, CORS, compression, middleware, graceful shutdown |
| **Admin** | `admin/` | Admin API routing, 30+ handler modules, web console |
| **App** | `app/` | Use-case orchestration: object_usecase, bucket_usecase, multipart_usecase |
| **Storage** | `storage/` | S3 API translation, erasure-coded FS, SSE encryption, RPC, concurrency |
| **Auth** | `auth.rs` | S3 signature verification, credential validation |
| **Config** | `config/` | CLI parsing, config struct, workload profiles |
A request flows **downward** through the layers. No layer should reach upward
(e.g., storage must not import from admin).
### Crate Reference
Crates are organized in a dependency DAG with 9 depth levels (0 = leaf, 8 = top):
```
Depth 0 — LEAF (no internal deps):
appauth, checksums, config, credentials, crypto, io-metrics,
madmin, mcp, s3-common, workers, zip
Depth 1:
io-core (→ io-metrics)
policy (→ config, credentials, crypto)
utils (→ config) ⚠️ inverted: utils should be leaf
Depth 2:
concurrency, filemeta, keystone, kms, lock, obs,
signer, targets, trusted-proxies
Depth 3:
common (→ filemeta, madmin) ⚠️ inverted: common should be leaf
Depth 4:
object-capacity, protos, rio
Depth 5 — CORE:
ecstore (16 internal deps, 11 dependents — the architectural heart)
Depth 6:
audit, heal, iam, metrics, notify, s3select-api, scanner
Depth 7:
object-io, protocols, s3select-query
Depth 8 — TOP:
rustfs (35 internal deps — the binary, depends on almost everything)
```
#### By Domain
**Core Infrastructure:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `config` | 3.3K | Configuration types and environment parsing |
| `utils` | 8.7K | Pure utilities (paths, compression, network, retry) |
| `common` | 4.4K | Shared runtime state, globals, data usage types, metrics |
| `madmin` | 5.5K | Admin API request/response types |
**I/O Pipeline:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `io-core` | 6.5K | Zero-copy I/O, buffer pool, direct I/O, scheduling, backpressure |
| `io-metrics` | 4.5K | I/O operation metrics and counters |
| `rio` | 6.9K | Composable reader chain (encrypt → compress → hash → limit) |
| `object-io` | 2.4K | High-level object read/write using rio + ecstore |
| `concurrency` | 1.8K | Concurrency control wrappers over io-core |
**Storage Engine:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `ecstore` | 87K | ⚠️ Erasure-coded storage: disks, pools, buckets, replication, lifecycle |
| `filemeta` | 10K | File/object metadata types and versioning |
| `checksums` | 732 | Checksum computation |
| `lock` | 7.1K | Distributed lock manager |
| `heal` | 5.9K | Data healing / bitrot repair |
| `scanner` | 5.4K | Background data usage scanner |
| `object-capacity` | 2.5K | Capacity tracking and management |
**Security & Auth:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `crypto` | 1.6K | Encryption primitives |
| `credentials` | 713 | Credential types (access key / secret key) |
| `signer` | 1.4K | S3 v4 request signing |
| `iam` | 9.0K | Identity and access management |
| `policy` | 8.8K | Policy engine (S3 bucket/IAM policies) |
| `kms` | 8.1K | Key management service integration |
| `keystone` | 1.9K | OpenStack Keystone auth |
| `appauth` | 143 | Application-level auth tokens |
**Protocol & API:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `protos` | 5.7K | Protobuf/gRPC definitions for inter-node RPC |
| `protocols` | 18K | FTP/FTPS, WebDAV, Swift API support |
| `s3-common` | 738 | Shared S3 types |
| `s3select-api` | 1.9K | S3 Select interface |
| `s3select-query` | 3.6K | S3 Select query engine |
**Observability:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `metrics` | 8.4K | Prometheus metric collectors |
| `io-metrics` | 4.5K | I/O-specific metrics |
| `obs` | 5.6K | OpenTelemetry tracing and telemetry |
| `audit` | 2.4K | Audit logging |
**Events:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `notify` | 5.5K | Event notification system |
| `targets` | 3.2K | Notification targets (Kafka, AMQP, webhook, etc.) |
**Other:**
| Crate | Lines | Purpose |
|-------|-------|---------|
| `trusted-proxies` | 4.0K | Trusted proxy / IP forwarding |
| `zip` | 986 | ZIP archive support for bulk downloads |
| `workers` | 136 | Simple worker abstraction |
| `mcp` | 2.0K | Model Context Protocol server (AI tooling) |
## Architecture Invariants
> These are rules that the codebase should follow. Some are currently violated
> (marked with ⚠️). Documenting them here makes the violations explicit and
> trackable.
1. **Layers flow downward.** Server → Admin/App → Storage → ecstore → rio/io-core.
No upward imports.
2. **Leaf crates have zero internal dependencies.** `config`, `credentials`, `crypto`,
`io-metrics`, `madmin`, `s3-common` should depend only on external crates.
- ⚠️ VIOLATED: `utils` depends on `config`, `common` depends on `filemeta` and `madmin`.
3. **Each type has exactly one definition.** Types shared across crates must be defined
in one crate and re-exported or imported by others.
- ⚠️ VIOLATED: `ReplicationStats` (4 copies), `LastMinuteLatency` (3 copies),
`BackpressureConfig` (3 copies), `DataUsageInfo` (2 copies).
4. **ecstore does not know about HTTP or S3 protocol details.** It operates on
storage-level abstractions (objects, buckets, disks, pools).
5. **The `rustfs` binary crate is the only place that wires everything together.**
Individual crates should be testable in isolation.
6. **Error types use `thiserror` with descriptive names** (e.g., `StorageError`,
not bare `Error`).
- ⚠️ VIOLATED: 6 crates use `pub enum Error`; 2 crates use `snafu`;
`mcp` and `heal` use `anyhow` in library code.
## Known Structural Issues
> This section documents known problems in the current architecture.
> It exists so the team can track and address them deliberately.
### Critical
- **common/scanner code duplication (~3K lines).** `scanner` depends on `common`
but maintains its own copies of `DataUsageInfo`, `LastMinuteLatency`, and related
types instead of importing them.
- **ecstore is a monolith (87K lines, 163 files).** It contains disk management,
bucket management, erasure coding, replication, lifecycle, RPC, and configuration
— all in one crate. It should be decomposed along its existing subdirectories.
### High
- **Dependency inversions.** `utils → config` and `common → filemeta/madmin` break
the layering model. These need to be untangled.
- **Three-layer BackpressureConfig/DeadlockConfig duplication** across io-core,
concurrency, and rustfs/storage. Should be defined once with builder/composition.
### Medium
- **Inconsistent error handling.** Three strategies (thiserror/snafu/anyhow) and
mixed naming (bare `Error` vs descriptive names).
- **Ambiguous common vs utils boundary.** Both described as "utilities and data
structures." Need clear ownership rules.
## Cross-Cutting Concerns
### Error Handling
The project convention is `thiserror` for typed errors with descriptive names.
See `AGENTS.md`: "Prefer thiserror for library-facing error types."
```rust
// GOOD
#[derive(Debug, thiserror::Error)]
pub enum StorageError {
#[error("disk not found: {0}")]
DiskNotFound(String),
}
// AVOID
pub enum Error { ... } // too generic
anyhow::Result<T> // in library code (OK in tests/CLI)
```
### Logging & Tracing
- Use `tracing` crate (`info!`, `warn!`, `error!`, `debug!`, `trace!`)
- Structured fields: `tracing::info!(bucket = %name, "created bucket")`
- Spans for request-scoped context
### Metrics
- Prometheus-style metrics via `rustfs-metrics` crate
- I/O-specific counters via `rustfs-io-metrics`
- Registration happens at crate level, collection in `metrics` crate
### Testing
- Unit tests: `#[cfg(test)] mod tests` in the same file
- Integration tests: inside respective crates (not top-level `tests/`)
- E2E tests: `crates/e2e_test/` — tests against a running server
- Run all: `make test` or `cargo nextest run`
## Startup Sequence
The binary (`main.rs`) boots in this order:
1. Environment variable compatibility (`MINIO_*``RUSTFS_*`)
2. Tokio runtime construction
3. CLI argument parsing
4. License, observability, TLS, trusted proxies initialization
5. Config parsing, server address resolution
6. Credentials, endpoints, local disks, lock client initialization
7. Capacity management initialization
8. HTTP server start (S3 API + optional console)
9. ECStore initialization (erasure coding storage engine)
10. Global config, background replication, KMS
11. Optional: FTP/FTPS/WebDAV servers
12. Event notifier, audit system, deadlock detector
13. Bucket metadata, IAM, Keystone, OIDC
14. Scanner and heal manager
15. Metrics system, mark `FullReady`
16. Wait for shutdown signal → graceful shutdown
## Dependency Diagram (Simplified)
```
┌─────────┐
│ rustfs │ (binary + lib, 75K lines)
│ main │
└────┬────┘
┌───────────────┼───────────────┐
│ │ │
┌────▼────┐ ┌────▼────┐ ┌─────▼─────┐
│ server │ │ admin │ │ app │
│ (HTTP) │ │(console)│ │(use-cases) │
└────┬────┘ └────┬────┘ └─────┬─────┘
│ │ │
└───────────────┼───────────────┘
┌──────▼──────┐
│ storage │
│ (ecfs, SSE, │
│ RPC, ACL) │
└──────┬──────┘
┌──────────────────┼──────────────────┐
│ │ │
┌─────▼─────┐ ┌──────▼──────┐ ┌──────▼──────┐
│ ecstore │ │ rio │ │ io-core │
│ (87K,core) │ │ (readers) │ │ (zero-copy) │
└─────┬──────┘ └─────────────┘ └─────────────┘
┌─────┬──┼──┬─────┬──────┐
│ │ │ │ │ │
common utils config policy filemeta ...
```
## How to Navigate
- **"Where does S3 PutObject go?"**
`server/` routes → `app/object_usecase` validates → `storage/ecfs` encodes →
`ecstore` distributes → `rio` encrypts/compresses → `io-core` writes
- **"Where are bucket policies enforced?"**
`app/bucket_usecase` calls into `crates/policy/`
- **"Where is replication configured?"**
`admin/handlers/replication.rs` and `admin/handlers/site_replication.rs` for API,
`ecstore/src/bucket/replication/` for engine
- **"Where do I add a new admin endpoint?"**
Add handler in `admin/handlers/`, register in `admin/router.rs`
- **"Where do I add a new metric?"**
Define in `crates/metrics/`, register collector, expose via `/minio/v2/metrics`
---
*Inspired by [matklad's ARCHITECTURE.md](https://matklad.github.io/2021/02/06/ARCHITECTURE.md.html)
and [rust-analyzer's architecture.md](https://github.com/rust-analyzer/rust-analyzer/blob/master/docs/book/src/contributing/architecture.md).*
Generated
+214 -211
View File
File diff suppressed because it is too large Load Diff
+15 -15
View File
@@ -52,7 +52,6 @@ members = [
"crates/workers", # Worker thread pools and task scheduling
"crates/io-metrics", # Zero-copy metrics collection for performance analysis
"crates/io-core", # Zero-copy core reader and writer implementations
"crates/object-io", # Object I/O policy and zero-copy helper primitives
"crates/zip", # ZIP file handling and compression
]
resolver = "3"
@@ -74,6 +73,8 @@ unsafe_code = "deny"
[workspace.lints.clippy]
all = "warn"
needless_collect = "warn"
redundant_clone = "warn"
[workspace.dependencies]
# RustFS Internal Crates
@@ -99,7 +100,6 @@ rustfs-metrics = { path = "crates/metrics", version = "0.0.5" }
rustfs-notify = { path = "crates/notify", version = "0.0.5" }
rustfs-io-metrics = { path = "crates/io-metrics", version = "0.0.5" }
rustfs-io-core = { path = "crates/io-core", version = "0.0.5" }
rustfs-object-io = { path = "crates/object-io", version = "0.0.5" }
rustfs-object-capacity = { path = "crates/object-capacity", version = "0.0.5" }
rustfs-obs = { path = "crates/obs", version = "0.0.5" }
rustfs-policy = { path = "crates/policy", version = "0.0.5" }
@@ -122,20 +122,20 @@ async-channel = "2.5.0"
async-compression = { version = "0.4.41" }
async-recursion = "1.1.1"
async-trait = "0.1.89"
axum = "0.8.8"
axum = "0.8.9"
futures = "0.3.32"
futures-core = "0.3.32"
futures-util = "0.3.32"
pollster = "0.4.0"
hyper = { version = "1.9.0", features = ["http2", "http1", "server"] }
hyper-rustls = { version = "0.27.7", default-features = false, features = ["native-tokio", "http1", "tls12", "logging", "http2", "aws-lc-rs", "webpki-roots"] }
hyper-rustls = { version = "0.27.9", default-features = false, features = ["native-tokio", "http1", "tls12", "logging", "http2", "aws-lc-rs", "webpki-roots"] }
hyper-util = { version = "0.1.20", features = ["tokio", "server-auto", "server-graceful", "tracing"] }
http = "1.4.0"
http-body = "1.0.1"
http-body-util = "0.1.3"
reqwest = { version = "0.13.2", default-features = false, features = ["rustls", "charset", "http2", "system-proxy", "stream", "json", "blocking", "query", "form"] }
socket2 = { version = "0.6.3", features = ["all"] }
tokio = { version = "1.51.1", features = ["fs", "rt-multi-thread"] }
tokio = { version = "1.52.0", features = ["fs", "rt-multi-thread", "io-uring"] }
tokio-rustls = { version = "0.26.4", default-features = false, features = ["logging", "tls12", "aws-lc-rs"] }
tokio-stream = { version = "0.1.18" }
tokio-test = "0.4.5"
@@ -154,7 +154,7 @@ flatbuffers = "25.12.19"
form_urlencoded = "1.2.2"
prost = "0.14.3"
quick-xml = "0.39.2"
rmcp = { version = "1.3.0" }
rmcp = { version = "1.4.0" }
rmp = { version = "0.8.15" }
rmp-serde = { version = "1.3.1" }
serde = { version = "1.0.228", features = ["derive"] }
@@ -173,7 +173,7 @@ jsonwebtoken = { version = "10.3.0", features = ["aws_lc_rs"] }
openidconnect = { version = "4.0", default-features = false }
pbkdf2 = "0.13.0-rc.10"
rsa = { version = "0.10.0-rc.17" }
rustls = { version = "0.23.37", default-features = false, features = ["aws-lc-rs", "logging", "tls12", "prefer-post-quantum", "std"] }
rustls = { version = "0.23.38", default-features = false, features = ["aws-lc-rs", "logging", "tls12", "prefer-post-quantum", "std"] }
rustls-pki-types = "1.14.0"
sha1 = "0.11.0"
sha2 = "0.11.0"
@@ -216,15 +216,15 @@ enumset = "1.1.10"
faster-hex = "0.10.0"
flate2 = "1.1.9"
glob = "0.3.3"
google-cloud-storage = "1.10.0"
google-cloud-auth = "1.8.0"
hashbrown = { version = "0.16.1", features = ["serde", "rayon"] }
google-cloud-storage = "1.11.0"
google-cloud-auth = "1.9.0"
hashbrown = { version = "0.17.0", features = ["serde", "rayon"] }
hex = "0.4.3"
hex-simd = "0.8.0"
highway = { version = "1.3.0" }
ipnetwork = { version = "0.21.1", features = ["serde"] }
lazy_static = "1.5.0"
libc = "0.2.184"
libc = "0.2.185"
libsystemd = "0.7.2"
local-ip-address = "0.6.11"
memmap2 = "0.9.10"
@@ -244,9 +244,9 @@ path-clean = "1.0.1"
percent-encoding = "2.3.2"
pin-project-lite = "0.2.17"
pretty_assertions = "1.4.1"
rand = { version = "0.10.0", features = ["serde"] }
rand = { version = "0.10.1", features = ["serde"] }
ratelimit = "0.10.1"
rayon = "1.11.0"
rayon = "1.12.0"
reed-solomon-erasure = { version = "6.0", default-features = false, features = ["std", "simd-accel"] }
reed-solomon-simd = "3.1.0"
regex = { version = "1.12.3" }
@@ -254,7 +254,7 @@ rumqttc = { package = "rumqttc-next", version = "0.29.0", features = ["websocket
rustix = { version = "1.1.4", features = ["fs"] }
rust-embed = { version = "8.11.0" }
rustc-hash = { version = "2.1.2" }
s3s = { git = "https://github.com/rustfs/s3s", rev = "010ae03bdd8c93a9c046108190987bb0e6cb0692", features = ["minio"] }
s3s = { git = "https://github.com/rustfs/s3s", rev = "0dd8fcaaa72eda68fafd49c38daea43bb8697558", features = ["minio"] }
serial_test = "3.4.0"
shadow-rs = { version = "1.7.1", default-features = false }
siphasher = "1.0.2"
@@ -288,7 +288,7 @@ zstd = "0.13.3"
# Observability and Metrics
metrics = "0.24.3"
dial9-tokio-telemetry = { version = "0.2", git = "https://github.com/dial9-rs/dial9-tokio-telemetry.git", rev = "60502082601b647c4a51962595721f631b7bbce1" }
dial9-tokio-telemetry = "0.2"
opentelemetry = { version = "0.31.0" }
opentelemetry-appender-tracing = { version = "0.31.1", features = ["experimental_use_tracing_span_context", "experimental_metadata_attributes", "spec_unstable_logs_enabled"] }
opentelemetry-otlp = { version = "0.31.1", features = ["gzip-http", "reqwest-rustls"] }
+52 -10
View File
@@ -100,28 +100,28 @@ To get started with RustFS, follow these steps:
curl -O https://rustfs.com/install_rustfs.sh && bash install_rustfs.sh
```
### 2\. Docker Quick Start (Option 2)
### 2. Docker Quick Start (Option 2)
The RustFS container runs as a non-root user `rustfs` (UID `10001`). If you run Docker with `-v` to mount a host directory, please ensure the host directory owner is set to `10001`, otherwise you will encounter permission denied errors.
```bash
# Create data and logs directories
mkdir -p data logs
# Create data and logs directories
mkdir -p data logs
# Change the owner of these directories
chown -R 10001:10001 data logs
# Change the owner of these directories
chown -R 10001:10001 data logs
# Using latest version
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
# Using latest version
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
# Using specific version
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-alpha.76
# Using specific version
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-alpha.76
```
If you use [podman](https://github.com/containers/podman) instead of docker, you can install the RustFS with the below command
```bash
podman run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
podman run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
```
You can also use Docker Compose. Using the `docker-compose.yml` file in the root directory:
@@ -219,6 +219,48 @@ rustfs --help
**NOTE**: To access the RustFS instance via `https`, please refer to the [TLS Configuration Docs](https://docs.rustfs.com/integration/tls-configured.html).
### OIDC Roles Claim (Microsoft Entra ID)
RustFS supports mapping an OIDC claim containing role values into the existing
authorization pipeline. The `roles_claim` setting is **optional**: when unset or
empty, only the `groups` claim contributes to authorization (same as older
RustFS releases). For Microsoft Entra ID app roles, set `roles_claim=roles` so
both console admin checks and bucket IAM policies can evaluate those roles.
Example environment configuration (opt-in roles claim):
```bash
RUSTFS_IDENTITY_OPENID_ENABLE=on
RUSTFS_IDENTITY_OPENID_CONFIG_URL="https://login.microsoftonline.com/<tenant-id>/v2.0/.well-known/openid-configuration"
RUSTFS_IDENTITY_OPENID_CLIENT_ID="<client-id>"
RUSTFS_IDENTITY_OPENID_CLIENT_SECRET="<client-secret>"
RUSTFS_IDENTITY_OPENID_SCOPES="openid,profile,email"
RUSTFS_IDENTITY_OPENID_GROUPS_CLAIM="groups"
RUSTFS_IDENTITY_OPENID_ROLES_CLAIM="roles"
```
Policy condition example (evaluate app roles directly with `jwt:roles`; when
`roles_claim` is configured, RustFS also merges those values into `jwt:groups`
for backward compatibility with older policies):
```json
{
"Version": "2012-10-17",
"Statement": [
{
"Effect": "Allow",
"Action": ["admin:*"],
"Resource": ["arn:aws:s3:::*"],
"Condition": {
"ForAnyValue:StringEquals": {
"jwt:roles": ["RustFS.ConsoleAdmin"]
}
}
}
]
}
```
## Documentation
For detailed documentation, including configuration options, API references, and advanced usage, please visit our [Documentation](https://docs.rustfs.com).
-1
View File
@@ -39,7 +39,6 @@ abd = "abd"
mak = "mak"
gae = "gae"
GAE = "GAE"
writeable = "writeable"
# s3-tests original test names (cannot be changed)
nonexisted = "nonexisted"
consts = "consts"
+1 -1
View File
@@ -1097,7 +1097,7 @@ impl DataUsageInfo {
/// Add bucket usage info
pub fn add_bucket_usage(&mut self, bucket: String, usage: BucketUsageInfo) {
self.buckets_usage.insert(bucket.clone(), usage);
self.buckets_usage.insert(bucket, usage);
self.buckets_count = self.buckets_usage.len() as u64;
self.last_update = Some(SystemTime::now());
}
+1 -1
View File
@@ -685,7 +685,7 @@ pub fn current_path_updater(disk: &str, initial: &str) -> (UpdateCurrentPathFn,
};
let done_fn: CloseDiskFn = {
let disk = disk_name.clone();
let disk = disk_name;
Arc::new(move || {
let disk = disk.clone();
Box::pin(async move {
+7 -1
View File
@@ -24,6 +24,7 @@ pub const OIDC_CLAIM_PREFIX: &str = "claim_prefix";
pub const OIDC_ROLE_POLICY: &str = "role_policy";
pub const OIDC_DISPLAY_NAME: &str = "display_name";
pub const OIDC_GROUPS_CLAIM: &str = "groups_claim";
pub const OIDC_ROLES_CLAIM: &str = "roles_claim";
pub const OIDC_EMAIL_CLAIM: &str = "email_claim";
pub const OIDC_USERNAME_CLAIM: &str = "username_claim";
@@ -40,11 +41,12 @@ pub const ENV_IDENTITY_OPENID_CLAIM_PREFIX: &str = "RUSTFS_IDENTITY_OPENID_CLAIM
pub const ENV_IDENTITY_OPENID_ROLE_POLICY: &str = "RUSTFS_IDENTITY_OPENID_ROLE_POLICY";
pub const ENV_IDENTITY_OPENID_DISPLAY_NAME: &str = "RUSTFS_IDENTITY_OPENID_DISPLAY_NAME";
pub const ENV_IDENTITY_OPENID_GROUPS_CLAIM: &str = "RUSTFS_IDENTITY_OPENID_GROUPS_CLAIM";
pub const ENV_IDENTITY_OPENID_ROLES_CLAIM: &str = "RUSTFS_IDENTITY_OPENID_ROLES_CLAIM";
pub const ENV_IDENTITY_OPENID_EMAIL_CLAIM: &str = "RUSTFS_IDENTITY_OPENID_EMAIL_CLAIM";
pub const ENV_IDENTITY_OPENID_USERNAME_CLAIM: &str = "RUSTFS_IDENTITY_OPENID_USERNAME_CLAIM";
/// List of all environment variable keys for an OIDC provider.
pub const ENV_IDENTITY_OPENID_KEYS: &[&str; 14] = &[
pub const ENV_IDENTITY_OPENID_KEYS: &[&str; 15] = &[
ENV_IDENTITY_OPENID_ENABLE,
ENV_IDENTITY_OPENID_CONFIG_URL,
ENV_IDENTITY_OPENID_CLIENT_ID,
@@ -57,6 +59,7 @@ pub const ENV_IDENTITY_OPENID_KEYS: &[&str; 14] = &[
ENV_IDENTITY_OPENID_ROLE_POLICY,
ENV_IDENTITY_OPENID_DISPLAY_NAME,
ENV_IDENTITY_OPENID_GROUPS_CLAIM,
ENV_IDENTITY_OPENID_ROLES_CLAIM,
ENV_IDENTITY_OPENID_EMAIL_CLAIM,
ENV_IDENTITY_OPENID_USERNAME_CLAIM,
];
@@ -75,6 +78,7 @@ pub const IDENTITY_OPENID_KEYS: &[&str] = &[
OIDC_ROLE_POLICY,
OIDC_DISPLAY_NAME,
OIDC_GROUPS_CLAIM,
OIDC_ROLES_CLAIM,
OIDC_EMAIL_CLAIM,
OIDC_USERNAME_CLAIM,
crate::COMMENT_KEY,
@@ -84,6 +88,8 @@ pub const IDENTITY_OPENID_KEYS: &[&str] = &[
pub const OIDC_DEFAULT_SCOPES: &str = "openid,profile,email";
pub const OIDC_DEFAULT_CLAIM_NAME: &str = "groups";
pub const OIDC_DEFAULT_GROUPS_CLAIM: &str = "groups";
/// Empty means do not merge a secondary claim into groups (legacy behavior). Set to e.g. `roles` to opt in.
pub const OIDC_DEFAULT_ROLES_CLAIM: &str = "";
pub const OIDC_DEFAULT_EMAIL_CLAIM: &str = "email";
pub const OIDC_DEFAULT_USERNAME_CLAIM: &str = "preferred_username";
-46
View File
@@ -17,24 +17,6 @@
//! This module defines environment variables and default values for zero-copy
//! read operations, which use memory mapping (mmap) to avoid data copying.
// =============================================================================
// GET Fast Path Configuration
// =============================================================================
/// Environment variable for the GetObject chunk fast path master switch.
///
/// When disabled, `GetObject` bypasses the chunk-streaming fast path entirely and
/// always uses the legacy reader path. This provides an operational stopgap for
/// regressions in the streaming data plane while keeping zero-copy internals
/// configurable independently for future opt-in validation.
pub const ENV_OBJECT_GET_CHUNK_FAST_PATH_ENABLE: &str = "RUSTFS_OBJECT_GET_CHUNK_FAST_PATH_ENABLE";
/// Default: GetObject chunk fast path is disabled.
///
/// The legacy reader path remains the safe default until the chunk-streaming
/// path has sufficient regression coverage for full-body delivery semantics.
pub const DEFAULT_OBJECT_GET_CHUNK_FAST_PATH_ENABLE: bool = false;
// =============================================================================
// Zero-Copy Configuration
// =============================================================================
@@ -67,34 +49,6 @@ pub const ENV_OBJECT_ZERO_COPY_ENABLE: &str = "RUSTFS_OBJECT_ZERO_COPY_ENABLE";
/// to regular I/O without errors.
pub const DEFAULT_OBJECT_ZERO_COPY_ENABLE: bool = true;
/// Environment variable for zero-copy read operating mode.
///
/// Supported values:
/// - `off`: disable mmap-backed chunk fast path and always use the compatibility path
/// - `conservative`: allow a single mmap window per request
/// - `balanced`: allow multiple mmap windows with the default size guardrails
/// - `aggressive`: allow multi-window mmap and relax the small-object cutoff
pub const ENV_OBJECT_ZERO_COPY_MODE: &str = "RUSTFS_OBJECT_ZERO_COPY_MODE";
/// Default zero-copy read mode.
pub const DEFAULT_OBJECT_ZERO_COPY_MODE: &str = "balanced";
/// Environment variable for the maximum mmap window size used by the chunk fast path.
///
/// This controls the visible bytes per mapped chunk before the implementation emits a new window.
pub const ENV_OBJECT_ZERO_COPY_MMAP_WINDOW_BYTES: &str = "RUSTFS_OBJECT_ZERO_COPY_MMAP_WINDOW_BYTES";
/// Default mmap window size for chunk fast path reads: 8 MiB.
pub const DEFAULT_OBJECT_ZERO_COPY_MMAP_WINDOW_BYTES: usize = 8 * 1024 * 1024;
/// Environment variable for the maximum total active mmap bytes.
///
/// Requests that would exceed this active window budget fall back to the compatibility path.
pub const ENV_OBJECT_ZERO_COPY_MAX_ACTIVE_MMAP_BYTES: &str = "RUSTFS_OBJECT_ZERO_COPY_MAX_ACTIVE_MMAP_BYTES";
/// Default maximum active mmap bytes across concurrent local chunk fast-path reads: 256 MiB.
pub const DEFAULT_OBJECT_ZERO_COPY_MAX_ACTIVE_MMAP_BYTES: usize = 256 * 1024 * 1024;
// =============================================================================
// Direct I/O Configuration
// =============================================================================
+14 -3
View File
@@ -204,10 +204,11 @@ pub fn gen_secret_key(length: usize) -> std::io::Result<String> {
let mut key = vec![0u8; URL_SAFE_NO_PAD.estimated_decoded_length(length)];
rng.fill_bytes(&mut key);
// URL_SAFE_NO_PAD uses "-" and "_" instead of "+" and "/", so "/" never
// appears in the output. The .replace("/", "+") was a dead no-op.
let encoded = URL_SAFE_NO_PAD.encode_to_string(&key);
let key_str = encoded.replace("/", "+");
Ok(key_str)
Ok(encoded)
}
/// Get the RPC authentication token from environment variable
@@ -447,7 +448,7 @@ mod tests {
// Initialize
let test_ak = "test_access_key".to_string();
let test_sk = "test_secret_key_123456".to_string();
init_global_action_credentials(Some(test_ak.clone()), Some(test_sk.clone())).ok();
init_global_action_credentials(Some(test_ak), Some(test_sk)).ok();
}
// Verify the state after initialization
@@ -473,6 +474,16 @@ mod tests {
}
}
#[test]
fn test_gen_secret_key_uses_url_safe_base64_without_padding() {
let key = gen_secret_key(32).expect("secret key should generate");
assert_eq!(key.len(), 32);
assert!(!key.contains('/'));
assert!(!key.contains('+'));
assert!(!key.contains('='));
}
#[test]
fn test_masked_debug() {
// Test None
-5
View File
@@ -20,11 +20,6 @@ license.workspace = true
repository.workspace = true
rust-version.workspace = true
[[bin]]
name = "small_put_bench"
path = "src/bin/small_put_bench.rs"
test = false
[lints]
workspace = true
@@ -15,6 +15,8 @@
#[cfg(test)]
mod tests {
use crate::common::{RustFSTestEnvironment, init_logging, local_http_client, rustfs_binary_path};
use aws_sdk_s3::Client as S3Client;
use aws_sdk_s3::error::ProvideErrorMetadata;
use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart};
use http::header::{CONTENT_TYPE, HOST};
@@ -216,12 +218,129 @@ mod tests {
Ok(builder.send().await?)
}
async fn assert_archive_object_content_encoding(
client: &S3Client,
bucket: &str,
key: &str,
expected_content_encoding: Option<&str>,
expected_body: &[u8],
) -> Result<(), Box<dyn Error + Send + Sync>> {
let head_resp = client.head_object().bucket(bucket).key(key).send().await?;
assert_eq!(head_resp.content_encoding(), expected_content_encoding);
let get_resp = client.get_object().bucket(bucket).key(key).send().await?;
assert_eq!(get_resp.content_encoding(), expected_content_encoding);
let body = get_resp.body.collect().await?.into_bytes();
assert_eq!(body.as_ref(), expected_body);
Ok(())
}
async fn complete_archive_multipart_upload_with_content_encoding(
client: &S3Client,
bucket: &str,
key: &str,
content_encoding: Option<&str>,
) -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
let payload = random_bytes(MULTIPART_PART_SIZE + 256 * 1024);
let zip_bytes = build_zip_bytes(&[("payload.bin", payload.as_slice())])?;
assert!(zip_bytes.len() > MULTIPART_PART_SIZE, "zip payload must exceed multipart threshold");
let mut create_builder = client
.create_multipart_upload()
.bucket(bucket)
.key(key)
.content_type("application/zip");
if let Some(content_encoding) = content_encoding {
create_builder = create_builder.content_encoding(content_encoding);
}
let create_output = create_builder.send().await?;
let upload_id = create_output.upload_id().expect("multipart upload id");
let first_part = zip_bytes[..MULTIPART_PART_SIZE].to_vec();
let second_part = zip_bytes[MULTIPART_PART_SIZE..].to_vec();
let upload_part_1 = client
.upload_part()
.bucket(bucket)
.key(key)
.upload_id(upload_id)
.part_number(1)
.body(ByteStream::from(first_part))
.send()
.await?;
let upload_part_2 = client
.upload_part()
.bucket(bucket)
.key(key)
.upload_id(upload_id)
.part_number(2)
.body(ByteStream::from(second_part))
.send()
.await?;
let completed_upload = CompletedMultipartUpload::builder()
.parts(
CompletedPart::builder()
.part_number(1)
.e_tag(upload_part_1.e_tag().unwrap_or_default())
.build(),
)
.parts(
CompletedPart::builder()
.part_number(2)
.e_tag(upload_part_2.e_tag().unwrap_or_default())
.build(),
)
.build();
client
.complete_multipart_upload()
.bucket(bucket)
.key(key)
.upload_id(upload_id)
.multipart_upload(completed_upload)
.send()
.await?;
Ok(zip_bytes)
}
#[tokio::test]
#[serial]
async fn test_archive_put_rejects_content_encoding_by_default() -> Result<(), Box<dyn Error + Send + Sync>> {
async fn test_archive_put_allows_content_encoding_by_default() -> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
env.start_rustfs_server_without_cleanup(vec![]).await?;
start_rustfs_server_with_env(&mut env, &[]).await?;
env.create_test_bucket(ARCHIVE_TEST_BUCKET).await?;
let zip_bytes = build_zip_bytes(&[("alpha.txt", b"archive-body")])?;
let object_url = format!("{}/{}/{}", env.url, ARCHIVE_TEST_BUCKET, "bundle.zip");
let response =
signed_put_request_with_headers(&object_url, &env.access_key, &env.secret_key, zip_bytes, "application/zip", "gzip")
.await?;
assert_eq!(response.status(), StatusCode::OK);
let client = env.create_s3_client();
let head_resp = client
.head_object()
.bucket(ARCHIVE_TEST_BUCKET)
.key("bundle.zip")
.send()
.await?;
assert_eq!(head_resp.content_encoding(), Some("gzip"));
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_archive_put_rejects_content_encoding_when_strict_mode_enabled() -> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING", "true")]).await?;
env.create_test_bucket(ARCHIVE_TEST_BUCKET).await?;
let zip_bytes = build_zip_bytes(&[("alpha.txt", b"archive-body")])?;
let object_url = format!("{}/{}/{}", env.url, ARCHIVE_TEST_BUCKET, "bundle.zip");
@@ -232,7 +351,145 @@ mod tests {
assert_eq!(response.status(), StatusCode::BAD_REQUEST);
let body = response.text().await?;
assert!(
body.contains("InvalidArgument") || body.contains("Content-Encoding"),
body.contains("InvalidArgument") || body.contains("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING"),
"unexpected error body: {body}"
);
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_archive_put_with_aws_chunked_does_not_persist_content_encoding_by_default()
-> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[]).await?;
env.create_test_bucket(ARCHIVE_TEST_BUCKET).await?;
let zip_bytes = build_zip_bytes(&[("alpha.txt", b"archive-body")])?;
let object_url = format!("{}/{}/{}", env.url, ARCHIVE_TEST_BUCKET, "bundle-aws-chunked.zip");
let response = signed_put_request_with_headers(
&object_url,
&env.access_key,
&env.secret_key,
zip_bytes.clone(),
"application/zip",
"aws-chunked",
)
.await?;
assert_eq!(response.status(), StatusCode::OK);
let client = env.create_s3_client();
assert_archive_object_content_encoding(
&client,
ARCHIVE_TEST_BUCKET,
"bundle-aws-chunked.zip",
None,
zip_bytes.as_slice(),
)
.await?;
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_archive_put_with_aws_chunked_and_effective_encoding_roundtrips_by_default()
-> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[]).await?;
env.create_test_bucket(ARCHIVE_TEST_BUCKET).await?;
let zip_bytes = build_zip_bytes(&[("alpha.txt", b"archive-body")])?;
let object_url = format!("{}/{}/{}", env.url, ARCHIVE_TEST_BUCKET, "bundle-aws-chunked-gzip.zip");
let response = signed_put_request_with_headers(
&object_url,
&env.access_key,
&env.secret_key,
zip_bytes.clone(),
"application/zip",
"aws-chunked,gzip",
)
.await?;
assert_eq!(response.status(), StatusCode::OK);
let client = env.create_s3_client();
assert_archive_object_content_encoding(
&client,
ARCHIVE_TEST_BUCKET,
"bundle-aws-chunked-gzip.zip",
Some("gzip"),
zip_bytes.as_slice(),
)
.await?;
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_archive_put_with_aws_chunked_allowed_when_strict_mode_enabled() -> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING", "true")]).await?;
env.create_test_bucket(ARCHIVE_TEST_BUCKET).await?;
let zip_bytes = build_zip_bytes(&[("alpha.txt", b"archive-body")])?;
let object_url = format!("{}/{}/{}", env.url, ARCHIVE_TEST_BUCKET, "bundle-strict-aws-chunked.zip");
let response = signed_put_request_with_headers(
&object_url,
&env.access_key,
&env.secret_key,
zip_bytes.clone(),
"application/zip",
"aws-chunked",
)
.await?;
assert_eq!(response.status(), StatusCode::OK);
let client = env.create_s3_client();
assert_archive_object_content_encoding(
&client,
ARCHIVE_TEST_BUCKET,
"bundle-strict-aws-chunked.zip",
None,
zip_bytes.as_slice(),
)
.await?;
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_archive_put_with_aws_chunked_and_effective_encoding_rejects_when_strict_mode_enabled()
-> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING", "true")]).await?;
env.create_test_bucket(ARCHIVE_TEST_BUCKET).await?;
let zip_bytes = build_zip_bytes(&[("alpha.txt", b"archive-body")])?;
let object_url = format!("{}/{}/{}", env.url, ARCHIVE_TEST_BUCKET, "bundle-strict-aws-chunked-gzip.zip");
let response = signed_put_request_with_headers(
&object_url,
&env.access_key,
&env.secret_key,
zip_bytes,
"application/zip",
"aws-chunked,gzip",
)
.await?;
assert_eq!(response.status(), StatusCode::BAD_REQUEST);
let body = response.text().await?;
assert!(
body.contains("InvalidArgument") || body.contains("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING"),
"unexpected error body: {body}"
);
@@ -400,11 +657,103 @@ mod tests {
#[tokio::test]
#[serial]
async fn test_presigned_get_and_reverse_proxy_preserve_multipart_bytes_with_fast_path()
async fn test_archive_multipart_with_aws_chunked_and_effective_encoding_roundtrips_by_default()
-> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[("RUSTFS_OBJECT_GET_CHUNK_FAST_PATH_ENABLE", "true")]).await?;
start_rustfs_server_with_env(&mut env, &[]).await?;
env.create_test_bucket(MULTIPART_ARCHIVE_TEST_BUCKET).await?;
let client = env.create_s3_client();
let zip_bytes = complete_archive_multipart_upload_with_content_encoding(
&client,
MULTIPART_ARCHIVE_TEST_BUCKET,
"multipart-aws-chunked-gzip.zip",
Some("aws-chunked,gzip"),
)
.await?;
assert_archive_object_content_encoding(
&client,
MULTIPART_ARCHIVE_TEST_BUCKET,
"multipart-aws-chunked-gzip.zip",
Some("gzip"),
zip_bytes.as_slice(),
)
.await?;
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_archive_multipart_with_aws_chunked_allowed_when_strict_mode_enabled() -> Result<(), Box<dyn Error + Send + Sync>>
{
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING", "true")]).await?;
env.create_test_bucket(MULTIPART_ARCHIVE_TEST_BUCKET).await?;
let client = env.create_s3_client();
let zip_bytes = complete_archive_multipart_upload_with_content_encoding(
&client,
MULTIPART_ARCHIVE_TEST_BUCKET,
"multipart-strict-aws-chunked.zip",
Some("aws-chunked"),
)
.await?;
assert_archive_object_content_encoding(
&client,
MULTIPART_ARCHIVE_TEST_BUCKET,
"multipart-strict-aws-chunked.zip",
None,
zip_bytes.as_slice(),
)
.await?;
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_archive_multipart_with_aws_chunked_and_effective_encoding_rejects_when_strict_mode_enabled()
-> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
start_rustfs_server_with_env(&mut env, &[("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING", "true")]).await?;
env.create_test_bucket(MULTIPART_ARCHIVE_TEST_BUCKET).await?;
let client = env.create_s3_client();
let create_result = client
.create_multipart_upload()
.bucket(MULTIPART_ARCHIVE_TEST_BUCKET)
.key("multipart-strict-aws-chunked-gzip.zip")
.content_type("application/zip")
.content_encoding("aws-chunked,gzip")
.send()
.await;
let err = create_result.expect_err("strict mode should reject effective archive content encoding");
assert_eq!(err.code(), Some("InvalidArgument"));
assert!(
err.message().is_some_and(|message| {
message.contains("Content-Encoding") && message.contains("RUSTFS_REJECT_ARCHIVE_CONTENT_ENCODING")
}),
"unexpected error metadata: code={:?}, message={:?}",
err.code(),
err.message()
);
env.stop_server();
Ok(())
}
#[tokio::test]
#[serial]
async fn test_presigned_get_and_reverse_proxy_preserve_multipart_bytes() -> Result<(), Box<dyn Error + Send + Sync>> {
init_logging();
let mut env = RustFSTestEnvironment::new().await?;
env.start_rustfs_server(vec![]).await?;
env.create_test_bucket(MULTIPART_ARCHIVE_TEST_BUCKET).await?;
let client = env.create_s3_client();
-441
View File
@@ -1,441 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use anyhow::{Context, Result, anyhow, bail};
use aws_sdk_s3::config::{Credentials, Region};
use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::{Delete, ObjectIdentifier};
use aws_sdk_s3::{Client, Config};
use aws_smithy_http_client::Builder as SmithyHttpClientBuilder;
use bytes::Bytes;
use clap::Parser;
use serde::Serialize;
use std::path::PathBuf;
use std::time::{Duration, Instant};
#[derive(Parser, Debug)]
#[command(name = "small_put_bench")]
#[command(about = "Rust-native small PUT benchmark for RustFS-compatible S3 endpoints")]
struct Args {
#[arg(long, env = "RUSTFS_BENCH_ENDPOINT")]
endpoint: String,
#[arg(long, env = "RUSTFS_BENCH_ACCESS_KEY", default_value = "rustfsadmin")]
access_key: String,
#[arg(long, env = "RUSTFS_BENCH_SECRET_KEY", default_value = "rustfsadmin")]
secret_key: String,
#[arg(long, env = "RUSTFS_BENCH_REGION", default_value = "us-east-1")]
region: String,
#[arg(long, env = "RUSTFS_BENCH_BUCKET", default_value = "small-put-benchmark")]
bucket: String,
#[arg(long, env = "RUSTFS_BENCH_SIZES", default_value = "4KiB,16KiB,64KiB,256KiB,1MiB")]
sizes: String,
#[arg(long, env = "RUSTFS_BENCH_CONCURRENCY", default_value_t = 8)]
concurrency: usize,
#[arg(long, env = "RUSTFS_BENCH_DURATION_SECS", default_value_t = 10)]
duration_secs: u64,
#[arg(long, env = "RUSTFS_BENCH_TIMEOUT_SECS", default_value_t = 15)]
timeout_secs: u64,
#[arg(long, env = "RUSTFS_BENCH_PREFIX")]
prefix: Option<String>,
#[arg(long)]
output_json: Option<PathBuf>,
#[arg(long, default_value_t = false)]
cleanup: bool,
}
#[derive(Clone, Debug)]
struct SizeSpec {
label: String,
slug: String,
bytes: usize,
}
#[derive(Debug)]
struct Sample {
ok: bool,
duration_ms: f64,
}
#[derive(Debug, Serialize)]
struct SizeSummary {
label: String,
bytes: usize,
total: usize,
succeeded: usize,
failed: usize,
wall_secs: f64,
object_rate: f64,
throughput_mib_per_sec: f64,
avg_ms: Option<f64>,
p50_ms: Option<f64>,
p90_ms: Option<f64>,
p99_ms: Option<f64>,
}
#[derive(Debug, Serialize)]
struct RunSummary {
run_id: String,
endpoint: String,
bucket: String,
concurrency: usize,
duration_secs: u64,
timeout_secs: u64,
sizes: Vec<SizeSummary>,
}
fn main() -> Result<()> {
let runtime = tokio::runtime::Builder::new_multi_thread()
.enable_all()
.build()
.context("failed to build tokio runtime")?;
runtime.block_on(async_main())
}
async fn async_main() -> Result<()> {
let args = Args::parse();
validate_args(&args)?;
let sizes = parse_size_list(&args.sizes)?;
let run_id = args.prefix.clone().unwrap_or_else(default_run_id);
let client = build_s3_client(&args.endpoint, &args.access_key, &args.secret_key, &args.region);
ensure_bucket(&client, &args.bucket).await?;
let mut size_summaries = Vec::with_capacity(sizes.len());
for size in &sizes {
let summary = run_size_benchmark(
client.clone(),
args.bucket.clone(),
run_id.clone(),
size.clone(),
args.concurrency,
Duration::from_secs(args.duration_secs),
Duration::from_secs(args.timeout_secs),
)
.await?;
print_size_summary(&summary);
size_summaries.push(summary);
}
if args.cleanup {
cleanup_prefix(&client, &args.bucket, &run_id).await?;
}
let summary = RunSummary {
run_id,
endpoint: args.endpoint,
bucket: args.bucket,
concurrency: args.concurrency,
duration_secs: args.duration_secs,
timeout_secs: args.timeout_secs,
sizes: size_summaries,
};
if let Some(path) = args.output_json {
let json = serde_json::to_vec_pretty(&summary).context("failed to serialize benchmark summary")?;
std::fs::write(&path, json).with_context(|| format!("failed to write benchmark summary to {}", path.display()))?;
println!("Wrote summary to {}", path.display());
}
Ok(())
}
fn validate_args(args: &Args) -> Result<()> {
if args.concurrency == 0 {
bail!("--concurrency must be greater than zero");
}
if args.duration_secs == 0 {
bail!("--duration-secs must be greater than zero");
}
if args.timeout_secs == 0 {
bail!("--timeout-secs must be greater than zero");
}
Ok(())
}
fn build_s3_client(endpoint: &str, access_key: &str, secret_key: &str, region: &str) -> Client {
let credentials = Credentials::new(access_key, secret_key, None, None, "small-put-bench");
let mut config = Config::builder()
.credentials_provider(credentials)
.region(Region::new(region.to_string()))
.endpoint_url(endpoint)
.force_path_style(true)
.behavior_version_latest();
if endpoint.starts_with("http://") {
config = config.http_client(SmithyHttpClientBuilder::new().build_http());
}
Client::from_conf(config.build())
}
async fn ensure_bucket(client: &Client, bucket: &str) -> Result<()> {
if client.head_bucket().bucket(bucket).send().await.is_ok() {
return Ok(());
}
match client.create_bucket().bucket(bucket).send().await {
Ok(_) => Ok(()),
Err(err) => {
let rendered = err.to_string();
if rendered.contains("BucketAlreadyOwnedByYou") || rendered.contains("BucketAlreadyExists") {
Ok(())
} else {
Err(err).with_context(|| format!("failed to create benchmark bucket {bucket}"))
}
}
}
}
async fn run_size_benchmark(
client: Client,
bucket: String,
run_id: String,
size: SizeSpec,
concurrency: usize,
duration: Duration,
timeout: Duration,
) -> Result<SizeSummary> {
let payload = Bytes::from(vec![0_u8; size.bytes]);
let deadline = Instant::now() + duration;
let wall_start = Instant::now();
let mut handles = Vec::with_capacity(concurrency);
for worker in 0..concurrency {
let client = client.clone();
let bucket = bucket.clone();
let payload = payload.clone();
let prefix = format!("{run_id}/{}/worker-{worker}", size.slug);
handles.push(tokio::spawn(async move {
let mut samples = Vec::new();
let mut idx = 0usize;
while Instant::now() < deadline {
let key = format!("{prefix}/obj-{idx}.bin");
let started_at = Instant::now();
let request = client
.put_object()
.bucket(&bucket)
.key(key)
.body(ByteStream::from(payload.clone()))
.content_type("application/octet-stream");
let ok = matches!(tokio::time::timeout(timeout, request.send()).await, Ok(Ok(_)));
samples.push(Sample {
ok,
duration_ms: started_at.elapsed().as_secs_f64() * 1000.0,
});
idx += 1;
}
samples
}));
}
let mut samples = Vec::new();
for handle in handles {
samples.extend(handle.await.map_err(|err| anyhow!("benchmark worker join error: {err}"))?);
}
Ok(build_size_summary(&size, samples, wall_start.elapsed()))
}
fn build_size_summary(size: &SizeSpec, mut samples: Vec<Sample>, wall_elapsed: Duration) -> SizeSummary {
let total = samples.len();
let succeeded = samples.iter().filter(|sample| sample.ok).count();
let failed = total.saturating_sub(succeeded);
let wall_secs = wall_elapsed.as_secs_f64();
let object_rate = if wall_secs > 0.0 { succeeded as f64 / wall_secs } else { 0.0 };
let throughput_mib_per_sec = if wall_secs > 0.0 {
((size.bytes * succeeded) as f64 / (1024.0 * 1024.0)) / wall_secs
} else {
0.0
};
let avg_ms = if total > 0 {
Some(samples.iter().map(|sample| sample.duration_ms).sum::<f64>() / total as f64)
} else {
None
};
samples.sort_by(|lhs, rhs| lhs.duration_ms.total_cmp(&rhs.duration_ms));
let durations: Vec<f64> = samples.into_iter().map(|sample| sample.duration_ms).collect();
SizeSummary {
label: size.label.clone(),
bytes: size.bytes,
total,
succeeded,
failed,
wall_secs,
object_rate,
throughput_mib_per_sec,
avg_ms,
p50_ms: percentile(&durations, 0.50),
p90_ms: percentile(&durations, 0.90),
p99_ms: percentile(&durations, 0.99),
}
}
async fn cleanup_prefix(client: &Client, bucket: &str, prefix: &str) -> Result<()> {
let mut continuation_token = None;
loop {
let response = client
.list_objects_v2()
.bucket(bucket)
.prefix(prefix)
.set_continuation_token(continuation_token.clone())
.send()
.await
.with_context(|| format!("failed to list objects for cleanup under {bucket}/{prefix}"))?;
let objects: Vec<ObjectIdentifier> = response
.contents
.unwrap_or_default()
.into_iter()
.filter_map(|object| object.key.map(|key| ObjectIdentifier::builder().key(key).build().ok()))
.flatten()
.collect();
for chunk in objects.chunks(1_000) {
if chunk.is_empty() {
continue;
}
client
.delete_objects()
.bucket(bucket)
.delete(
Delete::builder()
.set_objects(Some(chunk.to_vec()))
.quiet(true)
.build()
.context("failed to build delete request")?,
)
.send()
.await
.with_context(|| format!("failed to delete cleanup batch under {bucket}/{prefix}"))?;
}
if response.is_truncated.unwrap_or(false) {
continuation_token = response.next_continuation_token;
} else {
break;
}
}
Ok(())
}
fn parse_size_list(input: &str) -> Result<Vec<SizeSpec>> {
input
.split(',')
.map(str::trim)
.filter(|item| !item.is_empty())
.map(parse_size_spec)
.collect()
}
fn parse_size_spec(input: &str) -> Result<SizeSpec> {
let normalized = input.trim();
let lower = normalized.to_ascii_lowercase();
let (number_part, multiplier) = if let Some(value) = lower.strip_suffix("kib") {
(value, 1024usize)
} else if let Some(value) = lower.strip_suffix("mib") {
(value, 1024usize * 1024usize)
} else if let Some(value) = lower.strip_suffix('b') {
(value, 1usize)
} else {
(lower.as_str(), 1usize)
};
let value = number_part
.trim()
.parse::<usize>()
.with_context(|| format!("invalid size component: {input}"))?;
let bytes = value
.checked_mul(multiplier)
.ok_or_else(|| anyhow!("size overflow for {input}"))?;
Ok(SizeSpec {
label: normalized.to_string(),
slug: normalized
.chars()
.filter(|ch| ch.is_ascii_alphanumeric())
.collect::<String>()
.to_ascii_lowercase(),
bytes,
})
}
fn percentile(values: &[f64], percentile: f64) -> Option<f64> {
if values.is_empty() {
return None;
}
let index = ((values.len() - 1) as f64 * percentile).floor() as usize;
values.get(index).copied()
}
fn default_run_id() -> String {
format!("small-put-bench-{}", chrono::Utc::now().format("%Y%m%d-%H%M%S"))
}
fn print_size_summary(summary: &SizeSummary) {
println!(
"{}: success={} failed={} obj/s={:.3} MiB/s={:.3} avg={:.3?} p50={:.3?} p90={:.3?} p99={:.3?}",
summary.label,
summary.succeeded,
summary.failed,
summary.object_rate,
summary.throughput_mib_per_sec,
summary.avg_ms,
summary.p50_ms,
summary.p90_ms,
summary.p99_ms,
);
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn parse_size_spec_supports_binary_units() {
let four_kib = parse_size_spec("4KiB").expect("4KiB should parse");
assert_eq!(four_kib.bytes, 4 * 1024);
let one_mib = parse_size_spec("1MiB").expect("1MiB should parse");
assert_eq!(one_mib.bytes, 1024 * 1024);
}
#[test]
fn percentile_returns_expected_bucket() {
let values = vec![10.0, 20.0, 30.0, 40.0, 50.0];
assert_eq!(percentile(&values, 0.50), Some(30.0));
assert_eq!(percentile(&values, 0.90), Some(40.0));
}
}
+1 -275
View File
@@ -19,13 +19,9 @@
mod tests {
use crate::common::{RustFSTestEnvironment, init_logging};
use aws_sdk_s3::Client;
use aws_sdk_s3::primitives::{ByteStream, SdkBody};
use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::{ChecksumAlgorithm, ChecksumMode, CompletedMultipartUpload, CompletedPart};
use base64::Engine;
use bytes::Bytes;
use futures::StreamExt;
use http_body::Frame;
use http_body_util::StreamBody;
use rustfs_rio::{Checksum, ChecksumType as RioChecksumType};
use serial_test::serial;
use sha2::{Digest, Sha256};
@@ -68,53 +64,6 @@ mod tests {
.encoded
}
fn streamed_body_70kib_of_a() -> ByteStream {
let bytes = Bytes::from_static(&[b'a'; 1024]);
let stream = futures::stream::repeat_with(move || {
let frame = Frame::data(bytes.clone());
Ok::<_, std::io::Error>(frame)
});
let body = WithSizeHint::new(StreamBody::new(stream.take(70)), 70 * 1024);
ByteStream::new(SdkBody::from_body_1_x(body))
}
struct WithSizeHint<T> {
inner: T,
size_hint: usize,
}
impl<T> WithSizeHint<T> {
fn new(inner: T, size_hint: usize) -> Self {
Self { inner, size_hint }
}
}
impl<T> http_body::Body for WithSizeHint<T>
where
T: http_body::Body + Unpin,
{
type Data = T::Data;
type Error = T::Error;
fn poll_frame(
self: std::pin::Pin<&mut Self>,
cx: &mut std::task::Context<'_>,
) -> std::task::Poll<Option<Result<Frame<Self::Data>, Self::Error>>> {
let this = self.get_mut();
std::pin::Pin::new(&mut this.inner).poll_frame(cx)
}
fn is_end_stream(&self) -> bool {
self.inner.is_end_stream()
}
fn size_hint(&self) -> http_body::SizeHint {
let mut hint = self.inner.size_hint();
hint.set_exact(self.size_hint as u64);
hint
}
}
/// PutObject with Content-MD5: upload succeeds and GetObject returns same content.
#[tokio::test]
#[serial]
@@ -187,121 +136,6 @@ mod tests {
info!("PASSED: PutObject with checksum_sha256 and GetObject content match");
}
/// Mirrors `s3s-e2e` behavior: only request `checksum_algorithm`, then expect
/// both PutObject and GetObject(checksum_mode=enabled) to expose the same checksum.
#[tokio::test]
#[serial]
async fn test_put_object_with_checksum_algorithm_only() {
init_logging();
info!("TEST: PutObject with checksum_algorithm only");
let mut env = RustFSTestEnvironment::new().await.expect("Failed to create test environment");
env.start_rustfs_server(vec![]).await.expect("Failed to start RustFS");
let client = create_s3_client(&env);
let bucket = "test-checksum-algorithm-only";
create_bucket(&client, bucket).await.expect("Failed to create bucket");
let key = "obj-with-checksum-algorithm-only.txt";
let content = vec![b'a'; 70 * 1024];
let put_resp = client
.put_object()
.bucket(bucket)
.key(key)
.checksum_algorithm(ChecksumAlgorithm::Crc32)
.body(ByteStream::from(content.clone()))
.send()
.await
.expect("PutObject with checksum_algorithm should succeed");
let put_checksum = put_resp
.checksum_crc32()
.expect("PutObject should return checksum_crc32 when checksum_algorithm is used")
.to_string();
let mut get_resp = client
.get_object()
.bucket(bucket)
.key(key)
.checksum_mode(ChecksumMode::Enabled)
.send()
.await
.expect("GetObject should succeed");
let body_bytes = std::mem::replace(&mut get_resp.body, ByteStream::new(aws_sdk_s3::primitives::SdkBody::empty()))
.collect()
.await
.expect("collect body")
.into_bytes();
assert_eq!(body_bytes.as_ref(), content.as_slice(), "GetObject body must match uploaded content");
assert_eq!(
get_resp.checksum_crc32().map(str::to_string),
Some(put_checksum),
"GetObject(checksum_mode=enabled) should expose the stored CRC32 checksum"
);
}
/// Matches the `s3s-e2e` streaming upload shape more closely than `ByteStream::from(Vec<u8>)`.
#[tokio::test]
#[serial]
async fn test_put_object_with_checksum_algorithm_only_streaming_body() {
init_logging();
info!("TEST: PutObject with checksum_algorithm only using streaming body");
let mut env = RustFSTestEnvironment::new().await.expect("Failed to create test environment");
env.start_rustfs_server(vec![]).await.expect("Failed to start RustFS");
let client = create_s3_client(&env);
let bucket = "test-checksum-algorithm-streaming";
create_bucket(&client, bucket).await.expect("Failed to create bucket");
let key = "obj-with-checksum-algorithm-streaming.txt";
let expected_content = vec![b'a'; 70 * 1024];
let put_resp = client
.put_object()
.bucket(bucket)
.key(key)
.checksum_algorithm(ChecksumAlgorithm::Crc32)
.body(streamed_body_70kib_of_a())
.send()
.await
.expect("PutObject with streaming checksum_algorithm should succeed");
let put_checksum = put_resp
.checksum_crc32()
.expect("PutObject should return checksum_crc32 for streaming checksum_algorithm uploads")
.to_string();
let mut get_resp = client
.get_object()
.bucket(bucket)
.key(key)
.checksum_mode(ChecksumMode::Enabled)
.send()
.await
.expect("GetObject should succeed");
let body_bytes = std::mem::replace(&mut get_resp.body, ByteStream::new(SdkBody::empty()))
.collect()
.await
.expect("collect body")
.into_bytes();
assert_eq!(
body_bytes.as_ref(),
expected_content.as_slice(),
"GetObject body must match uploaded content"
);
assert_eq!(
get_resp.checksum_crc32().map(str::to_string),
Some(put_checksum),
"GetObject(checksum_mode=enabled) should expose the stored CRC32 checksum for streaming uploads"
);
}
/// Multipart upload with checksum: CreateMultipartUpload, UploadPart(s) with checksum_sha256, CompleteMultipartUpload; then GetObject verifies content.
/// Uses part size >= 5MB (server minimum) for two parts.
#[tokio::test]
@@ -400,114 +234,6 @@ mod tests {
info!("PASSED: MultipartUpload with checksum and GetObject content match");
}
/// Mirrors `s3s-e2e` multipart behavior: request checksum algorithm at MPU creation,
/// rely on auto checksum handling during UploadPart, and expect CompleteMultipartUpload to succeed.
#[tokio::test]
#[serial]
async fn test_multipart_upload_with_crc32_algorithm_only() {
init_logging();
info!("TEST: MultipartUpload with checksum_algorithm only (CRC32)");
let mut env = RustFSTestEnvironment::new().await.expect("Failed to create test environment");
env.start_rustfs_server(vec![]).await.expect("Failed to start RustFS");
let client = create_s3_client(&env);
let bucket = "test-multipart-checksum-crc32-auto";
create_bucket(&client, bucket).await.expect("Failed to create bucket");
let key = "multipart-with-crc32-auto.bin";
let part1_content = "a".repeat(5 * 1024 * 1024 + 1);
let part2_content = "b".repeat(1024);
let create_resp = client
.create_multipart_upload()
.bucket(bucket)
.key(key)
.checksum_algorithm(ChecksumAlgorithm::Crc32)
.send()
.await
.expect("CreateMultipartUpload should succeed");
let upload_id = create_resp.upload_id().expect("upload_id should be present");
let part1_resp = client
.upload_part()
.bucket(bucket)
.key(key)
.upload_id(upload_id)
.part_number(1)
.body(ByteStream::from(part1_content.clone().into_bytes()))
.send()
.await
.expect("UploadPart 1 should succeed");
let part1_checksum = part1_resp
.checksum_crc32()
.expect("UploadPart 1 should return checksum_crc32")
.to_string();
let part2_resp = client
.upload_part()
.bucket(bucket)
.key(key)
.upload_id(upload_id)
.part_number(2)
.body(ByteStream::from(part2_content.clone().into_bytes()))
.send()
.await
.expect("UploadPart 2 should succeed");
let part2_checksum = part2_resp
.checksum_crc32()
.expect("UploadPart 2 should return checksum_crc32")
.to_string();
let completed_upload = CompletedMultipartUpload::builder()
.parts(
CompletedPart::builder()
.part_number(1)
.e_tag(part1_resp.e_tag().expect("etag part 1"))
.checksum_crc32(part1_checksum)
.build(),
)
.parts(
CompletedPart::builder()
.part_number(2)
.e_tag(part2_resp.e_tag().expect("etag part 2"))
.checksum_crc32(part2_checksum)
.build(),
)
.build();
client
.complete_multipart_upload()
.bucket(bucket)
.key(key)
.upload_id(upload_id)
.multipart_upload(completed_upload)
.send()
.await
.expect("CompleteMultipartUpload should succeed");
let body_bytes = client
.get_object()
.bucket(bucket)
.key(key)
.send()
.await
.expect("GetObject should succeed")
.body
.collect()
.await
.expect("collect body")
.into_bytes();
let expected_content = format!("{part1_content}{part2_content}");
assert_eq!(
body_bytes.as_ref(),
expected_content.as_bytes(),
"completed multipart object must match concatenated parts"
);
}
/// Regression test for issue #2282:
/// CRC64NVME full-object checksum should match between direct PutObject and multipart upload.
#[tokio::test]
@@ -138,4 +138,63 @@ mod tests {
env.stop_server();
}
/// Issue #2475 / Route A: when aws-chunked is combined with an effective object encoding,
/// only the effective encoding should roundtrip through GET/HEAD.
#[tokio::test]
#[serial]
async fn test_content_encoding_aws_chunked_with_effective_encoding_roundtrip() {
init_logging();
info!("aws-chunked,gzip should persist only gzip");
let mut env = RustFSTestEnvironment::new().await.expect("Failed to create test environment");
env.start_rustfs_server(vec![]).await.expect("Failed to start RustFS");
let client = env.create_s3_client();
let bucket = "content-encoding-aws-chunked-gzip-test";
let key = "streamed/object.txt";
let content = b"streaming upload body with effective gzip encoding";
client
.create_bucket()
.bucket(bucket)
.send()
.await
.expect("Failed to create bucket");
client
.put_object()
.bucket(bucket)
.key(key)
.content_type("text/plain")
.content_encoding("aws-chunked,gzip")
.body(ByteStream::from_static(content))
.send()
.await
.expect("PUT failed");
let get_resp = client.get_object().bucket(bucket).key(key).send().await.expect("GET failed");
assert_eq!(
get_resp.content_encoding(),
Some("gzip"),
"GET must return only the effective content encoding after aws-chunked is stripped"
);
let body = get_resp.body.collect().await.unwrap().into_bytes();
assert_eq!(body.as_ref(), content, "Body content mismatch");
let head_resp = client
.head_object()
.bucket(bucket)
.key(key)
.send()
.await
.expect("HEAD failed");
assert_eq!(
head_resp.content_encoding(),
Some("gzip"),
"HEAD must return only the effective content encoding after aws-chunked is stripped"
);
env.stop_server();
}
}
-96
View File
@@ -344,9 +344,6 @@ async fn test_local_kms_multipart_upload() {
test_multipart_upload_with_sse_c(&s3_client, TEST_BUCKET)
.await
.expect("SSE-C multipart upload test failed");
test_multipart_download_with_wrong_sse_c_key_fails(&s3_client, TEST_BUCKET)
.await
.expect("SSE-C multipart wrong-key download test failed");
// Test 4: Large multipart upload (test streaming encryption with multiple blocks)
// TODO: Re-enable after fixing streaming encryption issues with large files
@@ -651,99 +648,6 @@ async fn test_multipart_upload_with_sse_c(
Ok(())
}
async fn test_multipart_download_with_wrong_sse_c_key_fails(
s3_client: &aws_sdk_s3::Client,
bucket: &str,
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
let object_key = "multipart-sse-c-bad-download-test";
let part_size = 5 * 1024 * 1024;
let total_parts = 2;
let total_size = part_size * total_parts;
let encryption_key = "01234567890123456789012345678901";
let key_b64 = base64::Engine::encode(&base64::engine::general_purpose::STANDARD, encryption_key);
let key_md5 = sse_customer_key_md5_base64(encryption_key);
let wrong_key = "abcdefghijklmnopqrstuvwxyz012345";
let wrong_key_b64 = base64::Engine::encode(&base64::engine::general_purpose::STANDARD, wrong_key);
let wrong_key_md5 = sse_customer_key_md5_base64(wrong_key);
let test_data: Vec<u8> = (0..total_size).map(|i| ((i * 5) % 256) as u8).collect();
let create_multipart_output = s3_client
.create_multipart_upload()
.bucket(bucket)
.key(object_key)
.sse_customer_algorithm("AES256")
.sse_customer_key(&key_b64)
.sse_customer_key_md5(&key_md5)
.send()
.await?;
let upload_id = create_multipart_output.upload_id().unwrap();
let mut completed_parts = Vec::new();
for part_number in 1..=total_parts {
let start = (part_number - 1) * part_size;
let end = std::cmp::min(start + part_size, total_size);
let part_data = &test_data[start..end];
let upload_part_output = s3_client
.upload_part()
.bucket(bucket)
.key(object_key)
.upload_id(upload_id)
.part_number(part_number as i32)
.body(aws_sdk_s3::primitives::ByteStream::from(part_data.to_vec()))
.sse_customer_algorithm("AES256")
.sse_customer_key(&key_b64)
.sse_customer_key_md5(&key_md5)
.send()
.await?;
completed_parts.push(
aws_sdk_s3::types::CompletedPart::builder()
.part_number(part_number as i32)
.e_tag(upload_part_output.e_tag().unwrap())
.build(),
);
}
let completed_multipart_upload = aws_sdk_s3::types::CompletedMultipartUpload::builder()
.set_parts(Some(completed_parts))
.build();
s3_client
.complete_multipart_upload()
.bucket(bucket)
.key(object_key)
.upload_id(upload_id)
.multipart_upload(completed_multipart_upload)
.send()
.await?;
let err = s3_client
.get_object()
.bucket(bucket)
.key(object_key)
.sse_customer_algorithm("AES256")
.sse_customer_key(&wrong_key_b64)
.sse_customer_key_md5(&wrong_key_md5)
.send()
.await
.expect_err("multipart SSE-C download with the wrong key should fail");
let service_err = err.into_service_error();
assert_eq!(
service_err.meta().code(),
Some("InvalidRequest"),
"wrong-key multipart SSE-C download should return InvalidRequest, got {:?}",
service_err.meta().code()
);
Ok(())
}
/// Test large multipart upload to verify streaming encryption works correctly
#[allow(dead_code)]
async fn test_large_multipart_upload(
-4
View File
@@ -100,10 +100,6 @@ mod cluster_concurrency_test;
#[cfg(test)]
mod checksum_upload_test;
// Range request regression tests
#[cfg(test)]
mod range_request_test;
// Group deletion tests
#[cfg(test)]
mod group_delete_test;
-80
View File
@@ -1,80 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! End-to-end regression test for invalid GET object ranges.
#[cfg(test)]
mod tests {
use crate::common::{RustFSTestEnvironment, init_logging};
use aws_sdk_s3::Client;
use aws_sdk_s3::error::SdkError;
use aws_sdk_s3::primitives::ByteStream;
use serial_test::serial;
use tracing::info;
fn create_s3_client(env: &RustFSTestEnvironment) -> Client {
env.create_s3_client()
}
async fn create_bucket(client: &Client, bucket: &str) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
match client.create_bucket().bucket(bucket).send().await {
Ok(_) => Ok(()),
Err(err) => {
if err.to_string().contains("BucketAlreadyOwnedByYou") || err.to_string().contains("BucketAlreadyExists") {
Ok(())
} else {
Err(Box::new(err))
}
}
}
}
#[tokio::test]
#[serial]
async fn test_get_object_invalid_range_returns_416_issue_s3_implemented_tests() {
init_logging();
info!("TEST: GetObject invalid range should return InvalidRange/416");
let mut env = RustFSTestEnvironment::new().await.expect("Failed to create test environment");
env.start_rustfs_server(vec![]).await.expect("Failed to start RustFS");
let client = create_s3_client(&env);
let bucket = "test-invalid-range";
let key = "range.txt";
let content = b"testcontent";
create_bucket(&client, bucket).await.expect("Failed to create bucket");
client
.put_object()
.bucket(bucket)
.key(key)
.body(ByteStream::from_static(content))
.send()
.await
.expect("PutObject should succeed");
let result = client.get_object().bucket(bucket).key(key).range("bytes=40-50").send().await;
let err = result.expect_err("GetObject with an unsatisfiable range should fail");
match err {
SdkError::ServiceError(service_err) => {
assert_eq!(service_err.raw().status().as_u16(), 416, "invalid range should return HTTP 416");
let s3_err = service_err.into_err();
assert_eq!(s3_err.meta().code(), Some("InvalidRange"), "invalid range should map to InvalidRange");
}
other_err => panic!("Expected S3 service error, got: {other_err:?}"),
}
}
}
+9 -4
View File
@@ -281,8 +281,11 @@ async fn test_select_object_content_csv_limit() -> Result<(), Box<dyn Error>> {
println!("CSV Limit result: {result_str}");
// Verify only first 2 records are returned
let lines: Vec<&str> = result_str.lines().filter(|line| !line.trim().is_empty()).collect();
assert_eq!(lines.len(), 2, "Should return exactly 2 records");
assert_eq!(
result_str.lines().filter(|line| !line.trim().is_empty()).count(),
2,
"Should return exactly 2 records"
);
Ok(())
}
@@ -321,8 +324,10 @@ async fn test_select_object_content_csv_order_by() -> Result<(), Box<dyn Error>>
println!("CSV Order By result: {result_str}");
// Verify ordered by age descending
let lines: Vec<&str> = result_str.lines().filter(|line| !line.trim().is_empty()).collect();
assert!(lines.len() >= 2, "Should return at least 2 records");
assert!(
result_str.lines().filter(|line| !line.trim().is_empty()).count() >= 2,
"Should return at least 2 records"
);
// Check if contains highest age records
assert!(result_str.contains("Charlie,35"));
-13
View File
@@ -70,7 +70,6 @@ reed-solomon-erasure = { workspace = true }
reed-solomon-simd = { workspace = true }
lazy_static.workspace = true
rustfs-lock.workspace = true
rustfs-io-core.workspace = true
rustfs-io-metrics.workspace = true
regex = { workspace = true }
path-absolutize = { workspace = true }
@@ -140,17 +139,5 @@ harness = false
name = "comparison_benchmark"
harness = false
[[bench]]
name = "direct_chunk_benchmark"
harness = false
[[bench]]
name = "reconstructed_chunk_benchmark"
harness = false
[[bench]]
name = "bitrot_chunk_benchmark"
harness = false
[lib]
doctest = false
-37
View File
@@ -32,43 +32,6 @@
For comprehensive documentation, examples, and usage guides, please visit the main [RustFS repository](https://github.com/rustfs/rustfs).
## 📈 Benchmarks
ECStore ships several Criterion benchmarks under [`crates/ecstore/benches/`](./benches/).
### Direct Chunk Path
Use the direct chunk benchmark to compare the current slice-forwarding path against the previous assembled-copy path:
```bash
cargo bench -p rustfs-ecstore --bench direct_chunk_benchmark
```
To run only the end-to-end ECStore range-read benchmark:
```bash
cargo bench -p rustfs-ecstore --bench direct_chunk_benchmark ecstore_get_object_chunks
```
To run the reconstructed multi-disk range-read benchmark:
```bash
cargo bench -p rustfs-ecstore --bench reconstructed_chunk_benchmark
```
### Saved Comparison Points
Latest local measurements on this branch:
- `direct_chunk_path/slice_forwarding/single_block_aligned`: about `477 ns`
- `direct_chunk_path/assembled_copy/single_block_aligned`: about `3.25 us`
- `direct_chunk_path/slice_forwarding/multi_block_unaligned`: about `963 ns`
- `direct_chunk_path/assembled_copy/multi_block_unaligned`: about `7.25 us`
- `ecstore_get_object_chunks/drain/multi_disk_range`: about `644-654 us`, throughput about `2.86-2.90 GiB/s`
- `reconstructed_chunk_path/drain/multi_disk_missing_shard`: about `1.292-1.304 ms`, throughput about `1.43-1.45 GiB/s`
These numbers are intended as branch-local reference points. Re-run the benchmark on your target machine before treating them as a regression baseline.
## 📄 License
This project is licensed under the Apache License 2.0 - see the [LICENSE](../../LICENSE) file for details.
@@ -1,106 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use bytes::Bytes;
use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_main};
use rustfs_ecstore::bitrot::decode_bitrot_chunk_source_for_bench;
use rustfs_io_core::IoChunk;
use rustfs_utils::HashAlgorithm;
use std::hint::black_box;
struct BitrotChunkBenchCase {
name: &'static str,
source_chunks: Vec<IoChunk>,
shard_size: usize,
expected_decoded_len: usize,
expected_copied: bool,
}
fn encode_shard(checksum_algo: HashAlgorithm, shard: &[u8]) -> Vec<u8> {
let mut encoded = Vec::with_capacity(checksum_algo.size() + shard.len());
encoded.extend_from_slice(checksum_algo.hash_encode(shard).as_ref());
encoded.extend_from_slice(shard);
encoded
}
fn bitrot_chunk_bench_cases() -> [BitrotChunkBenchCase; 2] {
let checksum_algo = HashAlgorithm::Md5;
let shard_one = b"abcd";
let shard_two = b"efgh";
let encoded_one = encode_shard(checksum_algo.clone(), shard_one);
let encoded_two = encode_shard(checksum_algo.clone(), shard_two);
let mut cross_chunk = Vec::with_capacity(encoded_one.len() + encoded_two.len());
cross_chunk.extend_from_slice(&encoded_one);
cross_chunk.extend_from_slice(&encoded_two);
let split = checksum_algo.size() + 2;
[
BitrotChunkBenchCase {
name: "aligned_multi_chunk_no_copy",
source_chunks: vec![
IoChunk::Shared(Bytes::from(encoded_one)),
IoChunk::Shared(Bytes::from(encoded_two)),
],
shard_size: shard_one.len(),
expected_decoded_len: shard_one.len() + shard_two.len(),
expected_copied: false,
},
BitrotChunkBenchCase {
name: "cross_chunk_frame_copy",
source_chunks: vec![
IoChunk::Shared(Bytes::copy_from_slice(&cross_chunk[..split])),
IoChunk::Shared(Bytes::copy_from_slice(&cross_chunk[split..])),
],
shard_size: shard_one.len(),
expected_decoded_len: shard_one.len() + shard_two.len(),
expected_copied: true,
},
]
}
fn bench_bitrot_chunk_decode(c: &mut Criterion) {
let checksum_algo = HashAlgorithm::Md5;
let mut group = c.benchmark_group("bitrot_chunk_decode");
group.sample_size(20);
for case in bitrot_chunk_bench_cases() {
let (decoded, copied) =
decode_bitrot_chunk_source_for_bench(&case.source_chunks, case.shard_size, checksum_algo.clone(), false)
.expect("decode bitrot source");
let decoded_len: usize = decoded.iter().map(IoChunk::len).sum();
assert_eq!(decoded_len, case.expected_decoded_len);
assert_eq!(copied, case.expected_copied);
group.throughput(Throughput::Bytes(case.expected_decoded_len as u64));
group.bench_with_input(BenchmarkId::new("decode", case.name), &case, |b, case| {
b.iter(|| {
let result = decode_bitrot_chunk_source_for_bench(
black_box(&case.source_chunks),
black_box(case.shard_size),
checksum_algo.clone(),
false,
)
.expect("decode bitrot source");
black_box(result);
});
});
}
group.finish();
}
criterion_group!(benches, bench_bitrot_chunk_decode);
criterion_main!(benches);
@@ -1,390 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use bytes::{Bytes, BytesMut};
use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_main};
use futures_util::StreamExt;
use futures_util::stream;
use http::HeaderMap;
use rustfs_ecstore::bucket::metadata_sys;
use rustfs_ecstore::disk::endpoint::Endpoint;
use rustfs_ecstore::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
use rustfs_ecstore::global::{GLOBAL_LOCAL_DISK_ID_MAP, GLOBAL_LOCAL_DISK_MAP, GLOBAL_LOCAL_DISK_SET_DRIVES};
use rustfs_ecstore::set_disk::collect_direct_data_shard_chunks_for_benchmark;
use rustfs_ecstore::store::{ECStore, init_local_disks};
use rustfs_ecstore::store_api::{
BucketOperations, BucketOptions, ChunkNativePutData, GetObjectChunkCopyMode, HTTPRangeSpec, MakeBucketOptions, ObjectIO,
ObjectOptions,
};
use rustfs_io_core::{BoxChunkStream, IoChunk, MappedChunk};
use std::hint::black_box;
use std::net::SocketAddr;
use std::sync::Arc;
use std::sync::atomic::{AtomicU16, Ordering};
use tempfile::TempDir;
use tokio::runtime::Runtime;
use tokio_util::sync::CancellationToken;
#[derive(Clone)]
struct BenchCase {
name: &'static str,
data_shards: usize,
block_size: usize,
blocks: usize,
offset: usize,
length: usize,
}
fn bench_cases() -> [BenchCase; 2] {
[
BenchCase {
name: "single_block_aligned",
data_shards: 4,
block_size: 256 * 1024,
blocks: 1,
offset: 0,
length: 256 * 1024,
},
BenchCase {
name: "multi_block_unaligned",
data_shards: 4,
block_size: 256 * 1024,
blocks: 8,
offset: 123_457,
length: 2 * 256 * 1024 + 33_333,
},
]
}
#[derive(Clone)]
struct EcstoreBenchCase {
name: &'static str,
disk_count: usize,
payload_len: usize,
range: HTTPRangeSpec,
expected_copy_mode: GetObjectChunkCopyMode,
}
struct EcstoreBenchEnv {
_temp_dir: TempDir,
store: Arc<ECStore>,
bucket: String,
key: String,
range: HTTPRangeSpec,
opts: ObjectOptions,
expected_len: usize,
expected_copy_mode: GetObjectChunkCopyMode,
}
fn ecstore_bench_cases() -> [EcstoreBenchCase; 1] {
[EcstoreBenchCase {
name: "multi_disk_range",
disk_count: 4,
payload_len: 3 * 1024 * 1024 + 137,
range: HTTPRangeSpec {
is_suffix_length: false,
start: 123_457,
end: 2 * 1024 * 1024 + 33_333,
},
expected_copy_mode: expected_direct_copy_mode(),
}]
}
#[cfg(unix)]
const fn expected_direct_copy_mode() -> GetObjectChunkCopyMode {
GetObjectChunkCopyMode::TrueZeroCopy
}
#[cfg(not(unix))]
const fn expected_direct_copy_mode() -> GetObjectChunkCopyMode {
GetObjectChunkCopyMode::SharedBytes
}
fn clone_range_spec(range: &HTTPRangeSpec) -> HTTPRangeSpec {
HTTPRangeSpec {
is_suffix_length: range.is_suffix_length,
start: range.start,
end: range.end,
}
}
fn next_loopback_addr() -> SocketAddr {
static NEXT_PORT: AtomicU16 = AtomicU16::new(39013);
let port = NEXT_PORT.fetch_add(1, Ordering::Relaxed);
SocketAddr::from(([127, 0, 0, 1], port))
}
fn build_endpoint_pools(paths: &[std::path::PathBuf]) -> EndpointServerPools {
let mut endpoints = Vec::with_capacity(paths.len());
for (idx, disk_path) in paths.iter().enumerate() {
let mut endpoint = Endpoint::try_from(disk_path.to_str().expect("utf8 path")).expect("endpoint");
endpoint.set_pool_index(0);
endpoint.set_set_index(0);
endpoint.set_disk_index(idx);
endpoints.push(endpoint);
}
EndpointServerPools(vec![PoolEndpoints {
legacy: false,
set_count: 1,
drives_per_set: paths.len(),
endpoints: Endpoints::from(endpoints),
cmd_line: "bench".to_string(),
platform: format!("OS: {} | Arch: {}", std::env::consts::OS, std::env::consts::ARCH),
}])
}
async fn build_ecstore_bench_env(case: &EcstoreBenchCase) -> EcstoreBenchEnv {
let temp_dir = tempfile::tempdir().expect("tempdir");
let mut disk_paths = Vec::with_capacity(case.disk_count);
for idx in 0..case.disk_count {
let path = temp_dir.path().join(format!("disk{}", idx + 1));
tokio::fs::create_dir_all(&path).await.expect("create disk dir");
disk_paths.push(path);
}
let endpoint_pools = build_endpoint_pools(&disk_paths);
GLOBAL_LOCAL_DISK_MAP.write().await.clear();
GLOBAL_LOCAL_DISK_ID_MAP.write().await.clear();
GLOBAL_LOCAL_DISK_SET_DRIVES.write().await.clear();
init_local_disks(endpoint_pools.clone()).await.expect("init local disks");
let store = ECStore::new(next_loopback_addr(), endpoint_pools, CancellationToken::new())
.await
.expect("create ecstore");
let buckets = store
.list_bucket(&BucketOptions {
no_metadata: true,
..Default::default()
})
.await
.expect("list buckets")
.into_iter()
.map(|bucket| bucket.name)
.collect();
metadata_sys::init_bucket_metadata_sys(store.clone(), buckets).await;
let object_id = case.name.replace('_', "-");
let bucket = format!("bench-direct-{object_id}");
let key = format!("objects/{object_id}.bin");
store
.make_bucket(
&bucket,
&MakeBucketOptions {
versioning_enabled: true,
..Default::default()
},
)
.await
.expect("make bucket");
let payload: Vec<u8> = (0..case.payload_len).map(|idx| (idx % 251) as u8).collect();
let mut reader = ChunkNativePutData::from_vec(payload);
let put_info = store
.put_object(&bucket, &key, &mut reader, &ObjectOptions::default())
.await
.expect("put object");
if case.disk_count > 1 {
assert!(put_info.data_blocks > 1, "expected multi-data-shard object");
}
let (_, expected_len) = case.range.get_offset_length(case.payload_len as i64).expect("range length");
EcstoreBenchEnv {
_temp_dir: temp_dir,
store,
bucket,
key,
range: clone_range_spec(&case.range),
opts: ObjectOptions::default(),
expected_len: expected_len as usize,
expected_copy_mode: case.expected_copy_mode,
}
}
async fn run_ecstore_get_object_chunks_bench(env: &EcstoreBenchEnv) -> (GetObjectChunkCopyMode, usize, usize) {
let mut result = env
.store
.get_object_chunks(&env.bucket, &env.key, Some(clone_range_spec(&env.range)), HeaderMap::new(), &env.opts)
.await
.expect("get object chunks");
let copy_mode = result.copy_mode;
let mut total_len = 0usize;
let mut chunk_count = 0usize;
while let Some(chunk) = result.stream.next().await {
let chunk = chunk.expect("chunk");
total_len += chunk.len();
chunk_count += 1;
}
(copy_mode, total_len, chunk_count)
}
fn build_shard_bytes(case: &BenchCase) -> Vec<Vec<Bytes>> {
let total_len = case.block_size * case.blocks;
let payload: Vec<u8> = (0..total_len).map(|idx| (idx % 251) as u8).collect();
let mut shards = vec![Vec::with_capacity(case.blocks); case.data_shards];
for block in 0..case.blocks {
let block_start = block * case.block_size;
let block_slice = &payload[block_start..block_start + case.block_size];
let shard_width = case.block_size / case.data_shards;
for (shard_idx, shard) in shards.iter_mut().enumerate().take(case.data_shards) {
let shard_start = shard_idx * shard_width;
let shard_end = shard_start + shard_width;
shard.push(Bytes::copy_from_slice(&block_slice[shard_start..shard_end]));
}
}
shards
}
fn build_mapped_streams(shards: &[Vec<Bytes>]) -> Vec<BoxChunkStream> {
shards
.iter()
.map(|shard_chunks| {
let chunks: Vec<_> = shard_chunks
.iter()
.map(|chunk| {
let mapped = MappedChunk::new(chunk.clone(), 0, chunk.len()).expect("mapped chunk");
Ok::<IoChunk, std::io::Error>(IoChunk::Mapped(mapped))
})
.collect();
Box::pin(stream::iter(chunks)) as BoxChunkStream
})
.collect()
}
fn collect_old_assembly(
shards: &[Vec<Bytes>],
data_shards: usize,
block_size: usize,
offset: usize,
length: usize,
) -> Vec<IoChunk> {
if length == 0 {
return Vec::new();
}
let start_block = offset / block_size;
let end_block = offset.saturating_add(length - 1) / block_size;
let mut result = Vec::with_capacity(end_block - start_block + 1);
for block_index in start_block..=end_block {
let block_offset = if block_index == start_block { offset % block_size } else { 0 };
let block_length = if start_block == end_block {
length
} else if block_index == start_block {
block_size - (offset % block_size)
} else if block_index == end_block {
(offset + length) % block_size
} else {
block_size
};
if block_length == 0 {
break;
}
let mut block = BytesMut::with_capacity(block_length);
let mut write_left = block_length;
let mut skip = block_offset;
for shard in shards.iter().take(data_shards) {
let shard_chunk = &shard[block_index];
if skip >= shard_chunk.len() {
skip -= shard_chunk.len();
continue;
}
let available = &shard_chunk[skip..];
skip = 0;
let take = available.len().min(write_left);
block.extend_from_slice(&available[..take]);
write_left -= take;
if write_left == 0 {
break;
}
}
result.push(IoChunk::Shared(block.freeze()));
}
result
}
fn bench_direct_chunk_path(c: &mut Criterion) {
let runtime = Runtime::new().expect("tokio runtime");
let mut group = c.benchmark_group("direct_chunk_path");
for case in bench_cases() {
let shard_bytes = build_shard_bytes(&case);
group.throughput(Throughput::Bytes(case.length as u64));
group.bench_with_input(BenchmarkId::new("slice_forwarding", case.name), &case, |b, case| {
b.iter(|| {
let streams = build_mapped_streams(&shard_bytes);
let chunks = runtime
.block_on(collect_direct_data_shard_chunks_for_benchmark(
streams,
case.data_shards,
case.block_size,
case.blocks * case.block_size,
false,
case.offset,
case.length,
))
.expect("collect direct chunks");
black_box(chunks);
});
});
group.bench_with_input(BenchmarkId::new("assembled_copy", case.name), &case, |b, case| {
b.iter(|| {
let chunks = collect_old_assembly(&shard_bytes, case.data_shards, case.block_size, case.offset, case.length);
black_box(chunks);
});
});
}
group.finish();
}
fn bench_ecstore_get_object_chunks(c: &mut Criterion) {
let runtime = Runtime::new().expect("tokio runtime");
let mut group = c.benchmark_group("ecstore_get_object_chunks");
group.sample_size(10);
for case in ecstore_bench_cases() {
let env = runtime.block_on(build_ecstore_bench_env(&case));
let (copy_mode, total_len, _) = runtime.block_on(run_ecstore_get_object_chunks_bench(&env));
assert_eq!(copy_mode, env.expected_copy_mode);
assert_eq!(total_len, env.expected_len);
group.throughput(Throughput::Bytes(env.expected_len as u64));
group.bench_with_input(BenchmarkId::new("drain", case.name), &env, |b, env| {
b.iter(|| {
let result = runtime.block_on(run_ecstore_get_object_chunks_bench(env));
black_box(result);
});
});
}
group.finish();
}
criterion_group!(benches, bench_direct_chunk_path, bench_ecstore_get_object_chunks);
criterion_main!(benches);
@@ -1,393 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use criterion::{BenchmarkId, Criterion, Throughput, criterion_group, criterion_main};
use futures_util::StreamExt;
use http::HeaderMap;
use rustfs_ecstore::bucket::metadata_sys;
use rustfs_ecstore::disk::endpoint::Endpoint;
use rustfs_ecstore::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
use rustfs_ecstore::global::{GLOBAL_LOCAL_DISK_ID_MAP, GLOBAL_LOCAL_DISK_MAP, GLOBAL_LOCAL_DISK_SET_DRIVES};
use rustfs_ecstore::store::{ECStore, init_local_disks};
use rustfs_ecstore::store_api::{
BucketOperations, BucketOptions, ChunkNativePutData, CompletePart, GetObjectChunkCopyMode, HTTPRangeSpec, MakeBucketOptions,
MultipartOperations, ObjectOperations, ObjectOptions,
};
use std::hint::black_box;
use std::net::SocketAddr;
use std::path::{Path, PathBuf};
use std::sync::Arc;
use std::sync::atomic::{AtomicU16, Ordering};
use tempfile::TempDir;
use tokio::runtime::Runtime;
use tokio_util::sync::CancellationToken;
#[derive(Clone)]
enum ReconstructedReadSpec {
PartNumber {
part_number: usize,
missing_part_name: &'static str,
},
Range {
start: u64,
end: u64,
missing_part_name: &'static str,
},
}
#[derive(Clone)]
struct ReconstructedBenchCase {
object_id: &'static str,
name: &'static str,
read_spec: ReconstructedReadSpec,
}
struct ReconstructedBenchEnv {
store: Arc<ECStore>,
bucket: String,
key: String,
range: HTTPRangeSpec,
opts: ObjectOptions,
expected_len: usize,
}
struct ReconstructedBenchSuite {
_temp_dir: TempDir,
store: Arc<ECStore>,
disk_paths: Vec<PathBuf>,
}
const MULTIPART_PART_ONE_LEN: usize = 5 * 1024 * 1024;
const MULTIPART_PART_TWO_LEN: usize = 5 * 1024 * 1024 + 137;
const MULTIPART_PART_THREE_LEN: usize = 1024 * 1024 + 77;
fn reconstructed_bench_cases() -> [ReconstructedBenchCase; 4] {
[
ReconstructedBenchCase {
object_id: "mp-part2",
name: "multi_disk_missing_shard_multipart_part2",
read_spec: ReconstructedReadSpec::PartNumber {
part_number: 2,
missing_part_name: "part.2",
},
},
ReconstructedBenchCase {
object_id: "mp-cross-range",
name: "multi_disk_missing_shard_multipart_cross_part_range",
read_spec: ReconstructedReadSpec::Range {
start: (MULTIPART_PART_ONE_LEN - 32 * 1024) as u64,
end: (MULTIPART_PART_ONE_LEN + 96 * 1024) as u64,
missing_part_name: "part.2",
},
},
ReconstructedBenchCase {
object_id: "mp-part3",
name: "multi_disk_missing_shard_multipart_part3",
read_spec: ReconstructedReadSpec::PartNumber {
part_number: 3,
missing_part_name: "part.3",
},
},
ReconstructedBenchCase {
object_id: "mp-cross-final-range",
name: "multi_disk_missing_shard_multipart_cross_final_part_range",
read_spec: ReconstructedReadSpec::Range {
start: (MULTIPART_PART_ONE_LEN + MULTIPART_PART_TWO_LEN - 32 * 1024) as u64,
end: (MULTIPART_PART_ONE_LEN + MULTIPART_PART_TWO_LEN + 96 * 1024) as u64,
missing_part_name: "part.3",
},
},
]
}
fn next_loopback_addr() -> SocketAddr {
static NEXT_PORT: AtomicU16 = AtomicU16::new(39113);
let port = NEXT_PORT.fetch_add(1, Ordering::Relaxed);
SocketAddr::from(([127, 0, 0, 1], port))
}
fn clone_range_spec(range: &HTTPRangeSpec) -> HTTPRangeSpec {
HTTPRangeSpec {
is_suffix_length: range.is_suffix_length,
start: range.start,
end: range.end,
}
}
fn build_endpoint_pools(paths: &[PathBuf]) -> EndpointServerPools {
let mut endpoints = Vec::with_capacity(paths.len());
for (idx, disk_path) in paths.iter().enumerate() {
let mut endpoint = Endpoint::try_from(disk_path.to_str().expect("utf8 path")).expect("endpoint");
endpoint.set_pool_index(0);
endpoint.set_set_index(0);
endpoint.set_disk_index(idx);
endpoints.push(endpoint);
}
EndpointServerPools(vec![PoolEndpoints {
legacy: false,
set_count: 1,
drives_per_set: paths.len(),
endpoints: Endpoints::from(endpoints),
cmd_line: "bench".to_string(),
platform: format!("OS: {} | Arch: {}", std::env::consts::OS, std::env::consts::ARCH),
}])
}
fn find_part_files(root: &Path, part_name: &str, out: &mut Vec<PathBuf>) {
let Ok(entries) = std::fs::read_dir(root) else {
return;
};
for entry in entries.flatten() {
let path = entry.path();
if path.is_dir() {
find_part_files(&path, part_name, out);
continue;
}
if path.file_name().and_then(|name| name.to_str()) == Some(part_name) {
out.push(path);
}
}
}
async fn remove_part_files(root: &Path, part_name: &str) -> Vec<(PathBuf, Vec<u8>)> {
let mut paths = Vec::new();
find_part_files(root, part_name, &mut paths);
let mut removed = Vec::with_capacity(paths.len());
for path in paths {
let content = tokio::fs::read(&path).await.expect("read part file before removal");
tokio::fs::remove_file(&path).await.expect("remove part file");
removed.push((path, content));
}
removed
}
async fn restore_part_files(files: Vec<(PathBuf, Vec<u8>)>) {
for (path, content) in files {
if let Some(parent) = path.parent() {
tokio::fs::create_dir_all(parent)
.await
.expect("ensure parent for part restore");
}
tokio::fs::write(&path, content).await.expect("restore part file");
}
}
async fn run_reconstructed_get_object_chunks(
store: &Arc<ECStore>,
bucket: &str,
key: &str,
range: &HTTPRangeSpec,
opts: &ObjectOptions,
) -> (GetObjectChunkCopyMode, usize, usize) {
let mut result = store
.get_object_chunks(bucket, key, Some(clone_range_spec(range)), HeaderMap::new(), opts)
.await
.expect("get object chunks");
let copy_mode = result.copy_mode;
let mut total_len = 0usize;
let mut chunk_count = 0usize;
while let Some(chunk) = result.stream.next().await {
let chunk = chunk.expect("chunk");
total_len += chunk.len();
chunk_count += 1;
}
(copy_mode, total_len, chunk_count)
}
async fn create_multipart_object(store: &Arc<ECStore>, bucket: &str, key: &str, parts: &[Vec<u8>]) -> usize {
let upload = store
.new_multipart_upload(bucket, key, &ObjectOptions::default())
.await
.expect("new multipart upload");
let mut completed_parts = Vec::with_capacity(parts.len());
for (idx, part) in parts.iter().enumerate() {
let mut reader = ChunkNativePutData::from_vec(part.clone());
let part_info = store
.put_object_part(bucket, key, &upload.upload_id, idx + 1, &mut reader, &ObjectOptions::default())
.await
.expect("put object part");
completed_parts.push(CompletePart {
part_num: idx + 1,
etag: part_info.etag,
..Default::default()
});
}
store
.clone()
.complete_multipart_upload(bucket, key, &upload.upload_id, completed_parts, &ObjectOptions::default())
.await
.expect("complete multipart upload");
parts.iter().map(Vec::len).sum()
}
async fn build_reconstructed_bench_suite() -> ReconstructedBenchSuite {
let temp_dir = tempfile::tempdir().expect("tempdir");
let disk_paths: Vec<_> = (1..=4).map(|idx| temp_dir.path().join(format!("disk{idx}"))).collect();
for disk_path in &disk_paths {
tokio::fs::create_dir_all(disk_path).await.expect("create disk dir");
}
let endpoint_pools = build_endpoint_pools(&disk_paths);
GLOBAL_LOCAL_DISK_MAP.write().await.clear();
GLOBAL_LOCAL_DISK_ID_MAP.write().await.clear();
GLOBAL_LOCAL_DISK_SET_DRIVES.write().await.clear();
init_local_disks(endpoint_pools.clone()).await.expect("init local disks");
let store = ECStore::new(next_loopback_addr(), endpoint_pools, CancellationToken::new())
.await
.expect("create ecstore");
let buckets = store
.list_bucket(&BucketOptions {
no_metadata: true,
..Default::default()
})
.await
.expect("list buckets")
.into_iter()
.map(|bucket| bucket.name)
.collect();
metadata_sys::init_bucket_metadata_sys(store.clone(), buckets).await;
ReconstructedBenchSuite {
_temp_dir: temp_dir,
store,
disk_paths,
}
}
async fn build_reconstructed_bench_env(suite: &ReconstructedBenchSuite, case: &ReconstructedBenchCase) -> ReconstructedBenchEnv {
let bucket = format!("bench-r-{}", case.object_id);
let key = format!("objects/{}.bin", case.object_id);
suite
.store
.make_bucket(
&bucket,
&MakeBucketOptions {
versioning_enabled: true,
..Default::default()
},
)
.await
.expect("make bucket");
let part_one: Vec<u8> = (0..MULTIPART_PART_ONE_LEN).map(|idx| (idx % 251) as u8).collect();
let part_two: Vec<u8> = (0..MULTIPART_PART_TWO_LEN).map(|idx| ((idx + 11) % 251) as u8).collect();
let part_three: Vec<u8> = (0..MULTIPART_PART_THREE_LEN).map(|idx| ((idx + 29) % 251) as u8).collect();
let payload_len = create_multipart_object(&suite.store, &bucket, &key, &[part_one, part_two, part_three]).await;
let info = suite
.store
.get_object_info(&bucket, &key, &ObjectOptions::default())
.await
.expect("get object info");
let (range, opts, missing_part_name) = match case.read_spec.clone() {
ReconstructedReadSpec::PartNumber {
part_number,
missing_part_name,
} => {
let range = HTTPRangeSpec::from_object_info(&info, part_number).expect("part_number range");
let opts = ObjectOptions {
part_number: Some(part_number),
..Default::default()
};
(range, opts, missing_part_name)
}
ReconstructedReadSpec::Range {
start,
end,
missing_part_name,
} => (
HTTPRangeSpec {
is_suffix_length: false,
start: start as i64,
end: end as i64,
},
ObjectOptions::default(),
missing_part_name,
),
};
let (_, expected_len) = range.get_offset_length(payload_len as i64).expect("range length");
let mut selected_reconstructed = false;
for disk_path in &suite.disk_paths {
let object_root = disk_path
.join(&bucket)
.join("objects")
.join(format!("{}.bin", case.object_id));
let removed_parts = remove_part_files(&object_root, missing_part_name).await;
if removed_parts.is_empty() {
continue;
}
let (copy_mode, total_len, _) = run_reconstructed_get_object_chunks(&suite.store, &bucket, &key, &range, &opts).await;
if copy_mode == GetObjectChunkCopyMode::Reconstructed && total_len == expected_len as usize {
selected_reconstructed = true;
break;
}
restore_part_files(removed_parts).await;
}
assert!(
selected_reconstructed,
"failed to select a missing shard placement that triggers reconstructed copy mode for benchmark case {}",
case.name
);
ReconstructedBenchEnv {
store: suite.store.clone(),
bucket,
key,
range: clone_range_spec(&range),
opts,
expected_len: expected_len as usize,
}
}
async fn run_reconstructed_get_object_chunks_bench(env: &ReconstructedBenchEnv) -> (GetObjectChunkCopyMode, usize, usize) {
run_reconstructed_get_object_chunks(&env.store, &env.bucket, &env.key, &env.range, &env.opts).await
}
fn bench_reconstructed_chunk_path(c: &mut Criterion) {
let runtime = Runtime::new().expect("tokio runtime");
let suite = runtime.block_on(build_reconstructed_bench_suite());
let mut group = c.benchmark_group("reconstructed_chunk_path");
group.sample_size(10);
for case in reconstructed_bench_cases() {
let env = runtime.block_on(build_reconstructed_bench_env(&suite, &case));
let (copy_mode, total_len, _) = runtime.block_on(run_reconstructed_get_object_chunks_bench(&env));
assert_eq!(copy_mode, GetObjectChunkCopyMode::Reconstructed);
assert_eq!(total_len, env.expected_len);
group.throughput(Throughput::Bytes(env.expected_len as u64));
group.bench_with_input(BenchmarkId::new("drain", case.name), &env, |b, env| {
b.iter(|| {
let result = runtime.block_on(run_reconstructed_get_object_chunks_bench(env));
black_box(result);
});
});
}
group.finish();
}
criterion_group!(benches, bench_reconstructed_chunk_path);
criterion_main!(benches);
-17
View File
@@ -119,16 +119,6 @@ run_large_data_test() {
print_success "Large-dataset tests completed"
}
# Run direct chunk path benchmarks
run_direct_chunk_benchmark() {
print_info "📦 Starting direct chunk path benchmarks..."
echo "================================================"
cargo bench --bench direct_chunk_benchmark
print_success "Direct chunk path benchmarks completed"
}
# Generate comparison report
generate_comparison_report() {
print_info "📊 Generating performance report..."
@@ -178,7 +168,6 @@ show_help() {
echo " full Run the full benchmark suite"
echo " performance Run detailed performance tests"
echo " simd Run the SIMD-only tests"
echo " direct Run the direct chunk path benchmarks"
echo " large Run large-dataset tests"
echo " clean Remove previous results"
echo " help Show this help message"
@@ -188,7 +177,6 @@ show_help() {
echo " $0 performance # Detailed performance test"
echo " $0 full # Full benchmark suite"
echo " $0 simd # SIMD-only benchmark"
echo " $0 direct # Direct chunk path benchmark"
echo " $0 large # Large-dataset benchmark"
echo ""
echo "Features:"
@@ -252,11 +240,6 @@ main() {
run_simd_benchmark
generate_comparison_report
;;
"direct")
cleanup
run_direct_chunk_benchmark
generate_comparison_report
;;
"large")
cleanup
run_large_data_test
+11 -781
View File
@@ -14,328 +14,13 @@
use crate::disk::{self, DiskAPI as _, DiskStore, error::DiskError};
use crate::erasure_coding::{BitrotReader, BitrotWriterWrapper, CustomWriter};
use crate::store_api::{GetObjectChunkCopyMode, GetObjectChunkPath, GetObjectChunkResult};
use bytes::{Bytes, BytesMut};
use futures_util::{StreamExt, stream};
use rustfs_io_core::{BoxChunkStream, IoChunk};
use bytes::Bytes;
use rustfs_utils::HashAlgorithm;
use std::collections::VecDeque;
use std::io::Cursor;
use std::time::Instant;
use tokio::io::AsyncRead;
use tracing::debug;
const BITROT_READ_OPERATION: &str = "bitrot_read";
fn classify_chunk_copy_mode(source_direct: bool, copied: bool) -> GetObjectChunkCopyMode {
if copied {
GetObjectChunkCopyMode::SingleCopy
} else if source_direct {
GetObjectChunkCopyMode::TrueZeroCopy
} else {
GetObjectChunkCopyMode::SharedBytes
}
}
struct ChunkSpan {
bytes: Bytes,
chunk: IoChunk,
copied: bool,
}
fn take_contiguous_chunk_span(chunk: &IoChunk, offset: usize, len: usize) -> std::io::Result<ChunkSpan> {
match chunk {
IoChunk::Shared(bytes) => {
let bytes = bytes.slice(offset..offset + len);
Ok(ChunkSpan {
bytes: bytes.clone(),
chunk: IoChunk::Shared(bytes),
copied: false,
})
}
IoChunk::Mapped(mapped) => {
let chunk = IoChunk::Mapped(mapped.slice(offset, len)?);
let bytes = chunk.as_bytes();
Ok(ChunkSpan {
bytes,
chunk,
copied: false,
})
}
IoChunk::Pooled(pooled) => {
let chunk = IoChunk::Pooled(pooled.slice(offset, len)?);
let bytes = chunk.as_bytes();
Ok(ChunkSpan {
bytes,
chunk,
copied: false,
})
}
}
}
struct BitrotChunkSource {
source_stream: BoxChunkStream,
source_chunks: VecDeque<IoChunk>,
source_chunk_offset: usize,
source_buffered_bytes: usize,
source_done: bool,
}
struct BitrotChunkStreamState {
source: BitrotChunkSource,
decoded_remaining: usize,
trim_prefix: usize,
output_remaining: usize,
shard_size: usize,
checksum_algo: HashAlgorithm,
skip_verify: bool,
}
struct ChunkCursor<'a> {
chunks: &'a [IoChunk],
chunk_index: usize,
chunk_offset: usize,
consumed: usize,
total_len: usize,
}
impl<'a> ChunkCursor<'a> {
fn new(chunks: &'a [IoChunk]) -> Self {
Self {
chunks,
chunk_index: 0,
chunk_offset: 0,
consumed: 0,
total_len: chunks.iter().map(IoChunk::len).sum(),
}
}
fn remaining(&self) -> usize {
self.total_len.saturating_sub(self.consumed)
}
fn skip_empty_chunks(&mut self) {
while let Some(chunk) = self.chunks.get(self.chunk_index) {
if self.chunk_offset < chunk.len() {
break;
}
self.chunk_index += 1;
self.chunk_offset = 0;
}
}
fn advance(&mut self, len: usize) {
self.consumed += len;
self.chunk_offset += len;
self.skip_empty_chunks();
}
fn take_span(&mut self, len: usize) -> std::io::Result<ChunkSpan> {
self.skip_empty_chunks();
if self.remaining() < len {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "truncated bitrot chunk source"));
}
let Some(chunk) = self.chunks.get(self.chunk_index) else {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "missing bitrot chunk source"));
};
let available = chunk.len().saturating_sub(self.chunk_offset);
if len <= available {
let span = take_contiguous_chunk_span(chunk, self.chunk_offset, len)?;
self.advance(len);
return Ok(span);
}
let mut aggregate = BytesMut::with_capacity(len);
let mut remaining = len;
while remaining > 0 {
self.skip_empty_chunks();
let Some(chunk) = self.chunks.get(self.chunk_index) else {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "truncated bitrot chunk source"));
};
let available = chunk.len().saturating_sub(self.chunk_offset);
let take = available.min(remaining);
aggregate.extend_from_slice(&chunk.as_bytes()[self.chunk_offset..self.chunk_offset + take]);
self.advance(take);
remaining -= take;
}
let bytes = aggregate.freeze();
Ok(ChunkSpan {
bytes: bytes.clone(),
chunk: IoChunk::Shared(bytes),
copied: true,
})
}
}
impl BitrotChunkSource {
fn new(source_stream: BoxChunkStream, source_chunks: VecDeque<IoChunk>, source_done: bool) -> Self {
let source_buffered_bytes = source_chunks.iter().map(IoChunk::len).sum();
Self {
source_stream,
source_chunks,
source_chunk_offset: 0,
source_buffered_bytes,
source_done,
}
}
fn skip_empty_chunks(&mut self) {
while let Some(chunk) = self.source_chunks.front() {
if self.source_chunk_offset < chunk.len() {
break;
}
self.source_chunks.pop_front();
self.source_chunk_offset = 0;
}
}
async fn fill(&mut self, min_bytes: usize) -> std::io::Result<()> {
while self.source_buffered_bytes < min_bytes && !self.source_done {
match self.source_stream.next().await {
Some(Ok(chunk)) => {
self.source_buffered_bytes += chunk.len();
self.source_chunks.push_back(chunk);
}
Some(Err(err)) => return Err(err),
None => self.source_done = true,
}
}
if self.source_buffered_bytes < min_bytes {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "truncated bitrot chunk source"));
}
Ok(())
}
fn advance(&mut self, len: usize) {
self.source_buffered_bytes = self.source_buffered_bytes.saturating_sub(len);
self.source_chunk_offset += len;
self.skip_empty_chunks();
}
fn take_span(&mut self, len: usize) -> std::io::Result<ChunkSpan> {
self.skip_empty_chunks();
if self.source_buffered_bytes < len {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "truncated bitrot chunk source"));
}
let Some(chunk) = self.source_chunks.front() else {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "missing bitrot chunk source"));
};
let available = chunk.len().saturating_sub(self.source_chunk_offset);
if len <= available {
let span = take_contiguous_chunk_span(chunk, self.source_chunk_offset, len)?;
self.advance(len);
return Ok(span);
}
let mut aggregate = BytesMut::with_capacity(len);
let mut remaining = len;
while remaining > 0 {
self.skip_empty_chunks();
let Some(chunk) = self.source_chunks.front() else {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "truncated bitrot chunk source"));
};
let available = chunk.len().saturating_sub(self.source_chunk_offset);
let take = available.min(remaining);
aggregate.extend_from_slice(&chunk.as_bytes()[self.source_chunk_offset..self.source_chunk_offset + take]);
self.advance(take);
remaining -= take;
}
let bytes = aggregate.freeze();
Ok(ChunkSpan {
bytes: bytes.clone(),
chunk: IoChunk::Shared(bytes),
copied: true,
})
}
}
impl BitrotChunkStreamState {
#[allow(clippy::too_many_arguments)]
fn new(
source_stream: BoxChunkStream,
source_chunks: VecDeque<IoChunk>,
source_done: bool,
decoded_remaining: usize,
trim_prefix: usize,
output_remaining: usize,
shard_size: usize,
checksum_algo: HashAlgorithm,
skip_verify: bool,
) -> Self {
Self {
source: BitrotChunkSource::new(source_stream, source_chunks, source_done),
decoded_remaining,
trim_prefix,
output_remaining,
shard_size,
checksum_algo,
skip_verify,
}
}
fn hash_size(&self) -> usize {
self.checksum_algo.size()
}
async fn next_verified_chunk(&mut self) -> std::io::Result<Option<IoChunk>> {
let hash_size = self.hash_size();
while self.output_remaining > 0 && self.decoded_remaining > 0 {
let data_len = self.shard_size.min(self.decoded_remaining);
let expected_hash = if hash_size > 0 {
self.source.fill(hash_size).await?;
Some(self.source.take_span(hash_size)?)
} else {
None
};
self.source.fill(data_len).await?;
let data_span = self.source.take_span(data_len)?;
if let Some(expected_hash) = expected_hash
&& !self.skip_verify
&& self.checksum_algo.hash_encode(data_span.bytes.as_ref()).as_ref() != expected_hash.bytes.as_ref()
{
return Err(std::io::Error::new(std::io::ErrorKind::InvalidData, "bitrot hash mismatch"));
}
self.decoded_remaining -= data_len;
if self.trim_prefix >= data_len {
self.trim_prefix -= data_len;
continue;
}
let start = self.trim_prefix;
self.trim_prefix = 0;
let take = (data_len - start).min(self.output_remaining);
self.output_remaining -= take;
let chunk = if start == 0 && take == data_len {
data_span.chunk
} else {
data_span.chunk.slice(start, take)?
};
if !chunk.is_empty() {
return Ok(Some(chunk));
}
}
Ok(None)
}
}
/// Create a BitrotReader from either inline data or disk file stream
///
/// # Parameters
@@ -379,40 +64,21 @@ pub async fn create_bitrot_reader(
Ok(Some(reader))
} else if let Some(disk) = disk {
// Read from disk
if use_zero_copy {
if !disk.is_local() {
rustfs_io_metrics::record_io_path_selected(BITROT_READ_OPERATION, rustfs_io_metrics::IoPath::Legacy);
rustfs_io_metrics::record_io_fallback(
rustfs_io_metrics::IoStage::ReadSetup,
rustfs_io_metrics::FallbackReason::NonLocalBackend,
);
let rd = disk.read_file_stream(bucket, path, offset, length).await?;
let reader = BitrotReader::new(rd, shard_size, checksum_algo, skip_verify);
return Ok(Some(reader));
}
if use_zero_copy && disk.is_local() {
// Try zero-copy read first (uses mmap on Unix)
let start = Instant::now();
match disk.read_file_zero_copy(bucket, path, offset, length).await {
Ok(bytes) => {
let duration_ms = start.elapsed().as_secs_f64() * 1000.0;
rustfs_io_metrics::record_io_path_selected(BITROT_READ_OPERATION, rustfs_io_metrics::IoPath::Fast);
// `read_file_zero_copy()` returns a shared `Bytes` view, but it may still
// internally aggregate multiple chunk windows. The exact chunk-native copy
// mode is only preserved by `create_bitrot_chunk_stream()`.
rustfs_io_metrics::record_io_copy_mode(
BITROT_READ_OPERATION,
rustfs_io_metrics::CopyMode::SharedBytes,
bytes.len(),
);
// Record zero-copy metrics
rustfs_io_metrics::record_zero_copy_read(bytes.len(), duration_ms);
// Log successful zero-copy read
debug!(
size = bytes.len(),
duration_ms,
path = %path,
"bitrot_fast_read_success"
"zero_copy_read_success"
);
// Wrap Bytes in Cursor for AsyncRead
@@ -427,16 +93,14 @@ pub async fn create_bitrot_reader(
Ok(Some(reader))
}
Err(e) => {
rustfs_io_metrics::record_io_path_selected(BITROT_READ_OPERATION, rustfs_io_metrics::IoPath::Legacy);
rustfs_io_metrics::record_io_fallback(
rustfs_io_metrics::IoStage::ReadSetup,
rustfs_io_metrics::FallbackReason::Unknown,
);
// Record zero-copy fallback
rustfs_io_metrics::record_zero_copy_fallback(&format!("{:?}", e));
// Log zero-copy fallback
debug!(
reason = %e,
reason = %format!("{:?}", e),
path = %path,
"bitrot_fast_read_fallback"
"zero_copy_fallback"
);
// Fall back to regular stream read on error
@@ -453,7 +117,6 @@ pub async fn create_bitrot_reader(
}
}
} else {
rustfs_io_metrics::record_io_path_selected(BITROT_READ_OPERATION, rustfs_io_metrics::IoPath::Legacy);
// Use regular stream read
match disk.read_file_stream(bucket, path, offset, length).await {
Ok(rd) => {
@@ -469,198 +132,6 @@ pub async fn create_bitrot_reader(
}
}
/// Create a chunk stream from bitrot-encoded data, preserving source chunk provenance when possible.
#[allow(clippy::too_many_arguments)]
pub async fn create_bitrot_chunk_stream(
inline_data: Option<&[u8]>,
disk: Option<&DiskStore>,
bucket: &str,
path: &str,
offset: usize,
length: usize,
total_data_size: usize,
shard_size: usize,
checksum_algo: HashAlgorithm,
skip_verify: bool,
use_zero_copy: bool,
) -> disk::error::Result<Option<GetObjectChunkResult>> {
let fetch_start = (offset / shard_size) * shard_size;
let fetch_end = (offset + length).div_ceil(shard_size) * shard_size;
let fetch_end = fetch_end.min(total_data_size);
let fetch_length = fetch_end.saturating_sub(fetch_start);
let trim_prefix = offset.saturating_sub(fetch_start);
let hash_size = checksum_algo.size();
let encoded_length = fetch_length.div_ceil(shard_size) * hash_size + fetch_length;
let encoded_offset = fetch_start.div_ceil(shard_size) * hash_size + fetch_start;
let mut source_done = false;
let (source_stream, mut prefetched_chunks, source_direct) = if let Some(data) = inline_data {
source_done = true;
let mut chunks = VecDeque::new();
chunks.push_back(IoChunk::Shared(
Bytes::copy_from_slice(data).slice(encoded_offset..encoded_offset + encoded_length),
));
let source_stream: BoxChunkStream = Box::pin(stream::empty::<std::io::Result<IoChunk>>());
(source_stream, chunks, false)
} else if let Some(disk) = disk {
if use_zero_copy {
let mut source_stream = disk.read_file_chunks(bucket, path, encoded_offset, encoded_length).await?;
let mut prefetched_chunks = VecDeque::new();
let mut direct = true;
while prefetched_chunks.len() < 2 {
let Some(chunk) = source_stream.next().await else {
source_done = true;
break;
};
let chunk = chunk?;
direct &= matches!(chunk, IoChunk::Mapped(_));
prefetched_chunks.push_back(chunk);
}
(source_stream, prefetched_chunks, direct)
} else {
source_done = true;
let bytes = disk.read_file_zero_copy(bucket, path, encoded_offset, encoded_length).await?;
let mut chunks = VecDeque::new();
chunks.push_back(IoChunk::Shared(bytes));
let source_stream: BoxChunkStream = Box::pin(stream::empty::<std::io::Result<IoChunk>>());
(source_stream, chunks, false)
}
} else {
return Ok(None);
};
let copied = predicted_stream_copy(encoded_length, shard_size, checksum_algo.size(), &prefetched_chunks, source_done);
let state = BitrotChunkStreamState::new(
source_stream,
std::mem::take(&mut prefetched_chunks),
source_done,
fetch_length,
trim_prefix,
length,
shard_size,
checksum_algo,
skip_verify,
);
let stream = stream::unfold(Some(state), |state| async move {
let mut state = match state {
Some(state) => state,
None => return None,
};
match state.next_verified_chunk().await {
Ok(Some(chunk)) => {
let next_state = if state.output_remaining == 0 { None } else { Some(state) };
Some((Ok::<IoChunk, std::io::Error>(chunk), next_state))
}
Ok(None) => None,
Err(err) => Some((Err(err), None)),
}
});
Ok(Some(GetObjectChunkResult {
stream: Box::pin(stream),
path: GetObjectChunkPath::Direct,
copy_mode: classify_chunk_copy_mode(source_direct, copied),
}))
}
fn predicted_stream_copy(
encoded_length: usize,
shard_size: usize,
hash_size: usize,
prefetched_chunks: &VecDeque<IoChunk>,
source_done: bool,
) -> bool {
if prefetched_chunks.is_empty() {
return false;
}
if source_done && prefetched_chunks.len() == 1 {
return false;
}
let full_frame_len = hash_size + shard_size;
if full_frame_len == 0 {
return false;
}
let first_window_len = prefetched_chunks.front().map(IoChunk::len).unwrap_or(encoded_length);
encoded_length > first_window_len && !first_window_len.is_multiple_of(full_frame_len)
}
fn trim_chunk_vec(chunks: Vec<IoChunk>, offset: usize, length: usize) -> std::io::Result<Vec<IoChunk>> {
let mut skip = offset;
let mut remaining = length;
let mut result = Vec::new();
for chunk in chunks {
if remaining == 0 {
break;
}
let chunk_len = chunk.len();
if skip >= chunk_len {
skip -= chunk_len;
continue;
}
let start = skip;
let take = (chunk_len - start).min(remaining);
result.push(chunk.slice(start, take)?);
remaining -= take;
skip = 0;
}
Ok(result)
}
fn decode_bitrot_chunk_source(
source_chunks: &[IoChunk],
shard_size: usize,
checksum_algo: HashAlgorithm,
skip_verify: bool,
) -> std::io::Result<(Vec<IoChunk>, bool)> {
let hash_size = checksum_algo.size();
let mut cursor = ChunkCursor::new(source_chunks);
let mut result = Vec::new();
let mut copied = false;
while cursor.remaining() > 0 {
let expected_hash = if hash_size > 0 {
Some(cursor.take_span(hash_size)?)
} else {
None
};
let data_len = shard_size.min(cursor.remaining());
if data_len == 0 {
break;
}
let data_span = cursor.take_span(data_len)?;
copied |= data_span.copied;
if let Some(expected_hash) = expected_hash {
copied |= expected_hash.copied;
if !skip_verify && checksum_algo.hash_encode(data_span.bytes.as_ref()).as_ref() != expected_hash.bytes.as_ref() {
return Err(std::io::Error::new(std::io::ErrorKind::InvalidData, "bitrot hash mismatch"));
}
}
result.push(data_span.chunk);
}
Ok((result, copied))
}
#[doc(hidden)]
pub fn decode_bitrot_chunk_source_for_bench(
source_chunks: &[IoChunk],
shard_size: usize,
checksum_algo: HashAlgorithm,
skip_verify: bool,
) -> std::io::Result<(Vec<IoChunk>, bool)> {
decode_bitrot_chunk_source(source_chunks, shard_size, checksum_algo, skip_verify)
}
/// Create a new BitrotWriterWrapper based on the provided parameters
///
/// # Parameters
@@ -705,7 +176,6 @@ pub async fn create_bitrot_writer(
#[cfg(test)]
mod tests {
use super::*;
use futures_util::StreamExt;
#[tokio::test]
async fn test_create_bitrot_reader_with_inline_data() {
@@ -756,246 +226,6 @@ mod tests {
assert!(result.unwrap().is_some());
}
#[tokio::test]
async fn test_create_bitrot_chunk_stream_with_inline_data() {
let shard_size = 4;
let checksum_algo = HashAlgorithm::HighwayHash256S;
let shard1 = b"abcd";
let shard2 = b"ef";
let mut encoded = Vec::new();
encoded.extend_from_slice(checksum_algo.hash_encode(shard1).as_ref());
encoded.extend_from_slice(shard1);
encoded.extend_from_slice(checksum_algo.hash_encode(shard2).as_ref());
encoded.extend_from_slice(shard2);
let mut stream = create_bitrot_chunk_stream(
Some(&encoded),
None,
"test-bucket",
"test-path",
0,
shard1.len() + shard2.len(),
shard1.len() + shard2.len(),
shard_size,
checksum_algo,
false,
false,
)
.await
.unwrap()
.unwrap()
.stream;
let mut collected = Vec::new();
while let Some(chunk) = stream.next().await {
collected.extend_from_slice(&chunk.unwrap().as_bytes());
}
assert_eq!(collected, b"abcdef");
}
#[tokio::test]
async fn test_create_bitrot_chunk_stream_detects_hash_mismatch() {
let shard_size = 4;
let checksum_algo = HashAlgorithm::HighwayHash256S;
let shard = b"abcd";
let mut encoded = Vec::new();
let mut bad_hash = checksum_algo.hash_encode(shard).as_ref().to_vec();
bad_hash[0] ^= 0xFF;
encoded.extend_from_slice(&bad_hash);
encoded.extend_from_slice(shard);
let result = create_bitrot_chunk_stream(
Some(&encoded),
None,
"test-bucket",
"test-path",
0,
shard.len(),
shard.len(),
shard_size,
checksum_algo,
false,
false,
)
.await;
let mut stream = result.unwrap().unwrap().stream;
let err = stream.next().await.unwrap().unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert!(err.to_string().contains("bitrot hash mismatch"));
}
#[tokio::test]
async fn test_create_bitrot_chunk_stream_trims_range_after_decode() {
let shard_size = 4;
let checksum_algo = HashAlgorithm::HighwayHash256S;
let shard1 = b"abcd";
let shard2 = b"efgh";
let mut encoded = Vec::new();
encoded.extend_from_slice(checksum_algo.hash_encode(shard1).as_ref());
encoded.extend_from_slice(shard1);
encoded.extend_from_slice(checksum_algo.hash_encode(shard2).as_ref());
encoded.extend_from_slice(shard2);
let mut stream = create_bitrot_chunk_stream(
Some(&encoded),
None,
"test-bucket",
"test-path",
1,
5,
shard1.len() + shard2.len(),
shard_size,
checksum_algo,
false,
false,
)
.await
.unwrap()
.unwrap()
.stream;
let mut collected = Vec::new();
while let Some(chunk) = stream.next().await {
collected.extend_from_slice(&chunk.unwrap().as_bytes());
}
assert_eq!(collected, b"bcdef");
}
#[test]
fn test_decode_bitrot_chunk_source_preserves_aligned_multi_chunk_slices() {
let shard_size = 4;
let checksum_algo = HashAlgorithm::Md5;
let shard1 = b"abcd";
let shard2 = b"efgh";
let mut encoded_chunk_one = Vec::new();
encoded_chunk_one.extend_from_slice(checksum_algo.hash_encode(shard1).as_ref());
encoded_chunk_one.extend_from_slice(shard1);
let mut encoded_chunk_two = Vec::new();
encoded_chunk_two.extend_from_slice(checksum_algo.hash_encode(shard2).as_ref());
encoded_chunk_two.extend_from_slice(shard2);
let source_chunks = vec![
IoChunk::Shared(Bytes::from(encoded_chunk_one)),
IoChunk::Shared(Bytes::from(encoded_chunk_two)),
];
let (decoded, copied) = decode_bitrot_chunk_source(&source_chunks, shard_size, checksum_algo, false).unwrap();
assert!(!copied, "frame-aligned multi-chunk source should not require aggregate copies");
assert_eq!(decoded.len(), 2);
assert_eq!(decoded[0].as_bytes(), Bytes::from_static(b"abcd"));
assert_eq!(decoded[1].as_bytes(), Bytes::from_static(b"efgh"));
}
#[test]
fn test_decode_bitrot_chunk_source_marks_cross_chunk_frame_as_copied() {
let shard_size = 4;
let checksum_algo = HashAlgorithm::Md5;
let shard1 = b"abcd";
let shard2 = b"efgh";
let hash1 = checksum_algo.hash_encode(shard1).as_ref().to_vec();
let hash2 = checksum_algo.hash_encode(shard2).as_ref().to_vec();
let mut encoded = Vec::new();
encoded.extend_from_slice(&hash1);
encoded.extend_from_slice(shard1);
encoded.extend_from_slice(&hash2);
encoded.extend_from_slice(shard2);
let split = hash1.len() + 2;
let source_chunks = vec![
IoChunk::Shared(Bytes::copy_from_slice(&encoded[..split])),
IoChunk::Shared(Bytes::copy_from_slice(&encoded[split..])),
];
let (decoded, copied) = decode_bitrot_chunk_source(&source_chunks, shard_size, checksum_algo, false).unwrap();
assert!(copied, "cross-chunk frame should be classified as requiring a copy");
assert_eq!(decoded.len(), 2);
assert_eq!(decoded[0].as_bytes(), Bytes::from_static(b"abcd"));
assert_eq!(decoded[1].as_bytes(), Bytes::from_static(b"efgh"));
}
#[test]
fn test_decode_bitrot_chunk_source_preserves_pooled_single_chunk_slice() {
let shard_size = 4;
let checksum_algo = HashAlgorithm::Md5;
let shard = b"abcd";
let mut encoded = Vec::new();
encoded.extend_from_slice(checksum_algo.hash_encode(shard).as_ref());
encoded.extend_from_slice(shard);
let source_chunks = vec![IoChunk::Pooled(rustfs_io_core::PooledChunk::from_vec(encoded))];
let (decoded, copied) = decode_bitrot_chunk_source(&source_chunks, shard_size, checksum_algo, false).unwrap();
assert!(!copied, "single pooled chunk slice should preserve provenance without copy");
assert_eq!(decoded.len(), 1);
assert!(matches!(&decoded[0], IoChunk::Pooled(_)));
assert_eq!(decoded[0].as_bytes(), Bytes::from_static(b"abcd"));
}
#[tokio::test]
async fn test_bitrot_chunk_source_marks_cross_chunk_take_as_copied() {
let source_stream: BoxChunkStream = Box::pin(stream::iter(vec![
Ok(IoChunk::Shared(Bytes::from_static(b"ab"))),
Ok(IoChunk::Shared(Bytes::from_static(b"cd"))),
]));
let mut source = BitrotChunkSource::new(source_stream, VecDeque::new(), false);
source.fill(4).await.expect("source fill should succeed");
let span = source.take_span(4).expect("cross-chunk take should succeed");
assert!(span.copied, "cross-chunk take should be classified as copied");
assert_eq!(span.bytes, Bytes::from_static(b"abcd"));
assert_eq!(span.chunk.as_bytes(), Bytes::from_static(b"abcd"));
}
#[tokio::test]
async fn test_bitrot_chunk_stream_state_yields_verified_prefix_before_later_truncation() {
let shard_size = 4;
let checksum_algo = HashAlgorithm::Md5;
let shard1 = b"abcd";
let shard2 = b"efgh";
let mut first_frame = Vec::new();
first_frame.extend_from_slice(checksum_algo.hash_encode(shard1).as_ref());
first_frame.extend_from_slice(shard1);
let mut second_frame_prefix = Vec::new();
second_frame_prefix.extend_from_slice(checksum_algo.hash_encode(shard2).as_ref());
second_frame_prefix.extend_from_slice(&shard2[..2]);
let source_stream: BoxChunkStream = Box::pin(stream::iter(vec![
Ok(IoChunk::Shared(Bytes::from(first_frame))),
Ok(IoChunk::Shared(Bytes::from(second_frame_prefix))),
]));
let mut state = BitrotChunkStreamState::new(
source_stream,
VecDeque::new(),
false,
shard1.len() + shard2.len(),
0,
shard1.len() + shard2.len(),
shard_size,
checksum_algo,
false,
);
let first = state.next_verified_chunk().await.unwrap().unwrap();
assert_eq!(first.as_bytes(), Bytes::from_static(b"abcd"));
let err = state.next_verified_chunk().await.unwrap_err();
assert_eq!(err.kind(), std::io::ErrorKind::UnexpectedEof);
assert!(err.to_string().contains("truncated bitrot chunk source"));
}
#[tokio::test]
async fn test_create_bitrot_reader_with_inline_offset_starts_at_requested_shard() {
let shard_size = 4;
@@ -593,7 +593,7 @@ impl BucketTargetSys {
};
if let Some(cli) = cli {
return Some(cli.clone());
return Some(cli);
}
// TODO: spawn a task to reload the target
@@ -821,7 +821,7 @@ pub async fn abort_incomplete_multipart_upload_due(
.as_ref()?
.days_after_initiation
.filter(|days| *days > 0)?;
Some((expected_expiry_time(initiated, days), rule.id.clone().unwrap_or_default()))
Some((expected_expiry_time(initiated, days), rule.id.unwrap_or_default()))
})
.min_by_key(|(due, _)| due.unix_timestamp_nanos())
}
@@ -83,7 +83,7 @@ impl LastDayTierStats {
#[allow(dead_code)]
fn merge(&self, m: LastDayTierStats) -> LastDayTierStats {
let mut cl = self.clone();
let mut cm = m.clone();
let mut cm = m;
let mut merged = LastDayTierStats::default();
if cl.updated_at.unix_timestamp() > cm.updated_at.unix_timestamp() {
+1 -1
View File
@@ -678,7 +678,7 @@ impl BucketMetadata {
// let x = data.clone();
// let str = std::str::from_utf8(&x).expect("Invalid UTF-8");
// println!("update config:{}", str);
self.bucket_targets_config_json = data.clone();
self.bucket_targets_config_json = data;
self.bucket_targets_config_updated_at = updated;
}
BUCKET_CORS_CONFIG => {
+3 -3
View File
@@ -17,7 +17,7 @@
use crate::bucket::metadata::BUCKET_METADATA_FILE;
use crate::bucket::replication::{decode_resync_file, encode_resync_file};
use crate::disk::{BUCKET_META_PREFIX, MIGRATING_META_BUCKET, RUSTFS_META_BUCKET};
use crate::store_api::{BucketOptions, ChunkNativePutData, ObjectOptions, StorageAPI};
use crate::store_api::{BucketOptions, ObjectOptions, PutObjReader, StorageAPI};
use http::HeaderMap;
use rustfs_policy::auth::UserIdentity;
use rustfs_policy::policy::PolicyDoc;
@@ -263,7 +263,7 @@ async fn migrate_one_if_missing<S: StorageAPI>(
}
};
let mut put_data = ChunkNativePutData::from_vec(data);
let mut put_data = PutObjReader::from_vec(data);
if let Err(e) = store.put_object(RUSTFS_META_BUCKET, path, &mut put_data, opts).await {
warn!("write {label}: {e}");
} else {
@@ -341,7 +341,7 @@ pub async fn try_migrate_iam_config<S: StorageAPI>(store: Arc<S>) {
continue;
}
};
let mut put_data = ChunkNativePutData::from_vec(data);
let mut put_data = PutObjReader::from_vec(data);
if let Err(e) = store.put_object(RUSTFS_META_BUCKET, path, &mut put_data, &opts).await {
warn!("write IAM config {path}: {e}");
} else {
@@ -754,7 +754,7 @@ impl<S: StorageAPI> ReplicationPool<S> {
buckets: Vec<String>,
) -> Result<(), EcstoreError> {
// Load bucket metadata system in background
let pool_clone = self.clone();
let pool_clone = self;
tokio::spawn(async move {
pool_clone.start_resync_routine(buckets, cancellation_token).await;
+8 -8
View File
@@ -16,7 +16,7 @@ use crate::config::{Config, GLOBAL_STORAGE_CLASS, KVS, audit, notify, oidc, stor
use crate::disk::{MIGRATING_META_BUCKET, RUSTFS_META_BUCKET};
use crate::error::{Error, Result};
use crate::global::is_first_cluster_node_local;
use crate::store_api::{ChunkNativePutData, ObjectInfo, ObjectOptions, StorageAPI};
use crate::store_api::{ObjectInfo, ObjectOptions, PutObjReader, StorageAPI};
use http::HeaderMap;
use rustfs_config::audit::{AUDIT_MQTT_KEYS, AUDIT_MQTT_SUB_SYS, AUDIT_WEBHOOK_KEYS, AUDIT_WEBHOOK_SUB_SYS};
use rustfs_config::notify::{NOTIFY_MQTT_KEYS, NOTIFY_MQTT_SUB_SYS, NOTIFY_WEBHOOK_KEYS, NOTIFY_WEBHOOK_SUB_SYS};
@@ -128,7 +128,7 @@ pub async fn delete_config<S: StorageAPI>(api: Arc<S>, file: &str) -> Result<()>
}
pub async fn save_config_with_opts<S: StorageAPI>(api: Arc<S>, file: &str, data: Vec<u8>, opts: &ObjectOptions) -> Result<()> {
let mut put_data = ChunkNativePutData::from_vec(data);
let mut put_data = PutObjReader::from_vec(data);
if let Err(err) = api.put_object(RUSTFS_META_BUCKET, file, &mut put_data, opts).await {
error!("save_config_with_opts: err: {:?}, file: {}", err, file);
return Err(err);
@@ -1076,10 +1076,10 @@ mod tests {
use crate::global::{is_dist_erasure, is_erasure, is_erasure_sd, update_erasure_type};
use crate::set_disk::SetDisks;
use crate::store_api::{
BucketInfo, BucketOperations, BucketOptions, ChunkNativePutData, CompletePart, DeleteBucketOptions, DeletedObject,
GetObjectReader, HTTPRangeSpec, HealOperations, ListMultipartsInfo, ListObjectVersionsInfo, ListObjectsV2Info,
ListOperations, MakeBucketOptions, MultipartInfo, MultipartOperations, MultipartUploadResult, ObjectIO, ObjectInfo,
ObjectOperations, ObjectOptions, ObjectToDelete, PartInfo, StorageAPI, WalkOptions,
BucketInfo, BucketOperations, BucketOptions, CompletePart, DeleteBucketOptions, DeletedObject, GetObjectReader,
HTTPRangeSpec, HealOperations, ListMultipartsInfo, ListObjectVersionsInfo, ListObjectsV2Info, ListOperations,
MakeBucketOptions, MultipartInfo, MultipartOperations, MultipartUploadResult, ObjectIO, ObjectInfo, ObjectOperations,
ObjectOptions, ObjectToDelete, PartInfo, PutObjReader, StorageAPI, WalkOptions,
};
use http::HeaderMap;
use rustfs_config::audit::{AUDIT_MQTT_SUB_SYS, AUDIT_WEBHOOK_SUB_SYS};
@@ -1302,7 +1302,7 @@ mod tests {
&self,
_bucket: &str,
_object: &str,
_data: &mut ChunkNativePutData,
_data: &mut PutObjReader,
_opts: &ObjectOptions,
) -> Result<ObjectInfo> {
panic!("unused in test")
@@ -1489,7 +1489,7 @@ mod tests {
_object: &str,
_upload_id: &str,
_part_id: usize,
_data: &mut ChunkNativePutData,
_data: &mut PutObjReader,
_opts: &ObjectOptions,
) -> Result<PartInfo> {
panic!("unused in test")
+1 -1
View File
@@ -144,7 +144,7 @@ impl KVS {
pub fn insert(&mut self, key: String, value: String) {
for kv in self.0.iter_mut() {
if kv.key == key {
kv.value = value.clone();
kv.value = value;
return;
}
}
+8 -3
View File
@@ -17,9 +17,9 @@ use rustfs_config::{
ENABLE_KEY, EnableState,
oidc::{
OIDC_CLAIM_NAME, OIDC_CLAIM_PREFIX, OIDC_CLIENT_ID, OIDC_CLIENT_SECRET, OIDC_CONFIG_URL, OIDC_DEFAULT_CLAIM_NAME,
OIDC_DEFAULT_EMAIL_CLAIM, OIDC_DEFAULT_GROUPS_CLAIM, OIDC_DEFAULT_SCOPES, OIDC_DEFAULT_USERNAME_CLAIM, OIDC_DISPLAY_NAME,
OIDC_EMAIL_CLAIM, OIDC_GROUPS_CLAIM, OIDC_REDIRECT_URI, OIDC_REDIRECT_URI_DYNAMIC, OIDC_ROLE_POLICY, OIDC_SCOPES,
OIDC_USERNAME_CLAIM,
OIDC_DEFAULT_EMAIL_CLAIM, OIDC_DEFAULT_GROUPS_CLAIM, OIDC_DEFAULT_ROLES_CLAIM, OIDC_DEFAULT_SCOPES,
OIDC_DEFAULT_USERNAME_CLAIM, OIDC_DISPLAY_NAME, OIDC_EMAIL_CLAIM, OIDC_GROUPS_CLAIM, OIDC_REDIRECT_URI,
OIDC_REDIRECT_URI_DYNAMIC, OIDC_ROLE_POLICY, OIDC_ROLES_CLAIM, OIDC_SCOPES, OIDC_USERNAME_CLAIM,
},
};
use std::sync::LazyLock;
@@ -87,6 +87,11 @@ pub static DEFAULT_IDENTITY_OPENID_KVS: LazyLock<KVS> = LazyLock::new(|| {
value: OIDC_DEFAULT_GROUPS_CLAIM.to_owned(),
hidden_if_empty: false,
},
KV {
key: OIDC_ROLES_CLAIM.to_owned(),
value: OIDC_DEFAULT_ROLES_CLAIM.to_owned(),
hidden_if_empty: false,
},
KV {
key: OIDC_EMAIL_CLAIM.to_owned(),
value: OIDC_DEFAULT_EMAIL_CLAIM.to_owned(),
+12 -39
View File
@@ -150,7 +150,18 @@ impl Config {
return false;
}
shard_size as usize <= self.inline_shard_limit_bytes(versioned)
let shard_size = shard_size as usize;
let mut inline_block = DEFAULT_INLINE_BLOCK;
if self.initialized {
inline_block = self.inline_block;
}
if versioned {
shard_size <= inline_block / 8
} else {
shard_size <= inline_block
}
}
pub fn inline_block(&self) -> usize {
@@ -161,15 +172,6 @@ impl Config {
}
}
pub fn inline_shard_limit_bytes(&self, versioned: bool) -> usize {
let inline_block = self.inline_block();
if versioned { inline_block / 8 } else { inline_block }
}
pub fn inline_object_limit_bytes(&self, data_shards: usize, versioned: bool) -> usize {
self.inline_shard_limit_bytes(versioned).saturating_mul(data_shards.max(1))
}
pub fn capacity_optimized(&self) -> bool {
if !self.initialized {
false
@@ -334,32 +336,3 @@ pub fn validate_parity_inner(ss_parity: usize, rrs_parity: usize, set_drive_coun
}
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn inline_object_limit_matches_default_non_versioned_budget() {
let cfg = Config {
initialized: true,
inline_block: DEFAULT_INLINE_BLOCK,
..Default::default()
};
assert_eq!(cfg.inline_shard_limit_bytes(false), DEFAULT_INLINE_BLOCK);
assert_eq!(cfg.inline_object_limit_bytes(8, false), DEFAULT_INLINE_BLOCK * 8);
}
#[test]
fn inline_object_limit_scales_down_for_versioned_objects() {
let cfg = Config {
initialized: true,
inline_block: DEFAULT_INLINE_BLOCK,
..Default::default()
};
assert_eq!(cfg.inline_shard_limit_bytes(true), DEFAULT_INLINE_BLOCK / 8);
assert_eq!(cfg.inline_object_limit_bytes(8, true), DEFAULT_INLINE_BLOCK);
}
}
+8 -15
View File
@@ -14,11 +14,9 @@
use crate::error::{Error, Result};
use crate::store::ECStore;
use crate::store_api::{
ChunkNativePutData, CompletePart, GetObjectReader, MultipartOperations, ObjectIO, ObjectInfo, ObjectOptions,
};
use crate::store_api::{CompletePart, GetObjectReader, MultipartOperations, ObjectIO, ObjectInfo, ObjectOptions, PutObjReader};
use bytes::Bytes;
use rustfs_rio::{BlockReadable, BoxReadBlockFuture, EtagResolvable, HashReader, HashReaderDetector, Index, TryGetIndex};
use rustfs_rio::{EtagResolvable, HashReader, HashReaderDetector, Index, TryGetIndex};
use std::io::Cursor;
use std::pin::Pin;
use std::sync::{
@@ -56,11 +54,6 @@ impl<R: AsyncRead + Unpin + Send + Sync> TryGetIndex for IndexedDataMovementRead
}
}
impl<R: AsyncRead + Unpin + Send + Sync> BlockReadable for IndexedDataMovementReader<R> {
fn read_block<'a>(&'a mut self, buf: &'a mut [u8]) -> BoxReadBlockFuture<'a> {
Box::pin(rustfs_utils::read_full(self, buf))
}
}
pub fn decode_part_index(index: Option<&Bytes>) -> Option<Index> {
let bytes = index?;
let mut decoded = Index::new();
@@ -71,7 +64,7 @@ pub fn decode_part_index(index: Option<&Bytes>) -> Option<Index> {
}
}
pub fn put_data_from_chunk(chunk: Vec<u8>, size: i64, actual_size: i64, index: Option<Index>) -> Result<ChunkNativePutData> {
pub fn put_obj_reader_from_chunk(chunk: Vec<u8>, size: i64, actual_size: i64, index: Option<Index>) -> Result<PutObjReader> {
use sha2::{Digest, Sha256};
let sha256hex = if !chunk.is_empty() {
@@ -81,8 +74,8 @@ pub fn put_data_from_chunk(chunk: Vec<u8>, size: i64, actual_size: i64, index: O
};
let reader = IndexedDataMovementReader::new(Cursor::new(chunk), index);
let hash_reader = HashReader::from_reader(reader, size, actual_size, None, sha256hex, false)?;
Ok(ChunkNativePutData::new(hash_reader))
let hash_reader = HashReader::from_stream(reader, size, actual_size, None, sha256hex, false)?;
Ok(PutObjReader::new(hash_reader))
}
pub fn new_multipart_abort_flag() -> Arc<AtomicBool> {
@@ -179,7 +172,7 @@ pub(crate) async fn migrate_object(
let part_size = i64::try_from(part.size).map_err(|_| Error::other("part size overflow"))?;
let part_actual_size = if part.actual_size > 0 { part.actual_size } else { part_size };
let index = decode_part_index(part.index.as_ref());
let mut data = put_data_from_chunk(chunk, part_size, part_actual_size, index)?;
let mut data = put_obj_reader_from_chunk(chunk, part_size, part_actual_size, index)?;
let pi = match store
.put_object_part(
@@ -261,8 +254,8 @@ pub(crate) async fn migrate_object(
.first()
.and_then(|part| decode_part_index(part.index.as_ref()));
let reader = IndexedDataMovementReader::new(BufReader::new(rd.stream), index);
let hrd = HashReader::from_reader(reader, object_info.size, actual_size, object_info.etag.clone(), None, false)?;
let mut data = ChunkNativePutData::new(hrd);
let hrd = HashReader::from_stream(reader, object_info.size, actual_size, object_info.etag.clone(), None, false)?;
let mut data = PutObjReader::new(hrd);
if let Err(err) = store
.put_object(
-9
View File
@@ -21,7 +21,6 @@ use crate::disk::{
use crate::global::GLOBAL_LOCAL_DISK_ID_MAP;
use bytes::Bytes;
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_io_core::BoxChunkStream;
use std::{
path::PathBuf,
sync::{
@@ -739,14 +738,6 @@ impl DiskAPI for LocalDiskWrapper {
.await
}
async fn read_file_chunks(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<BoxChunkStream> {
self.track_disk_health(
|| async { self.disk.read_file_chunks(volume, path, offset, length).await },
get_max_timeout_duration(),
)
.await
}
async fn append_file(&self, volume: &str, path: &str) -> Result<crate::disk::FileWriter> {
self.track_disk_health(|| async { self.disk.append_file(volume, path).await }, Duration::ZERO)
.await
+1 -4
View File
@@ -724,10 +724,7 @@ mod tests {
let path = PathBuf::from("/test/path");
let io_error = std::io::Error::new(std::io::ErrorKind::PermissionDenied, "permission denied");
let context_error = FileAccessDeniedWithContext {
path: path.clone(),
source: io_error,
};
let context_error = FileAccessDeniedWithContext { path, source: io_error };
let display_str = format!("{context_error}");
assert!(display_str.contains("/test/path"));
+5 -5
View File
@@ -67,7 +67,7 @@ pub fn reduce_errs(errors: &[Option<Error>], ignored_errs: &[Error]) -> (usize,
let (best_err, best_count) = err_counts
.into_iter()
.max_by(|(_, c1), (_, c2)| c1.cmp(c2))
.unwrap_or((nil_error.clone(), 0));
.unwrap_or((nil_error, 0));
// Compare nil errors with the top non-nil error and prefer the nil error
if nil_count > best_count || (nil_count == best_count && nil_count > 0) {
@@ -112,7 +112,7 @@ mod tests {
fn test_reduce_errs_basic() {
let e1 = err_io("a");
let e2 = err_io("b");
let errors = vec![Some(e1.clone()), Some(e1.clone()), Some(e2.clone()), None];
let errors = vec![Some(e1.clone()), Some(e1.clone()), Some(e2), None];
let ignored = vec![];
let (count, err) = reduce_errs(&errors, &ignored);
assert_eq!(count, 2);
@@ -124,7 +124,7 @@ mod tests {
let e1 = err_io("a");
let e2 = err_io("b");
let errors = vec![Some(e1.clone()), Some(e2.clone()), Some(e1.clone()), Some(e2.clone()), None];
let ignored = vec![e2.clone()];
let ignored = vec![e2];
let (count, err) = reduce_errs(&errors, &ignored);
assert_eq!(count, 2);
assert_eq!(err, Some(e1));
@@ -134,7 +134,7 @@ mod tests {
fn test_reduce_quorum_errs() {
let e1 = err_io("a");
let e2 = err_io("b");
let errors = vec![Some(e1.clone()), Some(e1.clone()), Some(e2.clone()), None];
let errors = vec![Some(e1.clone()), Some(e1.clone()), Some(e2), None];
let ignored = vec![];
let quorum_err = Error::FaultyDisk;
// quorum = 2, should return e1
@@ -167,7 +167,7 @@ mod tests {
fn test_reduce_errs_nil_tiebreak() {
// Error::Nil and another error have the same count, should prefer Nil
let e1 = err_io("a");
let errors = vec![Some(e1.clone()), None, Some(e1.clone()), None]; // e1:2, Nil:2
let errors = vec![Some(e1.clone()), None, Some(e1), None]; // e1:2, Nil:2
let ignored = vec![];
let (count, err) = reduce_errs(&errors, &ignored);
assert_eq!(count, 2);
File diff suppressed because it is too large Load Diff
-12
View File
@@ -41,7 +41,6 @@ use error::DiskError;
use error::{Error, Result};
use local::LocalDisk;
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_io_core::BoxChunkStream;
use rustfs_madmin::info_commands::DiskMetrics;
use serde::{Deserialize, Serialize};
use std::{fmt::Debug, path::PathBuf, sync::Arc};
@@ -296,14 +295,6 @@ impl DiskAPI for Disk {
}
}
#[tracing::instrument(skip(self))]
async fn read_file_chunks(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<BoxChunkStream> {
match self {
Disk::Local(local_disk) => local_disk.read_file_chunks(volume, path, offset, length).await,
Disk::Remote(remote_disk) => remote_disk.read_file_chunks(volume, path, offset, length).await,
}
}
#[tracing::instrument(skip(self))]
async fn append_file(&self, volume: &str, path: &str) -> Result<FileWriter> {
match self {
@@ -514,9 +505,6 @@ pub trait DiskAPI: Debug + Send + Sync + 'static {
/// On other platforms, falls back to efficient read operations.
async fn read_file_zero_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes>;
/// Chunk-based file read compatibility layer for the zero-copy data plane.
async fn read_file_chunks(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<BoxChunkStream>;
async fn append_file(&self, volume: &str, path: &str) -> Result<FileWriter>;
async fn create_file(&self, origvolume: &str, volume: &str, path: &str, file_size: i64) -> Result<FileWriter>;
// ReadFileStream
@@ -172,52 +172,6 @@ where
}
}
impl BitrotWriter<CustomWriter> {
fn write_inline_sync(&mut self, buf: &[u8]) -> std::io::Result<usize> {
if buf.is_empty() {
return Ok(0);
}
if self.finished {
return Err(std::io::Error::new(std::io::ErrorKind::InvalidInput, "bitrot writer already finished"));
}
if buf.len() > self.shard_size {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
format!("data size {} exceeds shard size {}", buf.len(), self.shard_size),
));
}
if buf.len() < self.shard_size {
self.finished = true;
}
match &mut self.inner {
CustomWriter::InlineBuffer(data) => {
if self.hash_algo.size() > 0 {
let hash = self.hash_algo.hash_encode(buf);
if hash.as_ref().is_empty() {
error!("bitrot writer write hash error: hash is empty");
return Err(std::io::Error::new(std::io::ErrorKind::InvalidInput, "hash is empty"));
}
data.extend_from_slice(hash.as_ref());
}
data.extend_from_slice(buf);
Ok(buf.len())
}
CustomWriter::Other(_) => Err(std::io::Error::other("inline sync write requires inline buffer writer")),
}
}
fn shutdown_inline_sync(&mut self) -> std::io::Result<()> {
match self.inner {
CustomWriter::InlineBuffer(_) => Ok(()),
CustomWriter::Other(_) => Err(std::io::Error::other("inline sync shutdown requires inline buffer writer")),
}
}
}
async fn write_all_vectored<W>(writer: &mut W, hash: &[u8], data: &[u8]) -> std::io::Result<()>
where
W: AsyncWrite + Unpin,
@@ -326,10 +280,6 @@ impl CustomWriter {
Self::Other(_) => None,
}
}
pub fn is_inline_buffer(&self) -> bool {
matches!(self, Self::InlineBuffer(_))
}
}
impl AsyncWrite for CustomWriter {
@@ -447,24 +397,6 @@ impl BitrotWriterWrapper {
self.bitrot_writer.shutdown().await
}
pub fn is_inline_buffer(&self) -> bool {
matches!(self.writer_type, WriterType::InlineBuffer)
}
pub fn write_inline_sync(&mut self, buf: &[u8]) -> std::io::Result<usize> {
if !self.is_inline_buffer() {
return Err(std::io::Error::other("inline sync write requires inline buffer writer"));
}
self.bitrot_writer.write_inline_sync(buf)
}
pub fn shutdown_inline_sync(&mut self) -> std::io::Result<()> {
if !self.is_inline_buffer() {
return Err(std::io::Error::other("inline sync shutdown requires inline buffer writer"));
}
self.bitrot_writer.shutdown_inline_sync()
}
/// Extract the inline buffer data, consuming the wrapper
pub fn into_inline_data(self) -> Option<Vec<u8>> {
match self.writer_type {
-235
View File
@@ -17,7 +17,6 @@ use crate::disk::error_reduce::reduce_errs;
use crate::erasure_coding::{BitrotReader, Erasure};
use futures::stream::{FuturesUnordered, StreamExt};
use pin_project_lite::pin_project;
use rustfs_io_core::{IoChunk, PooledChunk};
use std::io;
use std::io::ErrorKind;
use tokio::io::AsyncRead;
@@ -156,68 +155,6 @@ fn get_data_block_len(shards: &[Option<Vec<u8>>], data_blocks: usize) -> usize {
size
}
fn block_window(
offset: usize,
length: usize,
block_size: usize,
block_index: usize,
start_block: usize,
end_block: usize,
) -> (usize, usize) {
let end_remainder = offset.saturating_add(length) % block_size;
if start_block == end_block {
(offset % block_size, length)
} else if block_index == start_block {
(offset % block_size, block_size - (offset % block_size))
} else if block_index == end_block {
(0, if end_remainder == 0 { block_size } else { end_remainder })
} else {
(0, block_size)
}
}
fn take_data_blocks_as_chunks(
shards: &mut [Option<Vec<u8>>],
data_blocks: usize,
mut offset: usize,
length: usize,
) -> io::Result<Vec<IoChunk>> {
if get_data_block_len(shards, data_blocks) < length {
error!("take_data_blocks_as_chunks get_data_block_len < length");
return Err(io::Error::new(ErrorKind::UnexpectedEof, "Not enough data blocks to write"));
}
let mut chunks = Vec::new();
let mut remaining = length;
for block_op in shards.iter_mut().take(data_blocks) {
let Some(block) = block_op.take() else {
error!("take_data_blocks_as_chunks block_op.is_none()");
return Err(io::Error::new(ErrorKind::UnexpectedEof, "Missing data block"));
};
if offset >= block.len() {
offset -= block.len();
continue;
}
let start = offset;
offset = 0;
let take = (block.len() - start).min(remaining);
let chunk = if start == 0 && take == block.len() {
IoChunk::Pooled(PooledChunk::from_vec(block))
} else {
IoChunk::Pooled(PooledChunk::from_vec(block).slice(start, take)?)
};
chunks.push(chunk);
remaining -= take;
if remaining == 0 {
break;
}
}
Ok(chunks)
}
/// Write data blocks from encoded blocks to target, supporting offset and length
async fn write_data_blocks<W>(
writer: &mut W,
@@ -276,134 +213,6 @@ where
Ok(total_written)
}
pub(crate) struct ErasureChunkDecoder<R> {
erasure: Erasure,
reader: ParallelReader<R>,
offset: usize,
length: usize,
start_block: usize,
end_block: usize,
current_block: usize,
written: usize,
healable_error: Option<Error>,
finished: bool,
}
impl<R> ErasureChunkDecoder<R>
where
R: AsyncRead + Unpin + Send + Sync,
{
pub(crate) fn new(
erasure: Erasure,
readers: Vec<Option<BitrotReader<R>>>,
offset: usize,
length: usize,
total_length: usize,
) -> io::Result<Self> {
if readers.len() != erasure.data_shards + erasure.parity_shards {
return Err(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers"));
}
let end_offset = offset
.checked_add(length)
.ok_or_else(|| io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length"))?;
if end_offset > total_length {
return Err(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length"));
}
let start_block = offset / erasure.block_size;
let end_block = if length == 0 {
start_block
} else {
end_offset.saturating_sub(1) / erasure.block_size
};
let reader = ParallelReader::new(readers, erasure.clone(), offset, total_length);
Ok(Self {
erasure,
reader,
offset,
length,
start_block,
end_block,
current_block: start_block,
written: 0,
healable_error: None,
finished: length == 0,
})
}
pub(crate) async fn next_chunks(&mut self) -> io::Result<Option<Vec<IoChunk>>> {
if self.finished {
return Ok(None);
}
if self.current_block > self.end_block {
self.finished = true;
return Ok(None);
}
let block_index = self.current_block;
self.current_block += 1;
let (block_offset, block_length) = block_window(
self.offset,
self.length,
self.erasure.block_size,
block_index,
self.start_block,
self.end_block,
);
if block_length == 0 {
self.finished = true;
return Ok(None);
}
let (mut shards, errs) = self.reader.read().await;
if self.healable_error.is_none()
&& let (_, Some(err)) = reduce_errs(&errs, &[])
&& (err == Error::FileNotFound || err == Error::FileCorrupt)
{
self.healable_error = Some(err);
}
if !self.reader.can_decode(&shards) {
self.finished = true;
error!("reconstructed chunk decoder can_decode errs: {:?}", &errs);
return Err(Error::ErasureReadQuorum.into());
}
if let Err(err) = self.erasure.decode_data(&mut shards) {
self.finished = true;
error!("reconstructed chunk decoder decode_data err: {:?}", err);
return Err(err);
}
let chunks = take_data_blocks_as_chunks(&mut shards, self.erasure.data_shards, block_offset, block_length)?;
self.written += chunks.iter().map(IoChunk::len).sum::<usize>();
Ok(Some(chunks))
}
pub(crate) fn written(&self) -> usize {
self.written
}
pub(crate) fn finish_error(&self) -> Option<io::Error> {
if self.written < self.length {
Some(Error::LessData.into())
} else {
None
}
}
pub(crate) fn take_healable_error(&mut self) -> Option<Error> {
self.healable_error.take()
}
}
pub(crate) type ReconstructedChunkDecoder<R> = ErasureChunkDecoder<R>;
impl Erasure {
pub async fn decode<W, R>(
&self,
@@ -511,7 +320,6 @@ mod tests {
disk::error::DiskError,
erasure_coding::{BitrotReader, BitrotWriter},
};
use bytes::Bytes;
use rustfs_utils::HashAlgorithm;
use std::io::Cursor;
@@ -652,47 +460,4 @@ mod tests {
let reader_cursor = Cursor::new(buf);
BitrotReader::new(reader_cursor, shard_size, hash_algo.clone(), false)
}
async fn create_bitrot_reader_from_shard(
shard: Bytes,
shard_size: usize,
hash_algo: &HashAlgorithm,
) -> BitrotReader<Cursor<Vec<u8>>> {
let writer = Cursor::new(Vec::new());
let mut writer = BitrotWriter::new(writer, shard_size, hash_algo.clone());
writer.write(shard.as_ref()).await.unwrap();
let reader_cursor = Cursor::new(writer.into_inner().into_inner());
BitrotReader::new(reader_cursor, shard_size, hash_algo.clone(), false)
}
#[tokio::test]
async fn test_erasure_chunk_decoder_reconstructs_missing_data_shard_as_pooled_chunks() {
let erasure = Erasure::new(2, 1, 4);
let original = b"abcd";
let encoded = erasure.encode_data(original).unwrap();
let shard_size = erasure.shard_size();
let hash_algo = HashAlgorithm::None;
let readers = vec![
None,
Some(create_bitrot_reader_from_shard(encoded[1].clone(), shard_size, &hash_algo).await),
Some(create_bitrot_reader_from_shard(encoded[2].clone(), shard_size, &hash_algo).await),
];
let mut decoder = ErasureChunkDecoder::new(erasure, readers, 0, original.len(), original.len()).unwrap();
let first_batch = decoder.next_chunks().await.unwrap().unwrap();
assert!(
first_batch.iter().all(|chunk| matches!(chunk, IoChunk::Pooled(_))),
"reconstructed decoder should produce pooled chunks"
);
let collected = first_batch
.into_iter()
.flat_map(|chunk| chunk.as_bytes().to_vec())
.collect::<Vec<_>>();
assert_eq!(collected, original);
assert!(decoder.next_chunks().await.unwrap().is_none());
assert_eq!(decoder.written(), original.len());
assert!(decoder.finish_error().is_none());
}
}
+71 -308
View File
@@ -17,12 +17,11 @@ use crate::disk::error_reduce::count_errs;
use crate::disk::error_reduce::{OBJECT_OP_IGNORED_ERRS, reduce_write_quorum_errs};
use crate::erasure_coding::BitrotWriterWrapper;
use crate::erasure_coding::Erasure;
use crate::erasure_coding::erasure::{EncodeBlockBuffer, EncodedShardBlock, EncodedShardBufferPool};
use bytes::Bytes;
use futures::StreamExt;
use futures::stream::FuturesUnordered;
use rustfs_rio::BlockReadable;
use std::sync::Arc;
use std::vec;
use tokio::io::AsyncRead;
use tokio::sync::mpsc;
use tracing::error;
@@ -33,164 +32,6 @@ pub(crate) struct MultiWriter<'a> {
errs: Vec<Option<Error>>,
}
pub(crate) struct BlockAssembler<R> {
reader: R,
block_buffer: EncodeBlockBuffer,
total_bytes: usize,
}
impl<R> BlockAssembler<R>
where
R: AsyncRead + BlockReadable + Send + Sync + Unpin + 'static,
{
pub(crate) fn new(reader: R, block_size: usize) -> Self {
Self {
reader,
block_buffer: EncodeBlockBuffer::new(block_size),
total_bytes: 0,
}
}
pub(crate) async fn next_block(&mut self) -> std::io::Result<Option<Vec<u8>>> {
match self.block_buffer.read_from_block(&mut self.reader).await {
Ok(n) if n > 0 => {
self.total_bytes += n;
Ok(Some(self.block_buffer.filled(n).to_vec()))
}
Ok(_) => Ok(None),
Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => {
if let Some(inner) = e.get_ref()
&& rustfs_rio::is_checksum_mismatch(inner)
{
return Err(std::io::Error::new(std::io::ErrorKind::InvalidData, e.to_string()));
}
Ok(None)
}
Err(e) => Err(e),
}
}
pub(crate) fn total_bytes(&self) -> usize {
self.total_bytes
}
pub(crate) fn into_inner(self) -> R {
self.reader
}
}
#[derive(Clone)]
pub(crate) struct ErasureChunkEncoder {
erasure: Arc<Erasure>,
buffer_pool: EncodedShardBufferPool,
}
impl ErasureChunkEncoder {
pub(crate) async fn new(erasure: Arc<Erasure>) -> Self {
let reusable_capacity = erasure.shard_size() * erasure.total_shard_count();
Self {
erasure,
buffer_pool: EncodedShardBufferPool::with_prefill(reusable_capacity, 2).await,
}
}
pub(crate) async fn encode_block(&self, block: &[u8]) -> std::io::Result<EncodedShardBlock> {
let reusable_buffer = self.buffer_pool.acquire().await;
self.erasure.encode_data_block_with_buffer(block, reusable_buffer)
}
pub(crate) async fn release(&self, block: EncodedShardBlock) {
self.buffer_pool.release(block).await;
}
}
pub(crate) struct ErasureWritePipeline {
erasure: Arc<Erasure>,
write_quorum: usize,
}
impl ErasureWritePipeline {
pub(crate) fn new(erasure: Arc<Erasure>, write_quorum: usize) -> Self {
Self { erasure, write_quorum }
}
pub(crate) async fn run<R>(&self, reader: R, writers: &mut [Option<BitrotWriterWrapper>]) -> std::io::Result<(R, usize)>
where
R: AsyncRead + BlockReadable + Send + Sync + Unpin + 'static,
{
let (tx, mut rx) = mpsc::channel::<EncodedShardBlock>(8);
let producer = ErasureChunkEncoder::new(self.erasure.clone()).await;
let writer_pool = producer.clone();
let block_size = self.erasure.block_size;
let task = tokio::spawn(async move {
let mut assembler = BlockAssembler::new(reader, block_size);
while let Some(block) = assembler.next_block().await? {
let res = producer.encode_block(&block).await?;
if let Err(err) = tx.send(res).await {
return Err(std::io::Error::other(format!("Failed to send encoded data : {err}")));
}
}
let total = assembler.total_bytes();
Ok((assembler.into_inner(), total))
});
let mut writers = MultiWriter::new(writers, self.write_quorum);
let mut write_err = None;
while let Some(block) = rx.recv().await {
if block.is_empty() {
break;
}
let write_result = writers.write(&block).await;
writer_pool.release(block).await;
if let Err(err) = write_result {
write_err = Some(err);
break;
}
}
if let Some(err) = write_err {
task.abort();
let _ = task.await;
if let Err(shutdown_err) = writers.shutdown().await {
error!("failed to shutdown erasure writers after write error: {:?}", shutdown_err);
}
return Err(err);
}
let (reader, total) = task.await??;
writers.shutdown().await?;
Ok((reader, total))
}
}
pub(crate) trait ShardSource {
fn shard_count(&self) -> usize;
fn shard(&self, idx: usize) -> Bytes;
}
impl ShardSource for EncodedShardBlock {
fn shard_count(&self) -> usize {
self.shard_count()
}
fn shard(&self, idx: usize) -> Bytes {
self.shard(idx)
}
}
impl ShardSource for Vec<Bytes> {
fn shard_count(&self) -> usize {
self.len()
}
fn shard(&self, idx: usize) -> Bytes {
self[idx].clone()
}
}
impl<'a> MultiWriter<'a> {
pub fn new(writers: &'a mut [Option<BitrotWriterWrapper>], write_quorum: usize) -> Self {
let length = writers.len();
@@ -201,10 +42,10 @@ impl<'a> MultiWriter<'a> {
}
}
async fn write_shard(writer_opt: &mut Option<BitrotWriterWrapper>, err: &mut Option<Error>, shard: Bytes) {
async fn write_shard(writer_opt: &mut Option<BitrotWriterWrapper>, err: &mut Option<Error>, shard: &Bytes) {
match writer_opt {
Some(writer) => {
match writer.write(&shard).await {
match writer.write(shard).await {
Ok(n) => {
if n < shard.len() {
*err = Some(Error::ShortWrite);
@@ -224,40 +65,16 @@ impl<'a> MultiWriter<'a> {
}
}
fn write_shard_inline(writer_opt: &mut Option<BitrotWriterWrapper>, err: &mut Option<Error>, shard: Bytes) {
match writer_opt {
Some(writer) => match writer.write_inline_sync(&shard) {
Ok(n) => {
if n < shard.len() {
*err = Some(Error::ShortWrite);
*writer_opt = None;
} else {
*err = None;
}
}
Err(e) => {
*err = Some(Error::from(e));
}
},
None => {
*err = Some(Error::DiskNotFound);
}
}
}
pub async fn write<T>(&mut self, data: &T) -> std::io::Result<()>
where
T: ShardSource,
{
assert_eq!(data.shard_count(), self.writers.len());
pub async fn write(&mut self, data: Vec<Bytes>) -> std::io::Result<()> {
assert_eq!(data.len(), self.writers.len());
{
let mut futures = FuturesUnordered::new();
for (idx, (writer_opt, err)) in self.writers.iter_mut().zip(self.errs.iter_mut()).enumerate() {
for ((writer_opt, err), shard) in self.writers.iter_mut().zip(self.errs.iter_mut()).zip(data.iter()) {
if err.is_some() {
continue; // Skip if we already have an error for this writer
}
futures.push(Self::write_shard(writer_opt, err, data.shard(idx)));
futures.push(Self::write_shard(writer_opt, err, shard));
}
while let Some(()) = futures.next().await {}
}
@@ -295,45 +112,6 @@ impl<'a> MultiWriter<'a> {
)))
}
pub fn write_inline<T>(&mut self, data: &T) -> std::io::Result<()>
where
T: ShardSource,
{
assert_eq!(data.shard_count(), self.writers.len());
for (idx, (writer_opt, err)) in self.writers.iter_mut().zip(self.errs.iter_mut()).enumerate() {
if err.is_some() {
continue;
}
Self::write_shard_inline(writer_opt, err, data.shard(idx));
}
let nil_count = self.errs.iter().filter(|&e| e.is_none()).count();
if nil_count >= self.write_quorum {
return Ok(());
}
if let Some(write_err) = reduce_write_quorum_errs(&self.errs, OBJECT_OP_IGNORED_ERRS, self.write_quorum) {
return Err(std::io::Error::other(format!(
"Failed to write inline data: {} (offline-disks={}/{})",
write_err,
count_errs(&self.errs, &Error::DiskNotFound),
self.writers.len()
)));
}
Err(std::io::Error::other(format!(
"Failed to write inline data: (offline-disks={}/{}): {}",
count_errs(&self.errs, &Error::DiskNotFound),
self.writers.len(),
self.errs
.iter()
.map(|e| e.as_ref().map_or("<nil>".to_string(), |e| e.to_string()))
.collect::<Vec<_>>()
.join(", ")
)))
}
async fn shutdown_writer(writer_opt: &mut Option<BitrotWriterWrapper>, err: &mut Option<Error>) {
match writer_opt {
Some(writer) => match writer.shutdown().await {
@@ -351,23 +129,6 @@ impl<'a> MultiWriter<'a> {
}
}
fn shutdown_writer_inline(writer_opt: &mut Option<BitrotWriterWrapper>, err: &mut Option<Error>) {
match writer_opt {
Some(writer) => match writer.shutdown_inline_sync() {
Ok(()) => {
*err = None;
}
Err(e) => {
*err = Some(Error::from(e));
*writer_opt = None;
}
},
None => {
*err = Some(Error::DiskNotFound);
}
}
}
pub async fn shutdown(&mut self) -> std::io::Result<()> {
{
let mut futures = FuturesUnordered::new();
@@ -412,53 +173,80 @@ impl<'a> MultiWriter<'a> {
.join(", ")
)))
}
pub fn shutdown_inline(&mut self) -> std::io::Result<()> {
for (writer_opt, err) in self.writers.iter_mut().zip(self.errs.iter_mut()) {
if err.is_some() {
continue;
}
Self::shutdown_writer_inline(writer_opt, err);
}
let nil_count = self.errs.iter().filter(|&e| e.is_none()).count();
if nil_count >= self.write_quorum {
return Ok(());
}
if let Some(write_err) = reduce_write_quorum_errs(&self.errs, OBJECT_OP_IGNORED_ERRS, self.write_quorum) {
return Err(std::io::Error::other(format!(
"Failed to shutdown inline writers: {} (offline-disks={}/{})",
write_err,
count_errs(&self.errs, &Error::DiskNotFound),
self.writers.len()
)));
}
Err(std::io::Error::other(format!(
"Failed to shutdown inline writers: (offline-disks={}/{}): {}",
count_errs(&self.errs, &Error::DiskNotFound),
self.writers.len(),
self.errs
.iter()
.map(|e| e.as_ref().map_or("<nil>".to_string(), |e| e.to_string()))
.collect::<Vec<_>>()
.join(", ")
)))
}
}
impl Erasure {
pub async fn encode<R>(
self: Arc<Self>,
reader: R,
mut reader: R,
writers: &mut [Option<BitrotWriterWrapper>],
quorum: usize,
) -> std::io::Result<(R, usize)>
where
R: AsyncRead + BlockReadable + Send + Sync + Unpin + 'static,
R: AsyncRead + Send + Sync + Unpin + 'static,
{
ErasureWritePipeline::new(self, quorum).run(reader, writers).await
let (tx, mut rx) = mpsc::channel::<Vec<Bytes>>(8);
let task = tokio::spawn(async move {
let block_size = self.block_size;
let mut total = 0;
let mut buf = vec![0u8; block_size];
loop {
match rustfs_utils::read_full(&mut reader, &mut buf).await {
Ok(n) if n > 0 => {
total += n;
let res = self.encode_data(&buf[..n])?;
if let Err(err) = tx.send(res).await {
return Err(std::io::Error::other(format!("Failed to send encoded data : {err}")));
}
}
Ok(_) => {
break;
}
Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => {
// Check if the inner error is a checksum mismatch - if so, propagate it
if let Some(inner) = e.get_ref()
&& rustfs_rio::is_checksum_mismatch(inner)
{
return Err(std::io::Error::new(std::io::ErrorKind::InvalidData, e.to_string()));
}
break;
}
Err(e) => {
return Err(e);
}
}
}
Ok((reader, total))
});
let mut writers = MultiWriter::new(writers, quorum);
let mut write_err = None;
while let Some(block) = rx.recv().await {
if block.is_empty() {
break;
}
if let Err(err) = writers.write(block).await {
write_err = Some(err);
break;
}
}
if let Some(err) = write_err {
task.abort();
let _ = task.await;
if let Err(shutdown_err) = writers.shutdown().await {
error!("failed to shutdown erasure writers after write error: {:?}", shutdown_err);
}
return Err(err);
}
let (reader, total) = task.await??;
writers.shutdown().await?;
Ok((reader, total))
}
}
@@ -467,7 +255,6 @@ mod tests {
use super::*;
use crate::erasure_coding::{BitrotWriterWrapper, CustomWriter};
use rustfs_utils::HashAlgorithm;
use std::io::Cursor;
use std::pin::Pin;
use std::sync::{Arc, Mutex};
use std::task::{Context, Poll};
@@ -523,28 +310,4 @@ mod tests {
assert_eq!(written, b"small payload".len());
assert!(!committed.lock().unwrap().is_empty());
}
#[tokio::test]
async fn block_assembler_splits_input_into_erasure_blocks() {
let reader = tokio::io::BufReader::new(Cursor::new(b"abcdefghijkl".to_vec()));
let mut assembler = BlockAssembler::new(reader, 4);
assert_eq!(assembler.next_block().await.unwrap(), Some(b"abcd".to_vec()));
assert_eq!(assembler.next_block().await.unwrap(), Some(b"efgh".to_vec()));
assert_eq!(assembler.next_block().await.unwrap(), Some(b"ijkl".to_vec()));
assert_eq!(assembler.next_block().await.unwrap(), None);
assert_eq!(assembler.total_bytes(), 12);
}
#[tokio::test]
async fn erasure_chunk_encoder_produces_full_shard_block() {
let erasure = Arc::new(Erasure::new(2, 1, 4));
let encoder = ErasureChunkEncoder::new(erasure.clone()).await;
let block = encoder.encode_block(b"abcd").await.unwrap();
assert_eq!(block.shard_count(), 3);
assert_eq!(block.shard(0).len(), erasure.shard_size());
encoder.release(block).await;
}
}
+26 -253
View File
@@ -19,131 +19,12 @@
use bytes::{Bytes, BytesMut};
use reed_solomon_erasure::galois_8::ReedSolomon;
use reed_solomon_simd;
use rustfs_rio::BlockReadable;
use smallvec::SmallVec;
use std::io;
use std::sync::Arc;
use tokio::io::AsyncRead;
use tokio::sync::Mutex;
use tracing::warn;
use uuid::Uuid;
pub(crate) struct EncodeBlockBuffer {
buf: Vec<u8>,
}
impl EncodeBlockBuffer {
pub(crate) fn new(block_size: usize) -> Self {
Self {
buf: vec![0u8; block_size],
}
}
pub(crate) async fn read_from<R>(&mut self, reader: &mut R) -> io::Result<usize>
where
R: AsyncRead + Send + Sync + Unpin,
{
rustfs_utils::read_full(&mut *reader, &mut self.buf).await
}
pub(crate) async fn read_from_block<R>(&mut self, reader: &mut R) -> io::Result<usize>
where
R: BlockReadable + Send + Sync + Unpin,
{
reader.read_block(&mut self.buf).await
}
pub(crate) fn filled(&self, len: usize) -> &[u8] {
&self.buf[..len]
}
}
pub struct EncodedShardBlock {
data: Bytes,
shard_size: usize,
shard_count: usize,
}
impl EncodedShardBlock {
pub(crate) fn new(data: Bytes, shard_size: usize, shard_count: usize) -> Self {
Self {
data,
shard_size,
shard_count,
}
}
pub fn shard_count(&self) -> usize {
self.shard_count
}
pub fn len(&self) -> usize {
self.shard_count
}
pub fn is_empty(&self) -> bool {
self.shard_count == 0
}
pub fn shard(&self, idx: usize) -> Bytes {
let start = idx * self.shard_size;
let end = start + self.shard_size;
self.data.slice(start..end)
}
pub fn iter(&self) -> impl Iterator<Item = Bytes> + '_ {
(0..self.shard_count).map(|idx| self.shard(idx))
}
pub fn into_vec(self) -> Vec<Bytes> {
(0..self.shard_count).map(|idx| self.shard(idx)).collect()
}
pub fn into_reusable_buffer(self) -> BytesMut {
match self.data.try_into_mut() {
Ok(mut buf) => {
buf.clear();
buf
}
Err(data) => BytesMut::with_capacity(data.len()),
}
}
}
#[derive(Clone)]
pub(crate) struct EncodedShardBufferPool {
capacity: usize,
free: Arc<Mutex<Vec<BytesMut>>>,
}
impl EncodedShardBufferPool {
pub(crate) async fn with_prefill(capacity: usize, initial: usize) -> Self {
let mut free = Vec::with_capacity(initial);
for _ in 0..initial {
free.push(BytesMut::with_capacity(capacity));
}
Self {
capacity,
free: Arc::new(Mutex::new(free)),
}
}
pub(crate) async fn acquire(&self) -> BytesMut {
let mut free = self.free.lock().await;
free.pop().unwrap_or_else(|| BytesMut::with_capacity(self.capacity))
}
pub(crate) async fn release(&self, block: EncodedShardBlock) {
let mut free = self.free.lock().await;
let mut buf = block.into_reusable_buffer();
if buf.capacity() < self.capacity {
buf.reserve(self.capacity - buf.capacity());
}
free.push(buf);
}
}
/// Legacy calc_shard_size formula: (block_size.div_ceil(data_shards) + 1) & !1
/// Matches main branch and filemeta::ErasureInfo for old-version files.
pub fn calc_shard_size_legacy(block_size: usize, data_shards: usize) -> usize {
@@ -470,25 +351,6 @@ impl Erasure {
/// A vector of encoded shards as `Bytes`.
#[tracing::instrument(level = "debug", skip_all, fields(data_len=data.len()))]
pub fn encode_data(&self, data: &[u8]) -> io::Result<Vec<Bytes>> {
Ok(self.encode_data_block(data)?.into_vec())
}
/// Encode one logical block into an `EncodedShardBlock` using a caller-provided backing buffer.
///
/// This is the explicit reuse-oriented variant for non-hot paths that want to
/// thread a reusable `BytesMut` across multiple encode calls.
#[tracing::instrument(level = "debug", skip_all, fields(data_len=data.len()))]
pub fn encode_data_with_buffer(&self, data: &[u8], data_buffer: BytesMut) -> io::Result<EncodedShardBlock> {
self.encode_data_block_with_buffer(data, data_buffer)
}
#[tracing::instrument(level = "debug", skip_all, fields(data_len=data.len()))]
pub(crate) fn encode_data_block(&self, data: &[u8]) -> io::Result<EncodedShardBlock> {
self.encode_data_block_with_buffer(data, BytesMut::with_capacity(self.shard_size() * self.total_shard_count()))
}
#[tracing::instrument(level = "debug", skip_all, fields(data_len=data.len()))]
pub(crate) fn encode_data_block_with_buffer(&self, data: &[u8], mut data_buffer: BytesMut) -> io::Result<EncodedShardBlock> {
let shard_size_fn = if self.uses_legacy {
calc_shard_size_legacy
} else {
@@ -497,10 +359,7 @@ impl Erasure {
let per_shard_size = shard_size_fn(data.len(), self.data_shards);
let need_total_size = per_shard_size * self.total_shard_count();
data_buffer.clear();
if data_buffer.capacity() < need_total_size {
data_buffer.reserve(need_total_size - data_buffer.capacity());
}
let mut data_buffer = BytesMut::with_capacity(need_total_size);
data_buffer.extend_from_slice(data);
data_buffer.resize(need_total_size, 0u8);
@@ -523,7 +382,14 @@ impl Erasure {
}
// Zero-copy split, all shards reference data_buffer
Ok(EncodedShardBlock::new(data_buffer.freeze(), per_shard_size, self.total_shard_count()))
let mut data_buffer = data_buffer.freeze();
let mut shards = Vec::with_capacity(self.total_shard_count());
for _ in 0..self.total_shard_count() {
let shard = data_buffer.split_to(per_shard_size);
shards.push(shard);
}
Ok(shards)
}
/// Decode and reconstruct missing shards in-place.
@@ -612,8 +478,8 @@ impl Erasure {
///
/// # Arguments
/// * `reader` - An async reader implementing AsyncRead + Send + Sync + Unpin
/// * `mut on_block` - Async callback that receives encoded blocks and returns the block for reuse
/// * `F` - Callback type: FnMut(Result<EncodedShardBlock, std::io::Error>) -> Future<Output=Result<Option<EncodedShardBlock>, E>> + Send
/// * `mut on_block` - Async callback that receives encoded blocks and returns a Result
/// * `F` - Callback type: FnMut(Result<Vec<Bytes>, std::io::Error>) -> Future<Output=Result<(), E>> + Send
/// * `Fut` - Future type returned by the callback
/// * `E` - Error type returned by the callback
/// * `R` - Reader type implementing AsyncRead + Send + Sync + Unpin
@@ -630,24 +496,19 @@ impl Erasure {
) -> Result<usize, E>
where
R: AsyncRead + Send + Sync + Unpin,
F: FnMut(std::io::Result<EncodedShardBlock>) -> Fut + Send,
Fut: std::future::Future<Output = Result<Option<EncodedShardBlock>, E>> + Send,
F: FnMut(std::io::Result<Vec<Bytes>>) -> Fut + Send,
Fut: std::future::Future<Output = Result<(), E>> + Send,
{
let block_size = self.block_size;
let mut total = 0;
let mut block_buffer = EncodeBlockBuffer::new(block_size);
let reusable_capacity = self.shard_size() * self.total_shard_count();
let buffer_pool = EncodedShardBufferPool::with_prefill(reusable_capacity, 1).await;
let mut buf = vec![0u8; block_size];
loop {
match block_buffer.read_from(&mut *reader).await {
match rustfs_utils::read_full(&mut *reader, &mut buf).await {
Ok(n) if n > 0 => {
warn!("encode_stream_callback_async read n={}", n);
total += n;
let reusable_buffer = buffer_pool.acquire().await;
let res = self.encode_data_block_with_buffer(block_buffer.filled(n), reusable_buffer);
if let Some(block) = on_block(res).await? {
buffer_pool.release(block).await;
}
let res = self.encode_data(&buf[..n]);
on_block(res).await?
}
Ok(_) => {
warn!("encode_stream_callback_async read unexpected ok");
@@ -659,7 +520,7 @@ impl Erasure {
}
Err(e) => {
warn!("encode_stream_callback_async read error={:?}", e);
let _ = on_block(Err(e)).await?;
on_block(Err(e)).await?;
break;
}
}
@@ -885,8 +746,8 @@ mod tests {
let tx = tx.clone();
async move {
let shards = res.unwrap();
tx.send(shards.iter().collect()).await.unwrap();
Ok(Some(shards))
tx.send(shards).await.unwrap();
Ok(())
}
})
.await
@@ -898,36 +759,6 @@ mod tests {
assert_eq!(collected_shards.len(), data_shards + parity_shards);
}
#[test]
fn test_encode_data_with_buffer_supports_explicit_reuse() {
let erasure = Erasure::new(4, 2, 1024);
let reusable_capacity = erasure.shard_size() * erasure.total_shard_count();
let first_data = b"explicit reusable buffer path".repeat(32);
let first_block = erasure
.encode_data_with_buffer(&first_data, BytesMut::with_capacity(reusable_capacity))
.expect("first encode should succeed");
let reusable_buffer = first_block.into_reusable_buffer();
assert!(reusable_buffer.capacity() >= reusable_capacity);
let second_data = b"second encode through same reusable buffer".repeat(24);
let second_block = erasure
.encode_data_with_buffer(&second_data, reusable_buffer)
.expect("second encode should succeed");
let mut shards_opt: Vec<Option<Vec<u8>>> = second_block.iter().map(|shard| Some(shard.to_vec())).collect();
shards_opt[1] = None;
shards_opt[5] = None;
erasure.decode_data(&mut shards_opt).expect("decode should succeed");
let mut recovered = Vec::new();
for shard in shards_opt.iter().take(erasure.data_shards) {
recovered.extend_from_slice(shard.as_ref().expect("data shard should exist after decode"));
}
recovered.truncate(second_data.len());
assert_eq!(&recovered, &second_data);
}
#[tokio::test]
async fn test_encode_stream_callback_async_channel_decode() {
use std::io::Cursor;
@@ -954,8 +785,8 @@ mod tests {
let tx = tx.clone();
async move {
let shards = res.unwrap();
tx.send(shards.iter().collect()).await.unwrap();
Ok(Some(shards))
tx.send(shards).await.unwrap();
Ok(())
}
})
.await
@@ -968,8 +799,8 @@ mod tests {
// Test decode using the old API that operates in-place
let mut decode_input: Vec<Option<Vec<u8>>> = vec![None; data_shards + parity_shards];
for (i, shard) in shards.iter().enumerate().take(data_shards) {
decode_input[i] = Some(shard.to_vec());
for i in 0..data_shards {
decode_input[i] = Some(shards[i].to_vec());
}
erasure.decode_data(&mut decode_input).unwrap();
@@ -1366,8 +1197,8 @@ mod tests {
let tx = tx.clone();
async move {
let shards = res.unwrap();
tx.send(shards.iter().collect()).await.unwrap();
Ok(Some(shards))
tx.send(shards).await.unwrap();
Ok(())
}
})
.await
@@ -1401,63 +1232,5 @@ mod tests {
recovered.truncate(data_clone.len());
assert_eq!(&recovered, &data_clone);
}
#[tokio::test]
#[ignore]
async fn stress_simd_stream_callback_reuses_backing_buffers_across_many_blocks() {
use std::io::Cursor;
use std::sync::Arc;
use std::sync::atomic::{AtomicUsize, Ordering};
use tokio::sync::Mutex;
let data_shards = 4;
let parity_shards = 2;
let block_size = 1024;
let erasure = Arc::new(Erasure::new(data_shards, parity_shards, block_size));
let sample =
b"SIMD stress callback test payload that intentionally spans many blocks to exercise reusable backing buffers.";
let data = sample.repeat((4 * 1024 * 1024 / sample.len()).max(1));
let data_clone = data.clone();
let mut reader = Cursor::new(data);
let recovered = Arc::new(Mutex::new(Vec::with_capacity(data_clone.len())));
let block_count = Arc::new(AtomicUsize::new(0));
let erasure_for_callback = erasure.clone();
let recovered_for_callback = recovered.clone();
let block_count_for_callback = block_count.clone();
erasure
.clone()
.encode_stream_callback_async::<_, _, (), _>(&mut reader, move |res| {
let erasure_for_callback = erasure_for_callback.clone();
let recovered_for_callback = recovered_for_callback.clone();
let block_count_for_callback = block_count_for_callback.clone();
async move {
let shards = res.unwrap();
block_count_for_callback.fetch_add(1, Ordering::Relaxed);
let mut shards_opt: Vec<Option<Vec<u8>>> = shards.iter().map(|b| Some(b.to_vec())).collect();
shards_opt[1] = None;
shards_opt[5] = None;
erasure_for_callback.decode_data(&mut shards_opt).unwrap();
let mut recovered = recovered_for_callback.lock().await;
for shard in shards_opt.iter().take(data_shards) {
recovered.extend_from_slice(shard.as_ref().unwrap());
}
Ok(Some(shards))
}
})
.await
.unwrap();
assert!(block_count.load(Ordering::Relaxed) > 1024);
let mut recovered = recovered.lock().await;
recovered.truncate(data_clone.len());
assert_eq!(&*recovered, &data_clone);
}
}
}
+1 -1
View File
@@ -77,7 +77,7 @@ impl super::Erasure {
let available_writers = writers.iter().filter(|w| w.is_some()).count();
let write_quorum = available_writers.max(1); // At least 1 writer must succeed
let mut writers = MultiWriter::new(writers, write_quorum);
writers.write(&shards).await?;
writers.write(shards).await?;
}
Ok(())
+1 -1
View File
@@ -1173,7 +1173,7 @@ mod tests {
fn test_io_error_with_disk_error_inside() {
// Test io::Error containing DiskError -> StorageError conversion
let original_disk_error = DiskError::FileNotFound;
let io_with_disk_error = std::io::Error::other(original_disk_error.clone());
let io_with_disk_error = std::io::Error::other(original_disk_error);
// Convert io::Error to StorageError
let storage_error: StorageError = io_with_disk_error.into();
+1 -1
View File
@@ -3684,7 +3684,7 @@ mod pools_tests {
#[test]
fn test_take_decommission_canceler_takes_and_clears_slot() {
let token = CancellationToken::new();
let mut cancelers = vec![Some(token.clone())];
let mut cancelers = vec![Some(token)];
let taken = take_decommission_canceler(cancelers.as_mut_slice(), 0);
assert!(taken.is_some());
+2 -2
View File
@@ -357,7 +357,7 @@ impl S3PeerSys {
ress.into_iter()
.filter(|op| op.is_some())
.find_map(|op| op.clone())
.find_map(|op| op)
.ok_or(Error::VolumeNotFound)
}
@@ -575,7 +575,7 @@ pub struct RemotePeerS3Client {
impl RemotePeerS3Client {
pub fn new(node: Option<Node>, pools: Option<Vec<usize>>) -> Self {
let addr = node.as_ref().map(|v| v.url.to_string()).unwrap_or_default().to_string();
let addr = node.as_ref().map(|v| v.url.to_string()).unwrap_or_default();
let client = Self {
node,
pools,
+2 -31
View File
@@ -30,10 +30,8 @@ use crate::{
};
use bytes::Bytes;
use futures::lock::Mutex;
use futures_util::StreamExt;
use http::{HeaderMap, HeaderValue, Method, header::CONTENT_TYPE};
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_io_core::{BoxChunkStream, IoChunk};
use rustfs_protos::proto_gen::node_service::RenamePartRequest;
use rustfs_protos::proto_gen::node_service::{
CheckPartsRequest, DeletePathsRequest, DeleteRequest, DeleteVersionRequest, DeleteVersionsRequest, DeleteVolumeRequest,
@@ -42,7 +40,7 @@ use rustfs_protos::proto_gen::node_service::{
RenameFileRequest, StatVolumeRequest, UpdateMetadataRequest, VerifyFileRequest, WriteAllRequest, WriteMetadataRequest,
node_service_client::NodeServiceClient,
};
use rustfs_rio::{HttpReader, HttpWriter, open_http_byte_stream};
use rustfs_rio::{HttpReader, HttpWriter};
use serde::{Serialize, de::DeserializeOwned};
use std::{
io::Cursor,
@@ -111,7 +109,7 @@ impl RemoteDisk {
let disk = Self {
id: Mutex::new(None),
addr: addr.clone(),
addr,
endpoint: ep.clone(),
scanning: Arc::new(AtomicU32::new(0)),
health_check: opt.health_check && env_health_check,
@@ -1073,33 +1071,6 @@ impl DiskAPI for RemoteDisk {
Ok(Bytes::from(buffer))
}
#[tracing::instrument(level = "debug", skip(self))]
async fn read_file_chunks(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<BoxChunkStream> {
if self.health.is_faulty() {
return Err(DiskError::FaultyDisk);
}
let disk = self.disk_ref().await;
let url = format!(
"{}/rustfs/rpc/read_file_stream?disk={}&volume={}&path={}&offset={}&length={}",
self.endpoint.grid_host(),
urlencoding::encode(&disk),
urlencoding::encode(volume),
urlencoding::encode(path),
offset,
length
);
let mut headers = HeaderMap::new();
headers.insert(CONTENT_TYPE, HeaderValue::from_static("application/json"));
build_auth_headers(&url, &Method::GET, &mut headers);
let stream = open_http_byte_stream(url, Method::GET, headers, None)
.await?
.map(|result| result.map(IoChunk::Shared));
Ok(Box::pin(stream))
}
#[tracing::instrument(level = "debug", skip(self))]
async fn append_file(&self, volume: &str, path: &str) -> Result<FileWriter> {
info!("append_file {}/{}", volume, path);
+206 -343
View File
@@ -52,9 +52,9 @@ use crate::{
event_notification::{EventArgs, send_event},
global::{GLOBAL_LOCAL_DISK_MAP, GLOBAL_LOCAL_DISK_SET_DRIVES, get_global_deployment_id, is_dist_erasure},
store_api::{
BucketInfo, BucketOperations, BucketOptions, ChunkNativePutData, CompletePart, DeleteBucketOptions, DeletedObject,
GetObjectReader, HTTPRangeSpec, HealOperations, ListMultipartsInfo, ListObjectsV2Info, ListOperations, MakeBucketOptions,
MultipartInfo, MultipartOperations, MultipartUploadResult, ObjectIO, ObjectInfo, ObjectOperations, PartInfo, StorageAPI,
BucketInfo, BucketOperations, BucketOptions, CompletePart, DeleteBucketOptions, DeletedObject, GetObjectReader,
HTTPRangeSpec, HealOperations, ListMultipartsInfo, ListObjectsV2Info, ListOperations, MakeBucketOptions, MultipartInfo,
MultipartOperations, MultipartUploadResult, ObjectIO, ObjectInfo, ObjectOperations, PartInfo, PutObjReader, StorageAPI,
},
store_init::load_format_erasure,
};
@@ -119,19 +119,18 @@ use tokio::{
time::{interval, timeout},
};
use tokio_util::sync::CancellationToken;
use tracing::error;
use tracing::{debug, info, warn};
use uuid::Uuid;
const ENV_RUSTFS_PUT_INLINE_OBJECT_MAX_BYTES: &str = "RUSTFS_PUT_INLINE_OBJECT_MAX_BYTES";
const ENV_RUSTFS_PUT_FORCE_DISABLE_INLINE: &str = "RUSTFS_PUT_FORCE_DISABLE_INLINE";
const SLOW_PUT_STORAGE_PHASE_DEBUG_THRESHOLD_MS: u64 = 100;
const SLOW_PUT_STORAGE_PHASE_WARN_THRESHOLD_MS: u64 = 1_000;
const SLOW_PUT_STORAGE_PHASE_ERROR_THRESHOLD_MS: u64 = 5_000;
pub const DEFAULT_READ_BUFFER_SIZE: usize = MI_B; // 1 MiB = 1024 * 1024;
pub const MAX_PARTS_COUNT: usize = 10000;
pub(crate) const RUSTFS_MULTIPART_BUCKET_KEY: &str = "x-rustfs-internal-multipart-bucket";
pub(crate) const RUSTFS_MULTIPART_OBJECT_KEY: &str = "x-rustfs-internal-multipart-object";
fn env_flag_enabled(name: &str) -> bool {
rustfs_utils::get_env_bool(name, false)
}
fn env_non_negative_usize(name: &str) -> Option<usize> {
rustfs_utils::get_env_opt_usize(name)
pub(crate) fn strip_internal_multipart_metadata(metadata: &mut HashMap<String, String>) {
metadata.remove(RUSTFS_MULTIPART_BUCKET_KEY);
metadata.remove(RUSTFS_MULTIPART_OBJECT_KEY);
}
fn capacity_scope_from_disks(disks: &[Option<DiskStore>]) -> CapacityScope {
@@ -164,65 +163,6 @@ fn record_capacity_scope_if_needed(scope_token: Option<Uuid>, disks: &[Option<Di
}
}
fn resolved_put_inline_buffer_enabled(object_size: i64, inline_by_topology: bool) -> bool {
if !inline_by_topology || object_size < 0 {
return false;
}
if env_flag_enabled(ENV_RUSTFS_PUT_FORCE_DISABLE_INLINE) {
return false;
}
env_non_negative_usize(ENV_RUSTFS_PUT_INLINE_OBJECT_MAX_BYTES)
.map(|value| usize::try_from(object_size).is_ok_and(|size| size <= value))
.unwrap_or(inline_by_topology)
}
fn log_put_storage_phase(
bucket: &str,
object: &str,
phase: &str,
elapsed: Duration,
object_size: i64,
inline_selected: bool,
write_quorum: usize,
) {
let duration_ms = elapsed.as_millis() as u64;
if duration_ms < SLOW_PUT_STORAGE_PHASE_DEBUG_THRESHOLD_MS {
return;
}
if duration_ms >= SLOW_PUT_STORAGE_PHASE_ERROR_THRESHOLD_MS {
error!(
phase,
duration_ms, object_size, inline_selected, write_quorum, bucket, object, "PUT storage phase is critically slow"
);
} else if duration_ms >= SLOW_PUT_STORAGE_PHASE_WARN_THRESHOLD_MS {
warn!(
phase,
duration_ms, object_size, inline_selected, write_quorum, bucket, object, "PUT storage phase is slow"
);
} else {
debug!(
phase,
duration_ms, object_size, inline_selected, write_quorum, bucket, object, "PUT storage phase exceeded debug threshold"
);
}
}
use tracing::error;
use tracing::{debug, info, warn};
use uuid::Uuid;
pub const DEFAULT_READ_BUFFER_SIZE: usize = MI_B; // 1 MiB = 1024 * 1024;
pub const MAX_PARTS_COUNT: usize = 10000;
pub(crate) const RUSTFS_MULTIPART_BUCKET_KEY: &str = "x-rustfs-internal-multipart-bucket";
pub(crate) const RUSTFS_MULTIPART_OBJECT_KEY: &str = "x-rustfs-internal-multipart-object";
pub(crate) fn strip_internal_multipart_metadata(metadata: &mut HashMap<String, String>) {
metadata.remove(RUSTFS_MULTIPART_BUCKET_KEY);
metadata.remove(RUSTFS_MULTIPART_OBJECT_KEY);
}
/// Get the duplex buffer size from environment variable or use default.
///
/// This function reads `RUSTFS_DUPLEX_BUFFER_SIZE` environment variable
@@ -249,9 +189,6 @@ mod read;
mod replication;
mod write;
#[doc(hidden)]
pub use read::collect_direct_data_shard_chunks_for_benchmark;
/// Get lock acquire timeout from environment variable RUSTFS_LOCK_ACQUIRE_TIMEOUT (in seconds)
/// Defaults to 30 seconds if not set or invalid
pub fn get_lock_acquire_timeout() -> Duration {
@@ -793,13 +730,7 @@ impl ObjectIO for SetDisks {
}
#[tracing::instrument(skip(self, data,))]
async fn put_object(
&self,
bucket: &str,
object: &str,
data: &mut ChunkNativePutData,
opts: &ObjectOptions,
) -> Result<ObjectInfo> {
async fn put_object(&self, bucket: &str, object: &str, data: &mut PutObjReader, opts: &ObjectOptions) -> Result<ObjectInfo> {
let disks = self.get_disks_internal().await;
let mut object_lock_guard = None;
@@ -878,253 +809,205 @@ impl ObjectIO for SetDisks {
let tmp_object = format!("{}/{}/part.1", tmp_dir, fi.data_dir.unwrap());
let erasure = erasure_coding::Erasure::new(fi.erasure.data_blocks, fi.erasure.parity_blocks, fi.erasure.block_size);
let result: Result<ObjectInfo> = async {
let erasure = erasure_coding::Erasure::new(fi.erasure.data_blocks, fi.erasure.parity_blocks, fi.erasure.block_size);
let is_inline_buffer = {
let inline_by_topology = if let Some(sc) = GLOBAL_STORAGE_CLASS.get() {
sc.should_inline(erasure.shard_file_size(data.size()), opts.versioned)
} else {
false
let is_inline_buffer = {
if let Some(sc) = GLOBAL_STORAGE_CLASS.get() {
sc.should_inline(erasure.shard_file_size(data.size()), opts.versioned)
} else {
false
}
};
resolved_put_inline_buffer_enabled(data.size(), inline_by_topology)
};
if is_inline_buffer {
rustfs_io_metrics::record_put_inline_selected(data.size(), opts.versioned);
}
let writer_setup_start = Instant::now();
let mut writers = Vec::with_capacity(shuffle_disks.len());
let mut errors = Vec::with_capacity(shuffle_disks.len());
for disk_op in shuffle_disks.iter() {
if let Some(disk) = disk_op
&& disk.is_online().await
{
let writer = match create_bitrot_writer(
is_inline_buffer,
Some(disk),
RUSTFS_META_TMP_BUCKET,
&tmp_object,
erasure.shard_file_size(data.size()),
erasure.shard_size(),
HashAlgorithm::HighwayHash256S,
)
.await
let mut writers = Vec::with_capacity(shuffle_disks.len());
let mut errors = Vec::with_capacity(shuffle_disks.len());
for disk_op in shuffle_disks.iter() {
if let Some(disk) = disk_op
&& disk.is_online().await
{
Ok(writer) => writer,
Err(err) => {
warn!("create_bitrot_writer disk {}, err {:?}, skipping operation", disk.to_string(), err);
errors.push(Some(err));
writers.push(None);
continue;
}
};
let writer = match create_bitrot_writer(
is_inline_buffer,
Some(disk),
RUSTFS_META_TMP_BUCKET,
&tmp_object,
erasure.shard_file_size(data.size()),
erasure.shard_size(),
HashAlgorithm::HighwayHash256S,
)
.await
{
Ok(writer) => writer,
Err(err) => {
warn!("create_bitrot_writer disk {}, err {:?}, skipping operation", disk.to_string(), err);
errors.push(Some(err));
writers.push(None);
continue;
}
};
writers.push(Some(writer));
errors.push(None);
} else {
errors.push(Some(DiskError::DiskNotFound));
writers.push(None);
}
}
log_put_storage_phase(
bucket,
object,
"writer_setup",
writer_setup_start.elapsed(),
data.size(),
is_inline_buffer,
write_quorum,
);
let nil_count = errors.iter().filter(|&e| e.is_none()).count();
if nil_count < write_quorum {
error!("not enough disks to write: {:?}", errors);
if let Some(write_err) = reduce_write_quorum_errs(&errors, OBJECT_OP_IGNORED_ERRS, write_quorum) {
return Err(to_object_err(write_err.into(), vec![bucket, object]));
writers.push(Some(writer));
errors.push(None);
} else {
errors.push(Some(DiskError::DiskNotFound));
writers.push(None);
}
}
return Err(Error::other(format!("not enough disks to write: {errors:?}")));
}
let object_size = data.size();
let encode_write_start = Instant::now();
let w_size = match Self::write_chunk_native_put_data(data, Arc::new(erasure), &mut writers, write_quorum).await {
Ok(written) => written,
Err(e) => {
log_put_storage_phase(
bucket,
object,
"encode_write",
encode_write_start.elapsed(),
object_size,
is_inline_buffer,
write_quorum,
);
error!("encode err {:?}", e);
return Err(e.into());
}
}; // TODO: delete temporary directory on error
log_put_storage_phase(
bucket,
object,
"encode_write",
encode_write_start.elapsed(),
data.size(),
is_inline_buffer,
write_quorum,
);
if (w_size as i64) < data.size() {
warn!("put_object write size < data.size(), w_size={}, data.size={}", w_size, data.size());
return Err(Error::other(format!(
"put_object write size < data.size(), w_size={}, data.size={}",
w_size,
data.size()
)));
}
if contains_key_str(&user_defined, SUFFIX_COMPRESSION) {
insert_str(&mut user_defined, SUFFIX_COMPRESSION_SIZE, w_size.to_string());
}
let index_op = data.index_bytes();
//TODO: userDefined
let etag = data.resolve_etag().unwrap_or_default();
user_defined.insert("etag".to_owned(), etag.clone());
if !user_defined.contains_key("content-type") {
// get content-type
}
let mut actual_size = data.actual_size();
if actual_size < 0 {
let is_compressed = fi.is_compressed();
if !is_compressed {
actual_size = w_size as i64;
}
}
if fi.checksum.is_none()
&& let Some(content_hash) = data.content_hash_bytes()?
{
fi.checksum = Some(content_hash);
}
if let Some(sc) = user_defined.get(AMZ_STORAGE_CLASS)
&& sc == storageclass::STANDARD
{
let _ = user_defined.remove(AMZ_STORAGE_CLASS);
}
let mod_time = if let Some(mod_time) = opts.mod_time {
Some(mod_time)
} else {
Some(OffsetDateTime::now_utc())
};
for (i, pfi) in parts_metadatas.iter_mut().enumerate() {
pfi.metadata = user_defined.clone();
if is_inline_buffer {
if let Some(writer) = writers[i].take() {
pfi.data = Some(writer.into_inline_data().map(Bytes::from).unwrap_or_default());
let nil_count = errors.iter().filter(|&e| e.is_none()).count();
if nil_count < write_quorum {
error!("not enough disks to write: {:?}", errors);
if let Some(write_err) = reduce_write_quorum_errs(&errors, OBJECT_OP_IGNORED_ERRS, write_quorum) {
return Err(to_object_err(write_err.into(), vec![bucket, object]));
}
pfi.set_inline_data();
return Err(Error::other(format!("not enough disks to write: {errors:?}")));
}
pfi.mod_time = mod_time;
pfi.size = w_size as i64;
pfi.versioned = opts.versioned || opts.version_suspended;
pfi.add_object_part(1, etag.clone(), w_size, mod_time, actual_size, index_op.clone(), None);
pfi.checksum = fi.checksum.clone();
let stream = mem::replace(
&mut data.stream,
HashReader::from_stream(Cursor::new(Vec::new()), 0, 0, None, None, false)?,
);
if opts.data_movement {
pfi.set_data_moved();
let (reader, w_size) = match Arc::new(erasure).encode(stream, &mut writers, write_quorum).await {
Ok((r, w)) => (r, w),
Err(e) => {
error!("encode err {:?}", e);
return Err(e.into());
}
}; // TODO: delete temporary directory on error
let _ = mem::replace(&mut data.stream, reader);
// if let Err(err) = close_bitrot_writers(&mut writers).await {
// error!("close_bitrot_writers err {:?}", err);
// }
if (w_size as i64) < data.size() {
warn!("put_object write size < data.size(), w_size={}, data.size={}", w_size, data.size());
return Err(Error::other(format!(
"put_object write size < data.size(), w_size={}, data.size={}",
w_size,
data.size()
)));
}
}
drop(writers); // drop writers to close all files, this is to prevent FileAccessDenied errors when renaming data
if contains_key_str(&user_defined, SUFFIX_COMPRESSION) {
insert_str(&mut user_defined, SUFFIX_COMPRESSION_SIZE, w_size.to_string());
}
let post_write_lock_start = Instant::now();
if !opts.no_lock && object_lock_guard.is_none() {
let ns_lock = self.new_ns_lock(bucket, object).await?;
object_lock_guard = Some(ns_lock.get_write_lock(get_lock_acquire_timeout()).await.map_err(|e| {
StorageError::other(format!(
"Failed to acquire write lock: {}",
self.format_lock_error_from_error(bucket, object, "write", &e)
))
})?);
}
log_put_storage_phase(
bucket,
object,
"post_write_lock",
post_write_lock_start.elapsed(),
data.size(),
is_inline_buffer,
write_quorum,
);
let index_op = data.stream.try_get_index().map(|v| v.clone().into_vec());
let finalize_start = Instant::now();
let (online_disks, _, op_old_dir) = Self::rename_data(
&shuffle_disks,
RUSTFS_META_TMP_BUCKET,
tmp_dir.as_str(),
&parts_metadatas,
bucket,
object,
write_quorum,
)
.await
.inspect_err(|_| {
log_put_storage_phase(
//TODO: userDefined
let etag = data.stream.try_resolve_etag().unwrap_or_default();
user_defined.insert("etag".to_owned(), etag.clone());
if !user_defined.contains_key("content-type") {
// get content-type
}
let mut actual_size = data.actual_size();
if actual_size < 0 {
let is_compressed = fi.is_compressed();
if !is_compressed {
actual_size = w_size as i64;
}
}
if fi.checksum.is_none()
&& let Some(content_hash) = data.as_hash_reader().content_hash()
{
fi.checksum = Some(content_hash.to_bytes(&[]));
}
if let Some(sc) = user_defined.get(AMZ_STORAGE_CLASS)
&& sc == storageclass::STANDARD
{
let _ = user_defined.remove(AMZ_STORAGE_CLASS);
}
let mod_time = if let Some(mod_time) = opts.mod_time {
Some(mod_time)
} else {
Some(OffsetDateTime::now_utc())
};
for (i, pfi) in parts_metadatas.iter_mut().enumerate() {
pfi.metadata = user_defined.clone();
if is_inline_buffer {
if let Some(writer) = writers[i].take() {
pfi.data = Some(writer.into_inline_data().map(Bytes::from).unwrap_or_default());
}
pfi.set_inline_data();
}
pfi.mod_time = mod_time;
pfi.size = w_size as i64;
pfi.versioned = opts.versioned || opts.version_suspended;
pfi.add_object_part(1, etag.clone(), w_size, mod_time, actual_size, index_op.clone(), None);
pfi.checksum = fi.checksum.clone();
if opts.data_movement {
pfi.set_data_moved();
}
}
drop(writers); // drop writers to close all files, this is to prevent FileAccessDenied errors when renaming data
if !opts.no_lock && object_lock_guard.is_none() {
let ns_lock = self.new_ns_lock(bucket, object).await?;
object_lock_guard = Some(ns_lock.get_write_lock(get_lock_acquire_timeout()).await.map_err(|e| {
StorageError::other(format!(
"Failed to acquire write lock: {}",
self.format_lock_error_from_error(bucket, object, "write", &e)
))
})?);
}
let (online_disks, _, op_old_dir) = Self::rename_data(
&shuffle_disks,
RUSTFS_META_TMP_BUCKET,
tmp_dir.as_str(),
&parts_metadatas,
bucket,
object,
"finalize",
finalize_start.elapsed(),
data.size(),
is_inline_buffer,
write_quorum,
);
})?;
)
.await?;
if let Some(old_dir) = op_old_dir {
self.commit_rename_data_dir(&online_disks, bucket, object, &old_dir.to_string(), write_quorum)
.await?;
}
drop(object_lock_guard); // drop object lock guard to release the lock
self.delete_all(RUSTFS_META_TMP_BUCKET, &tmp_dir).await?;
log_put_storage_phase(
bucket,
object,
"finalize",
finalize_start.elapsed(),
data.size(),
is_inline_buffer,
write_quorum,
);
for (i, op_disk) in online_disks.iter().enumerate() {
if let Some(disk) = op_disk
&& disk.is_online().await
{
fi = parts_metadatas[i].clone();
break;
if let Some(old_dir) = op_old_dir {
self.commit_rename_data_dir(&online_disks, bucket, object, &old_dir.to_string(), write_quorum)
.await?;
}
drop(object_lock_guard); // drop object lock guard to release the lock
for (i, op_disk) in online_disks.iter().enumerate() {
if let Some(disk) = op_disk
&& disk.is_online().await
{
fi = parts_metadatas[i].clone();
break;
}
}
record_capacity_scope_if_needed(opts.capacity_scope_token, &online_disks);
fi.replication_state_internal = Some(opts.put_replication_state());
fi.is_latest = true;
Ok(ObjectInfo::from_file_info(&fi, bucket, object, opts.versioned || opts.version_suspended))
}
.await;
if let Err(err) = self.delete_all(RUSTFS_META_TMP_BUCKET, &tmp_dir).await {
warn!(tmp_dir = %tmp_dir, error = ?err, "failed to cleanup put_object temporary data");
}
record_capacity_scope_if_needed(opts.capacity_scope_token, &online_disks);
fi.replication_state_internal = Some(opts.put_replication_state());
fi.is_latest = true;
Ok(ObjectInfo::from_file_info(&fi, bucket, object, opts.versioned || opts.version_suspended))
result
}
}
@@ -2038,9 +1921,6 @@ impl ObjectOperations for SetDisks {
if let Some(ref version_id) = opts.version_id {
fi.version_id = Uuid::parse_str(version_id).ok();
}
if let Some(checksum) = &opts.resolved_checksum {
fi.checksum = Some(checksum.clone());
}
self.update_object_meta(bucket, object, fi.clone(), &online_disks)
.await
@@ -2269,7 +2149,7 @@ impl ObjectOperations for SetDisks {
let gr = gr.unwrap();
let reader = BufReader::new(gr.stream);
let hash_reader = HashReader::from_stream(reader, gr.object_info.size, gr.object_info.size, None, None, false)?;
let mut p_reader = ChunkNativePutData::new(hash_reader);
let mut p_reader = PutObjReader::new(hash_reader);
return match self_.clone().put_object(bucket, object, &mut p_reader, &ropts).await {
Ok(restored_info) => {
send_event(EventArgs {
@@ -2337,7 +2217,7 @@ impl ObjectOperations for SetDisks {
};
let reader = BufReader::new(gr.stream);
let hash_reader = HashReader::from_stream(reader, part_info.actual_size, part_info.actual_size, None, None, false)?;
let mut p_reader = ChunkNativePutData::new(hash_reader);
let mut p_reader = PutObjReader::new(hash_reader);
let p_info = self_
.clone()
.put_object_part(bucket, object, &res.upload_id, part_info.number, &mut p_reader, &ObjectOptions::default())
@@ -2561,7 +2441,7 @@ impl MultipartOperations for SetDisks {
object: &str,
upload_id: &str,
part_id: usize,
data: &mut ChunkNativePutData,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<PartInfo> {
let upload_id_path = Self::get_upload_id_dir(bucket, object, upload_id);
@@ -2573,8 +2453,9 @@ impl MultipartOperations for SetDisks {
if let Some(checksum) = fi.metadata.get(rustfs_rio::RUSTFS_MULTIPART_CHECKSUM)
&& !checksum.is_empty()
&& data
.as_hash_reader()
.content_crc_type()
.is_none_or(|v: rustfs_rio::ChecksumType| v.to_string() != *checksum)
.is_none_or(|v| v.to_string() != *checksum)
{
return Err(Error::other(format!("checksum mismatch: {checksum}")));
}
@@ -2639,7 +2520,14 @@ impl MultipartOperations for SetDisks {
return Err(Error::other(format!("not enough disks to write: {errors:?}")));
}
let w_size = Self::write_chunk_native_put_data(data, Arc::new(erasure), &mut writers, write_quorum).await?; // TODO: delete temporary directory on error
let stream = mem::replace(
&mut data.stream,
HashReader::from_stream(Cursor::new(Vec::new()), 0, 0, None, None, false)?,
);
let (reader, w_size) = Arc::new(erasure).encode(stream, &mut writers, write_quorum).await?; // TODO: delete temporary directory on error
let _ = mem::replace(&mut data.stream, reader);
if (w_size as i64) < data.size() {
warn!("put_object_part write size < data.size(), w_size={}, data.size={}", w_size, data.size());
@@ -2650,9 +2538,9 @@ impl MultipartOperations for SetDisks {
)));
}
let index_op = data.index_bytes();
let index_op = data.stream.try_get_index().map(|v| v.clone().into_vec());
let mut etag = data.resolve_etag().unwrap_or_default();
let mut etag = data.stream.try_resolve_etag().unwrap_or_default();
if let Some(ref tag) = opts.preserve_etag {
etag = tag.clone();
@@ -2666,7 +2554,7 @@ impl MultipartOperations for SetDisks {
}
}
let checksums = data.content_crc();
let checksums = data.as_hash_reader().content_crc();
let part_info = ObjectPartInfo {
etag: etag.clone(),
@@ -4396,31 +4284,6 @@ mod tests {
}
}
#[test]
#[serial]
fn resolved_put_inline_buffer_enabled_honors_disable_env() {
temp_env::with_var(ENV_RUSTFS_PUT_FORCE_DISABLE_INLINE, Some("true"), || {
assert!(!resolved_put_inline_buffer_enabled(4096, true));
});
}
#[test]
#[serial]
fn resolved_put_inline_buffer_enabled_honors_max_bytes_override() {
temp_env::with_var(ENV_RUSTFS_PUT_INLINE_OBJECT_MAX_BYTES, Some("4096"), || {
assert!(resolved_put_inline_buffer_enabled(4096, true));
assert!(!resolved_put_inline_buffer_enabled(4097, true));
});
}
#[test]
#[serial]
fn resolved_put_inline_buffer_enabled_ignores_invalid_override() {
temp_env::with_var(ENV_RUSTFS_PUT_INLINE_OBJECT_MAX_BYTES, Some("invalid"), || {
assert!(resolved_put_inline_buffer_enabled(4096, true));
});
}
async fn current_setup_type() -> SetupType {
if is_dist_erasure().await {
SetupType::DistErasure
File diff suppressed because it is too large Load Diff
-132
View File
@@ -13,87 +13,12 @@
// limitations under the License.
use super::*;
use crate::store_api::ChunkNativePutData;
impl SetDisks {
fn all_inline_bitrot_writers(writers: &[Option<crate::erasure_coding::BitrotWriterWrapper>]) -> bool {
writers.iter().all(|writer| {
writer
.as_ref()
.is_some_and(crate::erasure_coding::BitrotWriterWrapper::is_inline_buffer)
})
}
async fn write_chunk_native_put_data_inline(
data: &mut ChunkNativePutData,
erasure: Arc<erasure_coding::Erasure>,
writers: &mut [Option<crate::erasure_coding::BitrotWriterWrapper>],
write_quorum: usize,
) -> std::io::Result<usize> {
let stream = data.take_stream()?;
let mut assembler = erasure_coding::encode::BlockAssembler::new(stream, erasure.block_size);
let encoder = erasure_coding::encode::ErasureChunkEncoder::new(erasure).await;
let mut writer_group = erasure_coding::encode::MultiWriter::new(writers, write_quorum);
loop {
let block = match assembler.next_block().await {
Ok(Some(block)) => block,
Ok(None) => break,
Err(err) => {
data.restore_stream(assembler.into_inner());
return Err(err);
}
};
let encoded = match encoder.encode_block(&block).await {
Ok(encoded) => encoded,
Err(err) => {
data.restore_stream(assembler.into_inner());
return Err(err);
}
};
if let Err(err) = writer_group.write_inline(&encoded) {
encoder.release(encoded).await;
data.restore_stream(assembler.into_inner());
return Err(err);
}
encoder.release(encoded).await;
}
let total_bytes = assembler.total_bytes();
let stream = assembler.into_inner();
if let Err(err) = writer_group.shutdown_inline() {
data.restore_stream(stream);
return Err(err);
}
data.restore_stream(stream);
Ok(total_bytes)
}
pub(super) fn default_read_quorum(&self) -> usize {
self.set_drive_count - self.default_parity_count
}
pub(super) async fn write_chunk_native_put_data(
data: &mut ChunkNativePutData,
erasure: Arc<erasure_coding::Erasure>,
writers: &mut [Option<crate::erasure_coding::BitrotWriterWrapper>],
write_quorum: usize,
) -> std::io::Result<usize> {
if Self::all_inline_bitrot_writers(writers) {
return Self::write_chunk_native_put_data_inline(data, erasure, writers, write_quorum).await;
}
let stream = data.take_stream()?;
let (stream, written) = erasure_coding::encode::ErasureWritePipeline::new(erasure, write_quorum)
.run(stream, writers)
.await?;
data.restore_stream(stream);
Ok(written)
}
pub(super) fn default_write_quorum(&self) -> usize {
let mut data_count = self.set_drive_count - self.default_parity_count;
if data_count == self.default_parity_count {
@@ -702,60 +627,3 @@ impl SetDisks {
None
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::erasure_coding::{BitrotWriterWrapper, CustomWriter};
use crate::store_api::ChunkNativePutData;
#[tokio::test]
async fn write_chunk_native_put_data_restores_reader_state_after_encoding() {
let payload = b"chunk-native-put-payload".repeat(8);
let erasure = Arc::new(erasure_coding::Erasure::new(2, 1, 8));
let mut reader = ChunkNativePutData::from_vec(payload.clone());
let mut writers: Vec<Option<BitrotWriterWrapper>> = (0..erasure.total_shard_count())
.map(|_| {
Some(BitrotWriterWrapper::new(
CustomWriter::new_inline_buffer(),
erasure.shard_size(),
HashAlgorithm::HighwayHash256S,
))
})
.collect();
let written = SetDisks::write_chunk_native_put_data(&mut reader, erasure.clone(), &mut writers, 2)
.await
.unwrap();
assert_eq!(written, payload.len());
assert_eq!(reader.size(), payload.len() as i64);
assert_eq!(reader.actual_size(), payload.len() as i64);
assert!(
reader.resolve_etag().is_some(),
"restored reader should preserve computed etag state after chunk-native encode"
);
let inline_lengths: Vec<usize> = writers
.into_iter()
.map(|writer| writer.expect("writer").into_inline_data().expect("inline data").len())
.collect();
assert!(inline_lengths.iter().all(|len| *len > 0));
}
#[tokio::test]
async fn write_chunk_native_put_data_detects_inline_writer_set() {
let erasure = Arc::new(erasure_coding::Erasure::new(2, 1, 8));
let writers: Vec<Option<BitrotWriterWrapper>> = (0..erasure.total_shard_count())
.map(|_| {
Some(BitrotWriterWrapper::new(
CustomWriter::new_inline_buffer(),
erasure.shard_size(),
HashAlgorithm::HighwayHash256S,
))
})
.collect();
assert!(SetDisks::all_inline_bitrot_writers(&writers));
}
}
+7 -26
View File
@@ -15,7 +15,7 @@
use crate::disk::error_reduce::count_errs;
use crate::error::{Error, Result};
use crate::store_api::{GetObjectChunkResult, ListPartsInfo, ObjectInfoOrErr, WalkOptions};
use crate::store_api::{ListPartsInfo, ObjectInfoOrErr, WalkOptions};
use crate::{
disk::{
DiskAPI, DiskInfo, DiskOption, DiskStore,
@@ -28,10 +28,10 @@ use crate::{
global::{GLOBAL_LOCAL_DISK_SET_DRIVES, get_global_lock_clients, is_dist_erasure},
set_disk::SetDisks,
store_api::{
BucketInfo, BucketOperations, BucketOptions, ChunkNativePutData, CompletePart, DeleteBucketOptions, DeletedObject,
GetObjectReader, HTTPRangeSpec, HealOperations, ListMultipartsInfo, ListObjectVersionsInfo, ListObjectsV2Info,
ListOperations, MakeBucketOptions, MultipartInfo, MultipartOperations, MultipartUploadResult, ObjectIO, ObjectInfo,
ObjectOperations, ObjectOptions, ObjectToDelete, PartInfo, StorageAPI,
BucketInfo, BucketOperations, BucketOptions, CompletePart, DeleteBucketOptions, DeletedObject, GetObjectReader,
HTTPRangeSpec, HealOperations, ListMultipartsInfo, ListObjectVersionsInfo, ListObjectsV2Info, ListOperations,
MakeBucketOptions, MultipartInfo, MultipartOperations, MultipartUploadResult, ObjectIO, ObjectInfo, ObjectOperations,
ObjectOptions, ObjectToDelete, PartInfo, PutObjReader, StorageAPI,
},
store_init::{check_format_erasure_values, get_format_erasure_in_quorum, load_format_erasure_all, save_format_file},
};
@@ -287,19 +287,6 @@ impl Sets {
self.get_disks(self.get_hashed_set_index(key))
}
pub async fn get_object_chunks(
&self,
bucket: &str,
object: &str,
range: Option<HTTPRangeSpec>,
h: HeaderMap,
opts: &ObjectOptions,
) -> Result<GetObjectChunkResult> {
self.get_disks_by_key(object)
.get_object_chunks(bucket, object, range, h, opts)
.await
}
fn get_hashed_set_index(&self, input: &str) -> usize {
match self.distribution_algo {
DistributionAlgoVersion::V1 => crc_hash(input, self.disk_set.len()),
@@ -388,13 +375,7 @@ impl ObjectIO for Sets {
.await
}
#[tracing::instrument(level = "debug", skip(self, data))]
async fn put_object(
&self,
bucket: &str,
object: &str,
data: &mut ChunkNativePutData,
opts: &ObjectOptions,
) -> Result<ObjectInfo> {
async fn put_object(&self, bucket: &str, object: &str, data: &mut PutObjReader, opts: &ObjectOptions) -> Result<ObjectInfo> {
self.get_disks_by_key(object).put_object(bucket, object, data, opts).await
}
}
@@ -707,7 +688,7 @@ impl MultipartOperations for Sets {
object: &str,
upload_id: &str,
part_id: usize,
data: &mut ChunkNativePutData,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<PartInfo> {
self.get_disks_by_key(object)
+5 -26
View File
@@ -59,10 +59,9 @@ use crate::{
rpc::S3PeerSys,
sets::Sets,
store_api::{
BucketInfo, BucketOperations, BucketOptions, ChunkNativePutData, CompletePart, DeleteBucketOptions, DeletedObject,
GetObjectChunkResult, GetObjectReader, HTTPRangeSpec, HealOperations, ListObjectsV2Info, ListOperations,
MakeBucketOptions, MultipartOperations, MultipartUploadResult, ObjectInfo, ObjectOperations, ObjectOptions,
ObjectToDelete, PartInfo, StorageAPI,
BucketInfo, BucketOperations, BucketOptions, CompletePart, DeleteBucketOptions, DeletedObject, GetObjectReader,
HTTPRangeSpec, HealOperations, ListObjectsV2Info, ListOperations, MakeBucketOptions, MultipartOperations,
MultipartUploadResult, ObjectInfo, ObjectOperations, ObjectOptions, ObjectToDelete, PartInfo, PutObjReader, StorageAPI,
},
store_init,
};
@@ -260,31 +259,11 @@ impl ObjectIO for ECStore {
self.handle_get_object_reader(bucket, object, range, h, opts).await
}
#[instrument(level = "debug", skip(self, data))]
async fn put_object(
&self,
bucket: &str,
object: &str,
data: &mut ChunkNativePutData,
opts: &ObjectOptions,
) -> Result<ObjectInfo> {
async fn put_object(&self, bucket: &str, object: &str, data: &mut PutObjReader, opts: &ObjectOptions) -> Result<ObjectInfo> {
enqueue_transition_after_write(self.handle_put_object(bucket, object, data, opts).await, LcEventSrc::S3PutObject).await
}
}
impl ECStore {
#[instrument(level = "debug", skip(self))]
pub async fn get_object_chunks(
&self,
bucket: &str,
object: &str,
range: Option<HTTPRangeSpec>,
h: HeaderMap,
opts: &ObjectOptions,
) -> Result<GetObjectChunkResult> {
self.handle_get_object_chunks(bucket, object, range, h, opts).await
}
}
lazy_static! {
static ref enableObjcetLockConfig: ObjectLockConfiguration = ObjectLockConfiguration {
object_lock_enabled: Some(ObjectLockEnabled::from_static(ObjectLockEnabled::ENABLED)),
@@ -516,7 +495,7 @@ impl MultipartOperations for ECStore {
object: &str,
upload_id: &str,
part_id: usize,
data: &mut ChunkNativePutData,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<PartInfo> {
self.handle_put_object_part(bucket, object, upload_id, part_id, data, opts)
+1 -1
View File
@@ -176,7 +176,7 @@ impl ECStore {
object: &str,
upload_id: &str,
part_id: usize,
data: &mut ChunkNativePutData,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<PartInfo> {
check_put_object_part_args(bucket, object, upload_id)?;
+1 -31
View File
@@ -13,8 +13,6 @@
// limitations under the License.
use super::*;
use crate::store_api::GetObjectChunkResult;
fn select_data_movement_target_pool(
existing_pool_idx: Result<usize>,
src_pool_idx: usize,
@@ -214,40 +212,12 @@ impl ECStore {
.await
}
#[instrument(level = "debug", skip(self))]
pub(super) async fn handle_get_object_chunks(
&self,
bucket: &str,
object: &str,
range: Option<HTTPRangeSpec>,
h: HeaderMap,
opts: &ObjectOptions,
) -> Result<GetObjectChunkResult> {
check_get_obj_args(bucket, object)?;
let object = encode_dir_object(object);
if self.single_pool() {
return self.pools[0].get_object_chunks(bucket, object.as_str(), range, h, opts).await;
}
let mut opts = opts.clone();
opts.no_lock = true;
let (_, idx) = self
.get_latest_accessible_object_info_with_idx(bucket, &object, &opts)
.await?;
self.pools[idx]
.get_object_chunks(bucket, object.as_str(), range, h, &opts)
.await
}
#[instrument(level = "debug", skip(self, data))]
pub(super) async fn handle_put_object(
&self,
bucket: &str,
object: &str,
data: &mut ChunkNativePutData,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<ObjectInfo> {
check_put_object_args(bucket, object)?;
+3 -3
View File
@@ -24,9 +24,9 @@ fn pool_lookup_not_found_error(bucket: &str, object: &str, opts: &ObjectOptions)
let object = decode_dir_object(object);
if let Some(version_id) = &opts.version_id {
StorageError::VersionNotFound(bucket.to_owned(), object.to_owned(), version_id.clone())
StorageError::VersionNotFound(bucket.to_owned(), object, version_id.clone())
} else {
StorageError::ObjectNotFound(bucket.to_owned(), object.to_owned())
StorageError::ObjectNotFound(bucket.to_owned(), object)
}
}
@@ -418,7 +418,7 @@ impl ECStore {
if is_err_object_not_found(err)
&& let Err(err) = opts.precondition_check(&pinfo.object_info)
{
return Err(err.clone());
return Err(err);
}
if !is_err_object_not_found(err) && !is_err_version_not_found(err) {
-1
View File
@@ -31,7 +31,6 @@ use rustfs_filemeta::{
RestoreStatusOps as _, VersionPurgeStatusType, parse_restore_obj_status, replication_statuses_map,
version_purge_statuses_map,
};
use rustfs_io_core::BoxChunkStream;
use rustfs_lock::NamespaceLockWrapper;
use rustfs_madmin::heal_commands::HealResultItem;
use rustfs_rio::Checksum;
+79 -202
View File
@@ -1,101 +1,7 @@
use super::*;
use rustfs_rio::TryGetIndex;
pub struct ChunkNativePutData {
stream: Option<HashReader>,
size: i64,
actual_size: i64,
}
impl Debug for ChunkNativePutData {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.debug_struct("ChunkNativePutData").finish()
}
}
impl ChunkNativePutData {
pub fn new(stream: HashReader) -> Self {
let size = stream.size();
let actual_size = stream.actual_size();
Self {
stream: Some(stream),
size,
actual_size,
}
}
pub fn from_vec(data: Vec<u8>) -> Self {
use sha2::{Digest, Sha256};
let content_length = data.len() as i64;
let sha256hex = if content_length > 0 {
Some(hex_simd::encode_to_string(Sha256::digest(&data), hex_simd::AsciiCase::Lower))
} else {
None
};
Self::new(HashReader::from_stream(Cursor::new(data), content_length, content_length, None, sha256hex, false).unwrap())
}
pub fn take_stream(&mut self) -> std::io::Result<HashReader> {
self.stream
.take()
.ok_or_else(|| std::io::Error::other("ChunkNativePutData stream already taken"))
}
pub fn restore_stream(&mut self, stream: HashReader) {
self.size = stream.size();
self.actual_size = stream.actual_size();
self.stream = Some(stream);
}
pub fn as_hash_reader(&self) -> Option<&HashReader> {
self.stream.as_ref()
}
pub fn as_hash_reader_mut(&mut self) -> Option<&mut HashReader> {
self.stream.as_mut()
}
pub fn index_bytes(&self) -> Option<Bytes> {
self.as_hash_reader()
.and_then(|reader| reader.try_get_index().map(|index| index.clone().into_vec()))
}
pub fn resolve_etag(&mut self) -> Option<String> {
self.as_hash_reader_mut()
.and_then(rustfs_rio::EtagResolvable::try_resolve_etag)
}
pub fn content_hash_bytes(&mut self) -> std::io::Result<Option<Bytes>> {
let Some(reader) = self.as_hash_reader_mut() else {
return Ok(None);
};
Ok(reader
.finalize_content_hash()?
.as_ref()
.map(|checksum| checksum.to_bytes(&[])))
}
pub fn content_crc_type(&self) -> Option<rustfs_rio::ChecksumType> {
self.as_hash_reader().and_then(HashReader::content_crc_type)
}
pub fn content_crc(&self) -> HashMap<String, String> {
self.as_hash_reader().map_or_else(HashMap::new, HashReader::content_crc)
}
pub fn size(&self) -> i64 {
self.size
}
pub fn actual_size(&self) -> i64 {
self.actual_size
}
}
pub struct PutObjReader {
data: ChunkNativePutData,
pub stream: HashReader,
}
impl Debug for PutObjReader {
@@ -106,81 +12,32 @@ impl Debug for PutObjReader {
impl PutObjReader {
pub fn new(stream: HashReader) -> Self {
Self {
data: ChunkNativePutData::new(stream),
}
PutObjReader { stream }
}
pub fn chunk_native_data(&self) -> &ChunkNativePutData {
&self.data
}
pub fn chunk_native_data_mut(&mut self) -> &mut ChunkNativePutData {
&mut self.data
}
pub fn take_stream(&mut self) -> std::io::Result<HashReader> {
self.data.take_stream()
}
pub fn restore_stream(&mut self, stream: HashReader) {
self.data.restore_stream(stream);
}
pub fn as_hash_reader(&self) -> Option<&HashReader> {
self.data.as_hash_reader()
}
pub fn as_hash_reader_mut(&mut self) -> Option<&mut HashReader> {
self.data.as_hash_reader_mut()
}
pub fn index_bytes(&self) -> Option<Bytes> {
self.data.index_bytes()
}
pub fn resolve_etag(&mut self) -> Option<String> {
self.data.resolve_etag()
}
pub fn content_hash_bytes(&mut self) -> std::io::Result<Option<Bytes>> {
self.data.content_hash_bytes()
}
pub fn content_crc_type(&self) -> Option<rustfs_rio::ChecksumType> {
self.data.content_crc_type()
}
pub fn content_crc(&self) -> HashMap<String, String> {
self.data.content_crc()
pub fn as_hash_reader(&self) -> &HashReader {
&self.stream
}
pub fn from_vec(data: Vec<u8>) -> Self {
Self {
data: ChunkNativePutData::from_vec(data),
use sha2::{Digest, Sha256};
let content_length = data.len() as i64;
let sha256hex = if content_length > 0 {
Some(hex_simd::encode_to_string(Sha256::digest(&data), hex_simd::AsciiCase::Lower))
} else {
None
};
PutObjReader {
stream: HashReader::from_stream(Cursor::new(data), content_length, content_length, None, sha256hex, false).unwrap(),
}
}
pub fn size(&self) -> i64 {
self.data.size()
self.stream.size()
}
pub fn actual_size(&self) -> i64 {
self.data.actual_size()
}
}
impl std::ops::Deref for PutObjReader {
type Target = ChunkNativePutData;
fn deref(&self) -> &Self::Target {
&self.data
}
}
impl std::ops::DerefMut for PutObjReader {
fn deref_mut(&mut self) -> &mut Self::Target {
&mut self.data
self.stream.actual_size()
}
}
@@ -189,26 +46,6 @@ pub struct GetObjectReader {
pub object_info: ObjectInfo,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum GetObjectChunkPath {
Direct,
Bridge,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum GetObjectChunkCopyMode {
TrueZeroCopy,
SharedBytes,
SingleCopy,
Reconstructed,
}
pub struct GetObjectChunkResult {
pub stream: BoxChunkStream,
pub path: GetObjectChunkPath,
pub copy_mode: GetObjectChunkCopyMode,
}
impl GetObjectReader {
#[tracing::instrument(level = "debug", skip(reader, rs, opts, _h))]
pub fn new(
@@ -226,34 +63,31 @@ impl GetObjectReader {
rs = HTTPRangeSpec::from_object_info(oi, part_number);
}
let logical_size = oi.get_actual_size()?;
let encrypted_object = oi.user_defined.contains_key("x-rustfs-encryption-key")
|| oi
.user_defined
.contains_key("x-amz-server-side-encryption-customer-algorithm");
// TODO:Encrypted
let (algo, is_compressed) = oi.is_compressed_ok()?;
// TODO: check TRANSITION
if is_compressed {
let actual_size = oi.get_actual_size()?;
let (off, length, dec_off, dec_length) = if let Some(rs) = rs {
// Support range requests for compressed objects
let (dec_off, dec_length) = rs.get_offset_length(logical_size)?;
let (dec_off, dec_length) = rs.get_offset_length(actual_size)?;
(0, oi.size, dec_off, dec_length)
} else {
(0, oi.size, 0, logical_size)
(0, oi.size, 0, actual_size)
};
let dec_reader = DecompressReader::new(reader, algo);
let actual_size_usize = if logical_size > 0 {
logical_size as usize
let actual_size_usize = if actual_size > 0 {
actual_size as usize
} else {
return Err(Error::other(format!("invalid decompressed size {logical_size}")));
return Err(Error::other(format!("invalid decompressed size {actual_size}")));
};
let final_reader: Box<dyn AsyncRead + Unpin + Send + Sync> = if dec_off > 0 || dec_length != logical_size {
let final_reader: Box<dyn AsyncRead + Unpin + Send + Sync> = if dec_off > 0 || dec_length != actual_size {
// Use RangedDecompressReader for streaming range processing
// The new implementation supports any offset size by streaming and skipping data
match RangedDecompressReader::new(dec_reader, dec_off, dec_length, actual_size_usize) {
@@ -288,19 +122,8 @@ impl GetObjectReader {
));
}
if encrypted_object && rs.is_none() {
return Ok((
GetObjectReader {
stream: reader,
object_info: oi.clone(),
},
0,
oi.size,
));
}
if let Some(rs) = rs {
let (off, length) = rs.get_offset_length(logical_size)?;
let (off, length) = rs.get_offset_length(oi.size)?;
Ok((
GetObjectReader {
@@ -317,7 +140,7 @@ impl GetObjectReader {
object_info: oi.clone(),
},
0,
logical_size,
oi.size,
))
}
}
@@ -852,4 +675,58 @@ mod tests {
assert_eq!(n2, 1);
assert_eq!(&buf2[..1], b"e");
}
#[test]
fn test_get_object_reader_range_uses_stored_size_for_encrypted_metadata() {
let object_info = ObjectInfo {
size: 10,
user_defined: HashMap::from([("x-amz-server-side-encryption-customer-original-size".to_string(), "20".to_string())]),
..Default::default()
};
let range = HTTPRangeSpec {
is_suffix_length: false,
start: 8,
end: -1,
};
let (_, offset, length) = GetObjectReader::new(
Box::new(Cursor::new(b"0123456789".to_vec())),
Some(range),
&object_info,
&ObjectOptions::default(),
&HeaderMap::new(),
)
.unwrap();
assert_eq!(offset, 8);
assert_eq!(length, 2);
}
#[test]
fn test_get_object_reader_suffix_range_uses_stored_size_for_encrypted_metadata() {
let object_info = ObjectInfo {
size: 10,
user_defined: HashMap::from([("x-rustfs-encryption-original-size".to_string(), "20".to_string())]),
..Default::default()
};
let range = HTTPRangeSpec {
is_suffix_length: true,
start: 4,
end: -1,
};
let (_, offset, length) = GetObjectReader::new(
Box::new(Cursor::new(b"0123456789".to_vec())),
Some(range),
&object_info,
&ObjectOptions::default(),
&HeaderMap::new(),
)
.unwrap();
assert_eq!(offset, 6);
assert_eq!(length, 4);
}
}
+2 -8
View File
@@ -11,13 +11,7 @@ pub trait ObjectIO: Send + Sync + Debug + 'static {
opts: &ObjectOptions,
) -> Result<GetObjectReader>;
async fn put_object(
&self,
bucket: &str,
object: &str,
data: &mut ChunkNativePutData,
opts: &ObjectOptions,
) -> Result<ObjectInfo>;
async fn put_object(&self, bucket: &str, object: &str, data: &mut PutObjReader, opts: &ObjectOptions) -> Result<ObjectInfo>;
}
/// Bucket-level storage operations.
@@ -132,7 +126,7 @@ pub trait MultipartOperations: Send + Sync + Debug {
object: &str,
upload_id: &str,
part_id: usize,
data: &mut ChunkNativePutData,
data: &mut PutObjReader,
opts: &ObjectOptions,
) -> Result<PartInfo>;
async fn get_multipart_info(
+1 -15
View File
@@ -70,7 +70,6 @@ pub struct ObjectOptions {
pub eval_metadata: Option<HashMap<String, String>>,
pub resolved_checksum: Option<Bytes>,
pub want_checksum: Option<Checksum>,
pub skip_verify_bitrot: bool,
pub capacity_scope_token: Option<Uuid>,
@@ -285,7 +284,7 @@ pub struct ObjectInfo {
pub expires: Option<OffsetDateTime>,
pub num_versions: usize,
pub successor_mod_time: Option<OffsetDateTime>,
pub put_object_reader: Option<ChunkNativePutData>,
pub put_object_reader: Option<PutObjReader>,
pub etag: Option<String>,
pub inlined: bool,
pub metadata_only: bool,
@@ -511,18 +510,6 @@ impl ObjectInfo {
})
.collect();
let actual_size = fi
.parts
.iter()
.map(|part| {
if part.actual_size > 0 {
part.actual_size
} else {
i64::try_from(part.size).unwrap_or_default()
}
})
.sum();
// TODO: part checksums
ObjectInfo {
@@ -535,7 +522,6 @@ impl ObjectInfo {
delete_marker: fi.deleted,
mod_time: fi.mod_time,
size: fi.size,
actual_size,
parts,
is_latest: fi.is_latest,
user_tags,
+5 -5
View File
@@ -51,7 +51,7 @@ use crate::{
disk::{MIGRATING_META_BUCKET, RUSTFS_META_BUCKET},
global::is_first_cluster_node_local,
store::ECStore,
store_api::{ChunkNativePutData, ObjectIO as _, ObjectOptions},
store_api::{ObjectIO as _, ObjectOptions, PutObjReader},
};
use rustfs_rio::HashReader;
use rustfs_utils::path::{SLASH_SEPARATOR, path_join};
@@ -508,7 +508,7 @@ fn from_external_tier_config(name: String, ext: ExternalTierConfig) -> io::Resul
} else {
ext.version.clone()
},
name: if ext.name.is_empty() { name.clone() } else { ext.name.clone() },
name: if ext.name.is_empty() { name } else { ext.name.clone() },
..Default::default()
};
@@ -1046,7 +1046,7 @@ impl TierConfigMgr {
opts: &ObjectOptions,
) -> std::result::Result<(), std::io::Error> {
debug!("save tier config:{}", file);
let mut put_data = ChunkNativePutData::from_vec(data.to_vec());
let mut put_data = PutObjReader::from_vec(data.to_vec());
let _ = api.put_object(RUSTFS_META_BUCKET, file, &mut put_data, opts).await?;
Ok(())
}
@@ -1103,7 +1103,7 @@ async fn load_tier_config(api: Arc<ECStore>) -> std::result::Result<TierConfigMg
Ok(data) => {
let cfg = TierConfigMgr::unmarshal(&data)?;
let normalized = encode_external_tiering_config_blob(&cfg)?;
let mut put_data = ChunkNativePutData::from_vec(normalized.to_vec());
let mut put_data = PutObjReader::from_vec(normalized.to_vec());
let _ = api
.put_object(
RUSTFS_META_BUCKET,
@@ -1158,7 +1158,7 @@ async fn read_tier_config_from_bucket<S: StorageAPI>(
}
async fn write_tier_config_to_rustfs<S: StorageAPI>(api: Arc<S>, path: &str, data: Bytes) -> io::Result<()> {
let mut put_data = ChunkNativePutData::from_vec(data.to_vec());
let mut put_data = PutObjReader::from_vec(data.to_vec());
api.put_object(
RUSTFS_META_BUCKET,
path,
+32 -7
View File
@@ -230,13 +230,13 @@ impl FileMeta {
}
}
if let Some(checksum) = fi.checksum.as_ref() {
insert_bytes(&mut obj.meta_sys, SUFFIX_CRC, checksum.to_vec());
}
if let Some(mod_time) = fi.mod_time {
obj.mod_time = Some(mod_time);
}
if let Some(content_hash) = fi.checksum.as_ref() {
insert_bytes(&mut obj.meta_sys, SUFFIX_CRC, content_hash.to_vec());
}
}
// Update
@@ -1488,12 +1488,12 @@ mod test {
}
// Verify stable ordering
let original_order: Vec<_> = fm.versions.iter().map(|v| v.header.version_id).collect();
let original_order = fm.versions.iter().map(|v| v.header.version_id).len();
fm.sort_by_mod_time();
let sorted_order: Vec<_> = fm.versions.iter().map(|v| v.header.version_id).collect();
let sorted_order = fm.versions.iter().map(|v| v.header.version_id).len();
// Sorting should remain stable for identical timestamps
assert_eq!(original_order.len(), sorted_order.len());
assert_eq!(original_order, sorted_order);
}
#[test]
@@ -1725,6 +1725,30 @@ mod test {
assert_eq!(after, Some(Bytes::from_static(b"inline").to_vec()));
}
#[test]
fn test_update_object_version_persists_checksum_metadata() {
let mut fm = FileMeta::new();
let version_id = Some(Uuid::new_v4());
let mut fi = crate::fileinfo::FileInfo::new("test", 2, 1);
fi.version_id = version_id;
fi.mod_time = Some(OffsetDateTime::now_utc());
fm.add_version(fi).unwrap();
let checksum = Bytes::from_static(b"resolved-checksum");
let mut update = crate::fileinfo::FileInfo::new("test", 2, 1);
update.version_id = version_id;
update.metadata.insert("x-amz-meta-owner".to_string(), "alice".to_string());
update.checksum = Some(checksum.clone());
fm.update_object_version(update).unwrap();
let (_, version) = fm.find_version(version_id).unwrap();
let stored = version.into_fileinfo("bucket", "test", true);
assert_eq!(stored.metadata.get("x-amz-meta-owner"), Some(&"alice".to_string()));
assert_eq!(stored.checksum, Some(checksum));
}
#[test]
fn test_version_merge_scenarios() {
// Test various version merge scenarios
@@ -1962,6 +1986,7 @@ async fn test_read_xl_meta_no_data() {
let mut file = File::create(&filepath).await.unwrap();
// Write string data
file.write_all(&buff).await.unwrap();
file.flush().await.unwrap();
let mut f = File::open(&filepath).await.unwrap();
+1 -1
View File
@@ -208,7 +208,7 @@ impl HealStorageAPI for ECStoreHealStorage {
async fn put_object_data(&self, bucket: &str, object: &str, data: &[u8]) -> Result<()> {
debug!("Putting object data: {}/{} ({} bytes)", bucket, object, data.len());
let mut reader = rustfs_ecstore::store_api::ChunkNativePutData::from_vec(data.to_vec());
let mut reader = rustfs_ecstore::store_api::PutObjReader::from_vec(data.to_vec());
match (*self.ecstore)
.put_object(bucket, object, &mut reader, &Default::default())
.await
+2 -2
View File
@@ -18,7 +18,7 @@ use rustfs_ecstore::{
disk::endpoint::Endpoint,
endpoints::{EndpointServerPools, Endpoints, PoolEndpoints},
store::ECStore,
store_api::{BucketOperations, ChunkNativePutData, ObjectIO, ObjectOperations, ObjectOptions},
store_api::{BucketOperations, ObjectIO, ObjectOperations, ObjectOptions, PutObjReader},
};
use rustfs_heal::heal::{
manager::{HealConfig, HealManager},
@@ -162,7 +162,7 @@ async fn create_test_bucket(ecstore: &Arc<ECStore>, bucket_name: &str) {
/// Test helper: Upload test object
async fn upload_test_object(ecstore: &Arc<ECStore>, bucket: &str, object: &str, data: &[u8]) {
let mut reader = ChunkNativePutData::from_vec(data.to_vec());
let mut reader = PutObjReader::from_vec(data.to_vec());
let object_info = (**ecstore)
.put_object(bucket, object, &mut reader, &ObjectOptions::default())
.await
+4 -4
View File
@@ -450,7 +450,7 @@ where
self.cache.policy_docs.store(Arc::new(cache));
let items: Vec<_> = m.into_iter().map(|(k, v)| (k, v.policy.clone())).collect();
let items: Vec<_> = m.into_iter().map(|(k, v)| (k, v.policy)).collect();
let futures: Vec<_> = items.iter().map(|(_, policy)| policy.match_resource(bucket_name)).collect();
@@ -516,7 +516,7 @@ where
self.cache.policy_docs.store(Arc::new(cache));
let items: Vec<_> = m.into_iter().map(|(k, v)| (k, v.clone())).collect();
let items: Vec<_> = m.into_iter().collect();
let futures: Vec<_> = items
.iter()
@@ -1876,7 +1876,7 @@ fn set_default_canned_policies(policies: &mut HashMap<String, PolicyDoc>) {
pub fn get_token_signing_key() -> Option<String> {
if let Some(s) = get_global_action_cred() {
Some(s.secret_key.clone())
Some(s.secret_key)
} else {
None
}
@@ -2201,7 +2201,7 @@ mod tests {
name: Some("service-account-name".to_string()),
description: Some("Updated service account".to_string()),
expiration: None,
session_policy: Some(policy.clone()),
session_policy: Some(policy),
};
assert_eq!(opts.secret_key, Some("new-secret-key".to_string()));
+94 -2
View File
@@ -119,6 +119,7 @@ pub struct OidcProviderConfig {
pub role_policy: String,
pub display_name: String,
pub groups_claim: String,
pub roles_claim: String,
pub email_claim: String,
pub username_claim: String,
}
@@ -385,7 +386,7 @@ impl OidcSys {
sub: extract_string_claim(&raw, "sub"),
email: extract_string_claim(&raw, &config.email_claim),
username: extract_string_claim(&raw, &config.username_claim),
groups: extract_groups_claim(&raw, &config.groups_claim),
groups: extract_canonical_group_values(&raw, &config.groups_claim, &config.roles_claim),
raw,
};
@@ -509,7 +510,7 @@ impl OidcSys {
sub: extract_string_claim(&raw_claims, "sub"),
email: extract_string_claim(&raw_claims, &config.email_claim),
username: extract_string_claim(&raw_claims, &config.username_claim),
groups: extract_groups_claim(&raw_claims, &config.groups_claim),
groups: extract_canonical_group_values(&raw_claims, &config.groups_claim, &config.roles_claim),
raw: raw_claims,
};
@@ -719,6 +720,7 @@ impl OidcSys {
v
}
};
let roles_claim = get_env(ENV_IDENTITY_OPENID_ROLES_CLAIM);
let email_claim = {
let v = get_env(ENV_IDENTITY_OPENID_EMAIL_CLAIM);
if v.is_empty() {
@@ -762,6 +764,7 @@ impl OidcSys {
role_policy: get_env(ENV_IDENTITY_OPENID_ROLE_POLICY),
display_name,
groups_claim,
roles_claim,
email_claim,
username_claim,
})
@@ -800,6 +803,9 @@ impl OidcSys {
let groups_claim = kvs
.lookup(OIDC_GROUPS_CLAIM)
.unwrap_or_else(|| OIDC_DEFAULT_GROUPS_CLAIM.to_string());
let roles_claim = kvs
.lookup(OIDC_ROLES_CLAIM)
.unwrap_or_else(|| OIDC_DEFAULT_ROLES_CLAIM.to_string());
let email_claim = kvs
.lookup(OIDC_EMAIL_CLAIM)
.unwrap_or_else(|| OIDC_DEFAULT_EMAIL_CLAIM.to_string());
@@ -824,6 +830,7 @@ impl OidcSys {
role_policy: kvs.get(OIDC_ROLE_POLICY),
display_name,
groups_claim,
roles_claim,
email_claim,
username_claim,
})
@@ -1025,6 +1032,21 @@ fn extract_groups_claim(claims: &HashMap<String, serde_json::Value>, key: &str)
}
}
fn extract_canonical_group_values(
claims: &HashMap<String, serde_json::Value>,
groups_claim: &str,
roles_claim: &str,
) -> Vec<String> {
let mut groups = extract_groups_claim(claims, groups_claim);
if !roles_claim.is_empty() && roles_claim != groups_claim {
groups.extend(extract_groups_claim(claims, roles_claim));
}
groups.retain(|g| !g.is_empty());
groups.sort();
groups.dedup();
groups
}
#[cfg(test)]
mod tests {
use super::*;
@@ -1073,6 +1095,34 @@ mod tests {
assert!(groups.is_empty());
}
#[test]
fn test_extract_canonical_group_values_merges_groups_and_roles() {
let mut claims = HashMap::new();
claims.insert("groups".to_string(), serde_json::json!(["devs", "admins"]));
claims.insert("roles".to_string(), serde_json::json!(["admins", "consoleAdmin"]));
let merged = extract_canonical_group_values(&claims, "groups", "roles");
assert_eq!(merged, vec!["admins", "consoleAdmin", "devs"]);
}
#[test]
fn test_extract_canonical_group_values_skips_duplicate_claim_name() {
let mut claims = HashMap::new();
claims.insert("roles".to_string(), serde_json::json!(["consoleAdmin"]));
let merged = extract_canonical_group_values(&claims, "roles", "roles");
assert_eq!(merged, vec!["consoleAdmin"]);
}
#[test]
fn test_extract_canonical_group_values_roles_only() {
let mut claims = HashMap::new();
claims.insert("roles".to_string(), serde_json::json!(["consoleAdmin", "bucket-reader"]));
let merged = extract_canonical_group_values(&claims, "groups", "roles");
assert_eq!(merged, vec!["bucket-reader", "consoleAdmin"]);
}
#[test]
fn test_extract_string_claim_case_insensitive() {
let mut claims = HashMap::new();
@@ -1246,6 +1296,7 @@ mod tests {
role_policy: String::new(),
display_name: "mock-oidc".to_string(),
groups_claim: "groups".to_string(),
roles_claim: String::new(),
email_claim: "email".to_string(),
username_claim: "username".to_string(),
}
@@ -1491,6 +1542,7 @@ mod tests {
);
kvs.insert(OIDC_CLIENT_ID.to_string(), "console".to_string());
kvs.insert(ENABLE_KEY.to_string(), EnableState::On.to_string());
kvs.insert(OIDC_ROLES_CLAIM.to_string(), "app_roles".to_string());
cfg.0
.entry(IDENTITY_OPENID_SUB_SYS.to_string())
@@ -1502,6 +1554,44 @@ mod tests {
assert_eq!(parsed[0].id, "default");
assert_eq!(parsed[0].client_id, "console");
assert!(parsed[0].enabled);
assert_eq!(parsed[0].roles_claim, "app_roles");
}
#[test]
fn test_parse_persisted_provider_config_omitted_roles_claim_is_empty() {
let mut cfg = ServerConfig::new();
let mut kvs = KVS(vec![
rustfs_ecstore::config::KV {
key: ENABLE_KEY.to_string(),
value: EnableState::Off.to_string(),
hidden_if_empty: false,
},
rustfs_ecstore::config::KV {
key: OIDC_CONFIG_URL.to_string(),
value: String::new(),
hidden_if_empty: false,
},
rustfs_ecstore::config::KV {
key: OIDC_CLIENT_ID.to_string(),
value: String::new(),
hidden_if_empty: false,
},
]);
kvs.insert(
OIDC_CONFIG_URL.to_string(),
"https://example.com/.well-known/openid-configuration".to_string(),
);
kvs.insert(OIDC_CLIENT_ID.to_string(), "console".to_string());
kvs.insert(ENABLE_KEY.to_string(), EnableState::On.to_string());
cfg.0
.entry(IDENTITY_OPENID_SUB_SYS.to_string())
.or_default()
.insert(DEFAULT_DELIMITER.to_string(), kvs);
let parsed = OidcSys::parse_persisted_configs(&cfg);
assert_eq!(parsed.len(), 1);
assert_eq!(parsed[0].roles_claim, "");
}
#[test]
@@ -1554,6 +1644,7 @@ mod tests {
role_policy: "".to_string(),
display_name: id.to_string(),
groups_claim: "groups".to_string(),
roles_claim: String::new(),
email_claim: "email".to_string(),
username_claim: "preferred_username".to_string(),
}
@@ -1647,6 +1738,7 @@ mod tests {
role_policy: "readwrite".to_string(),
display_name: "Test Provider".to_string(),
groups_claim: "groups".to_string(),
roles_claim: String::new(),
email_claim: "email".to_string(),
username_claim: "preferred_username".to_string(),
};
+1 -1
View File
@@ -163,7 +163,7 @@ impl ObjectStore {
let password = if !cred.access_key.is_empty() && !cred.secret_key.is_empty() {
format!("{}:{}", cred.access_key, cred.secret_key).into_bytes()
} else {
cred.secret_key.clone().into_bytes()
cred.secret_key.into_bytes()
};
let en = rustfs_crypto::encrypt_stream_io(&password, data)?;
Ok(en)
+1 -1
View File
@@ -967,7 +967,7 @@ impl<T: Store> IamSys<T> {
let (effective_groups, groups_source) = match args.groups.as_ref() {
Some(g) if !g.is_empty() => (args.groups.clone(), "args"),
_ => match self.store.get_user(parent_user).await {
Some(u) => (u.credentials.groups.clone(), "parent_user_credentials"),
Some(u) => (u.credentials.groups, "parent_user_credentials"),
None => {
tracing::warn!(
parent_user = %parent_user,
-2
View File
@@ -29,14 +29,12 @@ workspace = true
[dependencies]
bytes = { workspace = true }
futures-core = { workspace = true }
thiserror = { workspace = true }
tokio = { workspace = true, features = ["io-util", "fs", "rt", "sync"] }
memmap2 = { workspace = true }
rustfs-io-metrics = { workspace = true }
[dev-dependencies]
futures-util = { workspace = true }
tokio = { workspace = true, features = ["rt-multi-thread", "macros"] }
[lib]
-124
View File
@@ -1,124 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! Compatibility adapter that exposes chunk streams as `AsyncRead`.
use crate::chunk::BoxChunkStream;
use bytes::Bytes;
use std::io;
use std::pin::Pin;
use std::task::{Context, Poll};
use tokio::io::{AsyncRead, ReadBuf};
/// `AsyncRead` adapter for boxed chunk streams.
pub struct ChunkStreamReader {
stream: BoxChunkStream,
current: Option<Bytes>,
offset: usize,
}
impl ChunkStreamReader {
#[must_use]
pub fn new(stream: BoxChunkStream) -> Self {
Self {
stream,
current: None,
offset: 0,
}
}
}
impl AsyncRead for ChunkStreamReader {
fn poll_read(mut self: Pin<&mut Self>, cx: &mut Context<'_>, buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
loop {
if let Some(current) = &self.current {
if self.offset < current.len() {
let remaining = &current[self.offset..];
let to_read = remaining.len().min(buf.remaining());
buf.put_slice(&remaining[..to_read]);
self.offset += to_read;
return Poll::Ready(Ok(()));
}
self.current = None;
self.offset = 0;
continue;
}
match self.stream.as_mut().poll_next(cx) {
Poll::Pending => return Poll::Pending,
Poll::Ready(Some(Ok(chunk))) => {
let next = chunk.as_bytes();
if next.is_empty() {
continue;
}
self.current = Some(next);
self.offset = 0;
}
Poll::Ready(Some(Err(err))) => return Poll::Ready(Err(err)),
Poll::Ready(None) => return Poll::Ready(Ok(())),
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::chunk::{IoChunk, MappedChunk, PooledChunk};
use bytes::Bytes;
use futures_util::stream;
use tokio::io::AsyncReadExt;
#[tokio::test]
async fn test_chunk_stream_reader_reads_single_chunk() {
let stream: BoxChunkStream = Box::pin(stream::iter(vec![Ok(IoChunk::Shared(Bytes::from_static(b"hello")))]));
let mut reader = ChunkStreamReader::new(stream);
let mut out = Vec::new();
reader.read_to_end(&mut out).await.unwrap();
assert_eq!(out, b"hello");
}
#[tokio::test]
async fn test_chunk_stream_reader_reads_multiple_chunks() {
let stream: BoxChunkStream = Box::pin(stream::iter(vec![
Ok(IoChunk::Shared(Bytes::from_static(b"he"))),
Ok(IoChunk::Mapped(MappedChunk::new(Bytes::from_static(b"llo!"), 0, 4).unwrap())),
Ok(IoChunk::Pooled(PooledChunk::from_bytes(Bytes::from_static(b" world")).unwrap())),
]));
let mut reader = ChunkStreamReader::new(stream);
let mut out = Vec::new();
reader.read_to_end(&mut out).await.unwrap();
assert_eq!(out, b"hello! world");
}
#[tokio::test]
async fn test_chunk_stream_reader_handles_empty_stream() {
let stream: BoxChunkStream = Box::pin(stream::iter(Vec::<io::Result<IoChunk>>::new()));
let mut reader = ChunkStreamReader::new(stream);
let mut out = Vec::new();
reader.read_to_end(&mut out).await.unwrap();
assert!(out.is_empty());
}
#[tokio::test]
async fn test_chunk_stream_reader_propagates_stream_error() {
let stream: BoxChunkStream = Box::pin(stream::iter(vec![Err(io::Error::other("chunk stream failure"))]));
let mut reader = ChunkStreamReader::new(stream);
let mut out = Vec::new();
let err = reader.read_to_end(&mut out).await.unwrap_err();
assert_eq!(err.kind(), io::ErrorKind::Other);
assert!(err.to_string().contains("chunk stream failure"));
}
}
-276
View File
@@ -1,276 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! Core chunk ownership abstractions for the zero-copy data plane.
use crate::pool::PooledBuffer;
use bytes::Bytes;
use futures_core::Stream;
use std::io;
use std::pin::Pin;
/// Boxed asynchronous stream of I/O chunks.
pub type BoxChunkStream = Pin<Box<dyn Stream<Item = io::Result<IoChunk>> + Send + Sync + 'static>>;
/// Source of chunked data.
pub trait ChunkSource {
fn into_chunk_stream(self) -> BoxChunkStream
where
Self: Sized;
}
/// Owned chunk variants used by the zero-copy data plane.
#[derive(Debug)]
pub enum IoChunk {
Shared(Bytes),
Mapped(MappedChunk),
Pooled(PooledChunk),
}
impl IoChunk {
/// Returns the visible length of this chunk.
#[must_use]
pub fn len(&self) -> usize {
match self {
Self::Shared(bytes) => bytes.len(),
Self::Mapped(chunk) => chunk.len(),
Self::Pooled(chunk) => chunk.len(),
}
}
/// Returns true when the chunk has no visible data.
#[must_use]
pub fn is_empty(&self) -> bool {
self.len() == 0
}
/// Returns a shared read-only view of the visible bytes.
#[must_use]
pub fn as_bytes(&self) -> Bytes {
match self {
Self::Shared(bytes) => bytes.clone(),
Self::Mapped(chunk) => chunk.as_bytes(),
Self::Pooled(chunk) => chunk.as_bytes(),
}
}
/// Returns a sliced view relative to the currently visible bytes.
pub fn slice(&self, offset: usize, len: usize) -> io::Result<Self> {
match self {
Self::Shared(bytes) => {
validate_slice_bounds(bytes.len(), offset, len)?;
Ok(Self::Shared(bytes.slice(offset..offset + len)))
}
Self::Mapped(chunk) => chunk.slice(offset, len).map(Self::Mapped),
Self::Pooled(chunk) => chunk.slice(offset, len).map(Self::Pooled),
}
}
}
/// Logical view into mapped file bytes.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct MappedChunk {
bytes: Bytes,
logical_offset: usize,
logical_len: usize,
}
impl MappedChunk {
pub fn new(bytes: Bytes, logical_offset: usize, logical_len: usize) -> io::Result<Self> {
validate_slice_bounds(bytes.len(), logical_offset, logical_len)?;
Ok(Self {
bytes,
logical_offset,
logical_len,
})
}
/// Returns the visible length of this mapped chunk.
#[must_use]
pub const fn len(&self) -> usize {
self.logical_len
}
/// Returns true when the chunk has no visible data.
#[must_use]
pub const fn is_empty(&self) -> bool {
self.logical_len == 0
}
/// Returns the visible bytes for this logical view.
#[must_use]
pub fn as_bytes(&self) -> Bytes {
self.bytes
.slice(self.logical_offset..self.logical_offset.saturating_add(self.logical_len))
}
/// Returns a sliced logical view relative to the current logical view.
pub fn slice(&self, offset: usize, len: usize) -> io::Result<Self> {
validate_slice_bounds(self.logical_len, offset, len)?;
Self::new(self.bytes.clone(), self.logical_offset + offset, len)
}
}
/// Placeholder pooled chunk variant for commit 4.
///
/// This is backed by a `PooledBuffer` and exposes a visible read-only window.
#[derive(Debug)]
pub struct PooledChunk {
bytes: Bytes,
}
#[derive(Debug)]
struct PooledChunkOwner {
buffer: PooledBuffer,
visible_len: usize,
}
impl AsRef<[u8]> for PooledChunkOwner {
fn as_ref(&self) -> &[u8] {
&self.buffer[..self.visible_len]
}
}
#[derive(Debug)]
struct DetachedVecChunkOwner {
bytes: Vec<u8>,
}
impl AsRef<[u8]> for DetachedVecChunkOwner {
fn as_ref(&self) -> &[u8] {
&self.bytes
}
}
impl PooledChunk {
pub fn new(buffer: PooledBuffer, len: usize) -> io::Result<Self> {
validate_slice_bounds(buffer.len(), 0, len)?;
Ok(Self {
bytes: Bytes::from_owner(PooledChunkOwner {
buffer,
visible_len: len,
}),
})
}
/// Convenience constructor for detached test and compatibility values.
pub fn from_bytes(bytes: Bytes) -> io::Result<Self> {
let len = bytes.len();
Self::new(PooledBuffer::from_bytes(bytes), len)
}
/// Detached constructor that takes ownership of an existing `Vec<u8>`
/// without introducing an additional copy.
pub fn from_vec(bytes: Vec<u8>) -> Self {
Self {
bytes: Bytes::from_owner(DetachedVecChunkOwner { bytes }),
}
}
/// Returns the visible length of this pooled chunk.
#[must_use]
pub fn len(&self) -> usize {
self.bytes.len()
}
/// Returns true when the chunk has no visible data.
#[must_use]
pub fn is_empty(&self) -> bool {
self.bytes.is_empty()
}
/// Returns the visible bytes for this pooled chunk.
#[must_use]
pub fn as_bytes(&self) -> Bytes {
self.bytes.clone()
}
/// Returns a sliced pooled chunk relative to the current visible view.
pub fn slice(&self, offset: usize, len: usize) -> io::Result<Self> {
validate_slice_bounds(self.bytes.len(), offset, len)?;
Ok(Self {
bytes: self.bytes.slice(offset..offset + len),
})
}
}
fn validate_slice_bounds(visible_len: usize, offset: usize, len: usize) -> io::Result<()> {
let end = offset
.checked_add(len)
.ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "chunk slice overflows"))?;
if end > visible_len {
return Err(io::Error::new(io::ErrorKind::InvalidInput, "chunk slice exceeds visible length"));
}
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
use crate::pool::BytesPool;
#[test]
fn test_shared_chunk_len_and_slice() {
let chunk = IoChunk::Shared(Bytes::from_static(b"abcdef"));
assert_eq!(chunk.len(), 6);
assert!(!chunk.is_empty());
assert_eq!(chunk.as_bytes(), Bytes::from_static(b"abcdef"));
assert_eq!(chunk.slice(1, 3).unwrap().as_bytes(), Bytes::from_static(b"bcd"));
}
#[test]
fn test_mapped_chunk_len_and_slice() {
let chunk = MappedChunk::new(Bytes::from_static(b"abcdefgh"), 2, 4).unwrap();
assert_eq!(chunk.len(), 4);
assert_eq!(chunk.as_bytes(), Bytes::from_static(b"cdef"));
assert_eq!(chunk.slice(1, 2).unwrap().as_bytes(), Bytes::from_static(b"de"));
}
#[test]
fn test_pooled_chunk_len_and_as_bytes() {
let chunk = PooledChunk::from_bytes(Bytes::from_static(b"hello")).unwrap();
assert_eq!(chunk.len(), 5);
assert_eq!(chunk.as_bytes(), Bytes::from_static(b"hello"));
assert_eq!(chunk.slice(1, 3).unwrap().as_bytes(), Bytes::from_static(b"ell"));
}
#[test]
fn test_io_chunk_as_bytes_for_all_variants() {
let shared = IoChunk::Shared(Bytes::from_static(b"s"));
let mapped = IoChunk::Mapped(MappedChunk::new(Bytes::from_static(b"mapped"), 0, 6).unwrap());
let pooled = IoChunk::Pooled(PooledChunk::from_bytes(Bytes::from_static(b"p")).unwrap());
assert_eq!(shared.as_bytes(), Bytes::from_static(b"s"));
assert_eq!(mapped.as_bytes(), Bytes::from_static(b"mapped"));
assert_eq!(pooled.as_bytes(), Bytes::from_static(b"p"));
}
#[tokio::test]
async fn test_pooled_chunk_keeps_owner_alive_until_last_view_drops() {
let pool = BytesPool::new_tiered();
let mut buffer = pool.acquire_buffer(16).await;
buffer.extend_from_slice(b"pooled-bytes");
let chunk = PooledChunk::new(buffer, "pooled-bytes".len()).unwrap();
let bytes = chunk.as_bytes();
assert_eq!(pool.available_buffers(), 0);
drop(chunk);
assert_eq!(pool.available_buffers(), 0);
assert_eq!(bytes, Bytes::from_static(b"pooled-bytes"));
drop(bytes);
assert_eq!(pool.available_buffers(), 1);
}
}
-4
View File
@@ -46,10 +46,8 @@
//! let mut buffer = pool.acquire_buffer(8192).await;
//! ```
pub mod adapter;
pub mod backpressure;
pub mod bufreader_optimizer;
pub mod chunk;
pub mod config;
pub mod deadlock_detector;
pub mod direct_io;
@@ -70,9 +68,7 @@ pub use reader::{ZeroCopyObjectReader, ZeroCopyReadError};
pub use writer::{ZeroCopyObjectWriter, ZeroCopyWriteError};
// BufReader optimizer exports
pub use adapter::ChunkStreamReader;
pub use bufreader_optimizer::{BufReaderConfig, BufReaderOptimizer, BufReaderStats, BufferedSource};
pub use chunk::{BoxChunkStream, ChunkSource, IoChunk, MappedChunk, PooledChunk};
// Shared memory exports
pub use shared_memory::{ArcData, ArcMetadata, SharedMemoryConfig, SharedMemoryPool, SharedMemoryStats};
+1 -41
View File
@@ -17,7 +17,7 @@
//! Migrated from rustfs-ecstore to provide unified buffer pooling
//! across rustfs and rustfs-ecstore without cyclic dependencies.
use bytes::{Bytes, BytesMut};
use bytes::BytesMut;
use std::mem::ManuallyDrop;
use std::sync::atomic::{AtomicU64, Ordering};
use std::sync::{Arc, Mutex};
@@ -108,7 +108,6 @@ pub struct BytesPoolMetrics {
/// A buffer managed by the BytesPool.
///
/// When dropped, the buffer is automatically returned to the pool for reuse.
#[derive(Debug)]
pub struct PooledBuffer {
/// The underlying buffer (ManuallyDrop to allow taking on drop)
pub buffer: ManuallyDrop<BytesMut>,
@@ -118,45 +117,6 @@ pub struct PooledBuffer {
_permit: Option<OwnedSemaphorePermit>,
}
impl PooledBuffer {
/// Create a detached pooled buffer from bytes.
///
/// This is primarily used for tests and transitional adapters where the
/// chunk abstraction needs a pool-shaped owner before a real pool-backed
/// producer exists.
#[must_use]
pub fn from_bytes(bytes: Bytes) -> Self {
Self {
buffer: ManuallyDrop::new(BytesMut::from(bytes.as_ref())),
tier: None,
_permit: None,
}
}
/// Current visible length of the underlying buffer.
#[must_use]
pub fn len(&self) -> usize {
self.buffer.len()
}
/// Total buffer capacity.
#[must_use]
pub fn capacity(&self) -> usize {
self.buffer.capacity()
}
/// Clear the visible contents while preserving capacity.
pub fn clear(&mut self) {
self.buffer.clear();
}
/// Returns true when the visible buffer is empty.
#[must_use]
pub fn is_empty(&self) -> bool {
self.buffer.is_empty()
}
}
/// BytesPool configuration.
///
/// Allows customization of buffer sizes and limits for each tier.
+4 -4
View File
@@ -249,7 +249,7 @@ mod tests {
#[test]
fn test_arc_data_clone() {
let data = vec![1u8, 2, 3, 4, 5];
let arc_data = ArcData::new(data.clone());
let arc_data = ArcData::new(data);
assert_eq!(arc_data.ref_count(), 1);
@@ -266,7 +266,7 @@ mod tests {
#[test]
fn test_arc_data_deref() {
let data = vec![1u8, 2, 3, 4, 5];
let arc_data = ArcData::new(data.clone());
let arc_data = ArcData::new(data);
// Test Deref trait
assert_eq!(arc_data.len(), 5);
@@ -289,7 +289,7 @@ mod tests {
let pool = SharedMemoryPool::with_defaults();
let data = vec![1u8, 2, 3, 4, 5];
let arc_data = pool.create(data.clone());
let arc_data = pool.create(data);
assert_eq!(arc_data.ref_count(), 1);
let shared = pool.share(&arc_data);
@@ -303,7 +303,7 @@ mod tests {
let pool = SharedMemoryPool::with_defaults();
let data = vec![1u8; 1024];
let arc_data = pool.create_with_size(data.clone(), 1024);
let arc_data = pool.create_with_size(data, 1024);
assert_eq!(arc_data.size(), Some(1024));
assert_eq!(pool.stats().current_memory.load(Ordering::Relaxed), 1024);
+1 -1
View File
@@ -84,7 +84,7 @@ mod tests {
assert!(Arc::ptr_eq(&metrics1, &metrics2));
// Create a MetricsCollector with the global metrics
let collector = MetricsCollector::new(metrics1.clone(), 100);
let collector = MetricsCollector::new(metrics1, 100);
// Record some data
let rt = tokio::runtime::Runtime::new().unwrap();
+225 -434
View File
@@ -121,131 +121,8 @@ pub use config::{
// Re-exports for convenience
pub use collector::MetricsCollector;
pub use metric_names::data_plane;
pub use performance::PerformanceMetrics;
/// High-level request path selected for an I/O operation.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum IoPath {
Fast,
Legacy,
}
impl IoPath {
#[must_use]
pub const fn as_str(self) -> &'static str {
match self {
Self::Fast => "fast",
Self::Legacy => "legacy",
}
}
}
/// Effective copy mode observed for an I/O operation.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum CopyMode {
TrueZeroCopy,
SharedBytes,
SingleCopy,
Reconstructed,
Transformed,
}
impl CopyMode {
#[must_use]
pub const fn as_str(self) -> &'static str {
match self {
Self::TrueZeroCopy => "true_zero_copy",
Self::SharedBytes => "shared_bytes",
Self::SingleCopy => "single_copy",
Self::Reconstructed => "reconstructed",
Self::Transformed => "transformed",
}
}
}
/// Stage where a data plane decision or fallback happened.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum IoStage {
Unknown,
ReadSetup,
HttpBridge,
LocalDiskChunk,
RangeGuard,
PutTransform,
}
impl IoStage {
#[must_use]
pub const fn as_str(self) -> &'static str {
match self {
Self::Unknown => "unknown",
Self::ReadSetup => "read_setup",
Self::HttpBridge => "http_bridge",
Self::LocalDiskChunk => "local_disk_chunk",
Self::RangeGuard => "range_guard",
Self::PutTransform => "put_transform",
}
}
}
/// Reason why the data plane fell back from a preferred path.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum FallbackReason {
Unknown,
FeatureDisabled,
ProbeFailed,
MmapDisabled,
MmapUnavailable,
SmallObject,
WindowLimitExceeded,
UnalignedWindow,
RangeNotSupported,
EncryptionEnabled,
CompressionEnabled,
TransformEncryptionLegacy,
TransformCompressionLegacy,
TransformCompressionEncryptionLegacy,
ChunkBridgeUnavailable,
NonLocalBackend,
}
impl FallbackReason {
#[must_use]
pub const fn as_str(self) -> &'static str {
match self {
Self::Unknown => "unknown",
Self::FeatureDisabled => "feature_disabled",
Self::ProbeFailed => "probe_failed",
Self::MmapDisabled => "mmap_disabled",
Self::MmapUnavailable => "mmap_unavailable",
Self::SmallObject => "small_object",
Self::WindowLimitExceeded => "window_limit_exceeded",
Self::UnalignedWindow => "unaligned_window",
Self::RangeNotSupported => "range_not_supported",
Self::EncryptionEnabled => "encryption_enabled",
Self::CompressionEnabled => "compression_enabled",
Self::TransformEncryptionLegacy => "transform_encryption_legacy",
Self::TransformCompressionLegacy => "transform_compression_legacy",
Self::TransformCompressionEncryptionLegacy => "transform_compression_encryption_legacy",
Self::ChunkBridgeUnavailable => "chunk_bridge_unavailable",
Self::NonLocalBackend => "non_local_backend",
}
}
}
#[inline(always)]
fn put_size_bucket_label(size_bytes: i64) -> &'static str {
match size_bytes {
..=0 => "unknown",
1..=16_384 => "le_16kib",
16_385..=65_536 => "le_64kib",
65_537..=262_144 => "le_256kib",
262_145..=1_048_576 => "le_1mib",
_ => "gt_1mib",
}
}
/// Record GetObject request start.
#[inline(always)]
pub fn record_get_object_request_start(concurrent_requests: usize) {
@@ -318,191 +195,40 @@ pub fn record_get_object_io_state(
counter!("rustfs_io_strategy_selected_total", "level" => load_level.to_string()).increment(1);
}
/// Record which request path was selected for an operation.
/// Record a zero-copy read operation.
///
/// # Arguments
///
/// * `size_bytes` - Size of the data read in bytes
/// * `duration_ms` - Time taken for the read operation in milliseconds
#[inline(always)]
pub fn record_io_path_selected(operation: &'static str, io_path: IoPath) {
counter!(
metric_names::data_plane::PATH_SELECTED_TOTAL,
"path" => operation,
"mode" => io_path.as_str()
)
.increment(1);
pub fn record_zero_copy_read(size_bytes: usize, duration_ms: f64) {
counter!("rustfs.zero_copy.reads.total").increment(1);
histogram!("rustfs.zero_copy.read.size.bytes").record(size_bytes as f64);
histogram!("rustfs.zero_copy.read.duration.ms").record(duration_ms);
}
/// Record the effective copy mode for an operation.
/// Record memory copies avoided by using zero-copy.
///
/// # Arguments
///
/// * `bytes_saved` - Number of bytes that would have been copied without zero-copy
#[inline(always)]
pub fn record_io_copy_mode(operation: &'static str, copy_mode: CopyMode, size_bytes: usize) {
counter!(
metric_names::data_plane::COPY_MODE_BYTES_TOTAL,
"path" => operation,
"mode" => copy_mode.as_str()
)
.increment(size_bytes as u64);
pub fn record_memory_copy_saved(bytes_saved: usize) {
counter!("rustfs.zero_copy.memory.saved.bytes").increment(bytes_saved as u64);
}
/// Record a data plane fallback decision.
/// Record a fallback from zero-copy to regular read.
///
/// This happens when zero-copy read fails (e.g., mmap not available,
/// file too large, etc.) and the system falls back to regular I/O.
///
/// # Arguments
///
/// * `reason` - Reason for the fallback (e.g., "mmap_unavailable", "file_too_large")
#[inline(always)]
pub fn record_io_fallback(stage: IoStage, reason: FallbackReason) {
counter!(
metric_names::data_plane::FALLBACK_TOTAL,
"stage" => stage.as_str(),
"reason" => reason.as_str()
)
.increment(1);
}
/// Record a selected GET chunk fast path.
#[inline(always)]
pub fn record_get_object_fast_path_selected(path: &'static str, copy_mode: CopyMode, promised_bytes: i64) {
counter!(
metric_names::data_plane::GET_FAST_PATH_SELECTED_TOTAL,
"path" => path.to_string(),
"copy_mode" => copy_mode.as_str().to_string()
)
.increment(1);
if promised_bytes >= 0 {
histogram!(metric_names::data_plane::GET_FAST_PATH_PROMISED_BYTES).record(promised_bytes as f64);
}
}
/// Record a failed GET chunk fast path probe before the response is committed.
#[inline(always)]
pub fn record_get_object_fast_path_probe_failed(path: &'static str, copy_mode: CopyMode, promised_bytes: i64) {
counter!(
metric_names::data_plane::GET_FAST_PATH_PROBE_FAILED_TOTAL,
"path" => path.to_string(),
"copy_mode" => copy_mode.as_str().to_string()
)
.increment(1);
if promised_bytes >= 0 {
histogram!(metric_names::data_plane::GET_FAST_PATH_PROMISED_BYTES).record(promised_bytes as f64);
}
}
/// Record a GET chunk fast path mid-stream error after headers have already been committed.
#[inline(always)]
pub fn record_get_object_fast_path_midstream_error(
path: &'static str,
copy_mode: CopyMode,
error_kind: &'static str,
sent_bytes: usize,
promised_bytes: i64,
) {
counter!(
metric_names::data_plane::GET_FAST_PATH_MIDSTREAM_ERROR_TOTAL,
"path" => path.to_string(),
"copy_mode" => copy_mode.as_str().to_string(),
"error_kind" => error_kind.to_string()
)
.increment(1);
histogram!(metric_names::data_plane::GET_FAST_PATH_MIDSTREAM_SENT_BYTES).record(sent_bytes as f64);
if promised_bytes >= 0 {
histogram!(metric_names::data_plane::GET_FAST_PATH_PROMISED_BYTES).record(promised_bytes as f64);
}
}
/// Record the currently active mmap bytes held by LocalDisk chunk streams.
#[inline(always)]
pub fn record_local_disk_active_mmap_bytes(active_bytes: usize) {
gauge!(metric_names::data_plane::LOCAL_DISK_ACTIVE_MMAP_BYTES).set(active_bytes as f64);
}
/// Record pooled chunk usage in LocalDisk compatibility paths.
#[inline(always)]
pub fn record_local_disk_pooled_chunk(source: &'static str, size_bytes: usize) {
counter!(
metric_names::data_plane::LOCAL_DISK_POOLED_CHUNKS_TOTAL,
"source" => source
)
.increment(1);
counter!(
metric_names::data_plane::LOCAL_DISK_POOLED_BYTES_TOTAL,
"source" => source
)
.increment(size_bytes as u64);
}
/// Record a compatibility chunk-stream aggregation performed by `read_file_zero_copy()`.
#[inline(always)]
pub fn record_local_disk_compat_collect(chunk_count: usize, total_bytes: usize) {
counter!(metric_names::data_plane::LOCAL_DISK_COMPAT_COLLECT_TOTAL).increment(1);
histogram!(metric_names::data_plane::LOCAL_DISK_COMPAT_COLLECT_CHUNKS).record(chunk_count as f64);
histogram!(metric_names::data_plane::LOCAL_DISK_COMPAT_COLLECT_BYTES).record(total_bytes as f64);
}
/// Record an attempted PUT fast path.
#[inline(always)]
pub fn record_put_object_attempted_fast_path(size_bytes: i64) {
counter!(metric_names::data_plane::PUT_FAST_PATH_ATTEMPTS_TOTAL).increment(1);
if size_bytes > 0 {
histogram!(metric_names::data_plane::PUT_FAST_PATH_ATTEMPT_SIZE_BYTES).record(size_bytes as f64);
}
}
/// Record which transformed PUT pipeline was selected.
#[inline(always)]
pub fn record_put_transform_selected(kind: &'static str, io_path: IoPath, size_bytes: usize) {
counter!(
metric_names::data_plane::PUT_TRANSFORM_SELECTED_TOTAL,
"kind" => kind,
"mode" => io_path.as_str()
)
.increment(1);
histogram!(
metric_names::data_plane::PUT_TRANSFORM_SIZE_BYTES,
"kind" => kind,
"mode" => io_path.as_str()
)
.record(size_bytes as f64);
}
/// Record PUT path selection with size-bucket context.
#[inline(always)]
pub fn record_put_path_selected(size_bytes: i64, io_path: IoPath) {
counter!(
"rustfs.s3.put_object.path.selected.total",
"mode" => io_path.as_str(),
"size_bucket" => put_size_bucket_label(size_bytes)
)
.increment(1);
}
/// Record PUT copy mode with size-bucket context.
#[inline(always)]
pub fn record_put_copy_mode(size_bytes: i64, copy_mode: CopyMode) {
counter!(
"rustfs.s3.put_object.copy_mode.total",
"mode" => copy_mode.as_str(),
"size_bucket" => put_size_bucket_label(size_bytes)
)
.increment(1);
}
/// Record PUT fallback with size-bucket context.
#[inline(always)]
pub fn record_put_fallback(size_bytes: i64, reason: FallbackReason) {
counter!(
"rustfs.s3.put_object.fallback.total",
"reason" => reason.as_str(),
"size_bucket" => put_size_bucket_label(size_bytes)
)
.increment(1);
}
/// Record inline-object selection for PUT with size-bucket context.
#[inline(always)]
pub fn record_put_inline_selected(size_bytes: i64, versioned: bool) {
counter!(
"rustfs.s3.put_object.inline.selected.total",
"versioned" => if versioned { "true" } else { "false" },
"size_bucket" => put_size_bucket_label(size_bytes)
)
.increment(1);
pub fn record_zero_copy_fallback(reason: &str) {
counter!("rustfs.zero_copy.fallback.total", "reason" => reason.to_string()).increment(1);
}
// ============================================================================
@@ -560,6 +286,41 @@ pub fn record_bytes_pool_hit_rate(tier: &str, hit_rate: f64) {
gauge!("rustfs.bytes.pool.hit.rate", "tier" => tier.to_string()).set(hit_rate * 100.0);
}
/// Record zero-copy write operation.
///
/// # Arguments
///
/// * `size_bytes` - Size of the data written in bytes
/// * `duration_ms` - Time taken for the write operation in milliseconds
#[inline(always)]
pub fn record_zero_copy_write(size_bytes: usize, duration_ms: f64) {
counter!("rustfs.zero_copy.write.total").increment(1);
histogram!("rustfs.zero_copy.write.size.bytes").record(size_bytes as f64);
histogram!("rustfs.zero_copy.write.duration.ms").record(duration_ms);
}
/// Record zero-copy write fallback.
///
/// This happens when zero-copy write fails and the system falls back to regular I/O.
///
/// # Arguments
///
/// * `reason` - Reason for the fallback
#[inline(always)]
pub fn record_zero_copy_write_fallback(reason: &str) {
counter!("rustfs.zero_copy.write.fallback.total", "reason" => reason.to_string()).increment(1);
}
/// Record bytes saved from zero-copy.
///
/// # Arguments
///
/// * `size_bytes` - Number of bytes saved from zero-copy
#[inline(always)]
pub fn record_bytes_saved(size_bytes: usize) {
counter!("rustfs.zero_copy.bytes.saved.total").increment(size_bytes as u64);
}
// ============================================================================
// S3 Operation Metrics (GetObject, PutObject, etc.)
// ============================================================================
@@ -589,33 +350,14 @@ pub fn record_get_object(duration_ms: f64, size_bytes: i64) {
///
/// * `duration_ms` - Operation duration in milliseconds
/// * `size_bytes` - Object size in bytes
/// * `zero_copy_enabled` - Legacy aggregate flag preserved for compatibility
///
/// Note: this function records aggregate S3 PUT metrics only. The definitive
/// outcome of request-level fast-path attempts must be tracked separately via
/// ADR 0001 data-plane helpers.
/// * `zero_copy_enabled` - Whether zero-copy was enabled for this operation
#[inline(always)]
pub fn record_put_object(duration_ms: f64, size_bytes: i64, zero_copy_enabled: bool) {
counter!("rustfs.s3.put_object.total").increment(1);
histogram!("rustfs.s3.put_object.duration.ms").record(duration_ms);
counter!(
"rustfs.s3.put_object.bucketed.total",
"size_bucket" => put_size_bucket_label(size_bytes)
)
.increment(1);
histogram!(
"rustfs.s3.put_object.bucketed.duration.ms",
"size_bucket" => put_size_bucket_label(size_bytes)
)
.record(duration_ms);
if size_bytes > 0 {
histogram!("rustfs.s3.put_object.size.bytes").record(size_bytes as f64);
histogram!(
"rustfs.s3.put_object.bucketed.size.bytes",
"size_bucket" => put_size_bucket_label(size_bytes)
)
.record(size_bytes as f64);
}
if zero_copy_enabled {
@@ -915,119 +657,10 @@ mod tests {
use super::*;
#[test]
fn test_io_path_as_str_values_stable() {
assert_eq!(IoPath::Fast.as_str(), "fast");
assert_eq!(IoPath::Legacy.as_str(), "legacy");
}
#[test]
fn test_copy_mode_as_str_values_stable() {
assert_eq!(CopyMode::TrueZeroCopy.as_str(), "true_zero_copy");
assert_eq!(CopyMode::SharedBytes.as_str(), "shared_bytes");
assert_eq!(CopyMode::SingleCopy.as_str(), "single_copy");
assert_eq!(CopyMode::Reconstructed.as_str(), "reconstructed");
assert_eq!(CopyMode::Transformed.as_str(), "transformed");
}
#[test]
fn test_fallback_reason_as_str_values_stable() {
assert_eq!(FallbackReason::Unknown.as_str(), "unknown");
assert_eq!(FallbackReason::FeatureDisabled.as_str(), "feature_disabled");
assert_eq!(FallbackReason::ProbeFailed.as_str(), "probe_failed");
assert_eq!(FallbackReason::MmapDisabled.as_str(), "mmap_disabled");
assert_eq!(FallbackReason::MmapUnavailable.as_str(), "mmap_unavailable");
assert_eq!(FallbackReason::SmallObject.as_str(), "small_object");
assert_eq!(FallbackReason::WindowLimitExceeded.as_str(), "window_limit_exceeded");
assert_eq!(FallbackReason::UnalignedWindow.as_str(), "unaligned_window");
assert_eq!(FallbackReason::RangeNotSupported.as_str(), "range_not_supported");
assert_eq!(FallbackReason::EncryptionEnabled.as_str(), "encryption_enabled");
assert_eq!(FallbackReason::CompressionEnabled.as_str(), "compression_enabled");
assert_eq!(FallbackReason::TransformEncryptionLegacy.as_str(), "transform_encryption_legacy");
assert_eq!(FallbackReason::TransformCompressionLegacy.as_str(), "transform_compression_legacy");
assert_eq!(
FallbackReason::TransformCompressionEncryptionLegacy.as_str(),
"transform_compression_encryption_legacy"
);
assert_eq!(FallbackReason::ChunkBridgeUnavailable.as_str(), "chunk_bridge_unavailable");
assert_eq!(FallbackReason::NonLocalBackend.as_str(), "non_local_backend");
}
#[test]
fn test_record_io_path_selected() {
record_io_path_selected("get", IoPath::Fast);
record_io_path_selected("put", IoPath::Legacy);
}
#[test]
fn test_record_io_copy_mode() {
record_io_copy_mode("get", CopyMode::SharedBytes, 1024);
record_io_copy_mode("put", CopyMode::Transformed, 2048);
}
#[test]
fn test_record_io_fallback() {
record_io_fallback(IoStage::ReadSetup, FallbackReason::MmapUnavailable);
record_io_fallback(IoStage::HttpBridge, FallbackReason::ChunkBridgeUnavailable);
}
#[test]
fn test_record_local_disk_active_mmap_bytes() {
record_local_disk_active_mmap_bytes(4096);
record_local_disk_active_mmap_bytes(0);
}
#[test]
fn test_record_local_disk_pooled_chunk() {
record_local_disk_pooled_chunk("fallback", 4096);
record_local_disk_pooled_chunk("compat_collect", 8192);
}
#[test]
fn test_record_local_disk_compat_collect() {
record_local_disk_compat_collect(3, 16384);
}
#[test]
fn test_record_get_object_fast_path_metrics() {
record_get_object_fast_path_selected("direct", CopyMode::TrueZeroCopy, 8192);
record_get_object_fast_path_probe_failed("bridge", CopyMode::SingleCopy, 4096);
record_get_object_fast_path_midstream_error("direct", CopyMode::Reconstructed, "unexpected_eof", 2048, 8192);
}
#[test]
fn test_record_put_object_attempted_fast_path() {
record_put_object_attempted_fast_path(1024 * 1024);
record_put_object_attempted_fast_path(0);
}
#[test]
fn test_record_put_transform_selected() {
record_put_transform_selected("compression", IoPath::Fast, 2048);
record_put_transform_selected("compression_encryption", IoPath::Legacy, 4096);
}
#[test]
fn test_record_put_path_selected() {
record_put_path_selected(8 * 1024, IoPath::Fast);
record_put_path_selected(2 * 1024 * 1024, IoPath::Legacy);
}
#[test]
fn test_record_put_copy_mode() {
record_put_copy_mode(8 * 1024, CopyMode::SingleCopy);
record_put_copy_mode(512 * 1024, CopyMode::Transformed);
}
#[test]
fn test_record_put_fallback() {
record_put_fallback(32 * 1024, FallbackReason::CompressionEnabled);
record_put_fallback(2 * 1024 * 1024, FallbackReason::EncryptionEnabled);
}
#[test]
fn test_record_put_inline_selected() {
record_put_inline_selected(8 * 1024, false);
record_put_inline_selected(32 * 1024, true);
fn test_record_zero_copy_read() {
record_zero_copy_read(1024, 10.5);
record_memory_copy_saved(1024);
record_zero_copy_fallback("test");
}
#[test]
@@ -1038,6 +671,13 @@ mod tests {
record_bytes_pool_hit_rate("small", 0.85);
}
#[test]
fn test_record_zero_copy_write() {
record_zero_copy_write(1024, 10.5);
record_zero_copy_write_fallback("test");
record_bytes_saved(1024);
}
// S3 Operation Metrics Tests
#[test]
fn test_record_get_object() {
@@ -1142,6 +782,157 @@ mod tests {
}
}
// ============================================================================
// Zero-Copy Optimization Metrics (Phase 1 Extension)
// ============================================================================
pub mod bandwidth;
pub mod global_metrics;
pub mod metric_names;
pub use metric_names::zero_copy;
/// Record a zero-copy buffer operation.
///
/// This function records metrics for zero-copy buffer operations,
/// including the operation type and size.
#[inline(always)]
pub fn record_zero_copy_buffer_operation(operation: &str, size: usize) {
counter!(
zero_copy::BUFFER_OPERATIONS_TOTAL,
"operation" => operation.to_string()
)
.increment(1);
counter!(
zero_copy::BUFFER_BYTES_TOTAL,
"operation" => operation.to_string()
)
.increment(size as u64);
}
/// Record memory copy operations.
///
/// This function tracks the number and size of memory copies,
/// which should be minimized in zero-copy paths.
#[inline(always)]
pub fn record_memory_copy(count: u32, size: usize) {
counter!(zero_copy::MEMORY_COPY_TOTAL).increment(count as u64);
counter!(zero_copy::MEMORY_COPY_BYTES_TOTAL).increment(size as u64);
histogram!("rustfs_memory_copy_size_bytes").record(size as f64);
}
/// Record a shared reference operation.
///
/// This function tracks operations that create or use shared references
/// for zero-copy data sharing.
#[inline(always)]
pub fn record_shared_ref_operation(operation: &str) {
counter!(
zero_copy::SHARED_REF_OPERATIONS_TOTAL,
"operation" => operation.to_string()
)
.increment(1);
}
/// Record BufReader optimization.
///
/// This function tracks BufReader layer elimination and buffer size
/// adjustments.
#[inline(always)]
pub fn record_bufreader_optimization(layers_eliminated: u32, buffer_size: usize) {
counter!(zero_copy::BUFREADER_LAYERS_ELIMINATED_TOTAL).increment(layers_eliminated as u64);
histogram!(zero_copy::BUFREADER_BUFFER_SIZE_BYTES).record(buffer_size as f64);
}
/// Record Direct I/O operation.
///
/// This function tracks Direct I/O operations and their success/fallback
/// status.
#[inline(always)]
pub fn record_direct_io_operation(operation: &str, size: usize, success: bool) {
let status = if success { "success" } else { "fallback" };
counter!(
zero_copy::DIRECT_IO_OPERATIONS_TOTAL,
"operation" => operation.to_string(),
"status" => status.to_string()
)
.increment(1);
counter!(
zero_copy::DIRECT_IO_BYTES_TOTAL,
"operation" => operation.to_string(),
"status" => status.to_string()
)
.increment(size as u64);
}
/// Update zero-copy performance metrics.
///
/// This function updates gauge metrics for overall zero-copy performance.
#[inline(always)]
pub fn update_zero_copy_performance_metrics(copy_count: u32, throughput_mbps: f64, memory_saved: u64) {
gauge!(zero_copy::AVG_COPY_COUNT).set(copy_count as f64);
gauge!(zero_copy::THROUGHPUT_MBPS).set(throughput_mbps);
gauge!(zero_copy::MEMORY_SAVED_BYTES).set(memory_saved as f64);
}
// ============================================================================
// Zero-Copy Metrics Tests
// ============================================================================
#[cfg(test)]
mod zero_copy_tests {
use super::*;
#[test]
fn test_record_zero_copy_buffer_operation() {
// This test verifies the function compiles and runs
// Actual metric verification requires a metrics recorder
record_zero_copy_buffer_operation("read", 1024);
record_zero_copy_buffer_operation("write", 2048);
}
#[test]
fn test_record_memory_copy() {
record_memory_copy(1, 1024);
record_memory_copy(2, 2048);
}
#[test]
fn test_record_shared_ref_operation() {
record_shared_ref_operation("create");
record_shared_ref_operation("share");
}
#[test]
fn test_record_bufreader_optimization() {
record_bufreader_optimization(1, 8192);
record_bufreader_optimization(2, 65536);
}
#[test]
fn test_record_direct_io_operation() {
record_direct_io_operation("read", 4096, true);
record_direct_io_operation("write", 8192, false);
}
#[test]
fn test_update_zero_copy_performance_metrics() {
update_zero_copy_performance_metrics(2, 150.5, 1024 * 1024);
}
#[test]
fn test_metric_names() {
// Verify metric names are defined
assert!(!zero_copy::BUFFER_OPERATIONS_TOTAL.is_empty());
assert!(!zero_copy::MEMORY_COPY_TOTAL.is_empty());
assert!(!zero_copy::THROUGHPUT_MBPS.is_empty());
}
}
+26 -44
View File
@@ -14,59 +14,41 @@
//! Metric name constants for consistent naming across the codebase.
/// Request-level data plane metric names introduced by ADR 0001.
pub mod data_plane {
/// Total number of selected request paths.
pub const PATH_SELECTED_TOTAL: &str = "rustfs.io.path.selected_total";
/// Zero-copy operation metric names.
pub mod zero_copy {
/// Total number of zero-copy buffer operations
pub const BUFFER_OPERATIONS_TOTAL: &str = "rustfs_zero_copy_buffer_operations_total";
/// Total bytes observed for a given effective copy mode.
pub const COPY_MODE_BYTES_TOTAL: &str = "rustfs.io.copy_mode.bytes_total";
/// Total bytes processed by zero-copy buffer operations
pub const BUFFER_BYTES_TOTAL: &str = "rustfs_zero_copy_buffer_bytes_total";
/// Total number of data plane fallbacks.
pub const FALLBACK_TOTAL: &str = "rustfs.io.zero_copy.fallback_total";
/// Total number of memory copies
pub const MEMORY_COPY_TOTAL: &str = "rustfs_memory_copy_total";
/// Current active local-disk mmap bytes held by chunk fast paths.
pub const LOCAL_DISK_ACTIVE_MMAP_BYTES: &str = "rustfs.io.local_disk.active_mmap.bytes";
/// Total bytes copied in memory
pub const MEMORY_COPY_BYTES_TOTAL: &str = "rustfs_memory_copy_bytes_total";
/// Total pooled chunks produced or consumed by LocalDisk compatibility paths.
pub const LOCAL_DISK_POOLED_CHUNKS_TOTAL: &str = "rustfs.io.local_disk.pooled_chunks.total";
/// Total number of shared reference operations
pub const SHARED_REF_OPERATIONS_TOTAL: &str = "rustfs_shared_ref_operations_total";
/// Total pooled bytes produced or consumed by LocalDisk compatibility paths.
pub const LOCAL_DISK_POOLED_BYTES_TOTAL: &str = "rustfs.io.local_disk.pooled_bytes.total";
/// Total number of BufReader layers eliminated
pub const BUFREADER_LAYERS_ELIMINATED_TOTAL: &str = "rustfs_bufreader_layers_eliminated_total";
/// Total number of compatibility chunk-stream aggregations performed for LocalDisk reads.
pub const LOCAL_DISK_COMPAT_COLLECT_TOTAL: &str = "rustfs.io.local_disk.compat_collect.total";
/// BufReader buffer size distribution
pub const BUFREADER_BUFFER_SIZE_BYTES: &str = "rustfs_bufreader_buffer_size_bytes";
/// Chunk count distribution for LocalDisk compatibility chunk aggregation.
pub const LOCAL_DISK_COMPAT_COLLECT_CHUNKS: &str = "rustfs.io.local_disk.compat_collect.chunks";
/// Total number of Direct I/O operations
pub const DIRECT_IO_OPERATIONS_TOTAL: &str = "rustfs_direct_io_operations_total";
/// Byte distribution for LocalDisk compatibility chunk aggregation.
pub const LOCAL_DISK_COMPAT_COLLECT_BYTES: &str = "rustfs.io.local_disk.compat_collect.bytes";
/// Total bytes processed by Direct I/O
pub const DIRECT_IO_BYTES_TOTAL: &str = "rustfs_direct_io_bytes_total";
/// Total number of attempted PUT fast paths.
pub const PUT_FAST_PATH_ATTEMPTS_TOTAL: &str = "rustfs.io.put.fast_path.attempts_total";
/// Average copy count per operation
pub const AVG_COPY_COUNT: &str = "rustfs_zero_copy_avg_copy_count";
/// Size distribution for attempted PUT fast paths.
pub const PUT_FAST_PATH_ATTEMPT_SIZE_BYTES: &str = "rustfs.io.put.fast_path.attempt.size.bytes";
/// Throughput in MB/s
pub const THROUGHPUT_MBPS: &str = "rustfs_zero_copy_throughput_mbps";
/// Total number of transformed PUT selections grouped by transform kind and ingress path.
pub const PUT_TRANSFORM_SELECTED_TOTAL: &str = "rustfs.io.put.transform.selected_total";
/// Size distribution for transformed PUT selections.
pub const PUT_TRANSFORM_SIZE_BYTES: &str = "rustfs.io.put.transform.size.bytes";
/// Total number of selected GET chunk fast paths.
pub const GET_FAST_PATH_SELECTED_TOTAL: &str = "rustfs.io.get.fast_path.selected_total";
/// Total number of GET chunk fast path probe failures before response commit.
pub const GET_FAST_PATH_PROBE_FAILED_TOTAL: &str = "rustfs.io.get.fast_path.probe_failed_total";
/// Total number of GET chunk fast path mid-stream errors after response commit.
pub const GET_FAST_PATH_MIDSTREAM_ERROR_TOTAL: &str = "rustfs.io.get.fast_path.midstream_error_total";
/// Byte distribution promised by GET chunk fast path selections or failures.
pub const GET_FAST_PATH_PROMISED_BYTES: &str = "rustfs.io.get.fast_path.promised.bytes";
/// Byte distribution already sent when a GET chunk fast path fails mid-stream.
pub const GET_FAST_PATH_MIDSTREAM_SENT_BYTES: &str = "rustfs.io.get.fast_path.midstream_sent.bytes";
/// Memory saved by zero-copy in bytes
pub const MEMORY_SAVED_BYTES: &str = "rustfs_zero_copy_memory_saved_bytes";
}
+1 -1
View File
@@ -267,7 +267,7 @@ impl KmsClient for LocalKmsClient {
key_id: uuid::Uuid::new_v4().to_string(),
master_key_id: request.master_key_id.clone(),
key_spec: request.key_spec.clone(),
encrypted_key: encrypted_key.clone(),
encrypted_key,
nonce,
encryption_context: request.encryption_context.clone(),
created_at: Zoned::now(),
+1 -1
View File
@@ -305,7 +305,7 @@ impl KmsClient for VaultKmsClient {
key_id: uuid::Uuid::new_v4().to_string(),
master_key_id: request.master_key_id.clone(),
key_spec: request.key_spec.clone(),
encrypted_key: encrypted_key.clone(),
encrypted_key,
nonce,
encryption_context: request.encryption_context.clone(),
created_at: Zoned::now(),
+1 -1
View File
@@ -80,7 +80,7 @@ impl KmsManager {
return Ok(GenerateDataKeyResponse {
key_id: request.key_id.clone(),
plaintext_key: cached_key.plaintext.clone(),
ciphertext_blob: cached_key.ciphertext.clone(),
ciphertext_blob: cached_key.ciphertext,
});
}
}
+2 -2
View File
@@ -1264,10 +1264,10 @@ mod tests {
// Test very long strings
let long_string = "a".repeat(1000);
let long_req = AddServiceAccountReq {
policy: Some(serde_json::json!({"Statement": [long_string.clone()]})),
policy: Some(serde_json::json!({"Statement": [long_string]})),
target_user: Some(long_string.clone()),
access_key: long_string.clone(),
secret_key: long_string.clone(),
secret_key: long_string,
name: Some("valid_name".to_string()),
description: Some("valid description".to_string()),
expiration: None,
+1 -1
View File
@@ -231,7 +231,7 @@ pub fn init_metrics_collectors(token: CancellationToken) {
.map(Duration::from_secs)
.unwrap_or(DEFAULT_SYSTEM_METRICS_INTERVAL);
let token_clone = token.clone();
let token_clone = token;
tokio::spawn(async move {
// Get current process PID
let pid = match sysinfo::get_current_pid() {
+1 -1
View File
@@ -59,7 +59,7 @@ quick-xml = { workspace = true, features = ["serialize", "serde-types", "encodin
tokio = { workspace = true, features = ["test-util"] }
tracing-subscriber = { workspace = true, features = ["env-filter"] }
axum = { workspace = true }
rustfs-utils = { workspace = true, features = ["path", "sys"] }
rustfs-utils = { workspace = true, features = ["path"] }
serde_json = { workspace = true }
time = { workspace = true }
@@ -83,7 +83,7 @@ mod pattern_rules_tests {
let target_id = TargetID::new("prefix-target".to_string(), "webhook".to_string());
rules.add("uploads/*".to_string(), target_id.clone());
rules.add("images/*".to_string(), target_id.clone());
rules.add("images/*".to_string(), target_id);
assert!(rules.match_simple("uploads/test.csv"));
assert!(rules.match_simple("uploads/subdir/test.csv"));
@@ -153,7 +153,7 @@ mod pattern_rules_tests {
let target2 = TargetID::new("target2".to_string(), "webhook".to_string());
rules1.add("*.csv".to_string(), target1.clone());
rules2.add("*.jpg".to_string(), target2.clone());
rules2.add("*.jpg".to_string(), target2);
let combined = rules1.union(&rules2);
@@ -177,10 +177,10 @@ mod pattern_rules_tests {
// Add same target to multiple patterns in rules1
rules1.add("*.csv".to_string(), target1.clone());
rules1.add("*.jpg".to_string(), target1.clone());
rules1.add("*.txt".to_string(), target1.clone());
rules1.add("*.txt".to_string(), target1);
// Add different target to .jpg pattern in rules2
rules2.add("*.jpg".to_string(), target2.clone());
rules2.add("*.jpg".to_string(), target2);
let diff = rules1.difference(&rules2);
@@ -207,7 +207,7 @@ mod pattern_rules_tests {
rules1.add("*.txt".to_string(), target1.clone());
// Add same target to .jpg pattern in rules2
rules2.add("*.jpg".to_string(), target1.clone());
rules2.add("*.jpg".to_string(), target1);
let diff = rules1.difference(&rules2);
@@ -227,7 +227,7 @@ mod pattern_rules_tests {
let target_id = TargetID::new("test-target".to_string(), "webhook".to_string());
rules.add("*.csv".to_string(), target_id.clone());
rules.add("*.jpg".to_string(), target_id.clone());
rules.add("*.jpg".to_string(), target_id);
assert!(rules.match_simple("test.csv"));
assert!(rules.match_simple("test.jpg"));
-36
View File
@@ -1,36 +0,0 @@
[package]
name = "rustfs-object-io"
version.workspace = true
edition.workspace = true
license.workspace = true
repository.workspace = true
rust-version.workspace = true
homepage.workspace = true
description = "Object I/O policy helpers and zero-copy support primitives for RustFS."
keywords.workspace = true
categories.workspace = true
[lints]
workspace = true
[dependencies]
atoi = { workspace = true }
bytes = { workspace = true }
futures-util = { workspace = true }
rustfs-ecstore = { workspace = true }
rustfs-io-core = { workspace = true }
rustfs-io-metrics = { workspace = true }
rustfs-rio = { workspace = true }
rustfs-s3select-api = { workspace = true }
rustfs-utils = { workspace = true ,features = ["http"]}
http = { workspace = true }
s3s.workspace = true
thiserror = { workspace = true }
time = { workspace = true, features = ["parsing", "formatting"] }
tokio = { workspace = true, features = ["io-util"] }
tokio-util = { workspace = true, features = ["io"] }
astral-tokio-tar = { workspace = true }
uuid = { workspace = true }
[lib]
doctest = false
File diff suppressed because it is too large Load Diff
-16
View File
@@ -1,16 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
pub mod get;
pub mod put;
File diff suppressed because it is too large Load Diff
+3 -3
View File
@@ -121,7 +121,7 @@ mod tests {
create_log_file(&dir, "other.log", 1024)?; // not managed
// Total managed = 3 072 bytes; limit = 2 048; keep_files = 2 → must delete 1.
let cleaner = make_cleaner(dir.clone(), 2, 2048);
let cleaner = make_cleaner(dir, 2, 2048);
let (deleted, freed) = cleaner.cleanup()?;
assert_eq!(deleted, 1, "should delete exactly one file");
@@ -138,7 +138,7 @@ mod tests {
create_log_file(&dir, &format!("app.log.2024-01-0{i}"), 1024)?;
}
let cleaner = make_cleaner(dir.clone(), 3, 0);
let cleaner = make_cleaner(dir, 3, 0);
let (deleted, _) = cleaner.cleanup()?;
// Updated expectation: keep_files acts as a limit (ceiling), so excess files are deleted.
@@ -213,7 +213,7 @@ mod tests {
create_log_file(&dir, "2026-03-01-06-22.rustfs.log", 1024)?;
create_log_file(&dir, "other.log", 1024)?; // not managed
let cleaner = LogCleaner::builder(dir.clone(), ".rustfs.log".to_string(), "current.log".to_string())
let cleaner = LogCleaner::builder(dir, ".rustfs.log".to_string(), "current.log".to_string())
.match_mode(FileMatchMode::Suffix)
.keep_files(1)
.max_total_size_bytes(1024)
+122 -3
View File
@@ -87,6 +87,69 @@ fn should_suppress_noisy_crates(logger_level: &str, default_level: Option<&str>,
!is_verbose_level(logger_level)
}
fn directive_applies_to_target(directive_target: &str, target: &str) -> bool {
let directive_target = directive_target.trim();
!directive_target.is_empty()
&& (directive_target == target
|| target
.strip_prefix(directive_target)
.is_some_and(|rest| rest.starts_with("::")))
}
fn effective_level_for_target<'a>(rust_log: &'a str, target: &str) -> Option<&'a str> {
let mut best_match: Option<(usize, usize, &'a str)> = None;
for (idx, directive) in rust_log.split(',').map(str::trim).filter(|d| !d.is_empty()).enumerate() {
if let Some((directive_target, level)) = directive.rsplit_once('=') {
let directive_target = directive_target.trim();
let level = level.trim();
if !is_level_token(level) {
continue;
}
let prefix_len = if directive_target.is_empty() {
0
} else if directive_applies_to_target(directive_target, target) {
directive_target.len()
} else {
continue;
};
if best_match.is_none_or(|(best_prefix_len, best_idx, _)| {
prefix_len > best_prefix_len || (prefix_len == best_prefix_len && idx >= best_idx)
}) {
best_match = Some((prefix_len, idx, level));
}
} else if is_level_token(directive)
&& best_match.is_none_or(|(best_prefix_len, best_idx, _)| best_prefix_len == 0 && idx >= best_idx)
{
best_match = Some((0, idx, directive));
}
}
best_match.map(|(_, _, level)| level)
}
fn should_demote_http_request_logs(logger_level: &str, default_level: Option<&str>, rust_log: Option<&str>) -> bool {
if let Some(level) = default_level {
let level = level.trim().to_ascii_lowercase();
return matches!(level.as_str(), "info" | "warn");
}
if let Some(rust_log) = rust_log {
if let Some(level) = effective_level_for_target(rust_log, "rustfs::server::http") {
let level = level.trim().to_ascii_lowercase();
return matches!(level.as_str(), "info" | "warn");
}
return false;
}
let level = logger_level.trim().to_ascii_lowercase();
matches!(level.as_str(), "info" | "warn")
}
pub(super) fn build_env_filter(logger_level: &str, default_level: Option<&str>) -> EnvFilter {
// 1. Determine the base filter source.
// If `default_level` is set (e.g. forced override), we use it.
@@ -111,16 +174,20 @@ pub(super) fn build_env_filter(logger_level: &str, default_level: Option<&str>)
// 2. Apply noisy crate suppression if needed.
// We only suppress if the effective configuration is NOT verbose (i.e. not debug/trace).
if should_suppress_noisy_crates(logger_level, default_level, rust_log_env.as_deref()) {
let directives = [
let mut directives = vec![
("hyper", LevelFilter::OFF),
("tonic", LevelFilter::OFF),
("h2", LevelFilter::OFF),
("reqwest", LevelFilter::OFF),
("tower", LevelFilter::OFF),
// HTTP request logs are demoted to WARN to reduce volume in production.
("rustfs::server::http", LevelFilter::WARN),
];
if should_demote_http_request_logs(logger_level, default_level, rust_log_env.as_deref()) {
// HTTP request logs are demoted to WARN to reduce volume in production,
// but only when the effective log level is not stricter than WARN.
directives.push(("rustfs::server::http", LevelFilter::WARN));
}
for (crate_name, level) in directives {
// We use `add_directive` which effectively appends to the filter.
// If RUST_LOG already specified `hyper=debug`, adding `hyper=off` later MIGHT override it
@@ -188,6 +255,22 @@ mod tests {
assert!(!should_suppress_noisy_crates("info", None, Some("rustfs=debug")));
}
#[test]
fn test_should_demote_http_request_logs() {
assert!(should_demote_http_request_logs("info", None, None));
assert!(should_demote_http_request_logs("warn", None, None));
assert!(!should_demote_http_request_logs("error", None, None));
assert!(!should_demote_http_request_logs("off", None, None));
assert!(!should_demote_http_request_logs("info", None, Some("ERROR")));
assert!(should_demote_http_request_logs("error", None, Some("WARN")));
assert!(!should_demote_http_request_logs("info", None, Some("foo=warn")));
assert!(!should_demote_http_request_logs("info", None, Some("rustfs=error")));
assert!(!should_demote_http_request_logs("info", None, Some("rustfs::server=error")));
assert!(!should_demote_http_request_logs("info", None, Some("rustfs::server::http=error")));
assert!(!should_demote_http_request_logs("info", None, Some("WARN,rustfs::server::http=error")));
assert!(should_demote_http_request_logs("error", None, Some("WARN,rustfs::server::http=warn")));
}
#[test]
fn test_build_env_filter_injects_suppressions_without_rust_log() {
// When RUST_LOG is not set and the base level is non-verbose ("info"),
@@ -258,4 +341,40 @@ mod tests {
);
});
}
#[test]
fn test_build_env_filter_does_not_promote_http_logs_above_error() {
temp_env::with_var("RUST_LOG", Some("ERROR"), || {
let filter = build_env_filter("info", None);
let filter_str = filter.to_string().to_ascii_lowercase();
assert!(
!filter_str.contains("rustfs::server::http=warn"),
"http logs must not be promoted above error level when RUST_LOG=ERROR overrides logger_level=info: {filter_str}"
);
});
temp_env::with_var("RUST_LOG", Some("rustfs=error"), || {
let filter = build_env_filter("info", None);
let filter_str = filter.to_string().to_ascii_lowercase();
assert!(
!filter_str.contains("rustfs::server::http=warn"),
"http logs must not be promoted above error level when RUST_LOG=rustfs=error overrides logger_level=info: {filter_str}"
);
});
}
#[test]
fn test_build_env_filter_does_not_fallback_to_logger_level_for_http_demotion() {
temp_env::with_var("RUST_LOG", Some("foo=warn"), || {
let filter = build_env_filter("info", None);
let filter_str = filter.to_string().to_ascii_lowercase();
assert!(
!filter_str.contains("rustfs::server::http=warn"),
"http log demotion must not fall back to logger_level when RUST_LOG only defines unrelated targets: {filter_str}"
);
});
}
}
@@ -79,6 +79,7 @@ impl KeyName {
KeyName::Jwt(JwtKeyName::JWTName),
KeyName::Jwt(JwtKeyName::JWTUpn),
KeyName::Jwt(JwtKeyName::JWTGroups),
KeyName::Jwt(JwtKeyName::JWTRoles),
KeyName::Jwt(JwtKeyName::JWTGivenName),
KeyName::Jwt(JwtKeyName::JWTFamilyName),
KeyName::Jwt(JwtKeyName::JWTMiddleName),
@@ -231,6 +232,9 @@ pub enum JwtKeyName {
#[strum(serialize = "jwt:groups")]
JWTGroups,
#[strum(serialize = "jwt:roles")]
JWTRoles,
#[strum(serialize = "jwt:given_name")]
JWTGivenName,
@@ -403,4 +407,9 @@ mod tests {
let data = serde_json::to_string(&TestCase { data: value }).expect("marshal failed");
assert_eq!(data, except);
}
#[test]
fn key_name_from_str_supports_jwt_roles() {
assert!(KeyName::try_from("jwt:roles").is_ok());
}
}
@@ -320,6 +320,31 @@ mod tests {
pollster::block_on(result) ^ negate
}
#[test]
fn test_jwt_roles_condition_uses_roles_values() {
assert!(test_eval(
new_fkv("jwt:roles", vec!["RustFS.ConsoleAdmin"]),
false,
false,
false,
vec![("roles", vec!["RustFS.ConsoleAdmin"])]
));
assert!(!test_eval(
new_fkv("jwt:roles", vec!["RustFS.ConsoleAdmin"]),
false,
false,
false,
vec![("roles", vec!["readonly"])]
));
assert!(!test_eval(
new_fkv("jwt:roles", vec!["RustFS.ConsoleAdmin"]),
false,
false,
false,
vec![("groups", vec!["RustFS.ConsoleAdmin"])]
));
}
#[test_case(new_fkv("s3:x-amz-copy-source", vec!["mybucket/myobject"]), false, vec![("x-amz-copy-source", vec!["mybucket/myobject"])] => true ; "1")]
#[test_case(new_fkv("s3:x-amz-copy-source", vec!["mybucket/myobject"]), false, vec![("x-amz-copy-source", vec!["yourbucket/myobject"])] => false ; "2")]
#[test_case(new_fkv("s3:x-amz-copy-source", vec!["mybucket/myobject"]), false, vec![] => false ; "3")]
+1
View File
@@ -61,6 +61,7 @@ rustfs-iam = { workspace = true }
rustfs-credentials = { workspace = true }
rustfs-policy = { workspace = true }
rustfs-utils = { workspace = true }
rustfs-config = { workspace = true }
# Async dependencies
tokio = { workspace = true, features = ["fs", "io-util", "sync", "time"] }
+5 -2
View File
@@ -18,6 +18,7 @@ use crate::common::client::s3::StorageBackend;
use crate::common::session::{Protocol, ProtocolPrincipal, SessionContext};
use crate::constants::{network::DEFAULT_SOURCE_IP, paths::ROOT_PATH};
use libunftp::options::FtpsRequired;
use rustfs_config::{RUSTFS_TLS_CERT, RUSTFS_TLS_KEY};
use std::fmt::{Debug, Display, Formatter};
use std::net::IpAddr;
use std::path::Path;
@@ -112,8 +113,10 @@ where
debug!("Enabling FTPS with multi-certificate support from directory: {}", cert_dir);
// Load all certificates from directory
let cert_key_pairs = rustfs_utils::load_all_certs_from_directory(cert_dir)
.map_err(|e| FtpsInitError::InvalidConfig(format!("Failed to load certificates: {}", e)))?;
let cert_key_pairs = rustfs_utils::load_all_certs_from_directory(
rustfs_utils::CertDirectoryLoadOptions::builder(cert_dir, RUSTFS_TLS_CERT, RUSTFS_TLS_KEY).build(),
)
.map_err(|e| FtpsInitError::InvalidConfig(format!("Failed to load certificates: {}", e)))?;
if cert_key_pairs.is_empty() {
return Err(FtpsInitError::InvalidConfig("No valid certificates found in directory".into()));

Some files were not shown because too many files have changed in this diff Show More