diff --git a/.docker/observability/README.md b/.docker/observability/README.md index ff5e0bbdb..cb4b9a881 100644 --- a/.docker/observability/README.md +++ b/.docker/observability/README.md @@ -60,6 +60,16 @@ The file `prometheus-rules/rustfs-get-optimization-alerts.yaml` contains pre-con | `CodecStreamingFallbackSpike` | Warning | Codec streaming fallback > 10x baseline for 10m | | `IoQueueSaturation` | Warning | IO queue utilization > 90% for 5m | +The file `prometheus-rules/rustfs-kms-alerts.yml` contains alerting rules for the KMS backend operation metrics. Thresholds are conservative defaults pending staging baseline calibration; response procedures live in `docs/operations/kms-observability-runbook.md`, and the matching dashboard is `deploy/observability/grafana/rustfs-kms-observability.json`. + +| Alert | Severity | Condition | +|-------|----------|-----------| +| `KmsBackendFatalErrors` | Critical | Fatal (non-retryable) attempt failures > 0 for 5m | +| `KmsBackendHighErrorRate` | Critical | Non-success operation ratio > 5% for 10m (with traffic guard) | +| `KmsBackendP99LatencyHigh` | Warning | Operation p99 duration (incl. retries) > 2s for 10m | +| `KmsBackendAttemptFailureSpike` | Warning | Attempt failure rate > 0.5/s for 10m | +| `KmsBackendRetryBudgetExhausted` | Warning | budget_exhausted / deadline_exceeded outcomes > 0.05/s for 10m | + ### Enabling Alert Rules Add the alert rules file to your Prometheus configuration: diff --git a/.docker/observability/README_ZH.md b/.docker/observability/README_ZH.md index 97253f20c..f42674f6c 100644 --- a/.docker/observability/README_ZH.md +++ b/.docker/observability/README_ZH.md @@ -60,6 +60,16 @@ | `CodecStreamingFallbackSpike` | 警告 | Codec streaming 回退 > 10x 基线,持续 10 分钟 | | `IoQueueSaturation` | 警告 | IO 队列利用率 > 90%,持续 5 分钟 | +文件 `prometheus-rules/rustfs-kms-alerts.yml` 包含 KMS 后端操作指标的告警规则。阈值为保守默认值,待 staging 基线校准;响应流程见 `docs/operations/kms-observability-runbook.md`,配套仪表盘为 `deploy/observability/grafana/rustfs-kms-observability.json`。 + +| 告警 | 级别 | 条件 | +|------|------|------| +| `KmsBackendFatalErrors` | 严重 | fatal(不可重试)尝试失败 > 0,持续 5 分钟 | +| `KmsBackendHighErrorRate` | 严重 | 非 success 操作占比 > 5%,持续 10 分钟(含流量下限保护) | +| `KmsBackendP99LatencyHigh` | 警告 | 操作 p99 耗时(含重试)> 2s,持续 10 分钟 | +| `KmsBackendAttemptFailureSpike` | 警告 | 尝试失败率 > 0.5/s,持续 10 分钟 | +| `KmsBackendRetryBudgetExhausted` | 警告 | budget_exhausted / deadline_exceeded 结果 > 0.05/s,持续 10 分钟 | + ### 启用告警规则 在 Prometheus 配置中添加告警规则文件: diff --git a/.docker/observability/prometheus-rules/rustfs-kms-alerts.yml b/.docker/observability/prometheus-rules/rustfs-kms-alerts.yml new file mode 100644 index 000000000..e4f1faf03 --- /dev/null +++ b/.docker/observability/prometheus-rules/rustfs-kms-alerts.yml @@ -0,0 +1,188 @@ +# Copyright 2024 RustFS Team +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# ============================================================================= +# RustFS KMS backend — Prometheus alerting rules +# ============================================================================= +# +# Metric source: the KMS operation-policy choke point in +# crates/kms/src/policy.rs. All label values are static enum strings +# (operation, op_class, outcome, error_class); key identifiers, key material, +# and tokens never appear in labels. +# +# Response procedures: docs/operations/kms-observability-runbook.md +# +# IMPORTANT — threshold status: every numeric threshold below is a +# conservative default chosen without a production baseline. Calibrate against +# a staging baseline before relying on these alerts for paging, and prefer +# loosening over tightening until the baseline exists. Formal SLO targets are +# deliberately not encoded here (see rustfs/backlog#1584). +# +# NOTE: prometheus.yml loads /etc/prometheus/rules/*.yml — keep the .yml +# extension or the file is silently ignored by the docker-compose stack. +# +# Validate: promtool check rules rustfs-kms-alerts.yml +# ============================================================================= + +groups: + # ========================================================================== + # Critical alerts — immediate action required + # ========================================================================== + - name: rustfs-kms-critical + interval: 30s + rules: + # ------------------------------------------------------------------ + # 1. KmsBackendFatalErrors + # Any attempt failure classified as fatal (non-retryable): auth + # or permission errors, malformed requests, missing keys. The + # policy never retries these, so even a low rate means real + # operations are failing right now. + # ------------------------------------------------------------------ + - alert: KmsBackendFatalErrors + expr: | + sum by (operation) (rate(rustfs_kms_backend_attempt_failures_total{error_class="fatal"}[5m])) > 0 + for: 5m + labels: + severity: critical + component: kms + annotations: + summary: "KMS backend fatal errors on operation {{ $labels.operation }}" + description: >- + Attempt failures classified as fatal are occurring at + {{ $value | printf "%.3f" }}/s on operation + {{ $labels.operation }}. Fatal failures are not retried: + each one is a KMS backend call that failed permanently + (authentication, permissions, malformed request, or a + missing key/version). + runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmsbackendfatalerrors" + + # ------------------------------------------------------------------ + # 2. KmsBackendHighErrorRate + # Sustained share of operations terminating without success + # (fatal, budget_exhausted, deadline_exceeded). The cancelled + # outcome is excluded because shutdowns legitimately produce it. + # The traffic guard keeps a single failure on a near-idle + # cluster from firing the alert. + # Threshold: 5% for 10m — conservative default, calibrate + # against a staging baseline. + # ------------------------------------------------------------------ + - alert: KmsBackendHighErrorRate + expr: | + ( + sum(rate(rustfs_kms_backend_operations_total{outcome!~"success|cancelled"}[5m])) + / + clamp_min(sum(rate(rustfs_kms_backend_operations_total[5m])), 1e-9) + ) > 0.05 + and + sum(rate(rustfs_kms_backend_operations_total[5m])) > 0.02 + for: 10m + labels: + severity: critical + component: kms + annotations: + summary: "KMS backend non-success ratio above 5% for 10m" + description: >- + {{ $value | humanizePercentage }} of KMS backend operations + are terminating in fatal, budget_exhausted, or + deadline_exceeded. Object encryption and decryption paths + depending on the KMS are degraded or failing. + runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmsbackendhigherrorrate" + + # ========================================================================== + # Warning alerts — investigation needed + # ========================================================================== + - name: rustfs-kms-warning + interval: 30s + rules: + # ------------------------------------------------------------------ + # 3. KmsBackendP99LatencyHigh + # p99 wall-clock duration of whole operations (attempts plus + # backoff) is sustained above 2 seconds. Because the histogram + # includes retries, a high p99 usually means the retry policy + # is absorbing backend failures, not that every call is slow. + # Threshold: 2s for 10m — conservative default, calibrate + # against a staging baseline. + # ------------------------------------------------------------------ + - alert: KmsBackendP99LatencyHigh + expr: | + histogram_quantile(0.99, + sum by (le) (rate(rustfs_kms_backend_operation_duration_seconds_bucket[5m])) + ) > 2 + for: 10m + labels: + severity: warning + component: kms + annotations: + summary: "KMS backend operation p99 latency above 2s for 10m" + description: >- + The 99th-percentile KMS backend operation duration is + {{ $value | humanizeDuration }}, including retries and + backoff. Encryption and decryption latency is leaking into + S3 request latency. + runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmsbackendp99latencyhigh" + + # ------------------------------------------------------------------ + # 4. KmsBackendAttemptFailureSpike + # Aggregate attempt-failure rate (all error classes) sustained + # above an absolute floor. An absolute threshold is used instead + # of an offset-1d baseline ratio because fresh deployments have + # no baseline and an empty offset vector would keep a ratio + # alert from ever firing; switch to a baseline-relative form + # (see rustfs-get-optimization-alerts.yaml for the pattern) + # once a stable staging baseline exists. + # Threshold: 0.5/s for 10m — conservative default, calibrate + # against a staging baseline. + # ------------------------------------------------------------------ + - alert: KmsBackendAttemptFailureSpike + expr: | + sum(rate(rustfs_kms_backend_attempt_failures_total[5m])) > 0.5 + for: 10m + labels: + severity: warning + component: kms + annotations: + summary: "KMS backend attempt failures above 0.5/s for 10m" + description: >- + KMS backend attempts are failing at + {{ $value | printf "%.2f" }}/s across all error classes. + The retry policy may still be masking these from callers — + check the error-class breakdown before it stops absorbing + them. + runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmsbackendattemptfailurespike" + + # ------------------------------------------------------------------ + # 5. KmsBackendRetryBudgetExhausted + # Operations are running out of retry budget (budget_exhausted) + # or operation deadline (deadline_exceeded). These surface to + # callers as failed KMS operations even though every individual + # failure was retryable — the backend is unhealthy for longer + # than the policy can bridge. + # Threshold: 0.05/s for 10m — conservative default, calibrate + # against a staging baseline. + # ------------------------------------------------------------------ + - alert: KmsBackendRetryBudgetExhausted + expr: | + sum by (outcome) (rate(rustfs_kms_backend_operations_total{outcome=~"budget_exhausted|deadline_exceeded"}[5m])) > 0.05 + for: 10m + labels: + severity: warning + component: kms + annotations: + summary: "KMS backend operations exhausting retry budget ({{ $labels.outcome }})" + description: >- + KMS backend operations are terminating as + {{ $labels.outcome }} at {{ $value | printf "%.3f" }}/s. + Retryable failures are outlasting the retry budget, so + callers are seeing hard failures. + runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmsbackendretrybudgetexhausted" diff --git a/Cargo.lock b/Cargo.lock index 97584df3a..416cbd5c4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3672,6 +3672,7 @@ dependencies = [ "flate2", "futures", "hex", + "hotpath", "http 1.5.0", "http-body-util", "hyper", @@ -9039,6 +9040,7 @@ dependencies = [ "const-str", "futures", "hashbrown 0.17.1", + "hotpath", "metrics", "rustfs-config", "rustfs-s3-types", @@ -9059,6 +9061,7 @@ dependencies = [ "base64-simd", "bytes", "crc-fast", + "hotpath", "http 1.5.0", "md-5 0.11.0", "pretty_assertions", @@ -9072,6 +9075,7 @@ name = "rustfs-common" version = "1.0.0-beta.12" dependencies = [ "chrono", + "hotpath", "metrics", "rmp-serde", "s3s", @@ -9086,6 +9090,7 @@ dependencies = [ name = "rustfs-concurrency" version = "1.0.0-beta.12" dependencies = [ + "hotpath", "insta", "rustfs-io-core", "serde", @@ -9099,6 +9104,7 @@ name = "rustfs-config" version = "1.0.0-beta.12" dependencies = [ "const-str", + "hotpath", "serde", "serde_json", ] @@ -9109,6 +9115,7 @@ version = "1.0.0-beta.12" dependencies = [ "base64-simd", "hmac 0.13.0", + "hotpath", "rand 0.10.2", "serde", "serde_json", @@ -9124,6 +9131,7 @@ dependencies = [ "argon2", "base64-simd", "chacha20poly1305", + "hotpath", "jsonwebtoken 11.0.0", "pbkdf2 0.13.0", "rand 0.10.2", @@ -9141,6 +9149,7 @@ name = "rustfs-data-usage" version = "1.0.0-beta.12" dependencies = [ "async-trait", + "hotpath", "rmp-serde", "rustfs-filemeta", "serde", @@ -9286,6 +9295,7 @@ dependencies = [ name = "rustfs-extension-schema" version = "1.0.0-beta.12" dependencies = [ + "hotpath", "serde", "serde_json", "thiserror 2.0.19", @@ -9324,6 +9334,7 @@ dependencies = [ "async-trait", "base64 0.23.0", "futures", + "hotpath", "http 1.5.0", "metrics", "rustfs-common", @@ -9355,6 +9366,7 @@ dependencies = [ "async-trait", "base64-simd", "futures", + "hotpath", "http 1.5.0", "jsonwebtoken 11.0.0", "moka", @@ -9389,6 +9401,7 @@ name = "rustfs-io-core" version = "1.0.0-beta.12" dependencies = [ "bytes", + "hotpath", "memmap2", "rustfs-io-metrics", "thiserror 2.0.19", @@ -9401,6 +9414,7 @@ name = "rustfs-io-metrics" version = "1.0.0-beta.12" dependencies = [ "criterion", + "hotpath", "metrics", "metrics-util", "num_cpus", @@ -9467,6 +9481,7 @@ version = "1.0.0-beta.12" dependencies = [ "bytes", "futures", + "hotpath", "http 1.5.0", "http-body 1.1.0", "http-body-util", @@ -9499,9 +9514,12 @@ dependencies = [ "base64 0.23.0", "chacha20poly1305", "hex", + "hotpath", "insta", "jiff", "md-5 0.11.0", + "metrics", + "metrics-util", "moka", "rand 0.10.2", "reqwest", @@ -9529,6 +9547,7 @@ name = "rustfs-lifecycle" version = "1.0.0-beta.12" dependencies = [ "async-trait", + "hotpath", "metrics", "metrics-util", "proptest", @@ -9553,6 +9572,7 @@ dependencies = [ "async-trait", "crossbeam-queue", "futures", + "hotpath", "parking_lot", "rand 0.10.2", "rustfs-io-metrics", @@ -9574,6 +9594,7 @@ version = "1.0.0-beta.12" dependencies = [ "chrono", "flate2", + "hotpath", "regex", "serde", "serde_json", @@ -9591,6 +9612,7 @@ name = "rustfs-madmin" version = "1.0.0-beta.12" dependencies = [ "chrono", + "hotpath", "humantime", "hyper", "rmp-serde", @@ -9611,6 +9633,7 @@ dependencies = [ "criterion", "form_urlencoded", "hashbrown 0.17.1", + "hotpath", "metrics", "percent-encoding", "quick-xml", @@ -9640,6 +9663,7 @@ version = "1.0.0-beta.12" dependencies = [ "criterion", "futures", + "hotpath", "rustfs-config", "rustfs-io-metrics", "rustfs-utils", @@ -9658,6 +9682,7 @@ version = "1.0.0-beta.12" dependencies = [ "bytes", "criterion", + "hotpath", "metrics", "metrics-util", "moka", @@ -9680,6 +9705,7 @@ dependencies = [ "flate2", "futures-util", "glob", + "hotpath", "jiff", "libc", "metrics", @@ -9765,6 +9791,7 @@ dependencies = [ "futures-util", "hex", "hmac 0.13.0", + "hotpath", "http 1.5.0", "http-body-util", "hyper", @@ -9817,6 +9844,7 @@ name = "rustfs-protos" version = "1.0.0-beta.12" dependencies = [ "flatbuffers", + "hotpath", "prost 0.14.4", "rmp-serde", "rustfs-common", @@ -9841,6 +9869,7 @@ version = "1.0.0-beta.12" dependencies = [ "byteorder", "bytes", + "hotpath", "regex", "rmp", "rmp-serde", @@ -9899,6 +9928,7 @@ dependencies = [ "chacha20poly1305", "hex", "hmac 0.13.0", + "hotpath", "minlz", "pin-project-lite", "rand 0.10.2", @@ -9916,6 +9946,7 @@ dependencies = [ name = "rustfs-s3-ops" version = "1.0.0-beta.12" dependencies = [ + "hotpath", "rustfs-s3-types", ] @@ -9923,6 +9954,7 @@ dependencies = [ name = "rustfs-s3-types" version = "1.0.0-beta.12" dependencies = [ + "hotpath", "serde", "serde_json", ] @@ -9937,6 +9969,7 @@ dependencies = [ "datafusion", "futures", "futures-core", + "hotpath", "http 1.5.0", "metrics", "parking_lot", @@ -9965,6 +9998,7 @@ dependencies = [ "datafusion", "derive_builder", "futures", + "hotpath", "parking_lot", "rustfs-s3select-api", "s3s", @@ -9982,6 +10016,7 @@ dependencies = [ "futures", "hex-simd", "hmac 0.13.0", + "hotpath", "http 1.5.0", "metrics", "rand 0.10.2", @@ -10014,6 +10049,7 @@ dependencies = [ name = "rustfs-security-governance" version = "1.0.0-beta.12" dependencies = [ + "hotpath", "thiserror 2.0.19", ] @@ -10023,6 +10059,7 @@ version = "1.0.0-beta.12" dependencies = [ "base64-simd", "bytes", + "hotpath", "http 1.5.0", "hyper", "rustfs-utils", @@ -10039,6 +10076,7 @@ name = "rustfs-storage-api" version = "1.0.0-beta.12" dependencies = [ "async-trait", + "hotpath", "insta", "rustfs-filemeta", "serde", @@ -10060,6 +10098,7 @@ dependencies = [ "deadpool-postgres", "futures-util", "hashbrown 0.17.1", + "hotpath", "hyper", "hyper-rustls", "lapin", @@ -10105,6 +10144,7 @@ dependencies = [ name = "rustfs-test-utils" version = "1.0.0-beta.12" dependencies = [ + "hotpath", "rustfs-data-usage", "rustfs-ecstore", "rustfs-storage-api", @@ -10121,6 +10161,7 @@ name = "rustfs-tls-runtime" version = "1.0.0-beta.12" dependencies = [ "arc-swap", + "hotpath", "metrics", "rcgen", "rustfs-common", @@ -10142,6 +10183,7 @@ version = "1.0.0-beta.12" dependencies = [ "async-trait", "axum", + "hotpath", "http 1.5.0", "ipnetwork", "metrics", @@ -10188,6 +10230,7 @@ dependencies = [ "hex-simd", "highway", "hmac 0.13.0", + "hotpath", "http 1.5.0", "hyper", "local-ip-address", @@ -10220,6 +10263,7 @@ dependencies = [ "astral-tokio-tar", "async-compression", "criterion", + "hotpath", "tempfile", "thiserror 2.0.19", "tokio", diff --git a/crates/audit/Cargo.toml b/crates/audit/Cargo.toml index d598bc281..421b2b3a2 100644 --- a/crates/audit/Cargo.toml +++ b/crates/audit/Cargo.toml @@ -25,7 +25,33 @@ documentation = "https://docs.rs/rustfs-audit/latest/rustfs_audit/" keywords = ["audit", "target", "management", "fan-out", "RustFS"] categories = ["web-programming", "development-tools", "asynchronous", "api-bindings"] +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "rustfs-config/hotpath", + "rustfs-s3-types/hotpath", + "rustfs-targets/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-s3-types/hotpath-alloc", + "rustfs-targets/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-s3-types/hotpath-cpu", + "rustfs-targets/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-targets = { workspace = true } rustfs-config = { workspace = true, features = ["audit", "server-config-model"] } rustfs-s3-types = { workspace = true } diff --git a/crates/checksums/Cargo.toml b/crates/checksums/Cargo.toml index 6af2106b1..0229dd3a8 100644 --- a/crates/checksums/Cargo.toml +++ b/crates/checksums/Cargo.toml @@ -28,7 +28,14 @@ documentation = "https://docs.rs/rustfs-checksums/latest/rustfs_checksum/" [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true bytes = { workspace = true, features = ["serde"] } crc-fast = { workspace = true } http = { workspace = true } diff --git a/crates/common/Cargo.toml b/crates/common/Cargo.toml index 93fe21df7..df98511d6 100644 --- a/crates/common/Cargo.toml +++ b/crates/common/Cargo.toml @@ -27,7 +27,14 @@ categories = ["web-programming", "development-tools", "data-structures"] [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true tokio = { workspace = true, features = ["fs", "rt-multi-thread"] } tonic = { workspace = true, features = ["gzip", "deflate"] } uuid = { workspace = true, features = ["v4", "fast-rng", "macro-diagnostics"] } diff --git a/crates/concurrency/Cargo.toml b/crates/concurrency/Cargo.toml index 55eb20eb9..ebc7ae095 100644 --- a/crates/concurrency/Cargo.toml +++ b/crates/concurrency/Cargo.toml @@ -13,7 +13,14 @@ categories = ["concurrency", "filesystem"] [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio", "rustfs-io-core/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-io-core/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-io-core/hotpath-cpu"] + [dependencies] +hotpath.workspace = true # Internal crates rustfs-io-core = { workspace = true } serde = { workspace = true, features = ["derive"] } diff --git a/crates/config/Cargo.toml b/crates/config/Cargo.toml index 9dcd0ec2f..7b61b1c5f 100644 --- a/crates/config/Cargo.toml +++ b/crates/config/Cargo.toml @@ -25,6 +25,7 @@ keywords = ["configuration", "settings", "management", "rustfs", "Minio"] categories = ["web-programming", "development-tools", "config"] [dependencies] +hotpath.workspace = true const-str = { workspace = true, optional = true, features = ["std", "proc"] } serde = { workspace = true, optional = true, features = ["derive"] } serde_json = { workspace = true, optional = true, features = ["raw_value"] } @@ -34,6 +35,9 @@ workspace = true [features] default = ["constants"] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] audit = ["dep:const-str", "constants"] constants = ["dep:const-str"] notify = ["dep:const-str", "constants"] diff --git a/crates/credentials/Cargo.toml b/crates/credentials/Cargo.toml index bc2eca866..605d257d5 100644 --- a/crates/credentials/Cargo.toml +++ b/crates/credentials/Cargo.toml @@ -24,7 +24,14 @@ description = "Credentials management utilities for RustFS, enabling secure hand keywords = ["rustfs", "Minio", "credentials", "authentication", "authorization"] categories = ["web-programming", "development-tools", "data-structures", "security"] +[features] +default = [] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true base64-simd = { workspace = true } hmac = { workspace = true } rand = { workspace = true, features = ["serde"] } diff --git a/crates/crypto/Cargo.toml b/crates/crypto/Cargo.toml index e7cdf3357..080db7364 100644 --- a/crates/crypto/Cargo.toml +++ b/crates/crypto/Cargo.toml @@ -29,6 +29,7 @@ documentation = "https://docs.rs/rustfs-crypto/latest/rustfs_crypto/" workspace = true [dependencies] +hotpath.workspace = true aes-gcm = { workspace = true, optional = true, features = ["rand_core"] } argon2 = { workspace = true, optional = true } chacha20poly1305 = { workspace = true, optional = true } @@ -49,6 +50,9 @@ time = { workspace = true, features = ["parsing", "formatting", "macros", "serde [features] default = ["crypto", "fips"] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] fips = [] crypto = [ "dep:aes-gcm", diff --git a/crates/data-usage/Cargo.toml b/crates/data-usage/Cargo.toml index b239e3dd6..3e33b530e 100644 --- a/crates/data-usage/Cargo.toml +++ b/crates/data-usage/Cargo.toml @@ -27,7 +27,14 @@ categories = ["data-structures", "filesystem"] [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath", "rustfs-filemeta/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-filemeta/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-filemeta/hotpath-cpu"] + [dependencies] +hotpath.workspace = true serde = { workspace = true, features = ["derive"] } rmp-serde = { workspace = true } async-trait = { workspace = true } diff --git a/crates/e2e_test/Cargo.toml b/crates/e2e_test/Cargo.toml index 00ddbc47a..b0bc6ed2a 100644 --- a/crates/e2e_test/Cargo.toml +++ b/crates/e2e_test/Cargo.toml @@ -25,10 +25,58 @@ workspace = true [features] default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/reqwest-0-13", + "rustfs-config/hotpath", + "rustfs-credentials/hotpath", + "rustfs-data-usage/hotpath", + "rustfs-ecstore/hotpath", + "rustfs-filemeta/hotpath", + "rustfs-lock/hotpath", + "rustfs-madmin/hotpath", + "rustfs-protos/hotpath", + "rustfs-rio/hotpath", + "rustfs-signer/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-data-usage/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-filemeta/hotpath-alloc", + "rustfs-lock/hotpath-alloc", + "rustfs-madmin/hotpath-alloc", + "rustfs-protos/hotpath-alloc", + "rustfs-rio/hotpath-alloc", + "rustfs-signer/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-data-usage/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-filemeta/hotpath-cpu", + "rustfs-lock/hotpath-cpu", + "rustfs-madmin/hotpath-cpu", + "rustfs-protos/hotpath-cpu", + "rustfs-rio/hotpath-cpu", + "rustfs-signer/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] ftps = [] sftp = [] [dependencies] +hotpath.workspace = true rustfs-config = { workspace = true, features = ["constants"] } rustfs-credentials.workspace = true rustfs-ecstore.workspace = true diff --git a/crates/e2e_test/src/copy_object_metadata_test.rs b/crates/e2e_test/src/copy_object_metadata_test.rs index cfd257f06..f56fe7f41 100644 --- a/crates/e2e_test/src/copy_object_metadata_test.rs +++ b/crates/e2e_test/src/copy_object_metadata_test.rs @@ -117,9 +117,14 @@ mod tests { .key("assets/explicit-copy.js") .copy_source(format!("{bucket}/{key}")) .metadata_directive(MetadataDirective::Copy) + .customize() + .mutate_request(|request| { + request.headers_mut().insert("content-type", "application/octet-stream"); + request.headers_mut().insert("x-amz-meta-request-only", "ignored"); + }) .send() .await - .expect("explicit COPY directive failed"); + .expect("explicit COPY directive with request metadata failed"); let explicit_copy_head = client .head_object() .bucket(bucket) @@ -128,6 +133,18 @@ mod tests { .await .expect("HEAD failed after explicit COPY"); assert_eq!(explicit_copy_head.cache_control(), Some("max-age=60")); + assert_eq!(explicit_copy_head.content_type(), Some("text/javascript; charset=utf-8")); + assert_eq!( + explicit_copy_head.metadata().and_then(|metadata| metadata.get("mtime")), + Some(&"1777992333".to_string()) + ); + assert_eq!( + explicit_copy_head + .metadata() + .and_then(|metadata| metadata.get("request-only")), + None, + "COPY must ignore request metadata" + ); assert_eq!( explicit_copy_head.website_redirect_location(), None, @@ -571,20 +588,6 @@ mod tests { Some("InvalidArgument") ); - let ignored_replacement = client - .copy_object() - .bucket(bucket) - .key(key) - .copy_source(format!("{bucket}/{key}")) - .content_type("application/ignored") - .send() - .await - .expect_err("Replacement fields without REPLACE should be rejected"); - assert_eq!( - ignored_replacement.as_service_error().and_then(|error| error.code()), - Some("InvalidRequest") - ); - let unchanged = client .get_object() .bucket(bucket) diff --git a/crates/ecstore/Cargo.toml b/crates/ecstore/Cargo.toml index 726c5f2a2..84ad18647 100644 --- a/crates/ecstore/Cargo.toml +++ b/crates/ecstore/Cargo.toml @@ -40,19 +40,83 @@ hotpath = [ "hotpath/async-channel", "hotpath/parking_lot", "hotpath/reqwest-0-13", + "rustfs-checksums/hotpath", + "rustfs-common/hotpath", + "rustfs-concurrency/hotpath", + "rustfs-config/hotpath", + "rustfs-credentials/hotpath", + "rustfs-data-usage/hotpath", "rustfs-filemeta/hotpath", + "rustfs-io-metrics/hotpath", + "rustfs-lifecycle/hotpath", + "rustfs-lock/hotpath", + "rustfs-madmin/hotpath", + "rustfs-object-capacity/hotpath", + "rustfs-policy/hotpath", + "rustfs-protos/hotpath", + "rustfs-replication/hotpath", "rustfs-rio/hotpath", + "rustfs-rio-v2?/hotpath", + "rustfs-s3-types/hotpath", + "rustfs-signer/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-tls-runtime/hotpath", + "rustfs-utils/hotpath", + "rustfs-crypto/hotpath", ] hotpath-alloc = [ + "hotpath", "hotpath/hotpath-alloc", + "rustfs-checksums/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-concurrency/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-data-usage/hotpath-alloc", "rustfs-filemeta/hotpath-alloc", + "rustfs-io-metrics/hotpath-alloc", + "rustfs-lifecycle/hotpath-alloc", + "rustfs-lock/hotpath-alloc", + "rustfs-madmin/hotpath-alloc", + "rustfs-object-capacity/hotpath-alloc", + "rustfs-policy/hotpath-alloc", + "rustfs-protos/hotpath-alloc", + "rustfs-replication/hotpath-alloc", "rustfs-rio/hotpath-alloc", + "rustfs-rio-v2?/hotpath-alloc", + "rustfs-s3-types/hotpath-alloc", + "rustfs-signer/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-tls-runtime/hotpath-alloc", + "rustfs-utils/hotpath-alloc", + "rustfs-crypto/hotpath-alloc", ] hotpath-cpu = [ "hotpath", "hotpath/hotpath-cpu", + "rustfs-checksums/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-concurrency/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-data-usage/hotpath-cpu", "rustfs-filemeta/hotpath-cpu", + "rustfs-io-metrics/hotpath-cpu", + "rustfs-lifecycle/hotpath-cpu", + "rustfs-lock/hotpath-cpu", + "rustfs-madmin/hotpath-cpu", + "rustfs-object-capacity/hotpath-cpu", + "rustfs-policy/hotpath-cpu", + "rustfs-protos/hotpath-cpu", + "rustfs-replication/hotpath-cpu", "rustfs-rio/hotpath-cpu", + "rustfs-rio-v2?/hotpath-cpu", + "rustfs-s3-types/hotpath-cpu", + "rustfs-signer/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-tls-runtime/hotpath-cpu", + "rustfs-utils/hotpath-cpu", + "rustfs-crypto/hotpath-cpu", ] # Exposes shared lifecycle/tier test utilities (MockWarmBackend, fault # injection, xl.meta transition assertions) via `api::tier::test_util`. diff --git a/crates/ecstore/src/lib.rs b/crates/ecstore/src/lib.rs index e0c7d7b6d..2ed643ead 100644 --- a/crates/ecstore/src/lib.rs +++ b/crates/ecstore/src/lib.rs @@ -12,6 +12,8 @@ // See the License for the specific language governing permissions and // limitations under the License. +#![recursion_limit = "256"] + /// Scope-based hotpath measurement for `#[async_trait]` methods, where /// `#[hotpath::measure]` would only time the boxed-future construction. /// The guard records wall time from this statement until the enclosing diff --git a/crates/extension-schema/Cargo.toml b/crates/extension-schema/Cargo.toml index 84d9449a1..5e6af760d 100644 --- a/crates/extension-schema/Cargo.toml +++ b/crates/extension-schema/Cargo.toml @@ -27,7 +27,14 @@ categories = ["web-programming", "development-tools"] [lib] doctest = false +[features] +default = [] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true serde = { workspace = true, features = ["derive"] } thiserror.workspace = true diff --git a/crates/filemeta/Cargo.toml b/crates/filemeta/Cargo.toml index f1d2bbad2..78d6e1c1c 100644 --- a/crates/filemeta/Cargo.toml +++ b/crates/filemeta/Cargo.toml @@ -27,9 +27,9 @@ documentation = "https://docs.rs/rustfs-filemeta/latest/rustfs_filemeta/" [features] default = [] -hotpath = ["hotpath/hotpath", "hotpath/tokio"] -hotpath-alloc = ["hotpath/hotpath-alloc"] -hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] +hotpath = ["hotpath/hotpath", "hotpath/tokio", "rustfs-utils/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-utils/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-utils/hotpath-cpu"] [dependencies] hotpath.workspace = true diff --git a/crates/heal/Cargo.toml b/crates/heal/Cargo.toml index e213c7da5..668c4af24 100644 --- a/crates/heal/Cargo.toml +++ b/crates/heal/Cargo.toml @@ -29,7 +29,48 @@ categories = ["web-programming", "development-tools", "filesystem"] [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "rustfs-common/hotpath", + "rustfs-concurrency/hotpath", + "rustfs-config/hotpath", + "rustfs-ecstore/hotpath", + "rustfs-madmin/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-utils/hotpath", + "rustfs-test-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-concurrency/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-madmin/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-utils/hotpath-alloc", + "rustfs-test-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-concurrency/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-madmin/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-utils/hotpath-cpu", + "rustfs-test-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-config = { workspace = true } rustfs-concurrency = { workspace = true } rustfs-ecstore = { workspace = true } diff --git a/crates/heal/src/heal/erasure_healer.rs b/crates/heal/src/heal/erasure_healer.rs index 2f4c6f87d..d08367f3f 100644 --- a/crates/heal/src/heal/erasure_healer.rs +++ b/crates/heal/src/heal/erasure_healer.rs @@ -159,6 +159,7 @@ impl ErasureSetHealer { /// execute erasure set heal with resume #[tracing::instrument(skip(self, buckets), fields(set_disk_id = %set_disk_id, bucket_count = buckets.len()))] + #[hotpath::measure] pub async fn heal_erasure_set(&self, buckets: &[String], set_disk_id: &str) -> Result<()> { debug!( target: "rustfs::heal::erasure_healer", diff --git a/crates/heal/src/heal/task.rs b/crates/heal/src/heal/task.rs index dbd505bb3..8057f6967 100644 --- a/crates/heal/src/heal/task.rs +++ b/crates/heal/src/heal/task.rs @@ -584,6 +584,7 @@ impl HealTask { } #[tracing::instrument(skip(self), fields(task_id = %self.id, heal_type = ?self.heal_type))] + #[hotpath::measure] pub async fn execute(&self) -> Result<()> { // update status and timestamps atomically to avoid race conditions let now = SystemTime::now(); @@ -759,6 +760,7 @@ impl HealTask { // specific heal implementation method #[tracing::instrument(skip(self), fields(bucket = %bucket, object = %object, version_id = ?version_id))] + #[hotpath::measure] async fn heal_object(&self, bucket: &str, object: &str, version_id: Option<&str>) -> Result<()> { debug!( target: "rustfs::heal::task", @@ -1404,6 +1406,7 @@ impl HealTask { self.heal_bucket_objects(bucket, prefix).await } + #[hotpath::measure] async fn heal_bucket_objects(&self, bucket: &str, prefix: &str) -> Result<()> { let mut continuation_token: Option = None; let mut scanned = 0u64; diff --git a/crates/iam/Cargo.toml b/crates/iam/Cargo.toml index 8c4a3c2aa..0ef8ef1e9 100644 --- a/crates/iam/Cargo.toml +++ b/crates/iam/Cargo.toml @@ -28,7 +28,55 @@ documentation = "https://docs.rs/rustfs-iam/latest/rustfs_iam/" [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/reqwest-0-13", + "rustfs-config/hotpath", + "rustfs-credentials/hotpath", + "rustfs-crypto/hotpath", + "rustfs-ecstore/hotpath", + "rustfs-io-metrics/hotpath", + "rustfs-madmin/hotpath", + "rustfs-policy/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-utils/hotpath", + "rustfs-test-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-crypto/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-io-metrics/hotpath-alloc", + "rustfs-madmin/hotpath-alloc", + "rustfs-policy/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-utils/hotpath-alloc", + "rustfs-test-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-crypto/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-io-metrics/hotpath-cpu", + "rustfs-madmin/hotpath-cpu", + "rustfs-policy/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-utils/hotpath-cpu", + "rustfs-test-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-credentials = { workspace = true } rustfs-config = { workspace = true, features = ["server-config-model"] } tokio = { workspace = true, features = ["fs", "rt-multi-thread"] } diff --git a/crates/iam/src/store/object.rs b/crates/iam/src/store/object.rs index f7e652118..03530084e 100644 --- a/crates/iam/src/store/object.rs +++ b/crates/iam/src/store/object.rs @@ -469,6 +469,7 @@ impl ObjectStore { }); } + #[hotpath::measure] async fn list_all_iamconfig_items(&self) -> Result>> { let (tx, mut rx) = mpsc::channel::(100); @@ -508,6 +509,7 @@ impl ObjectStore { Ok(res) } + #[hotpath::measure] async fn load_policy_doc_concurrent(&self, names: &[String], mode: LoadMode) -> Result> { let mut futures = Vec::with_capacity(names.len()); diff --git a/crates/io-core/Cargo.toml b/crates/io-core/Cargo.toml index 5d53c9288..9c309633b 100644 --- a/crates/io-core/Cargo.toml +++ b/crates/io-core/Cargo.toml @@ -27,7 +27,14 @@ categories = ["development-tools", "filesystem"] [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio", "rustfs-io-metrics/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-io-metrics/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-io-metrics/hotpath-cpu"] + [dependencies] +hotpath.workspace = true bytes = { workspace = true, features = ["serde"] } thiserror = { workspace = true } tokio = { workspace = true, features = ["io-util", "fs", "sync", "rt-multi-thread"] } diff --git a/crates/io-metrics/Cargo.toml b/crates/io-metrics/Cargo.toml index d77725ae5..64eac650b 100644 --- a/crates/io-metrics/Cargo.toml +++ b/crates/io-metrics/Cargo.toml @@ -28,7 +28,32 @@ categories = ["development-tools", "filesystem"] name = "metrics_pipeline" harness = false +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "rustfs-common/hotpath", + "rustfs-s3-ops/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-s3-ops/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-s3-ops/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true metrics = { workspace = true } rustfs-common = { workspace = true } rustfs-s3-ops = { workspace = true } diff --git a/crates/keystone/Cargo.toml b/crates/keystone/Cargo.toml index 50110221e..e67adf7f5 100644 --- a/crates/keystone/Cargo.toml +++ b/crates/keystone/Cargo.toml @@ -28,7 +28,34 @@ authors.workspace = true [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/reqwest-0-13", + "rustfs-credentials/hotpath", + "rustfs-policy/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-policy/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-policy/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true tokio = { workspace = true, features = ["rt", "sync"] } reqwest = { workspace = true, features = ["json"] } serde = { workspace = true, features = ["derive"] } diff --git a/crates/keystone/src/client.rs b/crates/keystone/src/client.rs index ffb5fcdd4..d5df169e8 100644 --- a/crates/keystone/src/client.rs +++ b/crates/keystone/src/client.rs @@ -95,6 +95,7 @@ impl KeystoneClient { } /// Validate a Keystone token + #[hotpath::measure] pub async fn validate_token(&self, token: &str) -> Result { match self.version { KeystoneVersion::V3 => self.validate_token_v3(token).await, @@ -238,6 +239,7 @@ impl KeystoneClient { } /// Get EC2 credentials for a user + #[hotpath::measure] pub async fn get_ec2_credentials(&self, user_id: &str, project_id: Option<&str>) -> Result> { let admin_token = self.get_admin_token().await?; diff --git a/crates/kms/Cargo.toml b/crates/kms/Cargo.toml index 3c46a63a3..1dd5b9385 100644 --- a/crates/kms/Cargo.toml +++ b/crates/kms/Cargo.toml @@ -28,6 +28,7 @@ categories = ["cryptography", "web-programming", "authentication"] workspace = true [dependencies] +hotpath.workspace = true # Core dependencies async-trait = { workspace = true } tokio = { workspace = true, features = ["fs", "io-util", "macros", "rt-multi-thread", "sync", "time"] } @@ -37,6 +38,8 @@ serde = { workspace = true, features = ["derive"] } serde_json = { workspace = true, features = ["raw_value"] } tracing = { workspace = true } thiserror = { workspace = true } +# Operation metrics emitted by the retry policy engine (crate::policy). +metrics = { workspace = true } # Cryptography aes-gcm = { workspace = true, features = ["rand_core"] } @@ -72,6 +75,8 @@ tokio-util = { workspace = true } [dev-dependencies] anyhow = { workspace = true } +# Debugging recorder for asserting emitted metrics in tests. +metrics-util = { version = "0.20", features = ["debugging"] } insta = { workspace = true, features = ["yaml", "json"] } tempfile = { workspace = true } temp-env = { workspace = true } @@ -80,3 +85,17 @@ tokio = { workspace = true, features = ["net", "test-util"] } [features] default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/reqwest-0-13", + "rustfs-security-governance/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-security-governance/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-security-governance/hotpath-cpu", "rustfs-utils/hotpath-cpu"] diff --git a/crates/kms/src/backends/contract_tests.rs b/crates/kms/src/backends/contract_tests.rs index a436a17e2..52d9beb18 100644 --- a/crates/kms/src/backends/contract_tests.rs +++ b/crates/kms/src/backends/contract_tests.rs @@ -27,11 +27,11 @@ //! server, so they are `#[ignore]`d in CI. Static is covered by its own //! stateless contract below. +use super::KmsBackend; use super::local::LocalKmsBackend; use super::static_kms::StaticKmsBackend; use super::vault::VaultKmsBackend; use super::vault_transit::VaultTransitKmsBackend; -use super::{KmsBackend, KmsClient}; use crate::config::KmsConfig; use crate::error::{KmsError, Result}; use crate::manager::KmsManager; @@ -46,6 +46,24 @@ use rand::RngExt as _; use std::collections::HashMap; use std::sync::Arc; +fn expect_unsupported(result: Result) { + match result { + Err(KmsError::UnsupportedCapability { .. }) => {} + other => panic!("expected UnsupportedCapability, got {other:?}"), + } +} + +/// Rotation while not Enabled: backends with rotation support must reject it +/// through the state machine; backends without it report the capability gap. +async fn expect_rotate_rejected(backend: &dyn KmsBackend, key_id: &str) { + let result = backend.rotate_key(key_id).await; + if backend.capabilities().rotate { + expect_invalid_key_state(result, ""); + } else { + expect_unsupported(result); + } +} + fn expect_invalid_key_state(result: Result, expected_fragment: &str) { match result { Err(KmsError::InvalidOperation { message }) => assert!( @@ -117,11 +135,9 @@ async fn assert_key_state(backend: &dyn KmsBackend, key_id: &str, expected: KeyS assert_eq!(described.key_metadata.key_state, expected, "unexpected state for key {key_id}"); } -/// Drives one freshly created (Enabled) key through the full state matrix. -/// -/// `backend` is the product surface; `client` drives the lifecycle -/// transitions not yet exposed through `KmsBackend`. -async fn assert_state_machine_contract(backend: &dyn KmsBackend, client: &dyn KmsClient, key_id: &str) { +/// Drives one freshly created (Enabled) key through the full state matrix, +/// entirely through the `KmsBackend` product surface. +async fn assert_state_machine_contract(backend: &dyn KmsBackend, key_id: &str) { // Enabled: cryptographic use is allowed. Keep an envelope around to prove // decryption keeps working in later states. let data_key = backend @@ -134,16 +150,13 @@ async fn assert_state_machine_contract(backend: &dyn KmsBackend, client: &dyn Km .expect("Enabled key must encrypt"); // Enabled -> Disabled. - client - .disable_key(key_id, None) - .await - .expect("disable from Enabled must succeed"); + backend.disable_key(key_id).await.expect("disable from Enabled must succeed"); assert_key_state(backend, key_id, KeyState::Disabled).await; // Disabled: new cryptographic use and rotation are rejected... expect_invalid_key_state(backend.encrypt(encrypt_request(key_id)).await, "disabled"); expect_invalid_key_state(backend.generate_data_key(generate_request(key_id)).await, "disabled"); - expect_invalid_key_state(client.rotate_key(key_id, None).await, ""); + expect_rotate_rejected(backend, key_id).await; // ...but decryption of existing data keeps working (explicit AWS deviation)... let decrypted = backend .decrypt(decrypt_request(data_key.ciphertext_blob.clone())) @@ -151,17 +164,14 @@ async fn assert_state_machine_contract(backend: &dyn KmsBackend, client: &dyn Km .expect("decrypt with a disabled key must keep working"); assert_eq!(decrypted.plaintext, data_key.plaintext_key, "decrypt must recover the original data key"); // ...disable stays idempotent, cancel has nothing to cancel, and enable recovers. - client.disable_key(key_id, None).await.expect("disable must be idempotent"); + backend.disable_key(key_id).await.expect("disable must be idempotent"); expect_invalid_key_state(backend.cancel_key_deletion(cancel_request(key_id)).await, "not pending deletion"); - client - .enable_key(key_id, None) - .await - .expect("enable from Disabled must succeed"); + backend.enable_key(key_id).await.expect("enable from Disabled must succeed"); assert_key_state(backend, key_id, KeyState::Enabled).await; // Disabled keys may still be scheduled for deletion. - client - .disable_key(key_id, None) + backend + .disable_key(key_id) .await .expect("disable before scheduling must succeed"); backend @@ -173,10 +183,9 @@ async fn assert_state_machine_contract(backend: &dyn KmsBackend, client: &dyn Km // PendingDeletion: everything except decryption and cancellation is rejected. expect_invalid_key_state(backend.encrypt(encrypt_request(key_id)).await, "pending deletion"); expect_invalid_key_state(backend.generate_data_key(generate_request(key_id)).await, "pending deletion"); - expect_invalid_key_state(client.enable_key(key_id, None).await, "pending deletion"); - expect_invalid_key_state(client.disable_key(key_id, None).await, "pending deletion"); - expect_invalid_key_state(client.rotate_key(key_id, None).await, ""); - expect_invalid_key_state(client.schedule_key_deletion(key_id, 7, None).await, "pending deletion"); + expect_invalid_key_state(backend.enable_key(key_id).await, "pending deletion"); + expect_invalid_key_state(backend.disable_key(key_id).await, "pending deletion"); + expect_rotate_rejected(backend, key_id).await; expect_invalid_key_state(backend.delete_key(schedule_request(key_id)).await, "pending deletion"); let decrypted = backend .decrypt(decrypt_request(data_key.ciphertext_blob.clone())) @@ -215,7 +224,7 @@ async fn local_fixture() -> (tempfile::TempDir, KmsConfig, LocalKmsBackend, Stri #[tokio::test] async fn local_backend_state_machine_contract() { let (_temp_dir, _config, backend, key_id) = local_fixture().await; - assert_state_machine_contract(&backend, backend.lifecycle_client(), &key_id).await; + assert_state_machine_contract(&backend, &key_id).await; } /// SSE-shaped regression: disabling a key must not break decryption of data @@ -256,10 +265,7 @@ async fn static_backend_stateless_contract() { rand::rng().fill(&mut raw_key[..]); let config = KmsConfig::static_kms(key_id.to_string(), BASE64.encode(raw_key)); let static_backend = StaticKmsBackend::new(config).await.expect("static backend should build"); - // StaticKmsBackend implements both traits with overlapping method names, - // so pin each surface once instead of qualifying every call. let backend: &dyn KmsBackend = &static_backend; - let client: &dyn KmsClient = &static_backend; let data_key = backend .generate_data_key(generate_request(key_id)) @@ -275,9 +281,11 @@ async fn static_backend_stateless_contract() { expect_invalid_key_state(backend.create_key(create_request("another-key".to_string())).await, "read-only"); expect_invalid_key_state(backend.delete_key(schedule_request(key_id)).await, "read-only"); expect_invalid_key_state(backend.cancel_key_deletion(cancel_request(key_id)).await, "read-only"); - expect_invalid_key_state(client.disable_key(key_id, None).await, "read-only"); - expect_invalid_key_state(client.schedule_key_deletion(key_id, 7, None).await, "read-only"); - expect_invalid_key_state(client.rotate_key(key_id, None).await, "read-only"); + // Enable/disable and rotation are capability gaps at the product + // surface, not state-machine rejections. + expect_unsupported(backend.enable_key(key_id).await); + expect_unsupported(backend.disable_key(key_id).await); + expect_unsupported(backend.rotate_key(key_id).await); } fn vault_dev_config(constructor: fn(url::Url, String) -> KmsConfig) -> KmsConfig { @@ -298,7 +306,15 @@ async fn vault_kv2_backend_state_machine_contract() { .await .expect("key should be created"); - assert_state_machine_contract(&backend, backend.lifecycle_client(), &created.key_id).await; + assert_state_machine_contract(&backend, &created.key_id).await; + + // KV2 additionally supports version-retaining rotation, which must only + // work while the key is Enabled (the shared matrix covered the + // rejections). + backend + .rotate_key(&created.key_id) + .await + .expect("rotation of an Enabled KV2 key must succeed"); // Cleanup: leave the key pending deletion so repeated runs stay tidy. let _ = backend.delete_key(schedule_request(&created.key_id)).await; @@ -316,13 +332,12 @@ async fn vault_transit_backend_state_machine_contract() { .await .expect("key should be created"); - assert_state_machine_contract(&backend, backend.lifecycle_client(), &created.key_id).await; + assert_state_machine_contract(&backend, &created.key_id).await; // Transit additionally supports rotation, which must only work while the // key is Enabled (the shared matrix already covered the rejections). backend - .lifecycle_client() - .rotate_key(&created.key_id, None) + .rotate_key(&created.key_id) .await .expect("rotation of an Enabled transit key must succeed"); diff --git a/crates/kms/src/backends/local.rs b/crates/kms/src/backends/local.rs index 2d410f6d7..1a12b8593 100644 --- a/crates/kms/src/backends/local.rs +++ b/crates/kms/src/backends/local.rs @@ -14,9 +14,7 @@ //! Local file-based KMS backend implementation -use crate::backends::{ - BackendCapabilities, BackendInfo, ExpiredKeyRemoval, KmsBackend, KmsClient, StateGatedOperation, ensure_key_status_permits, -}; +use crate::backends::{BackendCapabilities, ExpiredKeyRemoval, KmsBackend, StateGatedOperation, ensure_key_status_permits}; use crate::config::KmsConfig; use crate::config::LocalConfig; use crate::encryption::{AesDekCrypto, DataKeyEnvelope, DekCrypto, generate_key_material}; @@ -399,6 +397,20 @@ pub struct LocalKmsClient { /// Per-key write locks serializing read-modify-write updates within this /// process (see [`Self::lock_key_for_write`]). key_write_locks: Mutex>>>, + /// Directory-wide writer fence for backup export (see + /// [`Self::acquire_export_fence`]). Writers hold the read side; an export + /// snapshot holds the write side so it observes a single-generation view. + export_fence: Arc>, +} + +/// Guard pairing the export-fence read lock with a per-key write mutex. +/// +/// Dropping it releases both, so every existing `lock_key_for_write` call +/// site participates in the export fence without changes. +#[must_use] +struct KeyWriteGuard { + _fence: tokio::sync::OwnedRwLockReadGuard<()>, + _key: tokio::sync::OwnedMutexGuard<()>, } // pub(crate) so the backup contract tests can anchor the manifest's @@ -465,6 +477,7 @@ impl LocalKmsClient { legacy_master_cipher, dek_crypto: AesDekCrypto::new(), key_write_locks: Mutex::new(HashMap::new()), + export_fence: Arc::new(tokio::sync::RwLock::new(())), }; client.validate_existing_keys().await?; Ok(client) @@ -507,6 +520,7 @@ impl LocalKmsClient { legacy_master_cipher, dek_crypto: AesDekCrypto::new(), key_write_locks: Mutex::new(HashMap::new()), + export_fence: Arc::new(tokio::sync::RwLock::new(())), }) } @@ -517,12 +531,40 @@ impl LocalKmsClient { /// delete with a rewrite. Cross-process writers sharing a key directory /// remain unsupported. Entries live for the client's lifetime; the table /// is bounded by the number of distinct key ids this process touches. - async fn lock_key_for_write(&self, key_id: &str) -> tokio::sync::OwnedMutexGuard<()> { + async fn lock_key_for_write(&self, key_id: &str) -> KeyWriteGuard { + // Fence first, per-key mutex second: the ordering is uniform across + // all writers, so an export waiting on the write side can never + // deadlock with a writer holding a key mutex. + let fence = Arc::clone(&self.export_fence).read_owned().await; let lock = { let mut locks = self.key_write_locks.lock().expect("Local KMS key write lock table poisoned"); Arc::clone(locks.entry(key_id.to_string()).or_default()) }; - lock.lock_owned().await + KeyWriteGuard { + _fence: fence, + _key: lock.lock_owned().await, + } + } + + /// Block every key-directory writer while a backup export collects its + /// snapshot, so all records belong to one generation. + /// + /// Mutating operations hold the read side (via [`Self::lock_key_for_write`] + /// or [`Self::save_new_master_key`]); the export holds the write side only + /// for the collection phase, never while encrypting or writing the bundle. + pub(crate) async fn acquire_export_fence(&self) -> tokio::sync::OwnedRwLockWriteGuard<()> { + Arc::clone(&self.export_fence).write_owned().await + } + + /// Key directory root, exposed for the backup export module. + pub(crate) fn key_directory(&self) -> &Path { + &self.config.key_dir + } + + /// Absolute path of the master-key KDF salt file, exposed for the backup + /// export module. + pub(crate) fn master_key_salt_file(&self) -> PathBuf { + Self::master_key_salt_path(&self.config) } /// Derive a 256-bit key from the master key string using a persistent Argon2id salt. @@ -799,6 +841,11 @@ impl LocalKmsClient { } async fn save_new_master_key(&self, master_key: &MasterKeyInfo, key_material: &[u8]) -> Result<()> { + // Creates never take the per-key write lock (`NoClobber` publishing + // already linearizes them), so they join the export fence here. This + // must stay the only fence acquisition on the create path: the fence + // read lock is not reentrant while an export waits for the write side. + let _fence = Arc::clone(&self.export_fence).read_owned().await; let key_path = self.master_key_path(&master_key.key_id)?; let content = self.encode_master_key(master_key, key_material)?; let temp_path = key_path.with_extension(format!("tmp-{}", uuid::Uuid::new_v4())); @@ -937,9 +984,12 @@ impl LocalKmsClient { } } -#[async_trait] -impl KmsClient for LocalKmsClient { - async fn generate_data_key(&self, request: &GenerateKeyRequest, context: Option<&OperationContext>) -> Result { +impl LocalKmsClient { + pub(crate) async fn generate_data_key( + &self, + request: &GenerateKeyRequest, + context: Option<&OperationContext>, + ) -> Result { debug!("Generating data key for master key: {}", request.master_key_id); let key_info = self.describe_key(&request.master_key_id, context).await?; @@ -980,7 +1030,7 @@ impl KmsClient for LocalKmsClient { Ok(data_key) } - async fn encrypt(&self, request: &EncryptRequest, context: Option<&OperationContext>) -> Result { + pub(crate) async fn encrypt(&self, request: &EncryptRequest, context: Option<&OperationContext>) -> Result { debug!("Encrypting data with key: {}", request.key_id); // Verify key exists and its state allows encryption @@ -997,7 +1047,7 @@ impl KmsClient for LocalKmsClient { }) } - async fn decrypt(&self, request: &DecryptRequest, _context: Option<&OperationContext>) -> Result> { + pub(crate) async fn decrypt(&self, request: &DecryptRequest, _context: Option<&OperationContext>) -> Result> { debug!("Decrypting data"); // Parse the data key envelope from ciphertext @@ -1031,7 +1081,14 @@ impl KmsClient for LocalKmsClient { Ok(plaintext) } - async fn create_key(&self, key_id: &str, algorithm: &str, context: Option<&OperationContext>) -> Result { + /// Test-only lifecycle driver: the product path goes through [`KmsBackend`]. + #[cfg(test)] + pub(crate) async fn create_key( + &self, + key_id: &str, + algorithm: &str, + context: Option<&OperationContext>, + ) -> Result { debug!("Creating master key: {}", key_id); // Check if key already exists @@ -1060,14 +1117,18 @@ impl KmsClient for LocalKmsClient { Ok(master_key) } - async fn describe_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn describe_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { debug!("Describing key: {}", key_id); let master_key = self.load_master_key(key_id).await?; Ok(master_key.into()) } - async fn list_keys(&self, request: &ListKeysRequest, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn list_keys( + &self, + request: &ListKeysRequest, + _context: Option<&OperationContext>, + ) -> Result { debug!("Listing keys"); let mut keys = Vec::new(); @@ -1111,7 +1172,7 @@ impl KmsClient for LocalKmsClient { }) } - async fn enable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { + pub(crate) async fn enable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { debug!("Enabling key: {}", key_id); let _write_guard = self.lock_key_for_write(key_id).await; @@ -1129,7 +1190,7 @@ impl KmsClient for LocalKmsClient { Ok(()) } - async fn disable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { + pub(crate) async fn disable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { debug!("Disabling key: {}", key_id); let _write_guard = self.lock_key_for_write(key_id).await; @@ -1146,7 +1207,9 @@ impl KmsClient for LocalKmsClient { Ok(()) } - async fn schedule_key_deletion( + /// Test-only lifecycle driver: the product path goes through [`KmsBackend`]. + #[cfg(test)] + pub(crate) async fn schedule_key_deletion( &self, key_id: &str, pending_window_days: u32, @@ -1170,7 +1233,9 @@ impl KmsClient for LocalKmsClient { Ok(()) } - async fn cancel_key_deletion(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { + /// Test-only lifecycle driver: the product path goes through [`KmsBackend`]. + #[cfg(test)] + pub(crate) async fn cancel_key_deletion(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { debug!("Canceling deletion for key: {}", key_id); let _write_guard = self.lock_key_for_write(key_id).await; @@ -1190,7 +1255,9 @@ impl KmsClient for LocalKmsClient { Ok(()) } - async fn rotate_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { + /// Test-only lifecycle driver: the product path goes through [`KmsBackend`]. + #[cfg(test)] + pub(crate) async fn rotate_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { if !fs::try_exists(self.master_key_path(key_id)?).await? { return Err(KmsError::key_not_found(key_id)); } @@ -1199,7 +1266,7 @@ impl KmsClient for LocalKmsClient { )) } - async fn health_check(&self) -> Result<()> { + pub(crate) async fn health_check(&self) -> Result<()> { // Check if key directory is accessible if !self.config.key_dir.exists() { return Err(KmsError::backend_error("Key directory does not exist")); @@ -1210,17 +1277,6 @@ impl KmsClient for LocalKmsClient { Ok(()) } - - fn backend_info(&self) -> BackendInfo { - BackendInfo::new( - "local".to_string(), - env!("CARGO_PKG_VERSION").to_string(), - self.config.key_dir.to_string_lossy().to_string(), - true, // We'll assume healthy for now - ) - .with_metadata("key_dir".to_string(), self.config.key_dir.to_string_lossy().to_string()) - .with_metadata("encrypted_at_rest".to_string(), self.master_cipher.is_some().to_string()) - } } /// LocalKmsBackend wraps LocalKmsClient and implements the KmsBackend trait diff --git a/crates/kms/src/backends/mod.rs b/crates/kms/src/backends/mod.rs index a74ed9a38..bf9a54e55 100644 --- a/crates/kms/src/backends/mod.rs +++ b/crates/kms/src/backends/mod.rs @@ -19,7 +19,6 @@ use crate::types::*; use async_trait::async_trait; use jiff::Zoned; use serde::{Deserialize, Serialize}; -use std::collections::HashMap; #[cfg(test)] mod contract_tests; @@ -99,136 +98,6 @@ pub(crate) fn ensure_key_status_permits(key_id: &str, status: &KeyStatus, operat ensure_key_state_permits(key_id, &state, operation) } -/// Abstract KMS client interface that all backends must implement -#[async_trait] -pub trait KmsClient: Send + Sync { - /// Generate a new data encryption key (DEK) - /// - /// Creates a new data key using the specified master key. The returned DataKey - /// contains both the plaintext and encrypted versions of the key. - /// - /// # Arguments - /// * `request` - The key generation request - /// * `context` - Optional operation context for auditing - /// - /// # Returns - /// Returns a DataKey containing both plaintext and encrypted key material - async fn generate_data_key(&self, request: &GenerateKeyRequest, context: Option<&OperationContext>) -> Result; - - /// Encrypt data directly using a master key - /// - /// Encrypts the provided plaintext using the specified master key. - /// This is different from generate_data_key as it encrypts user data directly. - /// - /// # Arguments - /// * `request` - The encryption request containing plaintext and key ID - /// * `context` - Optional operation context for auditing - async fn encrypt(&self, request: &EncryptRequest, context: Option<&OperationContext>) -> Result; - - /// Decrypt data using a master key - /// - /// Decrypts the provided ciphertext. The KMS automatically determines - /// which key was used for encryption based on the ciphertext metadata. - /// - /// # Arguments - /// * `request` - The decryption request containing ciphertext - /// * `context` - Optional operation context for auditing - async fn decrypt(&self, request: &DecryptRequest, context: Option<&OperationContext>) -> Result>; - - /// Create a new master key - /// - /// Creates a new master key in the KMS with the specified ID. - /// Returns an error if a key with the same ID already exists. - /// - /// # Arguments - /// * `key_id` - Unique identifier for the new key - /// * `algorithm` - Key algorithm (e.g., "AES_256") - /// * `context` - Optional operation context for auditing - async fn create_key(&self, key_id: &str, algorithm: &str, context: Option<&OperationContext>) -> Result; - - /// Get information about a specific key - /// - /// Returns metadata and information about the specified key. - /// - /// # Arguments - /// * `key_id` - The key identifier - /// * `context` - Optional operation context for auditing - async fn describe_key(&self, key_id: &str, context: Option<&OperationContext>) -> Result; - - /// List available keys - /// - /// Returns a paginated list of keys available in the KMS. - /// - /// # Arguments - /// * `request` - List request parameters (pagination, filters) - /// * `context` - Optional operation context for auditing - async fn list_keys(&self, request: &ListKeysRequest, context: Option<&OperationContext>) -> Result; - - /// Enable a key - /// - /// Enables a previously disabled key, allowing it to be used for cryptographic operations. - /// - /// # Arguments - /// * `key_id` - The key identifier - /// * `context` - Optional operation context for auditing - async fn enable_key(&self, key_id: &str, context: Option<&OperationContext>) -> Result<()>; - - /// Disable a key - /// - /// Disables a key, preventing it from being used for new cryptographic operations. - /// Existing encrypted data can still be decrypted. - /// - /// # Arguments - /// * `key_id` - The key identifier - /// * `context` - Optional operation context for auditing - async fn disable_key(&self, key_id: &str, context: Option<&OperationContext>) -> Result<()>; - - /// Schedule key deletion - /// - /// Schedules a key for deletion after a specified number of days. - /// This allows for a grace period to recover the key if needed. - /// - /// # Arguments - /// * `key_id` - The key identifier - /// * `pending_window_days` - Number of days before actual deletion - /// * `context` - Optional operation context for auditing - async fn schedule_key_deletion( - &self, - key_id: &str, - pending_window_days: u32, - context: Option<&OperationContext>, - ) -> Result<()>; - - /// Cancel key deletion - /// - /// Cancels a previously scheduled key deletion. - /// - /// # Arguments - /// * `key_id` - The key identifier - /// * `context` - Optional operation context for auditing - async fn cancel_key_deletion(&self, key_id: &str, context: Option<&OperationContext>) -> Result<()>; - - /// Rotate a key - /// - /// Creates a new version of the specified key. Previous versions remain - /// available for decryption but new operations will use the new version. - /// - /// # Arguments - /// * `key_id` - The key identifier - /// * `context` - Optional operation context for auditing - async fn rotate_key(&self, key_id: &str, context: Option<&OperationContext>) -> Result; - - /// Health check - /// - /// Performs a health check on the KMS backend to ensure it's operational. - async fn health_check(&self) -> Result<()>; - - /// Get backend information - /// - /// Returns information about the KMS backend (type, version, etc.). - fn backend_info(&self) -> BackendInfo; -} - /// Simplified KMS backend interface for manager #[async_trait] pub trait KmsBackend: Send + Sync { @@ -327,58 +196,6 @@ pub enum ExpiredKeyRemoval { NotExpired, } -/// Information about a KMS backend -#[derive(Debug, Clone)] -pub struct BackendInfo { - /// Backend type name (e.g., "local", "vault") - pub backend_type: String, - /// Backend version - pub version: String, - /// Backend endpoint or location - pub endpoint: String, - /// Whether the backend is currently healthy - pub healthy: bool, - /// Additional metadata about the backend - pub metadata: HashMap, -} - -impl BackendInfo { - /// Create a new backend info - /// - /// # Arguments - /// * `backend_type` - The type of the backend - /// * `version` - The version of the backend - /// * `endpoint` - The endpoint or location of the backend - /// * `healthy` - Whether the backend is healthy - /// - /// # Returns - /// A new BackendInfo instance - /// - pub fn new(backend_type: String, version: String, endpoint: String, healthy: bool) -> Self { - Self { - backend_type, - version, - endpoint, - healthy, - metadata: HashMap::new(), - } - } - - /// Add metadata to the backend info - /// - /// # Arguments - /// * `key` - Metadata key - /// * `value` - Metadata value - /// - /// # Returns - /// Updated BackendInfo instance - /// - pub fn with_metadata(mut self, key: String, value: String) -> Self { - self.metadata.insert(key, value); - self - } -} - /// Set of operations a KMS backend supports. /// /// Reported by [`KmsBackend::capabilities`] so callers (manager, admin API) diff --git a/crates/kms/src/backends/snapshots/rustfs_kms__backends__tests__vault_kv2_backend_capabilities.snap b/crates/kms/src/backends/snapshots/rustfs_kms__backends__tests__vault_kv2_backend_capabilities.snap index 57be08b27..2c0bd7fce 100644 --- a/crates/kms/src/backends/snapshots/rustfs_kms__backends__tests__vault_kv2_backend_capabilities.snap +++ b/crates/kms/src/backends/snapshots/rustfs_kms__backends__tests__vault_kv2_backend_capabilities.snap @@ -8,7 +8,7 @@ expression: capabilities_snapshot(backend.capabilities()) "encrypt": true, "generate_data_key": true, "physical_delete": true, - "rotate": false, + "rotate": true, "schedule_deletion": true, - "versioning": false + "versioning": true } diff --git a/crates/kms/src/backends/static_kms.rs b/crates/kms/src/backends/static_kms.rs index 73b988eeb..f3fc065dc 100644 --- a/crates/kms/src/backends/static_kms.rs +++ b/crates/kms/src/backends/static_kms.rs @@ -21,7 +21,7 @@ //! //! encrypted_data(plaintext_len+16) || nonce (12 bytes) -use crate::backends::{BackendCapabilities, BackendInfo, KmsBackend, KmsClient}; +use crate::backends::{BackendCapabilities, KmsBackend}; use crate::config::{BackendConfig, KmsConfig}; use crate::encryption::DataKeyEnvelope; use crate::error::{KmsError, Result}; @@ -98,9 +98,10 @@ impl StaticKmsBackend { } } -#[async_trait] -impl KmsClient for StaticKmsBackend { - async fn generate_data_key(&self, request: &GenerateKeyRequest, _context: Option<&OperationContext>) -> Result { +impl StaticKmsBackend { + /// Generate a fresh data key and wrap it in the standard KMS envelope, + /// authenticated against the canonical encryption context. + pub(crate) fn generate_data_key_envelope(&self, request: &GenerateKeyRequest) -> Result { if request.master_key_id != self.key_id { return Err(KmsError::key_not_found(&request.master_key_id)); } @@ -151,7 +152,8 @@ impl KmsClient for StaticKmsBackend { )) } - async fn encrypt(&self, request: &EncryptRequest, _context: Option<&OperationContext>) -> Result { + /// Encrypt caller-provided plaintext into the standard KMS envelope. + pub(crate) fn encrypt_to_envelope(&self, request: &EncryptRequest) -> Result { if request.key_id != self.key_id { return Err(KmsError::key_not_found(&request.key_id)); } @@ -196,7 +198,8 @@ impl KmsClient for StaticKmsBackend { }) } - async fn decrypt(&self, request: &DecryptRequest, _context: Option<&OperationContext>) -> Result> { + /// Open a KMS envelope produced by this backend. + pub(crate) fn decrypt_envelope(&self, request: &DecryptRequest) -> Result> { let envelope: DataKeyEnvelope = serde_json::from_slice(&request.ciphertext) .map_err(|error| KmsError::cryptographic_error("parse", format!("Failed to parse data key envelope: {error}")))?; if envelope.master_key_id != self.key_id { @@ -235,14 +238,8 @@ impl KmsClient for StaticKmsBackend { Ok(plaintext) } - async fn create_key(&self, key_id: &str, _algorithm: &str, _context: Option<&OperationContext>) -> Result { - if key_id == self.key_id { - return Err(KmsError::key_already_exists(key_id)); - } - Err(KmsError::invalid_operation("Static KMS is read-only: cannot create new keys")) - } - - async fn describe_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { + /// Describe the single configured key. + pub(crate) fn configured_key_info(&self, key_id: &str) -> Result { if key_id != self.key_id { return Err(KmsError::key_not_found(key_id)); } @@ -263,7 +260,8 @@ impl KmsClient for StaticKmsBackend { }) } - async fn list_keys(&self, request: &ListKeysRequest, _context: Option<&OperationContext>) -> Result { + /// List the single configured key, honouring the pagination marker. + pub(crate) fn list_configured_key(&self, request: &ListKeysRequest) -> Result { let key_info = KeyInfo { key_id: self.key_id.clone(), description: Some("Static single-key KMS backend".to_string()), @@ -295,57 +293,6 @@ impl KmsClient for StaticKmsBackend { truncated: false, }) } - - async fn enable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { - if key_id != self.key_id { - return Err(KmsError::key_not_found(key_id)); - } - // Static KMS key is always enabled - Ok(()) - } - - async fn disable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { - if key_id != self.key_id { - return Err(KmsError::key_not_found(key_id)); - } - Err(KmsError::invalid_operation("Static KMS is read-only: cannot disable keys")) - } - - async fn schedule_key_deletion( - &self, - key_id: &str, - _pending_window_days: u32, - _context: Option<&OperationContext>, - ) -> Result<()> { - if key_id != self.key_id { - return Err(KmsError::key_not_found(key_id)); - } - Err(KmsError::invalid_operation("Static KMS is read-only: cannot schedule key deletion")) - } - - async fn cancel_key_deletion(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { - if key_id != self.key_id { - return Err(KmsError::key_not_found(key_id)); - } - Err(KmsError::invalid_operation("Static KMS is read-only: cannot cancel key deletion")) - } - - async fn rotate_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { - if key_id != self.key_id { - return Err(KmsError::key_not_found(key_id)); - } - Err(KmsError::invalid_operation("Static KMS is read-only: cannot rotate keys")) - } - - async fn health_check(&self) -> Result<()> { - // Static KMS is always healthy if it was successfully initialized - Ok(()) - } - - fn backend_info(&self) -> BackendInfo { - BackendInfo::new("static".to_string(), env!("CARGO_PKG_VERSION").to_string(), "local".to_string(), true) - .with_metadata("key_id".to_string(), self.key_id.clone()) - } } #[async_trait] @@ -359,12 +306,12 @@ impl KmsBackend for StaticKmsBackend { } async fn encrypt(&self, request: EncryptRequest) -> Result { - ::encrypt(self, &request, None).await + self.encrypt_to_envelope(&request) } async fn decrypt(&self, request: DecryptRequest) -> Result { let key_id = self.key_id.clone(); - let plaintext = ::decrypt(self, &request, None).await?; + let plaintext = self.decrypt_envelope(&request)?; Ok(DecryptResponse { plaintext, key_id, @@ -380,7 +327,7 @@ impl KmsBackend for StaticKmsBackend { encryption_context: request.encryption_context, grant_tokens: Vec::new(), }; - let data_key = ::generate_data_key(self, &gen_req, None).await?; + let data_key = self.generate_data_key_envelope(&gen_req)?; let plaintext_key = data_key .plaintext @@ -395,7 +342,7 @@ impl KmsBackend for StaticKmsBackend { } async fn describe_key(&self, request: DescribeKeyRequest) -> Result { - let key_info = ::describe_key(self, &request.key_id, None).await?; + let key_info = self.configured_key_info(&request.key_id)?; let key_metadata = KeyMetadata { key_id: key_info.key_id.clone(), key_state: if key_info.status == KeyStatus::Active { @@ -415,7 +362,7 @@ impl KmsBackend for StaticKmsBackend { } async fn list_keys(&self, request: ListKeysRequest) -> Result { - ::list_keys(self, &request, None).await + self.list_configured_key(&request) } async fn delete_key(&self, request: DeleteKeyRequest) -> Result { @@ -446,7 +393,7 @@ impl KmsBackend for StaticKmsBackend { #[cfg(test)] mod tests { use super::*; - use crate::backends::{KmsBackend as KmsBackendTrait, KmsClient}; + use crate::backends::KmsBackend as KmsBackendTrait; use crate::config::{BackendConfig, KmsBackend, StaticConfig}; use crate::encryption::is_data_key_envelope; use base64::Engine as _; @@ -491,8 +438,8 @@ mod tests { // Generate data key let request = GenerateKeyRequest::new(key_id.clone(), "AES_256".to_string()) .with_context("bucket".to_string(), "test-bucket".to_string()); - let data_key = KmsClient::generate_data_key(&backend, &request, None) - .await + let data_key = backend + .generate_data_key_envelope(&request) .expect("Failed to generate data key"); assert_eq!(data_key.key_id, key_id); @@ -509,9 +456,7 @@ mod tests { // Decrypt the data key let decrypt_request = DecryptRequest::new(data_key.ciphertext.clone()).with_context("bucket".to_string(), "test-bucket".to_string()); - let decrypted = KmsClient::decrypt(&backend, &decrypt_request, None) - .await - .expect("Failed to decrypt"); + let decrypted = backend.decrypt_envelope(&decrypt_request).expect("Failed to decrypt"); assert_eq!(decrypted.as_slice(), data_key.plaintext.as_deref().expect("plaintext should exist")); } @@ -523,8 +468,8 @@ mod tests { .with_context("bucket".to_string(), "source-bucket".to_string()) .with_context("object".to_string(), "source-object".to_string()); - let data_key = KmsClient::generate_data_key(&backend, &request, None) - .await + let data_key = backend + .generate_data_key_envelope(&request) .expect("generate static KMS data key"); assert!( @@ -568,8 +513,8 @@ mod tests { let (backend, key_id, _key) = create_test_backend().await; let request = GenerateKeyRequest::new(key_id, "AES_256".to_string()) .with_context("bucket".to_string(), "source-bucket".to_string()); - let generated = KmsClient::generate_data_key(&backend, &request, None) - .await + let generated = backend + .generate_data_key_envelope(&request) .expect("generate context-bound data key"); let mut envelope: DataKeyEnvelope = serde_json::from_slice(&generated.ciphertext).expect("parse static KMS envelope"); envelope @@ -578,8 +523,8 @@ mod tests { let decrypt_request = DecryptRequest::new(serde_json::to_vec(&envelope).expect("serialize tampered envelope")) .with_context("bucket".to_string(), "different-bucket".to_string()); - let error = KmsClient::decrypt(&backend, &decrypt_request, None) - .await + let error = backend + .decrypt_envelope(&decrypt_request) .expect_err("tampering with authenticated envelope context must fail"); assert!(matches!(error, KmsError::CryptographicError { .. })); @@ -590,7 +535,7 @@ mod tests { let (backend, _key_id, _key) = create_test_backend().await; let request = GenerateKeyRequest::new("wrong-key-id".to_string(), "AES_256".to_string()); - let result = KmsClient::generate_data_key(&backend, &request, None).await; + let result = backend.generate_data_key_envelope(&request); assert!(result.is_err()); assert!(result.expect_err("should be Err").to_string().contains("wrong-key-id")); } @@ -602,7 +547,7 @@ mod tests { // Ciphertext too short let short = vec![0u8; 10]; let request = DecryptRequest::new(short); - let result = KmsClient::decrypt(&backend, &request, None).await; + let result = backend.decrypt_envelope(&request); assert!(result.is_err()); } @@ -612,9 +557,7 @@ mod tests { // Generate a valid ciphertext first let gen_request = GenerateKeyRequest::new(key_id, "AES_256".to_string()); - let data_key = KmsClient::generate_data_key(&backend, &gen_request, None) - .await - .expect("generate"); + let data_key = backend.generate_data_key_envelope(&gen_request).expect("generate"); // Tamper with the ciphertext (flip a bit in the encrypted portion) let mut tampered = data_key.ciphertext.clone(); @@ -623,7 +566,7 @@ mod tests { } let request = DecryptRequest::new(tampered); - let result = KmsClient::decrypt(&backend, &request, None).await; + let result = backend.decrypt_envelope(&request); assert!(result.is_err()); } @@ -632,7 +575,14 @@ mod tests { let (backend, key_id, _key) = create_test_backend().await; // Creating the pre-configured key should return KeyAlreadyExists - let result = KmsClient::create_key(&backend, &key_id, "AES_256", None).await; + let result = KmsBackendTrait::create_key( + &backend, + CreateKeyRequest { + key_name: Some(key_id.clone()), + ..Default::default() + }, + ) + .await; assert!(result.is_err()); assert!(result.expect_err("should be Err").to_string().contains("already exists")); } @@ -642,7 +592,14 @@ mod tests { let (backend, _key_id, _key) = create_test_backend().await; // Creating any other key should return invalid operation (read-only) - let result = KmsClient::create_key(&backend, "other-key", "AES_256", None).await; + let result = KmsBackendTrait::create_key( + &backend, + CreateKeyRequest { + key_name: Some("other-key".to_string()), + ..Default::default() + }, + ) + .await; assert!(result.is_err()); let err_msg = result.expect_err("should be Err").to_string(); assert!(err_msg.contains("read-only") || err_msg.contains("cannot create")); @@ -652,15 +609,13 @@ mod tests { async fn test_describe_key() { let (backend, key_id, _key) = create_test_backend().await; - let key_info = KmsClient::describe_key(&backend, &key_id, None) - .await - .expect("describe_key should succeed"); + let key_info = backend.configured_key_info(&key_id).expect("describe_key should succeed"); assert_eq!(key_info.key_id, key_id); assert_eq!(key_info.status, KeyStatus::Active); assert_eq!(key_info.algorithm, "AES_256"); // Wrong key ID - let result = KmsClient::describe_key(&backend, "nonexistent", None).await; + let result = backend.configured_key_info("nonexistent"); assert!(result.is_err()); } @@ -668,8 +623,8 @@ mod tests { async fn test_list_keys() { let (backend, key_id, _key) = create_test_backend().await; - let response = KmsClient::list_keys(&backend, &ListKeysRequest::default(), None) - .await + let response = backend + .list_configured_key(&ListKeysRequest::default()) .expect("list_keys should succeed"); assert_eq!(response.keys.len(), 1); assert_eq!(response.keys[0].key_id, key_id); @@ -677,61 +632,45 @@ mod tests { } #[tokio::test] - async fn test_disable_key_returns_error() { + async fn lifecycle_mutations_are_unsupported_at_the_product_surface() { let (backend, key_id, _key) = create_test_backend().await; - let result = KmsClient::disable_key(&backend, &key_id, None).await; - assert!(result.is_err()); - assert!(result.expect_err("should be Err").to_string().contains("read-only")); - } - - #[tokio::test] - async fn test_enable_key_is_noop() { - let (backend, key_id, _key) = create_test_backend().await; - - // Enable should succeed (no-op for static KMS) - KmsClient::enable_key(&backend, &key_id, None) - .await - .expect("enable_key should be no-op"); - - // Wrong key should still fail - let result = KmsClient::enable_key(&backend, "wrong", None).await; - assert!(result.is_err()); + // The static backend advertises no enable/disable or rotation + // capability, so the shared KmsBackend defaults reject all three. + for result in [ + KmsBackendTrait::enable_key(&backend, &key_id).await, + KmsBackendTrait::disable_key(&backend, &key_id).await, + KmsBackendTrait::rotate_key(&backend, &key_id).await, + ] { + let error = result.expect_err("static lifecycle mutations must be rejected"); + assert!(matches!(error, KmsError::UnsupportedCapability { .. }), "got {error:?}"); + } } #[tokio::test] async fn test_delete_key_returns_error() { let (backend, key_id, _key) = create_test_backend().await; - let result = KmsClient::schedule_key_deletion(&backend, &key_id, 7, None).await; + let result = KmsBackendTrait::delete_key( + &backend, + DeleteKeyRequest { + key_id: key_id.clone(), + pending_window_in_days: Some(7), + force_immediate: None, + }, + ) + .await; assert!(result.is_err()); assert!(result.expect_err("should be Err").to_string().contains("read-only")); } - #[tokio::test] - async fn test_rotate_key_returns_error() { - let (backend, key_id, _key) = create_test_backend().await; - - let result = KmsClient::rotate_key(&backend, &key_id, None).await; - assert!(result.is_err()); - } - #[tokio::test] async fn test_health_check() { let (backend, _key_id, _key) = create_test_backend().await; - KmsClient::health_check(&backend).await.expect("health_check should succeed"); - } - - #[tokio::test] - async fn test_backend_info() { - let (backend, key_id, _key) = create_test_backend().await; - - let info = KmsClient::backend_info(&backend); - assert_eq!(info.backend_type, "static"); - assert_eq!(info.endpoint, "local"); - assert!(info.healthy); - assert_eq!(info.metadata.get("key_id"), Some(&key_id)); + KmsBackendTrait::health_check(&backend) + .await + .expect("health_check should succeed"); } #[tokio::test] @@ -740,17 +679,13 @@ mod tests { let plaintext = b"Hello, static KMS world!"; let enc_request = EncryptRequest::new(key_id.clone(), plaintext.to_vec()); - let enc_response = KmsClient::encrypt(&backend, &enc_request, None) - .await - .expect("encrypt should succeed"); + let enc_response = backend.encrypt_to_envelope(&enc_request).expect("encrypt should succeed"); assert_eq!(enc_response.key_id, key_id); assert!(!enc_response.ciphertext.is_empty()); let dec_request = DecryptRequest::new(enc_response.ciphertext); - let decrypted = KmsClient::decrypt(&backend, &dec_request, None) - .await - .expect("decrypt should succeed"); + let decrypted = backend.decrypt_envelope(&dec_request).expect("decrypt should succeed"); assert_eq!(decrypted, plaintext); } diff --git a/crates/kms/src/backends/vault.rs b/crates/kms/src/backends/vault.rs index ebc417833..24d9c9c39 100644 --- a/crates/kms/src/backends/vault.rs +++ b/crates/kms/src/backends/vault.rs @@ -19,8 +19,7 @@ use crate::backends::vault_credentials::{ token_source_for, }; use crate::backends::{ - BackendCapabilities, BackendInfo, ExpiredKeyRemoval, KmsBackend, KmsClient, StateGatedOperation, ensure_key_state_permits, - ensure_key_status_permits, + BackendCapabilities, ExpiredKeyRemoval, KmsBackend, StateGatedOperation, ensure_key_state_permits, ensure_key_status_permits, }; use crate::config::{KmsConfig, VaultConfig}; use crate::encryption::{AesDekCrypto, DataKeyEnvelope, DekCrypto, generate_key_material}; @@ -42,7 +41,6 @@ use vaultrs::{api::kv2::requests::SetSecretRequestOptions, error::ClientError, k /// Vault KMS client implementation pub struct VaultKmsClient { credentials: Arc, - config: VaultConfig, /// Mount path for the KV engine (typically "kv" or "secret") kv_mount: String, /// Path prefix for storing keys @@ -200,7 +198,6 @@ impl VaultKmsClient { credentials, kv_mount: config.kv_mount.clone(), key_path_prefix: config.key_path_prefix.clone(), - config, dek_crypto: AesDekCrypto::new(), retry: RetryPolicy::from_config(kms_config), cancel: CancellationToken::new(), @@ -254,13 +251,6 @@ impl VaultKmsClient { Ok(general_purpose::STANDARD.encode(key_material)) } - /// Decode key material from KV2 storage (plain Base64, see `encrypt_key_material`). - async fn decrypt_key_material(&self, encrypted_material: &str) -> Result> { - general_purpose::STANDARD - .decode(encrypted_material) - .map_err(|e| KmsError::cryptographic_error("decrypt", e.to_string())) - } - /// Read the immutable material record of one key version. /// /// A missing record fails closed with [`KmsError::KeyVersionNotFound`]; falling @@ -589,9 +579,12 @@ impl VaultKmsClient { } } -#[async_trait] -impl KmsClient for VaultKmsClient { - async fn generate_data_key(&self, request: &GenerateKeyRequest, _context: Option<&OperationContext>) -> Result { +impl VaultKmsClient { + pub(crate) async fn generate_data_key( + &self, + request: &GenerateKeyRequest, + _context: Option<&OperationContext>, + ) -> Result { debug!("Generating data key for master key: {}", request.master_key_id); let key_data = self.get_key_data(&request.master_key_id).await?; @@ -632,20 +625,32 @@ impl KmsClient for VaultKmsClient { Ok(data_key) } - async fn encrypt(&self, request: &EncryptRequest, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn encrypt(&self, request: &EncryptRequest, _context: Option<&OperationContext>) -> Result { debug!("Encrypting data with key: {}", request.key_id); - // Get the master key and verify its state allows encryption + // Single read of the key record: the material we wrap with and the + // version stamped into the envelope must come from the same snapshot + // (see generate_data_key). let key_data = self.get_key_data(&request.key_id).await?; ensure_key_status_permits(&request.key_id, &key_data.status, StateGatedOperation::Encrypt)?; - let key_material = self.decrypt_key_material(&key_data.encrypted_key_material).await?; + let key_material = decode_stored_key_material(&request.key_id, &key_data.encrypted_key_material) + .inspect_err(|error| warn!(key_id = %request.key_id, %error, "Vault KMS key material failed validation"))?; + let (encrypted_key, nonce) = self.dek_crypto.encrypt(&key_material, &request.plaintext).await?; - // For simplicity, we'll use a basic encryption approach - // In practice, you'd use proper AEAD encryption - let mut ciphertext = request.plaintext.clone(); - for (i, byte) in ciphertext.iter_mut().enumerate() { - *byte ^= key_material[i % key_material.len()]; - } + // Wrap the ciphertext in the same authenticated envelope that + // generate_data_key emits, so decrypt() round-trips it and resolves + // the wrapping master key version after rotations. + let envelope = DataKeyEnvelope { + key_id: uuid::Uuid::new_v4().to_string(), + master_key_id: request.key_id.clone(), + key_spec: "AES_256".to_string(), + encrypted_key, + nonce, + encryption_context: request.encryption_context.clone(), + created_at: Zoned::now(), + master_key_version: Some(key_data.version), + }; + let ciphertext = serde_json::to_vec(&envelope)?; Ok(EncryptResponse { ciphertext, @@ -655,7 +660,7 @@ impl KmsClient for VaultKmsClient { }) } - async fn decrypt(&self, request: &DecryptRequest, _context: Option<&OperationContext>) -> Result> { + pub(crate) async fn decrypt(&self, request: &DecryptRequest, _context: Option<&OperationContext>) -> Result> { debug!("Decrypting data"); // Parse the data key envelope from ciphertext @@ -697,7 +702,12 @@ impl KmsClient for VaultKmsClient { Ok(plaintext) } - async fn create_key(&self, key_id: &str, algorithm: &str, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn create_key( + &self, + key_id: &str, + algorithm: &str, + _context: Option<&OperationContext>, + ) -> Result { debug!("Creating master key: {} with algorithm: {}", key_id, algorithm); // Existence pre-check with read-confirm recovery: a create whose @@ -779,7 +789,7 @@ impl KmsClient for VaultKmsClient { Ok(master_key) } - async fn describe_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn describe_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { debug!("Describing key: {}", key_id); let key_data = self.get_key_data(key_id).await?; @@ -799,7 +809,11 @@ impl KmsClient for VaultKmsClient { }) } - async fn list_keys(&self, request: &ListKeysRequest, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn list_keys( + &self, + request: &ListKeysRequest, + _context: Option<&OperationContext>, + ) -> Result { debug!("Listing keys with limit: {:?}", request.limit); let all_keys = self.list_vault_keys().await?; @@ -836,7 +850,7 @@ impl KmsClient for VaultKmsClient { }) } - async fn enable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { + pub(crate) async fn enable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { debug!("Enabling key: {}", key_id); let mut key_data = self.get_key_data(key_id).await?; @@ -848,7 +862,7 @@ impl KmsClient for VaultKmsClient { Ok(()) } - async fn disable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { + pub(crate) async fn disable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { debug!("Disabling key: {}", key_id); let mut key_data = self.get_key_data(key_id).await?; @@ -860,39 +874,6 @@ impl KmsClient for VaultKmsClient { Ok(()) } - async fn schedule_key_deletion( - &self, - key_id: &str, - pending_window_days: u32, - _context: Option<&OperationContext>, - ) -> Result<()> { - debug!("Scheduling key deletion: {}", key_id); - - let mut key_data = self.get_key_data(key_id).await?; - ensure_key_status_permits(key_id, &key_data.status, StateGatedOperation::ScheduleDeletion)?; - key_data.status = KeyStatus::PendingDeletion; - key_data.deletion_date = Some(Zoned::now() + Duration::from_secs(pending_window_days as u64 * 86400)); - self.store_key_data(key_id, &key_data).await?; - - debug!(key_id, "Vault KMS key deletion scheduled"); - Ok(()) - } - - async fn cancel_key_deletion(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { - debug!("Canceling key deletion: {}", key_id); - - let mut key_data = self.get_key_data(key_id).await?; - if key_data.status != KeyStatus::PendingDeletion { - return Err(KmsError::invalid_key_state(format!("Key {key_id} is not pending deletion"))); - } - key_data.status = KeyStatus::Active; - key_data.deletion_date = None; - self.store_key_data(key_id, &key_data).await?; - - debug!(key_id, "Vault KMS key deletion canceled"); - Ok(()) - } - /// Rotate the master key while keeping every historical version decryptable. /// /// Commit protocol (all writes check-and-set, in this order): @@ -908,10 +889,11 @@ impl KmsClient for VaultKmsClient { /// or interrupted rotation never exposes half-committed material. Concurrent /// rotations are serialized by the check-and-set writes: at most one caller /// commits each version and the losers fail without side effects on current. - async fn rotate_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn rotate_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { debug!("Rotating master key: {}", key_id); let (mut cas, mut key_data) = self.get_key_data_versioned(key_id).await?; + ensure_key_status_permits(key_id, &key_data.status, StateGatedOperation::Rotate)?; // The material about to be frozen must be decodable: freezing poisoned // material would give legacy envelopes a permanently broken baseline. This @@ -992,7 +974,7 @@ impl KmsClient for VaultKmsClient { }) } - async fn health_check(&self) -> Result<()> { + pub(crate) async fn health_check(&self) -> Result<()> { debug!("Performing Vault health check"); // Use list_vault_keys but handle the case where no keys exist (which is normal) @@ -1014,15 +996,6 @@ impl KmsClient for VaultKmsClient { } } } - - fn backend_info(&self) -> BackendInfo { - BackendInfo::new("vault-kv2".to_string(), "0.1.0".to_string(), self.config.address.clone(), true) - .with_metadata("kv_mount".to_string(), self.kv_mount.clone()) - .with_metadata("key_prefix".to_string(), self.key_path_prefix.clone()) - // Master key material is protected only by Vault ACLs and KV2 at-rest - // encryption; there is no additional cryptographic wrapping. - .with_metadata("at_rest_protection".to_string(), "vault-kv2-acl".to_string()) - } } /// VaultKmsBackend wraps VaultKmsClient and implements the KmsBackend trait @@ -1031,12 +1004,6 @@ pub struct VaultKmsBackend { } impl VaultKmsBackend { - /// Lifecycle driver for the shared state-machine contract tests. - #[cfg(test)] - pub(crate) fn lifecycle_client(&self) -> &VaultKmsClient { - &self.client - } - /// Create a new VaultKmsBackend pub async fn new(config: KmsConfig) -> Result { config.validate()?; @@ -1296,17 +1263,32 @@ impl KmsBackend for VaultKmsBackend { }) } + async fn enable_key(&self, key_id: &str) -> Result<()> { + self.client.enable_key(key_id, None).await + } + + async fn disable_key(&self, key_id: &str) -> Result<()> { + self.client.disable_key(key_id, None).await + } + + async fn rotate_key(&self, key_id: &str) -> Result<()> { + self.client.rotate_key(key_id, None).await.map(|_| ()) + } + async fn health_check(&self) -> Result { self.client.health_check().await.map(|_| true) } fn capabilities(&self) -> BackendCapabilities { - // Rotation is unadvertised: the KV2 backend cannot rotate without - // replacing key material in place, and no historical versions are - // retained, so versioning is unsupported as well. + // Rotation freezes the outgoing material as an immutable version + // record before switching the current pointer, and envelopes resolve + // their wrapping version on decrypt, so every historical version + // stays decryptable after a rotation. BackendCapabilities::minimal() + .with_rotate(true) .with_enable_disable(true) .with_schedule_deletion(true) + .with_versioning(true) .with_physical_delete(true) } @@ -1720,19 +1702,6 @@ mod tests { assert!(!is_cas_conflict(¬_found)); } - #[tokio::test] - async fn test_vault_kv2_backend_info_reports_at_rest_protection() { - let client = VaultKmsClient::new(integration_vault_config(), &KmsConfig::default()) - .await - .expect("client"); - - let info = client.backend_info(); - assert_eq!(info.backend_type, "vault-kv2"); - assert_eq!(info.metadata.get("at_rest_protection").map(String::as_str), Some("vault-kv2-acl")); - // The KV2 backend must not present itself as Transit-backed. - assert!(!format!("{info:?}").contains("Transit")); - } - fn integration_generate_request(key_id: &str) -> GenerateKeyRequest { GenerateKeyRequest { master_key_id: key_id.to_string(), @@ -2081,4 +2050,150 @@ mod tests { let legacy: VaultKeyData = serde_json::from_value(value).expect("legacy record must deserialize"); assert!(legacy.deletion_date.is_none()); } + + /// KV2 write acknowledgement (`SecretVersionMetadata`) for `kv2::set`. + fn kv2_write_ack() -> serde_json::Value { + serde_json::json!({ + "created_time": "2026-01-01T00:00:00Z", + "custom_metadata": null, + "deletion_time": "", + "destroyed": false, + "version": 2, + }) + } + + #[tokio::test] + async fn wired_kv2_encrypt_round_trips_through_decrypt() { + // One key-record read for the encrypt, one for the decrypt. + let (_vault, client) = scripted_client(vec![ + ScriptedResponse::ok(kv2_read_data(&healthy_key_data())), + ScriptedResponse::ok(kv2_read_data(&healthy_key_data())), + ]) + .await; + let context = HashMap::from([("bucket".to_string(), "kv2".to_string())]); + + let encrypted = client + .encrypt( + &EncryptRequest { + key_id: "wired-key".to_string(), + plaintext: b"kv2-direct-encrypt".to_vec(), + encryption_context: context.clone(), + grant_tokens: Vec::new(), + }, + None, + ) + .await + .expect("encrypt must produce an envelope"); + + // The ciphertext is a real KMS envelope wrapping AEAD output that + // decrypt() can open, not an XOR of the plaintext with the master key + // material. + let envelope: DataKeyEnvelope = serde_json::from_slice(&encrypted.ciphertext).expect("envelope must parse"); + assert_eq!(envelope.master_key_id, "wired-key"); + assert_eq!(envelope.master_key_version, Some(1)); + + let decrypted = client + .decrypt( + &DecryptRequest { + ciphertext: encrypted.ciphertext.clone(), + encryption_context: context, + grant_tokens: Vec::new(), + }, + None, + ) + .await + .expect("decrypt must round-trip the envelope"); + assert_eq!(decrypted, b"kv2-direct-encrypt".to_vec()); + + // A different object context must not decrypt (checked before any + // Vault read, so no scripted response is consumed). + let error = client + .decrypt( + &DecryptRequest { + ciphertext: encrypted.ciphertext, + encryption_context: HashMap::from([("bucket".to_string(), "other".to_string())]), + grant_tokens: Vec::new(), + }, + None, + ) + .await + .expect_err("a different context must not decrypt"); + assert!(matches!(error, KmsError::ContextMismatch { .. }), "got {error:?}"); + } + + /// KV2 secret-metadata read payload (`kv2::read_metadata`) pinning the + /// current secret version used as the rotation check-and-set base. + fn kv2_metadata_read_data(current_version: u64) -> serde_json::Value { + serde_json::json!({ + "cas_required": false, + "created_time": "2026-01-01T00:00:00Z", + "current_version": current_version, + "delete_version_after": "0s", + "max_versions": 0, + "oldest_version": 0, + "updated_time": "2026-01-01T00:00:00Z", + "custom_metadata": null, + "versions": {}, + }) + } + + #[tokio::test] + async fn wired_kv2_rotate_rejected_while_disabled() { + let mut key_data = healthy_key_data(); + key_data.status = KeyStatus::Disabled; + let (vault, client) = scripted_client(vec![ + ScriptedResponse::ok(kv2_metadata_read_data(1)), + ScriptedResponse::ok(kv2_read_data(&key_data)), + ]) + .await; + + let error = client + .rotate_key("wired-key", None) + .await + .expect_err("rotation of a disabled key must be rejected"); + assert!(matches!(error, KmsError::InvalidOperation { .. }), "got {error:?}"); + + let requests = vault.requests(); + assert_eq!( + requests.len(), + 2, + "the state gate must reject after the versioned read, before any write: {requests:?}" + ); + assert!(requests.iter().all(|line| line.starts_with("GET ")), "{requests:?}"); + } + + #[tokio::test] + async fn wired_backend_lifecycle_overrides_reach_the_client() { + let mut disabled = healthy_key_data(); + disabled.status = KeyStatus::Disabled; + let vault = ScriptedVault::serve(vec![ + // disable: read the Active record, persist it Disabled. + ScriptedResponse::ok(kv2_read_data(&healthy_key_data())), + ScriptedResponse::ok(kv2_write_ack()), + // enable: read the Disabled record, persist it Active. + ScriptedResponse::ok(kv2_read_data(&disabled)), + ScriptedResponse::ok(kv2_write_ack()), + ]) + .await; + let config = KmsConfig::vault( + url::Url::parse(&vault.address).expect("scripted vault address should parse"), + "scripted-token".to_string(), + ) + .with_insecure_development_defaults(); + let backend = VaultKmsBackend::new(config).await.expect("vault kv2 backend should build"); + + backend + .disable_key("wired-key") + .await + .expect("KmsBackend::disable_key must persist through the client"); + backend + .enable_key("wired-key") + .await + .expect("KmsBackend::enable_key must persist through the client"); + + let requests = vault.requests(); + assert_eq!(requests.len(), 4, "each transition is one read plus one write: {requests:?}"); + assert!(requests[0].starts_with("GET ") && requests[2].starts_with("GET "), "{requests:?}"); + assert!(requests[1].starts_with("POST ") && requests[3].starts_with("POST "), "{requests:?}"); + } } diff --git a/crates/kms/src/backends/vault_transit.rs b/crates/kms/src/backends/vault_transit.rs index 79d5219b7..2b01e8478 100644 --- a/crates/kms/src/backends/vault_transit.rs +++ b/crates/kms/src/backends/vault_transit.rs @@ -18,9 +18,7 @@ use crate::backends::vault_credentials::{ CredentialTaskHandle, VaultClientHandle, VaultConnectionSettings, VaultCredentialPolicy, VaultCredentialProvider, token_source_for, }; -use crate::backends::{ - BackendCapabilities, BackendInfo, ExpiredKeyRemoval, KmsBackend, KmsClient, StateGatedOperation, ensure_key_state_permits, -}; +use crate::backends::{BackendCapabilities, ExpiredKeyRemoval, KmsBackend, StateGatedOperation, ensure_key_state_permits}; use crate::config::{KmsConfig, VaultTransitConfig}; use crate::encryption::{DataKeyEnvelope, generate_key_material}; use crate::error::{KmsError, Result}; @@ -506,9 +504,12 @@ impl VaultTransitKmsClient { } } -#[async_trait] -impl KmsClient for VaultTransitKmsClient { - async fn generate_data_key(&self, request: &GenerateKeyRequest, _context: Option<&OperationContext>) -> Result { +impl VaultTransitKmsClient { + pub(crate) async fn generate_data_key( + &self, + request: &GenerateKeyRequest, + _context: Option<&OperationContext>, + ) -> Result { self.ensure_key_state_allows(&request.master_key_id, StateGatedOperation::GenerateDataKey) .await?; @@ -540,7 +541,7 @@ impl KmsClient for VaultTransitKmsClient { )) } - async fn encrypt(&self, request: &EncryptRequest, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn encrypt(&self, request: &EncryptRequest, _context: Option<&OperationContext>) -> Result { let metadata = self .ensure_key_state_allows(&request.key_id, StateGatedOperation::Encrypt) .await?; @@ -556,7 +557,7 @@ impl KmsClient for VaultTransitKmsClient { }) } - async fn decrypt(&self, request: &DecryptRequest, _context: Option<&OperationContext>) -> Result> { + pub(crate) async fn decrypt(&self, request: &DecryptRequest, _context: Option<&OperationContext>) -> Result> { let envelope: DataKeyEnvelope = serde_json::from_slice(&request.ciphertext) .map_err(|e| KmsError::cryptographic_error("parse", format!("Failed to parse data key envelope: {e}")))?; @@ -578,7 +579,14 @@ impl KmsClient for VaultTransitKmsClient { .await } - async fn create_key(&self, key_id: &str, algorithm: &str, _context: Option<&OperationContext>) -> Result { + /// Test-only lifecycle driver: the product path goes through [`KmsBackend`]. + #[cfg(test)] + pub(crate) async fn create_key( + &self, + key_id: &str, + algorithm: &str, + _context: Option<&OperationContext>, + ) -> Result { if algorithm != "AES_256" { return Err(KmsError::unsupported_algorithm(algorithm)); } @@ -645,11 +653,17 @@ impl KmsClient for VaultTransitKmsClient { }) } - async fn describe_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { + /// Test-only lifecycle driver: the product path goes through [`KmsBackend`]. + #[cfg(test)] + pub(crate) async fn describe_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { self.key_info(key_id).await } - async fn list_keys(&self, request: &ListKeysRequest, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn list_keys( + &self, + request: &ListKeysRequest, + _context: Option<&OperationContext>, + ) -> Result { let all_keys = self .run("vault_transit_list_keys", OpClass::ReadIdempotent, move || async move { let vault = self.vault().map_err(AttemptError::fatal)?; @@ -692,7 +706,7 @@ impl KmsClient for VaultTransitKmsClient { }) } - async fn enable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { + pub(crate) async fn enable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { // A pending deletion must be reverted through cancel_key_deletion, not // silently by enabling, so the gate rejects PendingDeletion here. let mut metadata = self.ensure_key_state_allows(key_id, StateGatedOperation::Enable).await?; @@ -701,13 +715,15 @@ impl KmsClient for VaultTransitKmsClient { self.store_key_metadata(key_id, &metadata).await } - async fn disable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { + pub(crate) async fn disable_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { let mut metadata = self.ensure_key_state_allows(key_id, StateGatedOperation::Disable).await?; metadata.key_state = KeyState::Disabled; self.store_key_metadata(key_id, &metadata).await } - async fn schedule_key_deletion( + /// Test-only lifecycle driver: the product path goes through [`KmsBackend`]. + #[cfg(test)] + pub(crate) async fn schedule_key_deletion( &self, key_id: &str, pending_window_days: u32, @@ -721,17 +737,7 @@ impl KmsClient for VaultTransitKmsClient { self.store_key_metadata(key_id, &metadata).await } - async fn cancel_key_deletion(&self, key_id: &str, _context: Option<&OperationContext>) -> Result<()> { - let mut metadata = self.get_key_metadata(key_id).await?; - if metadata.key_state != KeyState::PendingDeletion { - return Err(KmsError::invalid_key_state(format!("Key {key_id} is not pending deletion"))); - } - metadata.key_state = KeyState::Enabled; - metadata.deletion_date = None; - self.store_key_metadata(key_id, &metadata).await - } - - async fn rotate_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { + pub(crate) async fn rotate_key(&self, key_id: &str, _context: Option<&OperationContext>) -> Result { self.ensure_key_state_allows(key_id, StateGatedOperation::Rotate).await?; // Single attempt, never retried: replaying a rotate whose response was @@ -768,7 +774,7 @@ impl KmsClient for VaultTransitKmsClient { }) } - async fn health_check(&self) -> Result<()> { + pub(crate) async fn health_check(&self) -> Result<()> { self.run("vault_transit_health_check", OpClass::ReadIdempotent, move || async move { let vault = self.vault().map_err(AttemptError::fatal)?; key::list(&vault.client, &self.config.mount_path) @@ -780,11 +786,6 @@ impl KmsClient for VaultTransitKmsClient { }) .await } - - fn backend_info(&self) -> BackendInfo { - BackendInfo::new("vault-transit".to_string(), "0.1.0".to_string(), self.config.address.clone(), true) - .with_metadata("mount_path".to_string(), self.config.mount_path.clone()) - } } pub struct VaultTransitKmsBackend { @@ -792,14 +793,6 @@ pub struct VaultTransitKmsBackend { } impl VaultTransitKmsBackend { - /// Lifecycle driver for the shared state-machine contract tests. Using the - /// backend's own client keeps its in-process metadata cache coherent with - /// the transitions the tests perform. - #[cfg(test)] - pub(crate) fn lifecycle_client(&self) -> &VaultTransitKmsClient { - &self.client - } - pub async fn new(config: KmsConfig) -> Result { config.validate()?; @@ -1000,6 +993,18 @@ impl KmsBackend for VaultTransitKmsBackend { }) } + async fn enable_key(&self, key_id: &str) -> Result<()> { + self.client.enable_key(key_id, None).await + } + + async fn disable_key(&self, key_id: &str) -> Result<()> { + self.client.disable_key(key_id, None).await + } + + async fn rotate_key(&self, key_id: &str) -> Result<()> { + self.client.rotate_key(key_id, None).await.map(|_| ()) + } + async fn health_check(&self) -> Result { self.client.health_check().await.map(|_| true) } @@ -1437,4 +1442,58 @@ mod tests { assert_eq!(metadata.key_state, KeyState::Enabled); assert!(metadata.deletion_date.is_none()); } + + /// KV2 write acknowledgement (`SecretVersionMetadata`) for `kv2::set`. + fn kv2_write_ack() -> serde_json::Value { + serde_json::json!({ + "created_time": "2026-01-01T00:00:00Z", + "custom_metadata": null, + "deletion_time": "", + "destroyed": false, + "version": 2, + }) + } + + #[tokio::test] + async fn wired_backend_lifecycle_overrides_reach_the_client() { + let metadata = TransitKeyMetadata::from_create_request(&CreateKeyRequest::default()); + let vault = ScriptedVault::serve(vec![ + // disable: metadata cache miss reads KV, then persists Disabled. + ScriptedResponse::ok(metadata_read_data(&metadata)), + ScriptedResponse::ok(kv2_write_ack()), + // enable: the state gate hits the metadata cache, so only the + // persisting write goes out. + ScriptedResponse::ok(kv2_write_ack()), + // rotate: the gate hits the cache again; the single rotate + // attempt fails and must not be retried. + ScriptedResponse::error(503, "standby"), + ]) + .await; + let config = KmsConfig::vault_transit( + url::Url::parse(&vault.address).expect("scripted vault address should parse"), + "scripted-token".to_string(), + ) + .with_insecure_development_defaults(); + let backend = VaultTransitKmsBackend::new(config) + .await + .expect("vault transit backend should build"); + + backend + .disable_key("wired-key") + .await + .expect("KmsBackend::disable_key must persist through the client"); + backend + .enable_key("wired-key") + .await + .expect("KmsBackend::enable_key must persist through the client"); + let error = backend + .rotate_key("wired-key") + .await + .expect_err("the scripted 503 must fail the rotation"); + assert!(matches!(error, KmsError::BackendError { .. }), "got {error:?}"); + + let requests = vault.requests(); + assert_eq!(requests.len(), 4, "gated reads, two writes and one rotate attempt: {requests:?}"); + assert_eq!(requests[3], "POST /v1/transit/keys/wired-key/rotate", "{requests:?}"); + } } diff --git a/crates/kms/src/backup/local_export.rs b/crates/kms/src/backup/local_export.rs new file mode 100644 index 000000000..00bb905e8 --- /dev/null +++ b/crates/kms/src/backup/local_export.rs @@ -0,0 +1,998 @@ +// Copyright 2024 RustFS Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Local backend backup export: sealed, KEK-protected bundle production. +//! +//! This is the producer side only; restore lives in a follow-up change. The +//! admin API is not wired here either — callers construct the request and +//! supply the backup KEK explicitly. +//! +//! # Bundle layout +//! +//! A bundle is a directory (simple to produce, artifacts stream one file at a +//! time, and partial output is trivially recognizable because the manifest is +//! written last): +//! +//! ```text +//! / +//! manifest.json # sealed BackupManifest, written last +//! artifacts/keys/.key.enc # one per stored key record +//! artifacts/master-key.salt.enc # present when the salt file exists +//! ``` +//! +//! # Artifact payload framing +//! +//! Every artifact payload is `nonce (12 bytes) || AES-256-GCM ciphertext`, +//! encrypted under the caller-supplied backup KEK with an AAD binding of +//! `(context, backup_id, snapshot_generation, artifact path)`, so an artifact +//! cannot be swapped into another bundle or renamed within its own bundle. +//! Records already encrypted at rest stay encrypted inside the wrap; +//! plaintext-dev-only records become ciphertext-only in the bundle, which is +//! their mandatory re-wrap under the backup KEK. +//! +//! # Write protocol +//! +//! Artifacts are written and fsynced first, re-read and digest-verified, and +//! only then is the sealed manifest (completeness marker plus final digest) +//! published. A crash at any earlier point leaves a bundle without a +//! manifest, which decodes as an incomplete bundle and can never be restored. + +use crate::backends::local::{LocalKmsClient, StoredKeyProtection}; +use crate::backup::capability::{AtRestProtection, BackupBackendKind, BackupResponsibility}; +use crate::backup::error::BackupError; +use crate::backup::manifest::{ + AeadAlgorithm, ArtifactDescriptor, ArtifactKind, BackupKekDescriptor, BackupManifest, CompletenessState, ContentDigest, + DigestAlgorithm, LocalKdfDescriptor, LocalKeyDerivation, +}; +use crate::error::{KmsError, Result}; +use aes_gcm::{ + Aes256Gcm, Key, Nonce, + aead::{Aead, KeyInit, Payload}, +}; +use jiff::Zoned; +use rand::RngExt; +use serde::Deserialize; +use std::path::{Path, PathBuf}; +use tokio::fs; +use tokio::io::AsyncWriteExt; +use zeroize::Zeroizing; + +/// File name of the sealed manifest inside a bundle directory. +pub const LOCAL_BUNDLE_MANIFEST_FILE: &str = "manifest.json"; +const ARTIFACTS_DIR: &str = "artifacts"; +const KEYS_DIR: &str = "artifacts/keys"; +const SALT_ARTIFACT_PATH: &str = "artifacts/master-key.salt.enc"; +const AEAD_NONCE_LEN: usize = 12; +/// Domain-separation context for the artifact AAD binding. +const BUNDLE_AAD_CONTEXT: &str = "rustfs-kms-local-backup:v1"; + +/// Caller-supplied backup KEK: a trust root deliberately separate from the +/// business KMS hierarchy (it must not be a key that is itself part of the +/// state being backed up). Where the KEK comes from is the admin layer's +/// concern; this module only consumes it. +pub struct BackupKek { + kek_id: String, + kek_version: u32, + key: Zeroizing<[u8; 32]>, +} + +impl BackupKek { + /// Wrap 32 bytes of KEK material. The material is zeroized on drop; + /// callers should zeroize their own copy of the input. + pub fn new(kek_id: impl Into, kek_version: u32, key: [u8; 32]) -> Result { + let kek_id = kek_id.into(); + if kek_id.is_empty() { + return Err(KmsError::validation_error("backup KEK id must not be empty")); + } + Ok(Self { + kek_id, + kek_version, + key: Zeroizing::new(key), + }) + } + + /// Manifest descriptor for this KEK. + pub fn descriptor(&self) -> BackupKekDescriptor { + BackupKekDescriptor { + kek_id: self.kek_id.clone(), + kek_version: self.kek_version, + aead_algorithm: AeadAlgorithm::Aes256Gcm, + } + } + + fn cipher(&self) -> Aes256Gcm { + Aes256Gcm::new(&Key::::from(*self.key)) + } +} + +/// Parameters of one export run. +/// +/// `snapshot_generation` is injected by the caller: the contract only +/// requires it to be monotonic per deployment, and the source (persisted +/// counter, coordinated clock) is decided by the admin layer, which keeps +/// this module free of ambient time or state lookups. +#[derive(Debug, Clone)] +pub struct LocalBackupExportRequest { + /// Unique identifier for this backup. + pub backup_id: String, + /// Opaque identity of the producing deployment. + pub deployment_identity: String, + /// RustFS version string recorded in the manifest. + pub rustfs_version: String, + /// Monotonic snapshot generation this bundle belongs to. + pub snapshot_generation: u64, + /// Bundle output directory; must not exist yet or must be empty. + pub destination: PathBuf, +} + +impl LocalBackupExportRequest { + fn validate(&self) -> Result<()> { + for (field, value) in [ + ("backup_id", &self.backup_id), + ("deployment_identity", &self.deployment_identity), + ("rustfs_version", &self.rustfs_version), + ] { + if value.is_empty() { + return Err(KmsError::validation_error(format!("backup export {field} must not be empty"))); + } + } + Ok(()) + } +} + +/// Minimal projection of a stored key record: only the fields the exporter +/// needs. Unknown fields are ignored on purpose — the record travels into the +/// bundle byte-identical, so the exporter must not constrain its schema. +#[derive(Deserialize)] +struct StoredRecordProbe { + key_id: String, + #[serde(default)] + at_rest_protection: StoredKeyProtection, +} + +struct CollectedRecord { + key_id: String, + protection: StoredKeyProtection, + /// Raw record bytes exactly as stored. Zeroized on drop because + /// plaintext-dev-only records embed key material. + raw: Zeroizing>, +} + +struct CollectedSnapshot { + records: Vec, + salt: Option>, +} + +/// Export the Local backend's key directory as a sealed backup bundle. +/// +/// The directory scan runs under the export fence, so concurrent +/// create/update/delete operations are either fully included or fully +/// excluded — never half a record. Encryption and bundle writing happen after +/// the fence is released to keep it short. +/// +/// Returns the sealed manifest that was written to the bundle. +pub async fn export_local_backup( + client: &LocalKmsClient, + kek: &BackupKek, + request: &LocalBackupExportRequest, +) -> Result { + request.validate()?; + prepare_destination(&request.destination).await?; + + let snapshot = collect_snapshot(client).await?; + if snapshot.records.is_empty() { + return Err(KmsError::invalid_operation( + "Local backup export found no key records; refusing to publish an empty bundle", + )); + } + + let has_encrypted = snapshot + .records + .iter() + .any(|record| record.protection == StoredKeyProtection::EncryptedMasterKey); + if has_encrypted && snapshot.salt.is_none() { + return Err(KmsError::invalid_operation( + "key directory contains encrypted-master-key records but the master key salt file is missing; \ + the bundle would be unrestorable", + )); + } + + let manifest = build_and_write_bundle(kek, request, &snapshot).await?; + Ok(manifest) +} + +/// Read and fully validate the manifest of a local bundle directory. +/// +/// A directory without a manifest is an interrupted export: the manifest is +/// written last, so its absence means the bundle never sealed. +pub async fn read_local_bundle_manifest(bundle_dir: &Path) -> Result { + let manifest_path = bundle_dir.join(LOCAL_BUNDLE_MANIFEST_FILE); + let bytes = match fs::read(&manifest_path).await { + Ok(bytes) => bytes, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + return Err(BackupError::incomplete_bundle("bundle has no manifest; the export never sealed it").into()); + } + Err(error) => return Err(error.into()), + }; + let manifest = BackupManifest::decode(&bytes)?; + if manifest.backend != BackupBackendKind::Local { + return Err( + BackupError::corrupted(format!("bundle manifest declares backend {:?}, expected Local", manifest.backend)).into(), + ); + } + Ok(manifest) +} + +/// Read, verify, and decrypt one artifact of a local bundle. +/// +/// Fail-closed order: KEK identity, artifact presence, declared length, +/// encrypted digest, then AEAD authentication. The returned plaintext is +/// zeroized on drop. +pub async fn decrypt_bundle_artifact( + bundle_dir: &Path, + manifest: &BackupManifest, + descriptor: &ArtifactDescriptor, + kek: &BackupKek, +) -> Result>> { + manifest.backup_kek.ensure_matches(&kek.kek_id, kek.kek_version)?; + if descriptor.aead_algorithm != AeadAlgorithm::Aes256Gcm { + return Err(KmsError::unsupported_algorithm(format!( + "{:?} (local bundles are produced with AES-256-GCM)", + descriptor.aead_algorithm + ))); + } + + let artifact_path = bundle_dir.join(&descriptor.path); + let payload = match fs::read(&artifact_path).await { + Ok(payload) => payload, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + return Err(BackupError::missing_artifact(descriptor.path.clone()).into()); + } + Err(error) => return Err(error.into()), + }; + + if (payload.len() as u64) < descriptor.len { + return Err(BackupError::truncated(format!( + "artifact '{}' is {} bytes, manifest declares {}", + descriptor.path, + payload.len(), + descriptor.len + )) + .into()); + } + if payload.len() as u64 != descriptor.len { + return Err(BackupError::corrupted(format!( + "artifact '{}' is {} bytes, manifest declares {}", + descriptor.path, + payload.len(), + descriptor.len + )) + .into()); + } + if ContentDigest::sha256_of(&payload) != descriptor.encrypted_digest { + return Err(BackupError::corrupted(format!("artifact '{}' does not match its manifest digest", descriptor.path)).into()); + } + if payload.len() < AEAD_NONCE_LEN { + return Err(BackupError::corrupted(format!("artifact '{}' is too short to carry a nonce", descriptor.path)).into()); + } + + let (nonce_bytes, ciphertext) = payload.split_at(AEAD_NONCE_LEN); + let mut nonce = [0u8; AEAD_NONCE_LEN]; + nonce.copy_from_slice(nonce_bytes); + let aad = artifact_aad(&manifest.backup_id, manifest.snapshot_generation, &descriptor.path); + let plaintext = kek + .cipher() + .decrypt( + &Nonce::from(nonce), + Payload { + msg: ciphertext, + aad: &aad, + }, + ) + .map_err(|_| { + KmsError::from(BackupError::corrupted(format!( + "artifact '{}' failed authenticated decryption under the supplied backup KEK", + descriptor.path + ))) + })?; + Ok(Zeroizing::new(plaintext)) +} + +/// Scan the key directory under the export fence. +async fn collect_snapshot(client: &LocalKmsClient) -> Result { + let _fence = client.acquire_export_fence().await; + + let mut records = Vec::new(); + let mut entries = fs::read_dir(client.key_directory()).await?; + while let Some(entry) = entries.next_entry().await? { + let path = entry.path(); + if !path.extension().is_some_and(|extension| extension == "key") { + continue; + } + let stem = path + .file_stem() + .and_then(|stem| stem.to_str()) + .ok_or_else(|| KmsError::configuration_error("Local KMS key file name must be valid UTF-8"))? + .to_string(); + + let raw = Zeroizing::new(fs::read(&path).await?); + // Any unreadable record aborts the export: a bundle silently missing + // one key is worse than no bundle at all. + let probe: StoredRecordProbe = serde_json::from_slice(&raw) + .map_err(|error| KmsError::material_corrupt(&stem, format!("stored key record does not deserialize: {error}")))?; + if probe.key_id != stem { + return Err(KmsError::invalid_key(format!( + "Local KMS key file identity mismatch: expected {stem:?}, found {:?}", + probe.key_id + ))); + } + + records.push(CollectedRecord { + key_id: stem, + protection: probe.at_rest_protection, + raw, + }); + } + + let salt_path = client.master_key_salt_file(); + let salt = match fs::read(&salt_path).await { + Ok(bytes) => Some(bytes), + Err(error) if error.kind() == std::io::ErrorKind::NotFound => None, + Err(error) => return Err(error.into()), + }; + + records.sort_by(|a, b| a.key_id.cmp(&b.key_id)); + Ok(CollectedSnapshot { records, salt }) +} + +async fn build_and_write_bundle( + kek: &BackupKek, + request: &LocalBackupExportRequest, + snapshot: &CollectedSnapshot, +) -> Result { + let mut artifacts = Vec::with_capacity(snapshot.records.len() + 1); + for record in &snapshot.records { + let artifact_path = format!("{KEYS_DIR}/{}.key.enc", record.key_id); + let descriptor = encrypt_and_write_artifact(kek, request, ArtifactKind::KeyMaterial, &artifact_path, &record.raw).await?; + artifacts.push(descriptor); + } + if let Some(salt) = &snapshot.salt { + let descriptor = encrypt_and_write_artifact(kek, request, ArtifactKind::MasterKeySalt, SALT_ARTIFACT_PATH, salt).await?; + artifacts.push(descriptor); + } + + // Make the artifact directory entries durable before sealing: the sealed + // manifest must never survive a crash that its artifacts did not. + fsync_dir(&request.destination.join(KEYS_DIR)).await?; + fsync_dir(&request.destination.join(ARTIFACTS_DIR)).await?; + + let manifest = BackupManifest { + format_version: BackupManifest::FORMAT_VERSION, + backup_id: request.backup_id.clone(), + // Normalized to UTC so the stored spelling is host-independent: the + // local zone's name (or a POSIX TZ string in minimal containers) has + // no business inside a portable bundle. + created_at: Zoned::now().with_time_zone(jiff::tz::TimeZone::UTC), + rustfs_version: request.rustfs_version.clone(), + deployment_identity: request.deployment_identity.clone(), + backend: BackupBackendKind::Local, + at_rest_protection: weakest_observed_protection(&snapshot.records), + responsibility: BackupResponsibility::FullMaterial, + snapshot_generation: request.snapshot_generation, + backup_kek: kek.descriptor(), + artifacts, + local_kdf: Some(local_kdf_descriptor(snapshot)), + key_versions: None, + capability_discovery: None, + completeness: CompletenessState::InProgress, + manifest_digest: ContentDigest { + algorithm: DigestAlgorithm::Sha256, + hex: String::new(), + }, + }; + let manifest = manifest.seal()?; + let manifest_bytes = manifest.encode()?; + + write_new_file(&request.destination.join(LOCAL_BUNDLE_MANIFEST_FILE), &manifest_bytes).await?; + fsync_dir(&request.destination).await?; + Ok(manifest) +} + +/// Encrypt one artifact, write it durably, and re-read it to verify the +/// digest before it is allowed into the manifest. +async fn encrypt_and_write_artifact( + kek: &BackupKek, + request: &LocalBackupExportRequest, + kind: ArtifactKind, + artifact_path: &str, + plaintext: &[u8], +) -> Result { + let mut nonce = [0u8; AEAD_NONCE_LEN]; + rand::rng().fill(&mut nonce[..]); + let aad = artifact_aad(&request.backup_id, request.snapshot_generation, artifact_path); + let ciphertext = kek + .cipher() + .encrypt( + &Nonce::from(nonce), + Payload { + msg: plaintext, + aad: &aad, + }, + ) + .map_err(|error| KmsError::cryptographic_error("backup_artifact_encrypt", error.to_string()))?; + + let mut payload = Vec::with_capacity(AEAD_NONCE_LEN + ciphertext.len()); + payload.extend_from_slice(&nonce); + payload.extend_from_slice(&ciphertext); + + let absolute_path = request.destination.join(artifact_path); + write_new_file(&absolute_path, &payload).await?; + + // Verify what actually landed on disk, not the in-memory buffer. + let written = fs::read(&absolute_path).await?; + let digest = ContentDigest::sha256_of(&written); + if written != payload { + return Err(KmsError::internal_error(format!( + "bundle artifact '{artifact_path}' read back differently than written" + ))); + } + + Ok(ArtifactDescriptor { + kind, + path: artifact_path.to_string(), + len: payload.len() as u64, + aead_algorithm: AeadAlgorithm::Aes256Gcm, + encrypted_digest: digest, + }) +} + +/// AAD binding an artifact to its bundle identity and path. A JSON tuple +/// gives unambiguous field boundaries without a hand-rolled framing format. +fn artifact_aad(backup_id: &str, snapshot_generation: u64, artifact_path: &str) -> Vec { + serde_json::to_vec(&(BUNDLE_AAD_CONTEXT, backup_id, snapshot_generation, artifact_path)) + .expect("AAD tuple of strings and integers always serializes") +} + +/// The bundle-level protection label is the weakest state observed across +/// records: any plaintext-dev-only record marks the whole bundle, then any +/// legacy-unspecified marker (unknown until read), and only a uniformly +/// encrypted directory is labeled encrypted-master-key. +fn weakest_observed_protection(records: &[CollectedRecord]) -> AtRestProtection { + let mut has_legacy = false; + for record in records { + match record.protection { + StoredKeyProtection::PlaintextDevOnly => return AtRestProtection::PlaintextDevOnly, + StoredKeyProtection::LegacyUnspecified => has_legacy = true, + StoredKeyProtection::EncryptedMasterKey => {} + } + } + if has_legacy { + AtRestProtection::LegacyUnspecified + } else { + AtRestProtection::EncryptedMasterKey + } +} + +fn local_kdf_descriptor(snapshot: &CollectedSnapshot) -> LocalKdfDescriptor { + let mut modes = Vec::new(); + for (marker, mode) in [ + (StoredKeyProtection::EncryptedMasterKey, AtRestProtection::EncryptedMasterKey), + (StoredKeyProtection::PlaintextDevOnly, AtRestProtection::PlaintextDevOnly), + (StoredKeyProtection::LegacyUnspecified, AtRestProtection::LegacyUnspecified), + ] { + if snapshot.records.iter().any(|record| record.protection == marker) { + modes.push(mode); + } + } + + // With a salt on disk the backend derives via Argon2id; without one only + // the pre-beta.9 SHA-256 derivation can apply. For plaintext-only + // directories the derivation is informational. + let derivation = if snapshot.salt.is_some() { + LocalKeyDerivation::current_argon2id() + } else { + LocalKeyDerivation::LegacySha256 + }; + + LocalKdfDescriptor { + derivation, + protection_modes: modes, + // The verifier shape is left to the restore change; the schema keeps + // it optional so bundles without one stay valid. + master_key_verifier: None, + } +} + +async fn prepare_destination(destination: &Path) -> Result<()> { + if fs::try_exists(destination).await? { + let mut entries = fs::read_dir(destination) + .await + .map_err(|error| KmsError::invalid_operation(format!("backup destination is not a readable directory: {error}")))?; + if entries.next_entry().await?.is_some() { + return Err(KmsError::invalid_operation( + "backup destination directory is not empty; refusing to mix bundles", + )); + } + } + fs::create_dir_all(destination.join(KEYS_DIR)).await?; + Ok(()) +} + +async fn write_new_file(path: &Path, bytes: &[u8]) -> Result<()> { + let mut file = fs::OpenOptions::new() + .write(true) + .create_new(true) + .open(path) + .await + .map_err(|error| KmsError::io_error(format!("failed to create bundle file {}: {error}", path.display())))?; + file.write_all(bytes).await?; + file.sync_all().await?; + Ok(()) +} + +/// Fsync a directory so freshly created bundle entries survive power loss. +/// No-op on non-Unix platforms where directories cannot be opened for +/// syncing (mirrors the local backend's durable commit helper). +async fn fsync_dir(path: &Path) -> Result<()> { + #[cfg(unix)] + { + let path = path.to_path_buf(); + tokio::task::spawn_blocking(move || std::fs::File::open(&path)?.sync_all()) + .await + .map_err(|error| KmsError::io_error(error.to_string()))??; + } + #[cfg(not(unix))] + let _ = path; + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::config::LocalConfig; + use std::sync::Arc; + use tempfile::TempDir; + + async fn encrypted_client() -> (LocalKmsClient, TempDir) { + let temp = TempDir::new().expect("temp dir"); + let client = LocalKmsClient::new(LocalConfig { + key_dir: temp.path().to_path_buf(), + master_key: Some("test-master-key".to_string()), + file_permissions: Some(0o600), + }) + .await + .expect("client should initialize"); + (client, temp) + } + + async fn dev_client() -> (LocalKmsClient, TempDir) { + let temp = TempDir::new().expect("temp dir"); + let client = LocalKmsClient::new(LocalConfig { + key_dir: temp.path().to_path_buf(), + master_key: None, + file_permissions: Some(0o600), + }) + .await + .expect("client should initialize"); + (client, temp) + } + + fn test_kek() -> BackupKek { + BackupKek::new("backup-kek-test", 1, [0x42; 32]).expect("kek") + } + + fn export_request(destination: PathBuf) -> LocalBackupExportRequest { + LocalBackupExportRequest { + backup_id: "backup-0001".to_string(), + deployment_identity: "deployment-test".to_string(), + rustfs_version: "1.0.0-test".to_string(), + snapshot_generation: 7, + destination, + } + } + + fn walk_files(dir: &Path, out: &mut Vec) { + for entry in std::fs::read_dir(dir).expect("read dir") { + let path = entry.expect("dir entry").path(); + if path.is_dir() { + walk_files(&path, out); + } else { + out.push(path); + } + } + } + + fn contains_subslice(haystack: &[u8], needle: &[u8]) -> bool { + !needle.is_empty() && haystack.windows(needle.len()).any(|window| window == needle) + } + + #[tokio::test] + async fn export_round_trips_and_decrypts_to_source_records() { + let (client, _key_dir) = encrypted_client().await; + client.create_key("alpha", "AES_256", None).await.expect("create alpha"); + client.create_key("beta", "AES_256", None).await.expect("create beta"); + + let bundle = TempDir::new().expect("bundle dir"); + let destination = bundle.path().join("bundle"); + let kek = test_kek(); + let manifest = export_local_backup(&client, &kek, &export_request(destination.clone())) + .await + .expect("export should succeed"); + + assert_eq!(manifest.backend, BackupBackendKind::Local); + assert_eq!(manifest.responsibility, BackupResponsibility::FullMaterial); + assert_eq!(manifest.at_rest_protection, AtRestProtection::EncryptedMasterKey); + assert_eq!(manifest.snapshot_generation, 7); + let kdf = manifest.local_kdf.as_ref().expect("local kdf descriptor"); + assert_eq!(kdf.derivation, LocalKeyDerivation::current_argon2id()); + assert_eq!(kdf.protection_modes, vec![AtRestProtection::EncryptedMasterKey]); + + // alpha, beta (sorted), then the salt artifact. + assert_eq!(manifest.artifacts.len(), 3); + assert_eq!(manifest.artifacts[0].path, "artifacts/keys/alpha.key.enc"); + assert_eq!(manifest.artifacts[1].path, "artifacts/keys/beta.key.enc"); + assert_eq!(manifest.artifacts[2].kind, ArtifactKind::MasterKeySalt); + + let reread = read_local_bundle_manifest(&destination) + .await + .expect("manifest should decode"); + assert_eq!(reread, manifest); + + for (artifact, key_id) in [(&manifest.artifacts[0], "alpha"), (&manifest.artifacts[1], "beta")] { + let decrypted = decrypt_bundle_artifact(&destination, &manifest, artifact, &kek) + .await + .expect("artifact should decrypt"); + let source = fs::read(client.key_directory().join(format!("{key_id}.key"))) + .await + .expect("source record"); + assert_eq!(decrypted.as_slice(), source.as_slice(), "record {key_id} must round-trip verbatim"); + } + + let salt = decrypt_bundle_artifact(&destination, &manifest, &manifest.artifacts[2], &kek) + .await + .expect("salt should decrypt"); + let source_salt = fs::read(client.master_key_salt_file()).await.expect("source salt"); + assert_eq!(salt.as_slice(), source_salt.as_slice()); + } + + #[tokio::test] + async fn plaintext_dev_only_material_is_rewrapped_and_absent_from_bundle() { + let (client, _key_dir) = dev_client().await; + client.create_key("dev-key", "AES_256", None).await.expect("create key"); + + let material = client + .decrypt_key_material_for_export("dev-key") + .await + .expect("material should be readable"); + let source_record = fs::read(client.key_directory().join("dev-key.key")).await.expect("record"); + let record_json: serde_json::Value = serde_json::from_slice(&source_record).expect("record parses"); + let material_base64 = record_json + .get("encrypted_key_material") + .and_then(|value| value.as_str()) + .expect("material field") + .to_string(); + + let bundle = TempDir::new().expect("bundle dir"); + let destination = bundle.path().join("bundle"); + let kek = test_kek(); + let manifest = export_local_backup(&client, &kek, &export_request(destination.clone())) + .await + .expect("export should succeed"); + + assert_eq!(manifest.at_rest_protection, AtRestProtection::PlaintextDevOnly); + let kdf = manifest.local_kdf.as_ref().expect("local kdf descriptor"); + assert_eq!(kdf.protection_modes, vec![AtRestProtection::PlaintextDevOnly]); + assert_eq!(kdf.derivation, LocalKeyDerivation::LegacySha256); + assert!( + !manifest.artifacts.iter().any(|a| a.kind == ArtifactKind::MasterKeySalt), + "dev-mode directory has no salt to bundle" + ); + + // Byte-level: neither the raw material nor its base64 form may appear + // anywhere in the bundle. The mandatory KEK re-wrap is what hides it. + let mut files = Vec::new(); + walk_files(&destination, &mut files); + assert!(!files.is_empty()); + for file in files { + let bytes = std::fs::read(&file).expect("bundle file"); + assert!( + !contains_subslice(&bytes, material.as_ref()), + "raw key material leaked into {}", + file.display() + ); + assert!( + !contains_subslice(&bytes, material_base64.as_bytes()), + "base64 key material leaked into {}", + file.display() + ); + } + + // The wrapped record still round-trips for restore. + let decrypted = decrypt_bundle_artifact(&destination, &manifest, &manifest.artifacts[0], &kek) + .await + .expect("artifact should decrypt"); + assert_eq!(decrypted.as_slice(), source_record.as_slice()); + } + + #[tokio::test] + async fn export_fence_blocks_writers_until_released() { + let (client, _key_dir) = encrypted_client().await; + client.create_key("existing", "AES_256", None).await.expect("create key"); + let client = Arc::new(client); + + let fence = client.acquire_export_fence().await; + + let writer = { + let client = Arc::clone(&client); + tokio::spawn(async move { + client.create_key("new-key", "AES_256", None).await.expect("create"); + client.disable_key("existing", None).await.expect("disable"); + }) + }; + + for _ in 0..64 { + tokio::task::yield_now().await; + } + assert!(!writer.is_finished(), "writers must stay blocked while the export fence is held"); + + drop(fence); + writer.await.expect("writer should finish after fence release"); + assert!( + fs::try_exists(client.key_directory().join("new-key.key")) + .await + .expect("exists") + ); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] + async fn concurrent_writers_yield_complete_records() { + let (client, _key_dir) = encrypted_client().await; + for index in 0..5 { + client + .create_key(&format!("seed-{index}"), "AES_256", None) + .await + .expect("seed key"); + } + let client = Arc::new(client); + + let writer = { + let client = Arc::clone(&client); + tokio::spawn(async move { + for index in 0..30 { + client + .create_key(&format!("concurrent-{index}"), "AES_256", None) + .await + .expect("create"); + let target = format!("seed-{}", index % 5); + if index % 2 == 0 { + client.disable_key(&target, None).await.expect("disable"); + } else { + client.enable_key(&target, None).await.expect("enable"); + } + } + }) + }; + + let bundle = TempDir::new().expect("bundle dir"); + let destination = bundle.path().join("bundle"); + let kek = test_kek(); + let manifest = export_local_backup(&client, &kek, &export_request(destination.clone())) + .await + .expect("export should succeed under concurrent writers"); + writer.await.expect("writer task"); + + // Whatever subset of writers landed before the fence, every record in + // the bundle must be complete: parseable, self-identifying, and with + // non-empty material. No torn records, no half-updates. + let reread = read_local_bundle_manifest(&destination).await.expect("manifest decodes"); + assert_eq!(reread, manifest); + for artifact in manifest.artifacts.iter().filter(|a| a.kind == ArtifactKind::KeyMaterial) { + let record = decrypt_bundle_artifact(&destination, &manifest, artifact, &kek) + .await + .expect("record decrypts"); + let value: serde_json::Value = serde_json::from_slice(&record).expect("record is complete JSON"); + let key_id = value.get("key_id").and_then(|v| v.as_str()).expect("key_id present"); + assert_eq!(artifact.path, format!("artifacts/keys/{key_id}.key.enc")); + let material = value + .get("encrypted_key_material") + .and_then(|v| v.as_str()) + .expect("material present"); + assert!(!material.is_empty()); + } + } + + #[tokio::test] + async fn tampered_and_truncated_bundles_fail_closed() { + let (client, _key_dir) = encrypted_client().await; + client.create_key("victim", "AES_256", None).await.expect("create key"); + + let bundle = TempDir::new().expect("bundle dir"); + let destination = bundle.path().join("bundle"); + let kek = test_kek(); + let manifest = export_local_backup(&client, &kek, &export_request(destination.clone())) + .await + .expect("export should succeed"); + let artifact = &manifest.artifacts[0]; + let artifact_file = destination.join(&artifact.path); + let original_artifact = std::fs::read(&artifact_file).expect("artifact bytes"); + + // Tampered artifact byte: digest verification rejects it. + let mut tampered = original_artifact.clone(); + let last = tampered.len() - 1; + tampered[last] ^= 0x01; + std::fs::write(&artifact_file, &tampered).expect("write tampered"); + let error = decrypt_bundle_artifact(&destination, &manifest, artifact, &kek) + .await + .expect_err("tampered artifact must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::Corrupted { .. })), "got {error:?}"); + + // Truncated artifact: typed truncation error. + std::fs::write(&artifact_file, &original_artifact[..original_artifact.len() - 4]).expect("truncate"); + let error = decrypt_bundle_artifact(&destination, &manifest, artifact, &kek) + .await + .expect_err("truncated artifact must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::Truncated { .. })), "got {error:?}"); + std::fs::write(&artifact_file, &original_artifact).expect("restore artifact"); + + // Tampered manifest (generation flip): sealed digest mismatch. + let manifest_file = destination.join(LOCAL_BUNDLE_MANIFEST_FILE); + let original_manifest = std::fs::read(&manifest_file).expect("manifest bytes"); + let tampered_manifest = String::from_utf8(original_manifest.clone()) + .expect("manifest is utf-8") + .replace("\"snapshot_generation\":7", "\"snapshot_generation\":8"); + assert_ne!(tampered_manifest.as_bytes(), original_manifest.as_slice(), "tamper must apply"); + std::fs::write(&manifest_file, tampered_manifest).expect("write tampered manifest"); + let error = read_local_bundle_manifest(&destination) + .await + .expect_err("tampered manifest must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::Corrupted { .. })), "got {error:?}"); + + // Truncated manifest. + std::fs::write(&manifest_file, &original_manifest[..original_manifest.len() / 2]).expect("truncate manifest"); + let error = read_local_bundle_manifest(&destination) + .await + .expect_err("truncated manifest must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::Truncated { .. })), "got {error:?}"); + + // Missing manifest: the bundle never sealed. + std::fs::remove_file(&manifest_file).expect("remove manifest"); + let error = read_local_bundle_manifest(&destination) + .await + .expect_err("bundle without manifest must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::IncompleteBundle { .. })), "got {error:?}"); + } + + #[tokio::test] + async fn wrong_kek_is_rejected_before_decryption() { + let (client, _key_dir) = encrypted_client().await; + client.create_key("victim", "AES_256", None).await.expect("create key"); + + let bundle = TempDir::new().expect("bundle dir"); + let destination = bundle.path().join("bundle"); + let kek = test_kek(); + let manifest = export_local_backup(&client, &kek, &export_request(destination.clone())) + .await + .expect("export should succeed"); + let artifact = &manifest.artifacts[0]; + + let wrong_id = BackupKek::new("other-kek", 1, [0x42; 32]).expect("kek"); + let error = decrypt_bundle_artifact(&destination, &manifest, artifact, &wrong_id) + .await + .expect_err("mismatched KEK id must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::WrongKek { .. })), "got {error:?}"); + + let wrong_version = BackupKek::new("backup-kek-test", 2, [0x42; 32]).expect("kek"); + let error = decrypt_bundle_artifact(&destination, &manifest, artifact, &wrong_version) + .await + .expect_err("mismatched KEK version must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::WrongKek { .. })), "got {error:?}"); + + // Right identity, wrong material: AEAD authentication fails closed. + let wrong_material = BackupKek::new("backup-kek-test", 1, [0x24; 32]).expect("kek"); + let error = decrypt_bundle_artifact(&destination, &manifest, artifact, &wrong_material) + .await + .expect_err("wrong KEK material must be rejected"); + assert!(matches!(error, KmsError::Backup(BackupError::Corrupted { .. })), "got {error:?}"); + } + + #[tokio::test] + async fn missing_salt_with_encrypted_records_fails_export() { + let (client, _key_dir) = encrypted_client().await; + client.create_key("victim", "AES_256", None).await.expect("create key"); + fs::remove_file(client.master_key_salt_file()).await.expect("remove salt"); + + let bundle = TempDir::new().expect("bundle dir"); + let error = export_local_backup(&client, &test_kek(), &export_request(bundle.path().join("bundle"))) + .await + .expect_err("export without salt must fail"); + assert!(matches!(error, KmsError::InvalidOperation { .. }), "got {error:?}"); + assert!(error.to_string().contains("salt"), "got {error}"); + } + + #[tokio::test] + async fn refuses_empty_key_dir_and_nonempty_destination() { + let (client, _key_dir) = dev_client().await; + let bundle = TempDir::new().expect("bundle dir"); + let error = export_local_backup(&client, &test_kek(), &export_request(bundle.path().join("bundle"))) + .await + .expect_err("empty key dir must not produce a bundle"); + assert!(matches!(error, KmsError::InvalidOperation { .. }), "got {error:?}"); + + client.create_key("dev-key", "AES_256", None).await.expect("create key"); + let occupied = bundle.path().join("occupied"); + std::fs::create_dir_all(&occupied).expect("mkdir"); + std::fs::write(occupied.join("stale"), b"leftover").expect("occupy"); + let error = export_local_backup(&client, &test_kek(), &export_request(occupied)) + .await + .expect_err("non-empty destination must be refused"); + assert!(matches!(error, KmsError::InvalidOperation { .. }), "got {error:?}"); + } + + #[tokio::test] + async fn legacy_records_export_verbatim_with_weakest_protection_label() { + let (client, _key_dir) = encrypted_client().await; + client.create_key("modern", "AES_256", None).await.expect("create key"); + client.create_key("legacy-key", "AES_256", None).await.expect("create key"); + + // Strip the protection marker to fabricate a pre-beta.9 record, the + // same way the local backend's own legacy-compat tests do. + let legacy_path = client.key_directory().join("legacy-key.key"); + let mut record: serde_json::Value = + serde_json::from_slice(&fs::read(&legacy_path).await.expect("record")).expect("record parses"); + record + .as_object_mut() + .expect("record is an object") + .remove("at_rest_protection"); + let legacy_bytes = serde_json::to_vec_pretty(&record).expect("record serializes"); + fs::write(&legacy_path, &legacy_bytes).await.expect("write legacy record"); + + let bundle = TempDir::new().expect("bundle dir"); + let destination = bundle.path().join("bundle"); + let kek = test_kek(); + let manifest = export_local_backup(&client, &kek, &export_request(destination.clone())) + .await + .expect("export should succeed"); + + assert_eq!(manifest.at_rest_protection, AtRestProtection::LegacyUnspecified); + let kdf = manifest.local_kdf.as_ref().expect("local kdf descriptor"); + assert_eq!( + kdf.protection_modes, + vec![AtRestProtection::EncryptedMasterKey, AtRestProtection::LegacyUnspecified] + ); + + let legacy_artifact = manifest + .artifacts + .iter() + .find(|a| a.path == "artifacts/keys/legacy-key.key.enc") + .expect("legacy artifact"); + let decrypted = decrypt_bundle_artifact(&destination, &manifest, legacy_artifact, &kek) + .await + .expect("legacy artifact decrypts"); + assert_eq!(decrypted.as_slice(), legacy_bytes.as_slice(), "legacy record must travel verbatim"); + } + + #[tokio::test] + async fn record_identity_mismatch_aborts_export() { + let (client, _key_dir) = encrypted_client().await; + client.create_key("good", "AES_256", None).await.expect("create key"); + std::fs::copy(client.key_directory().join("good.key"), client.key_directory().join("evil.key")) + .expect("plant mismatched record"); + + let bundle = TempDir::new().expect("bundle dir"); + let error = export_local_backup(&client, &test_kek(), &export_request(bundle.path().join("bundle"))) + .await + .expect_err("identity mismatch must abort the export"); + assert!(matches!(error, KmsError::InvalidKey { .. }), "got {error:?}"); + } +} diff --git a/crates/kms/src/backup/manifest.rs b/crates/kms/src/backup/manifest.rs index 052642428..3eff4fad0 100644 --- a/crates/kms/src/backup/manifest.rs +++ b/crates/kms/src/backup/manifest.rs @@ -355,8 +355,10 @@ struct ManifestProbe { /// /// The manifest is the authoritative description of one backup bundle: what /// was captured, under which snapshot generation, protected by which backup -/// KEK, and which restore responsibility applies. Field order is part of the -/// canonical digest form and is frozen for this format version. +/// KEK, and which restore responsibility applies. The digest's canonical +/// form is the manifest's JSON value with the digest hex emptied (see +/// [`Self::compute_digest`]); decoders verify it against the raw stored +/// bytes and never re-serialize parsed fields. #[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] #[serde(deny_unknown_fields)] pub struct BackupManifest { @@ -415,8 +417,14 @@ impl BackupManifest { /// /// Fail-closed order: truncated or malformed input, then unknown format /// version, then a missing completeness marker, then schema decoding - /// (unknown fields, duplicate fields, missing fields), then semantic - /// validation including digest verification. + /// (unknown fields, duplicate fields, missing fields), then digest + /// verification against the raw input bytes, then semantic validation. + /// + /// The digest is verified against the bytes as stored — parsed typed + /// fields are never re-serialized for verification, so a field whose + /// string form does not round-trip byte-identically through its parsed + /// representation (timestamps in environment-dependent time zone + /// spellings, for example) cannot produce a spurious mismatch. pub fn decode(bytes: &[u8]) -> Result { let probe: ManifestProbe = serde_json::from_slice(bytes).map_err(map_serde_error)?; if probe.format_version != Self::FORMAT_VERSION { @@ -429,7 +437,9 @@ impl BackupManifest { return Err(BackupError::incomplete_bundle("manifest has no completeness marker")); } let manifest: Self = serde_json::from_slice(bytes).map_err(map_serde_error)?; - manifest.validate()?; + manifest.validate_pre_digest()?; + Self::verify_digest_in_bytes(bytes, &manifest.manifest_digest)?; + manifest.validate_content()?; Ok(manifest) } @@ -449,23 +459,30 @@ impl BackupManifest { Ok(self) } - /// Compute the digest over the canonical manifest bytes. + /// Compute the digest over the canonical manifest form. /// - /// Canonical form: compact JSON serialization of this manifest with the - /// digest hex emptied. Field order is struct declaration order and is - /// frozen for format version 1, so the same manifest content always - /// hashes to the same value. + /// Canonical form: the JSON *value* of the manifest with the digest hex + /// emptied, object keys rebuilt in bytewise-sorted order at every level + /// (see [`canonicalize_value`]), then serialized compactly. The value + /// layer is what makes sealing and decoding agree byte-for-byte — a + /// decoder recovers the identical value from the raw stored bytes + /// without round-tripping any typed field through parse-and-reprint — + /// and the explicit key sort makes the bytes independent of + /// `serde_json`'s map implementation (`preserve_order` on or off). pub fn compute_digest(&self) -> Result { let mut unsealed = self.clone(); unsealed.manifest_digest = ContentDigest::placeholder(self.manifest_digest.algorithm); - let canonical = serde_json::to_vec(&unsealed) + let value = serde_json::to_value(&unsealed) .map_err(|error| BackupError::corrupted(format!("manifest canonicalization failed: {error}")))?; - match self.manifest_digest.algorithm { - DigestAlgorithm::Sha256 => Ok(ContentDigest::sha256_of(&canonical)), - } + Self::digest_of_canonical_value(value, self.manifest_digest.algorithm) } - /// Verify the sealed digest against the current manifest content. + /// Verify the sealed digest against the current in-memory content. + /// + /// This is the producer-side check (sealing and [`Self::encode`]). + /// Decoders must use the raw stored bytes instead (see [`Self::decode`]): + /// re-serializing parsed fields is not guaranteed to reproduce the + /// stored spelling byte-for-byte. pub fn verify_digest(&self) -> Result<(), BackupError> { if !self.manifest_digest.is_well_formed() { return Err(BackupError::corrupted("manifest digest is not a well-formed digest value")); @@ -478,6 +495,33 @@ impl BackupManifest { Ok(()) } + /// Verify a declared digest against raw manifest bytes, normalizing only + /// through the JSON value layer and emptying the digest slot in place. + fn verify_digest_in_bytes(bytes: &[u8], declared: &ContentDigest) -> Result<(), BackupError> { + if !declared.is_well_formed() { + return Err(BackupError::corrupted("manifest digest is not a well-formed digest value")); + } + let mut value: serde_json::Value = serde_json::from_slice(bytes).map_err(map_serde_error)?; + let Some(slot) = value.get_mut("manifest_digest").and_then(|digest| digest.get_mut("hex")) else { + return Err(BackupError::corrupted("manifest has no digest slot")); + }; + *slot = serde_json::Value::String(String::new()); + if Self::digest_of_canonical_value(value, declared.algorithm)? != *declared { + return Err(BackupError::corrupted( + "manifest digest mismatch: content does not match the sealed digest", + )); + } + Ok(()) + } + + fn digest_of_canonical_value(value: serde_json::Value, algorithm: DigestAlgorithm) -> Result { + let canonical = serde_json::to_vec(&canonicalize_value(value)) + .map_err(|error| BackupError::corrupted(format!("manifest canonicalization failed: {error}")))?; + match algorithm { + DigestAlgorithm::Sha256 => Ok(ContentDigest::sha256_of(&canonical)), + } + } + /// Look up a required artifact by kind, failing closed when absent. pub fn require_artifact(&self, kind: ArtifactKind) -> Result<&ArtifactDescriptor, BackupError> { self.artifacts @@ -486,11 +530,21 @@ impl BackupManifest { .ok_or_else(|| BackupError::missing_artifact(artifact_kind_name(kind))) } - /// Validate the full manifest contract. + /// Validate the full manifest contract against the in-memory content. /// - /// This is decode-side validation and also guards [`Self::encode`], so a - /// producer cannot publish a manifest a decoder would reject. + /// This guards [`Self::encode`], so a producer cannot publish a manifest + /// a decoder would reject. [`Self::decode`] runs the same checks but + /// verifies the digest against the raw input bytes instead. pub fn validate(&self) -> Result<(), BackupError> { + self.validate_pre_digest()?; + self.verify_digest()?; + self.validate_content() + } + + /// Checks that must run before any digest verification: an unknown + /// version or an unsealed bundle is reported as its own typed error, not + /// as a digest mismatch. + fn validate_pre_digest(&self) -> Result<(), BackupError> { if self.format_version != Self::FORMAT_VERSION { return Err(BackupError::UnknownVersion { found: self.format_version, @@ -500,7 +554,12 @@ impl BackupManifest { if self.completeness != CompletenessState::Complete { return Err(BackupError::incomplete_bundle("completeness marker records an in-progress bundle")); } - self.verify_digest()?; + Ok(()) + } + + /// Semantic validation of everything except version, completeness, and + /// digest integrity. + fn validate_content(&self) -> Result<(), BackupError> { require_non_empty("backup_id", &self.backup_id)?; require_non_empty("rustfs_version", &self.rustfs_version)?; require_non_empty("deployment_identity", &self.deployment_identity)?; @@ -645,6 +704,29 @@ fn map_serde_error(error: serde_json::Error) -> BackupError { } } +/// Rebuild a JSON value with object keys in bytewise-sorted order at every +/// nesting level (array element order is preserved). +/// +/// `serde_json`'s map keeps keys sorted by default but preserves insertion +/// order when the `preserve_order` feature is unified into the build by any +/// other crate. Digest bytes must not depend on that, so the ordering is +/// imposed explicitly here instead of being inherited from the map type. +fn canonicalize_value(value: serde_json::Value) -> serde_json::Value { + match value { + serde_json::Value::Object(map) => { + let mut entries: Vec<(String, serde_json::Value)> = map.into_iter().collect(); + entries.sort_by(|a, b| a.0.cmp(&b.0)); + let mut sorted = serde_json::Map::with_capacity(entries.len()); + for (key, entry) in entries { + sorted.insert(key, canonicalize_value(entry)); + } + serde_json::Value::Object(sorted) + } + serde_json::Value::Array(items) => serde_json::Value::Array(items.into_iter().map(canonicalize_value).collect()), + other => other, + } +} + #[cfg(test)] mod tests { use super::*; @@ -736,7 +818,7 @@ mod tests { /// canonical form and frozen. If serialization layout or field order /// changes, this value changes and the fixture test fails — which is the /// point: that is a format-version bump, not a patch. - const FIXTURE_DIGEST_HEX: &str = "01accb3e2bc51e12d17d1efc52ad6ab9441c50bec52c3fb29e0a9f92b725cdaa"; + const FIXTURE_DIGEST_HEX: &str = "a8b104d61c358cd75cc9a7691d2f7d2d96f73c53653bef8940a49e96dbdf7775"; fn fixture() -> String { FIXTURE.replace("SEALED_DIGEST_HEX", FIXTURE_DIGEST_HEX) @@ -1010,12 +1092,39 @@ mod tests { let with_discovery = fixture().replace("\"completeness\"", "\"capability_discovery\": [1], \"completeness\""); expect_corrupted(BackupManifest::decode(with_discovery.as_bytes()), "reserved"); - // Explicit null carries no data and is tolerated as absence; digest - // verification still passes because null slots are skipped on - // serialization. + // An explicit null slot decodes as absence at the schema layer, but + // sealed bundles never contain the key (`skip_serializing_if`), so + // inserting one after sealing is a byte-level modification and the + // raw-bytes digest check rejects it. let with_null = fixture().replace("\"completeness\"", "\"key_versions\": null, \"completeness\""); - let decoded = BackupManifest::decode(with_null.as_bytes()).expect("null reserved slot should decode"); - assert_eq!(decoded.key_versions, None); + expect_corrupted(BackupManifest::decode(with_null.as_bytes()), "digest mismatch"); + } + + #[test] + fn digest_verification_survives_non_round_tripping_timestamp_spellings() { + // A legacy `created_at` spelling (no time zone annotation) parses via + // the compat fallback and re-serializes differently ("+00:00[UTC]"), + // and host-dependent zone spellings can do the same. Digest + // verification therefore operates on the raw stored bytes and must + // never re-serialize parsed fields. + let sealed = seal(local_manifest_unsealed()); + let mut value = serde_json::to_value(&sealed).expect("manifest should convert to a value"); + value["created_at"] = serde_json::Value::String("2026-07-30T00:00:00+00:00".to_string()); + value["manifest_digest"]["hex"] = serde_json::Value::String(String::new()); + let digest = BackupManifest::digest_of_canonical_value(value.clone(), DigestAlgorithm::Sha256) + .expect("canonical digest should compute"); + value["manifest_digest"]["hex"] = serde_json::Value::String(digest.hex); + let bytes = serde_json::to_vec(&value).expect("manifest bytes"); + + let decoded = BackupManifest::decode(&bytes).expect("a non-round-tripping timestamp spelling must not break decoding"); + + // Precondition: the spelling really does not survive a typed + // round-trip — otherwise this test is vacuous. + let reserialized = serde_json::to_value(&decoded).expect("decoded manifest should convert to a value"); + assert_ne!(reserialized["created_at"], value["created_at"]); + // Which is exactly why the producer-side (in-memory) digest check + // cannot be used on decoded manifests. + assert!(decoded.verify_digest().is_err()); } #[test] diff --git a/crates/kms/src/backup/mod.rs b/crates/kms/src/backup/mod.rs index 59cdba350..d3f9134e9 100644 --- a/crates/kms/src/backup/mod.rs +++ b/crates/kms/src/backup/mod.rs @@ -12,13 +12,13 @@ // See the License for the specific language governing permissions and // limitations under the License. -//! Backup/restore contract types for KMS state. +//! Backup/restore contracts and backup production for KMS state. //! -//! This module is contract-only: it defines the versioned backup manifest, -//! the per-backend responsibility matrix, typed failure modes, and the -//! restore dry-run report. Nothing here is wired into handlers or backends; -//! backup export, restore orchestration, and the admin API build on these -//! types in follow-up changes. +//! The contract side defines the versioned backup manifest, the per-backend +//! responsibility matrix, typed failure modes, and the restore dry-run +//! report. [`local_export`] implements the producer side for the Local +//! backend as a crate-internal API; restore orchestration and the admin API +//! build on these pieces in follow-up changes. //! //! # Bundle model //! @@ -50,6 +50,7 @@ mod capability; mod dry_run; mod error; +pub mod local_export; mod manifest; pub use capability::{AtRestProtection, BackupBackendKind, BackupResponsibility}; @@ -57,6 +58,10 @@ pub use dry_run::{ ExternalDependencyMismatch, RestoreBlocker, RestoreBlockerCode, RestoreConflict, RestoreConflictKind, RestoreDryRunReport, }; pub use error::BackupError; +pub use local_export::{ + BackupKek, LOCAL_BUNDLE_MANIFEST_FILE, LocalBackupExportRequest, decrypt_bundle_artifact, export_local_backup, + read_local_bundle_manifest, +}; pub use manifest::{ AeadAlgorithm, ArtifactDescriptor, ArtifactKind, BackupKekDescriptor, BackupManifest, CompletenessState, ContentDigest, DigestAlgorithm, LocalKdfDescriptor, LocalKeyDerivation, ReservedSlot, diff --git a/crates/kms/src/deletion_worker.rs b/crates/kms/src/deletion_worker.rs index fba80b121..3fca15b25 100644 --- a/crates/kms/src/deletion_worker.rs +++ b/crates/kms/src/deletion_worker.rs @@ -182,7 +182,6 @@ impl DeletionWorker { #[cfg(test)] mod tests { use super::*; - use crate::backends::KmsClient as _; use crate::backends::local::LocalKmsBackend; use crate::config::KmsConfig; use crate::error::KmsError; diff --git a/crates/kms/src/manager.rs b/crates/kms/src/manager.rs index 7ae6bc055..9d77e07b4 100644 --- a/crates/kms/src/manager.rs +++ b/crates/kms/src/manager.rs @@ -66,16 +66,19 @@ impl KmsManager { } /// Encrypt data with a master key + #[hotpath::measure] pub async fn encrypt(&self, request: EncryptRequest) -> Result { self.backend.encrypt(request).await } /// Decrypt data with a master key + #[hotpath::measure] pub async fn decrypt(&self, request: DecryptRequest) -> Result { self.backend.decrypt(request).await } /// Generate a data encryption key + #[hotpath::measure] pub async fn generate_data_key(&self, request: GenerateDataKeyRequest) -> Result { self.backend.generate_data_key(request).await } diff --git a/crates/kms/src/policy.rs b/crates/kms/src/policy.rs index 58411a51b..bc7fcc664 100644 --- a/crates/kms/src/policy.rs +++ b/crates/kms/src/policy.rs @@ -27,6 +27,12 @@ //! retried automatically: a response lost after the server applied the write //! would otherwise be replayed into duplicate side effects (extra key versions, //! repeated deletes). +//! +//! Every execution also records operation metrics (attempt failures by retry +//! class, terminal outcome, attempts used, wall-clock duration) through the +//! process-global `metrics` recorder. Metric labels carry only static enum +//! values — operation names, classes, outcomes — never key identifiers, key +//! material, ciphertext, or tokens. use std::future::Future; use std::time::Duration; @@ -200,6 +206,130 @@ fn equal_jitter(rng: &mut impl RngExt, cap: Duration) -> Duration { half + Duration::from_nanos(rng.random_range(0..=spread)) } +// --------------------------------------------------------------------------- +// Metrics +// +// Every execution is recorded here, at the single choke point all backend +// calls flow through, so instrumenting a new call site costs nothing beyond +// naming its operation. Label values are exclusively static enum strings +// (operation names, classes, outcomes) — key identifiers, key material, +// ciphertext, and tokens must never reach a metric label. +// --------------------------------------------------------------------------- + +/// Counter: operations executed, by `operation`, `op_class`, and `outcome`. +const METRIC_OPERATIONS_TOTAL: &str = "rustfs_kms_backend_operations_total"; +/// Counter: failed attempts, by `operation` and `error_class` (including +/// `attempt_timeout` for attempts cut off by the per-attempt timeout). +const METRIC_ATTEMPT_FAILURES_TOTAL: &str = "rustfs_kms_backend_attempt_failures_total"; +/// Histogram: wall-clock duration of a whole operation (attempts plus +/// backoff), in seconds, by `operation` and `outcome`. +const METRIC_OPERATION_DURATION_SECONDS: &str = "rustfs_kms_backend_operation_duration_seconds"; +/// Histogram: attempts one operation used before completing, by `operation` +/// and `outcome`. +const METRIC_OPERATION_ATTEMPTS: &str = "rustfs_kms_backend_operation_attempts"; + +impl OpClass { + fn as_label(self) -> &'static str { + match self { + OpClass::ReadIdempotent => "read_idempotent", + OpClass::MutatingNonIdempotent => "mutating_non_idempotent", + OpClass::Auth => "auth", + } + } +} + +impl ErrorClass { + fn as_label(self) -> &'static str { + match self { + ErrorClass::RetryableConn => "retryable_conn", + ErrorClass::RetryableStatus => "retryable_status", + ErrorClass::Fatal => "fatal", + } + } +} + +/// How one policy execution terminated, for the `outcome` metric label. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Outcome { + Success, + /// A fatal-classified failure ended the operation on its first observation. + Fatal, + /// The attempt budget ran out; the last failure was retryable (including a + /// timed-out final attempt). + BudgetExhausted, + /// The operation deadline ran out before another attempt could complete. + DeadlineExceeded, + Cancelled, +} + +impl Outcome { + fn as_label(self) -> &'static str { + match self { + Outcome::Success => "success", + Outcome::Fatal => "fatal", + Outcome::BudgetExhausted => "budget_exhausted", + Outcome::DeadlineExceeded => "deadline_exceeded", + Outcome::Cancelled => "cancelled", + } + } +} + +/// Register metric descriptions once per process. +fn describe_metrics() { + static DESCRIBE: std::sync::Once = std::sync::Once::new(); + DESCRIBE.call_once(|| { + metrics::describe_counter!( + METRIC_OPERATIONS_TOTAL, + "Total KMS backend operations executed under the operation policy, by operation, operation class, and outcome" + ); + metrics::describe_counter!( + METRIC_ATTEMPT_FAILURES_TOTAL, + "Total failed KMS backend attempts, by operation and retry classification" + ); + metrics::describe_histogram!( + METRIC_OPERATION_DURATION_SECONDS, + "Wall-clock duration of KMS backend operations including retries and backoff, in seconds" + ); + metrics::describe_histogram!( + METRIC_OPERATION_ATTEMPTS, + "Number of attempts a KMS backend operation used before completing" + ); + }); +} + +/// Record one failed attempt with its retry classification. +fn record_attempt_failure(operation: &'static str, error_class: &'static str) { + metrics::counter!( + METRIC_ATTEMPT_FAILURES_TOTAL, + "operation" => operation, + "error_class" => error_class + ) + .increment(1); +} + +/// Record the terminal outcome of one policy execution. +fn record_operation(operation: &'static str, class: OpClass, outcome: Outcome, attempts: u32, elapsed: Duration) { + metrics::counter!( + METRIC_OPERATIONS_TOTAL, + "operation" => operation, + "op_class" => class.as_label(), + "outcome" => outcome.as_label() + ) + .increment(1); + metrics::histogram!( + METRIC_OPERATION_DURATION_SECONDS, + "operation" => operation, + "outcome" => outcome.as_label() + ) + .record(elapsed.as_secs_f64()); + metrics::histogram!( + METRIC_OPERATION_ATTEMPTS, + "operation" => operation, + "outcome" => outcome.as_label() + ) + .record(f64::from(attempts)); +} + /// Run `attempt` under the policy. /// /// Each attempt is bounded by `attempt_timeout` (further capped by whatever is @@ -230,13 +360,37 @@ where /// [`execute`] with an injectable jitter source so tests can pin deterministic /// backoff durations instead of asserting around random sleeps. pub(crate) async fn execute_with_jitter( + operation: &'static str, + class: OpClass, + policy: &RetryPolicy, + cancel: &CancellationToken, + jitter: J, + attempt: F, +) -> Result +where + F: FnMut() -> Fut, + Fut: Future>, + J: FnMut(Duration) -> Duration, +{ + describe_metrics(); + let started = Instant::now(); + let mut attempts_made = 0u32; + let (outcome, result) = drive_attempts(operation, class, policy, cancel, jitter, attempt, &mut attempts_made).await; + record_operation(operation, class, outcome, attempts_made, started.elapsed()); + result +} + +/// The attempt loop behind [`execute_with_jitter`], returning the terminal +/// outcome alongside the result so the caller can record it exactly once. +async fn drive_attempts( operation: &'static str, class: OpClass, policy: &RetryPolicy, cancel: &CancellationToken, mut jitter: J, mut attempt: F, -) -> Result + attempts_made: &mut u32, +) -> (Outcome, Result) where F: FnMut() -> Fut, Fut: Future>, @@ -247,48 +401,68 @@ where let mut attempt_no = 0u32; loop { - attempt_no += 1; if cancel.is_cancelled() { - return Err(KmsError::operation_cancelled(format!( - "{operation} cancelled before attempt {attempt_no}" - ))); + return ( + Outcome::Cancelled, + Err(KmsError::operation_cancelled(format!( + "{operation} cancelled before attempt {}", + attempt_no + 1 + ))), + ); } let remaining = deadline.saturating_duration_since(Instant::now()); if remaining.is_zero() { - return Err(KmsError::operation_timed_out(format!( - "{operation} exceeded operation deadline of {:?}", - policy.op_deadline - ))); + return ( + Outcome::DeadlineExceeded, + Err(KmsError::operation_timed_out(format!( + "{operation} exceeded operation deadline of {:?}", + policy.op_deadline + ))), + ); } + attempt_no += 1; + *attempts_made = attempt_no; let attempt_budget = policy.attempt_timeout.min(remaining); let outcome = tokio::select! { biased; _ = cancel.cancelled() => { - return Err(KmsError::operation_cancelled(format!("{operation} cancelled during attempt {attempt_no}"))); + return ( + Outcome::Cancelled, + Err(KmsError::operation_cancelled(format!("{operation} cancelled during attempt {attempt_no}"))), + ); } outcome = tokio::time::timeout(attempt_budget, attempt()) => outcome, }; let failure = match outcome { - Ok(Ok(value)) => return Ok(value), - Ok(Err(failure)) => failure, - Err(_) => AttemptError { - class: ErrorClass::RetryableConn, - error: KmsError::operation_timed_out(format!( - "{operation} attempt {attempt_no} timed out after {attempt_budget:?}" - )), - }, + Ok(Ok(value)) => return (Outcome::Success, Ok(value)), + Ok(Err(failure)) => { + record_attempt_failure(operation, failure.class.as_label()); + failure + } + Err(_) => { + record_attempt_failure(operation, "attempt_timeout"); + AttemptError { + class: ErrorClass::RetryableConn, + error: KmsError::operation_timed_out(format!( + "{operation} attempt {attempt_no} timed out after {attempt_budget:?}" + )), + } + } }; - if failure.class == ErrorClass::Fatal || attempt_no >= max_attempts { - return Err(failure.error); + if failure.class == ErrorClass::Fatal { + return (Outcome::Fatal, Err(failure.error)); + } + if attempt_no >= max_attempts { + return (Outcome::BudgetExhausted, Err(failure.error)); } let backoff = jitter(backoff_cap(policy, attempt_no)); if backoff >= deadline.saturating_duration_since(Instant::now()) { // Not enough deadline budget left for another attempt. - return Err(failure.error); + return (Outcome::DeadlineExceeded, Err(failure.error)); } tracing::warn!( operation, @@ -300,7 +474,10 @@ where tokio::select! { biased; _ = cancel.cancelled() => { - return Err(KmsError::operation_cancelled(format!("{operation} cancelled during retry backoff"))); + return ( + Outcome::Cancelled, + Err(KmsError::operation_cancelled(format!("{operation} cancelled during retry backoff"))), + ); } _ = tokio::time::sleep(backoff) => {} } @@ -628,4 +805,313 @@ mod tests { assert_eq!(classify_vaultrs(&ClientError::ResponseEmptyError), ErrorClass::Fatal); assert_eq!(classify_vaultrs(&ClientError::InvalidLoginMethodError), ErrorClass::Fatal); } + + // -- Metric emission ---------------------------------------------------- + // + // Each test installs a thread-local debugging recorder and drives a + // paused-clock current-thread runtime inside it, so the emitted metrics + // (including virtual-clock durations) are fully deterministic. + + use metrics_util::MetricKind; + use metrics_util::debugging::{DebugValue, DebuggingRecorder}; + + type MetricEntry = ( + metrics_util::CompositeKey, + Option, + Option, + DebugValue, + ); + + /// Run `test` on a paused current-thread runtime under a debugging + /// recorder and return one snapshot of everything it emitted. + /// + /// A single snapshot per test on purpose: `Snapshotter::snapshot` drains + /// the recorded state, so taking it per assertion would only show the + /// first assertion any data. + fn record_metrics(test: impl FnOnce() -> std::pin::Pin>>) -> (Vec, Out) { + let recorder = DebuggingRecorder::new(); + let snapshotter = recorder.snapshotter(); + let out = metrics::with_local_recorder(&recorder, || { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_time() + .start_paused(true) + .build() + .expect("current-thread runtime must build"); + runtime.block_on(test()) + }); + (snapshotter.snapshot().into_vec(), out) + } + + fn labels_match(key: &metrics::Key, labels: &[(&str, &str)]) -> bool { + labels + .iter() + .all(|(label, expected)| key.labels().any(|l| l.key() == *label && l.value() == *expected)) + } + + fn counter_value(snapshot: &[MetricEntry], name: &str, labels: &[(&str, &str)]) -> u64 { + snapshot + .iter() + .filter_map(|(composite, _unit, _description, value)| { + let matches = composite.kind() == MetricKind::Counter + && composite.key().name() == name + && labels_match(composite.key(), labels); + match (matches, value) { + (true, DebugValue::Counter(count)) => Some(*count), + _ => None, + } + }) + .sum() + } + + fn histogram_values(snapshot: &[MetricEntry], name: &str, labels: &[(&str, &str)]) -> Vec { + snapshot + .iter() + .filter_map(|(composite, _unit, _description, value)| { + let matches = composite.kind() == MetricKind::Histogram + && composite.key().name() == name + && labels_match(composite.key(), labels); + match (matches, value) { + (true, DebugValue::Histogram(values)) => Some(values), + _ => None, + } + }) + .flatten() + .map(|value| value.into_inner()) + .collect() + } + + #[test] + fn metrics_record_retried_success_with_attempts_and_duration() { + let calls_in_test = Arc::new(AtomicU32::new(0)); + let (snapshot, ()) = record_metrics(move || { + Box::pin(async move { + let policy = policy_of(1_000, 60_000, 3, 100, 2_000); + let cancel = CancellationToken::new(); + let calls_in_attempt = calls_in_test.clone(); + execute_with_jitter("metrics_read", OpClass::ReadIdempotent, &policy, &cancel, full_jitter, move || { + let calls = calls_in_attempt.clone(); + async move { + if calls.fetch_add(1, Ordering::SeqCst) < 2 { + Err(AttemptError { + class: ErrorClass::RetryableStatus, + error: KmsError::backend_error("throttled (429)"), + }) + } else { + Ok(()) + } + } + }) + .await + .expect("retries within budget must succeed"); + }) + }); + + assert_eq!( + counter_value( + &snapshot, + METRIC_OPERATIONS_TOTAL, + &[ + ("operation", "metrics_read"), + ("op_class", "read_idempotent"), + ("outcome", "success") + ] + ), + 1 + ); + assert_eq!( + counter_value( + &snapshot, + METRIC_ATTEMPT_FAILURES_TOTAL, + &[("operation", "metrics_read"), ("error_class", "retryable_status")] + ), + 2 + ); + assert_eq!( + histogram_values( + &snapshot, + METRIC_OPERATION_ATTEMPTS, + &[("operation", "metrics_read"), ("outcome", "success")] + ), + vec![3.0] + ); + // Full-cap backoffs of 100ms and 200ms on the paused clock. + let durations = histogram_values( + &snapshot, + METRIC_OPERATION_DURATION_SECONDS, + &[("operation", "metrics_read"), ("outcome", "success")], + ); + assert_eq!(durations.len(), 1); + assert!((durations[0] - 0.3).abs() < 1e-9, "expected 0.3s of virtual backoff, got {durations:?}"); + } + + #[test] + fn metrics_record_fatal_outcome_with_single_attempt() { + let (snapshot, ()) = record_metrics(|| { + Box::pin(async { + let policy = policy_of(1_000, 60_000, 5, 100, 2_000); + let cancel = CancellationToken::new(); + let result: Result<()> = + execute_with_jitter("metrics_fatal", OpClass::ReadIdempotent, &policy, &cancel, full_jitter, || async { + Err(AttemptError { + class: ErrorClass::Fatal, + error: KmsError::access_denied("permission denied (403)"), + }) + }) + .await; + result.expect_err("a fatal failure must end the operation"); + }) + }); + + assert_eq!( + counter_value( + &snapshot, + METRIC_OPERATIONS_TOTAL, + &[("operation", "metrics_fatal"), ("outcome", "fatal")] + ), + 1 + ); + assert_eq!( + counter_value( + &snapshot, + METRIC_ATTEMPT_FAILURES_TOTAL, + &[("operation", "metrics_fatal"), ("error_class", "fatal")] + ), + 1 + ); + assert_eq!( + histogram_values( + &snapshot, + METRIC_OPERATION_ATTEMPTS, + &[("operation", "metrics_fatal"), ("outcome", "fatal")] + ), + vec![1.0] + ); + } + + #[test] + fn metrics_record_mutating_budget_exhausted_after_one_attempt() { + let (snapshot, ()) = record_metrics(|| { + Box::pin(async { + let policy = policy_of(1_000, 60_000, 5, 100, 2_000); + let cancel = CancellationToken::new(); + let result: Result<()> = execute_with_jitter( + "metrics_rotate", + OpClass::MutatingNonIdempotent, + &policy, + &cancel, + full_jitter, + || async { Err(retryable_conn_error()) }, + ) + .await; + result.expect_err("a mutating operation must not retry a retryable failure"); + }) + }); + + assert_eq!( + counter_value( + &snapshot, + METRIC_OPERATIONS_TOTAL, + &[ + ("operation", "metrics_rotate"), + ("op_class", "mutating_non_idempotent"), + ("outcome", "budget_exhausted") + ] + ), + 1 + ); + assert_eq!( + histogram_values( + &snapshot, + METRIC_OPERATION_ATTEMPTS, + &[("operation", "metrics_rotate"), ("outcome", "budget_exhausted")] + ), + vec![1.0] + ); + } + + #[test] + fn metrics_record_timeouts_and_deadline_outcome() { + let (snapshot, ()) = record_metrics(|| { + Box::pin(async { + // Hung attempts: 10s each against a 25s deadline (see + // total_duration_never_exceeds_deadline for the timeline). + let policy = policy_of(10_000, 25_000, 5, 100, 2_000); + let cancel = CancellationToken::new(); + let result: Result<()> = + execute_with_jitter("metrics_hung", OpClass::ReadIdempotent, &policy, &cancel, full_jitter, || { + std::future::pending::>() + }) + .await; + result.expect_err("hung attempts must exhaust the deadline"); + }) + }); + + assert_eq!( + counter_value( + &snapshot, + METRIC_OPERATIONS_TOTAL, + &[("operation", "metrics_hung"), ("outcome", "deadline_exceeded")] + ), + 1 + ); + assert_eq!( + counter_value( + &snapshot, + METRIC_ATTEMPT_FAILURES_TOTAL, + &[("operation", "metrics_hung"), ("error_class", "attempt_timeout")] + ), + 3 + ); + let durations = histogram_values( + &snapshot, + METRIC_OPERATION_DURATION_SECONDS, + &[("operation", "metrics_hung"), ("outcome", "deadline_exceeded")], + ); + assert_eq!(durations.len(), 1); + assert!((durations[0] - 25.0).abs() < 1e-9, "expected the full 25s deadline, got {durations:?}"); + } + + #[test] + fn metrics_record_cancelled_outcome() { + let calls_in_test = Arc::new(AtomicU32::new(0)); + let (snapshot, ()) = record_metrics(move || { + Box::pin(async move { + let policy = policy_of(1_000, 600_000, 5, 10_000, 10_000); + let cancel = CancellationToken::new(); + let canceller = cancel.clone(); + tokio::spawn(async move { + tokio::time::sleep(Duration::from_millis(500)).await; + canceller.cancel(); + }); + let calls_in_attempt = calls_in_test.clone(); + let result: Result<()> = + execute_with_jitter("metrics_cancel", OpClass::ReadIdempotent, &policy, &cancel, full_jitter, move || { + let calls = calls_in_attempt.clone(); + async move { + calls.fetch_add(1, Ordering::SeqCst); + Err(retryable_conn_error()) + } + }) + .await; + result.expect_err("cancellation must abort the backoff"); + }) + }); + + assert_eq!( + counter_value( + &snapshot, + METRIC_OPERATIONS_TOTAL, + &[("operation", "metrics_cancel"), ("outcome", "cancelled")] + ), + 1 + ); + assert_eq!( + histogram_values( + &snapshot, + METRIC_OPERATION_ATTEMPTS, + &[("operation", "metrics_cancel"), ("outcome", "cancelled")] + ), + vec![1.0] + ); + } } diff --git a/crates/kms/tests/vault_fault_injection.rs b/crates/kms/tests/vault_fault_injection.rs new file mode 100644 index 000000000..f7f86113a --- /dev/null +++ b/crates/kms/tests/vault_fault_injection.rs @@ -0,0 +1,262 @@ +// Copyright 2024 RustFS Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! Fault-injection matrix for the Vault backend operation policy. +//! +//! Offline cases run against locally injected transport faults (a closed +//! port, a listener that never responds) — deterministic, no external +//! dependencies. Real-Vault cases are `#[ignore]`d and need a dev Vault +//! (default `http://127.0.0.1:8200`, override with `RUSTFS_KMS_VAULT_ADDR`). +//! +//! Throttling (429) and recoverable 5xx responses cannot be forced on a stock +//! dev Vault, so their retry and metric behavior is pinned deterministically +//! by the scripted-Vault wiring tests in `backends::vault` and the engine +//! tests in `policy.rs`. Pointing `RUSTFS_KMS_VAULT_ADDR` at a +//! fault-injecting proxy reuses the ignored cases here unchanged. +//! +//! Every case installs a thread-local debugging metrics recorder and drives a +//! current-thread runtime inside it, so the policy metrics double as the +//! request-count assertion even against a real server. + +use std::time::Duration; + +use metrics_util::MetricKind; +use metrics_util::debugging::{DebugValue, DebuggingRecorder}; +use rustfs_kms::backends::KmsBackend as KmsBackendTrait; +use rustfs_kms::backends::vault::VaultKmsBackend; +use rustfs_kms::{ + BackendConfig, DescribeKeyRequest, KmsBackend as KmsBackendKind, KmsConfig, KmsError, VaultAuthMethod, VaultConfig, +}; + +const OPERATIONS_TOTAL: &str = "rustfs_kms_backend_operations_total"; +const ATTEMPT_FAILURES_TOTAL: &str = "rustfs_kms_backend_attempt_failures_total"; + +fn vault_config(address: &str, token: &str) -> VaultConfig { + VaultConfig { + address: address.to_string(), + auth_method: VaultAuthMethod::Token { + token: token.to_string(), + }, + namespace: None, + mount_path: "transit".to_string(), + kv_mount: "secret".to_string(), + key_path_prefix: "rustfs/kms/fault-injection".to_string(), + tls: None, + } +} + +fn kms_config(vault_config: VaultConfig, attempt_timeout: Duration, retry_attempts: u32) -> KmsConfig { + KmsConfig { + backend: KmsBackendKind::VaultKv2, + backend_config: BackendConfig::VaultKv2(Box::new(vault_config)), + allow_insecure_dev_defaults: true, + timeout: attempt_timeout, + retry_attempts, + ..KmsConfig::default() + } +} + +fn describe_key_request(key_id: &str) -> DescribeKeyRequest { + DescribeKeyRequest { + key_id: key_id.to_string(), + } +} + +type MetricEntry = ( + metrics_util::CompositeKey, + Option, + Option, + DebugValue, +); + +/// Run `test` on a current-thread runtime under a debugging metrics recorder +/// and return one snapshot of everything it emitted. +/// +/// A single snapshot per test on purpose: `Snapshotter::snapshot` drains the +/// recorded state, so taking it per assertion would only show the first +/// assertion any data. +fn record_metrics(test: impl FnOnce() -> std::pin::Pin>>) -> Vec { + let recorder = DebuggingRecorder::new(); + let snapshotter = recorder.snapshotter(); + metrics::with_local_recorder(&recorder, || { + let runtime = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("current-thread runtime must build"); + runtime.block_on(test()); + }); + snapshotter.snapshot().into_vec() +} + +/// Sum of counters with `name` whose labels include every `(key, value)` pair. +fn counter_value(snapshot: &[MetricEntry], name: &str, labels: &[(&str, &str)]) -> u64 { + snapshot + .iter() + .filter_map(|(composite, _unit, _description, value)| { + let key = composite.key(); + let matches = composite.kind() == MetricKind::Counter + && key.name() == name + && labels + .iter() + .all(|(label, expected)| key.labels().any(|l| l.key() == *label && l.value() == *expected)); + match (matches, value) { + (true, DebugValue::Counter(count)) => Some(*count), + _ => None, + } + }) + .sum() +} + +/// Connection refused: connection-class failures are retried up to the +/// configured budget, then surface as a backend error. +#[test] +fn connection_refused_is_retried_within_budget() { + // Reserve a loopback port and release it so nothing is listening there. + let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("reserve a loopback port"); + let address = format!("http://{}", listener.local_addr().expect("reserved port addr")); + drop(listener); + + let snapshot = record_metrics(|| { + Box::pin(async move { + let client = VaultKmsBackend::new(kms_config(vault_config(&address, "unused"), Duration::from_secs(2), 2)) + .await + .expect("client construction performs no network calls"); + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-refused")) + .await + .expect_err("a refused connection must fail the operation"); + assert!(matches!(error, KmsError::BackendError { .. }), "got {error:?}"); + }) + }); + + assert_eq!( + counter_value(&snapshot, ATTEMPT_FAILURES_TOTAL, &[("error_class", "retryable_conn")]), + 2, + "both budgeted attempts must observe the refused connection" + ); + assert_eq!(counter_value(&snapshot, OPERATIONS_TOTAL, &[("outcome", "budget_exhausted")]), 1); + // The static-token login records its own success; the Vault read must not. + assert_eq!( + counter_value( + &snapshot, + OPERATIONS_TOTAL, + &[("operation", "vault_kv2_read_key"), ("outcome", "success")] + ), + 0 + ); +} + +/// Stalled connection: a server that accepts but never responds is cut off by +/// the per-attempt timeout (either the policy timer or the equally sized HTTP +/// client timeout, whichever fires first) instead of hanging forever. +#[test] +fn stalled_connection_is_cut_off_by_the_attempt_timeout() { + let snapshot = record_metrics(|| { + Box::pin(async move { + let listener = tokio::net::TcpListener::bind("127.0.0.1:0") + .await + .expect("bind stall listener"); + let address = format!("http://{}", listener.local_addr().expect("stall listener addr")); + // Accept and park every connection without ever responding. + tokio::spawn(async move { + let mut parked = Vec::new(); + loop { + let Ok((socket, _)) = listener.accept().await else { return }; + parked.push(socket); + } + }); + + let client = VaultKmsBackend::new(kms_config(vault_config(&address, "unused"), Duration::from_millis(250), 1)) + .await + .expect("client construction performs no network calls"); + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-stalled")) + .await + .expect_err("a stalled request must be cut off by the attempt timeout"); + assert!( + matches!(error, KmsError::OperationTimedOut { .. } | KmsError::BackendError { .. }), + "got {error:?}" + ); + }) + }); + + // The policy timer reports attempt_timeout; the client-level HTTP timeout + // surfaces as a connection-class failure. Either way it is exactly one + // attempt that was cut off. + let cut_off = counter_value(&snapshot, ATTEMPT_FAILURES_TOTAL, &[("error_class", "attempt_timeout")]) + + counter_value(&snapshot, ATTEMPT_FAILURES_TOTAL, &[("error_class", "retryable_conn")]); + assert_eq!(cut_off, 1, "the single budgeted attempt must be cut off by a timeout"); + assert_eq!(counter_value(&snapshot, OPERATIONS_TOTAL, &[("outcome", "budget_exhausted")]), 1); +} + +fn real_vault_address() -> String { + std::env::var("RUSTFS_KMS_VAULT_ADDR").unwrap_or_else(|_| "http://127.0.0.1:8200".to_string()) +} + +/// Invalid token against a real Vault: the 403 is fatal — exactly one +/// attempt, no retry, and the operation fails closed. +#[test] +#[ignore] // Requires a running Vault dev server +fn real_vault_invalid_token_is_fatal_and_never_retried() { + let snapshot = record_metrics(|| { + Box::pin(async { + let config = vault_config(&real_vault_address(), "fault-injection-invalid-token"); + let client = VaultKmsBackend::new(kms_config(config, Duration::from_secs(5), 3)) + .await + .expect("client construction performs no network calls"); + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-forbidden")) + .await + .expect_err("an invalid token must be rejected"); + assert!(matches!(error, KmsError::BackendError { .. }), "got {error:?}"); + }) + }); + + assert_eq!( + counter_value(&snapshot, ATTEMPT_FAILURES_TOTAL, &[("error_class", "fatal")]), + 1, + "a 403 must be observed by exactly one attempt" + ); + assert_eq!(counter_value(&snapshot, OPERATIONS_TOTAL, &[("outcome", "fatal")]), 1); + assert_eq!( + counter_value(&snapshot, ATTEMPT_FAILURES_TOTAL, &[("error_class", "retryable_status")]), + 0, + "an auth failure must never be classified as retryable" + ); +} + +/// Healthy read against a real Vault: a missing key resolves in one attempt +/// (404 is fatal for retry purposes) and records a fatal outcome rather than +/// burning the retry budget. +#[test] +#[ignore] // Requires a running Vault dev server +fn real_vault_missing_key_is_resolved_in_one_attempt() { + let token = std::env::var("RUSTFS_KMS_VAULT_TOKEN").unwrap_or_else(|_| "dev-only-token".to_string()); + let snapshot = record_metrics(|| { + Box::pin(async move { + let config = vault_config(&real_vault_address(), &token); + let client = VaultKmsBackend::new(kms_config(config, Duration::from_secs(5), 3)) + .await + .expect("client construction performs no network calls"); + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-definitely-missing")) + .await + .expect_err("a missing key must resolve to key-not-found"); + assert!(matches!(error, KmsError::KeyNotFound { .. }), "got {error:?}"); + }) + }); + + assert_eq!( + counter_value(&snapshot, ATTEMPT_FAILURES_TOTAL, &[("error_class", "fatal")]), + 1, + "a 404 must be observed by exactly one attempt" + ); + assert_eq!(counter_value(&snapshot, OPERATIONS_TOTAL, &[("outcome", "fatal")]), 1); +} diff --git a/crates/lifecycle/Cargo.toml b/crates/lifecycle/Cargo.toml index 43dfac6aa..8f5f63556 100644 --- a/crates/lifecycle/Cargo.toml +++ b/crates/lifecycle/Cargo.toml @@ -25,7 +25,35 @@ keywords = ["lifecycle", "storage", "rustfs", "Minio"] categories = ["web-programming", "development-tools", "filesystem"] documentation = "https://docs.rs/rustfs-lifecycle/latest/rustfs_lifecycle/" +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "rustfs-common/hotpath", + "rustfs-config/hotpath", + "rustfs-replication/hotpath", + "rustfs-storage-api/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-replication/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-replication/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true async-trait.workspace = true metrics.workspace = true rustfs-common.workspace = true diff --git a/crates/lock/Cargo.toml b/crates/lock/Cargo.toml index 7b0993adb..dcaccf4a6 100644 --- a/crates/lock/Cargo.toml +++ b/crates/lock/Cargo.toml @@ -28,7 +28,21 @@ documentation = "https://docs.rs/rustfs-lock/latest/rustfs_lock/" [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/parking_lot", + "rustfs-io-metrics/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-io-metrics/hotpath-alloc", "rustfs-utils/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-io-metrics/hotpath-cpu", "rustfs-utils/hotpath-cpu"] + [dependencies] +hotpath.workspace = true rustfs-io-metrics = { workspace = true } rustfs-utils = { workspace = true } async-trait.workspace = true diff --git a/crates/lock/src/distributed_lock.rs b/crates/lock/src/distributed_lock.rs index 97b8af06d..e3465d732 100644 --- a/crates/lock/src/distributed_lock.rs +++ b/crates/lock/src/distributed_lock.rs @@ -540,6 +540,7 @@ impl DistributedLock { } /// Acquire a lock and return a RAII guard + #[hotpath::measure] pub(crate) async fn acquire_guard(&self, request: &LockRequest) -> Result> { if self.clients.is_empty() { return Err(LockError::internal("No lock clients available")); @@ -626,6 +627,7 @@ impl DistributedLock { } /// Convenience: acquire exclusive lock as a guard + #[hotpath::measure] pub async fn lock_guard( &self, resource: ObjectKey, @@ -640,6 +642,7 @@ impl DistributedLock { } /// Convenience: acquire exclusive lock with expected contention logs suppressed + #[hotpath::measure] pub async fn lock_guard_quiet( &self, resource: ObjectKey, @@ -655,6 +658,7 @@ impl DistributedLock { } /// Convenience: acquire shared lock as a guard + #[hotpath::measure] pub async fn rlock_guard( &self, resource: ObjectKey, diff --git a/crates/log-analyzer/Cargo.toml b/crates/log-analyzer/Cargo.toml index b8830e4d0..2c0eebcb4 100644 --- a/crates/log-analyzer/Cargo.toml +++ b/crates/log-analyzer/Cargo.toml @@ -32,7 +32,14 @@ doctest = false name = "la-dump-anchors" path = "src/bin/la_dump_anchors.rs" +[features] +default = [] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true chrono = { workspace = true, features = ["serde"] } flate2 = { workspace = true } regex = { workspace = true } diff --git a/crates/madmin/Cargo.toml b/crates/madmin/Cargo.toml index e5b395735..2d704d2d4 100644 --- a/crates/madmin/Cargo.toml +++ b/crates/madmin/Cargo.toml @@ -28,7 +28,14 @@ documentation = "https://docs.rs/rustfs-madmin/latest/rustfs_madmin/" [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true chrono = { workspace = true, features = ["serde"] } humantime.workspace = true hyper = { workspace = true, features = ["http2", "http1", "server"] } diff --git a/crates/notify/Cargo.toml b/crates/notify/Cargo.toml index e6808f364..ddbb5d6de 100644 --- a/crates/notify/Cargo.toml +++ b/crates/notify/Cargo.toml @@ -26,9 +26,40 @@ categories = ["web-programming", "development-tools", "filesystem"] documentation = "https://docs.rs/rustfs-notify/latest/rustfs_notify/" [features] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "rustfs-config/hotpath", + "rustfs-ecstore/hotpath", + "rustfs-s3-ops/hotpath", + "rustfs-s3-types/hotpath", + "rustfs-targets/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-s3-ops/hotpath-alloc", + "rustfs-s3-types/hotpath-alloc", + "rustfs-targets/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-s3-ops/hotpath-cpu", + "rustfs-s3-types/hotpath-cpu", + "rustfs-targets/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] demo-examples = [] [dependencies] +hotpath.workspace = true rustfs-config = { workspace = true, features = ["notify", "server-config-model"] } rustfs-ecstore = { workspace = true } rustfs-s3-types = { workspace = true } diff --git a/crates/object-capacity/Cargo.toml b/crates/object-capacity/Cargo.toml index 0ae8161df..fd17815f4 100644 --- a/crates/object-capacity/Cargo.toml +++ b/crates/object-capacity/Cargo.toml @@ -34,7 +34,33 @@ harness = false [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "rustfs-config/hotpath", + "rustfs-io-metrics/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-io-metrics/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-io-metrics/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-config = { workspace = true, features = ["constants"] } rustfs-io-metrics = { workspace = true } rustfs-utils = { workspace = true, features = ["os"] } diff --git a/crates/object-capacity/src/capacity_manager.rs b/crates/object-capacity/src/capacity_manager.rs index 876fa9be6..ed3107b9b 100644 --- a/crates/object-capacity/src/capacity_manager.rs +++ b/crates/object-capacity/src/capacity_manager.rs @@ -1035,6 +1035,7 @@ impl HybridCapacityManager { /// /// Joiners subscribe to the watch channel *before* releasing the mutex, which guarantees /// they cannot miss the completion notification even if the leader finishes very quickly. + #[hotpath::measure] pub async fn refresh_or_join(&self, source: DataSource, refresh_fn: F) -> Result where F: FnOnce() -> Fut, @@ -1142,6 +1143,7 @@ impl HybridCapacityManager { } /// Start a background refresh if one is not already in flight. + #[hotpath::measure] pub async fn spawn_refresh_if_needed(self: Arc, source: DataSource, refresh_fn: F) -> bool where F: FnOnce() -> Fut + Send + 'static, diff --git a/crates/object-capacity/src/scan.rs b/crates/object-capacity/src/scan.rs index 2021dd84b..62bb01b93 100644 --- a/crates/object-capacity/src/scan.rs +++ b/crates/object-capacity/src/scan.rs @@ -338,6 +338,7 @@ pub async fn select_capacity_refresh_disks( } } +#[hotpath::measure] pub async fn refresh_capacity_with_scope(disks: Vec, dirty_subset: bool) -> Result { let scan_started_at = Instant::now(); let report = calculate_data_dir_used_capacity_report(&disks) diff --git a/crates/object-data-cache/Cargo.toml b/crates/object-data-cache/Cargo.toml index a89ce6a50..53a7ad5d3 100644 --- a/crates/object-data-cache/Cargo.toml +++ b/crates/object-data-cache/Cargo.toml @@ -27,7 +27,14 @@ categories = ["web-programming", "development-tools"] [lib] doctest = false +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true bytes = { workspace = true, features = ["serde"] } metrics = { workspace = true } moka = { workspace = true, features = ["future"] } diff --git a/crates/obs/Cargo.toml b/crates/obs/Cargo.toml index 6f42090e5..8b627c353 100644 --- a/crates/obs/Cargo.toml +++ b/crates/obs/Cargo.toml @@ -27,6 +27,50 @@ documentation = "https://docs.rs/rustfs-obs/latest/rustfs_obs/" [features] default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/crossbeam", + "rustfs-audit/hotpath", + "rustfs-common/hotpath", + "rustfs-config/hotpath", + "rustfs-ecstore/hotpath", + "rustfs-iam/hotpath", + "rustfs-io-metrics/hotpath", + "rustfs-notify/hotpath", + "rustfs-security-governance/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-audit/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-iam/hotpath-alloc", + "rustfs-io-metrics/hotpath-alloc", + "rustfs-notify/hotpath-alloc", + "rustfs-security-governance/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-audit/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-iam/hotpath-cpu", + "rustfs-io-metrics/hotpath-cpu", + "rustfs-notify/hotpath-cpu", + "rustfs-security-governance/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] # Tokio runtime-level telemetry. Requires a `--cfg tokio_unstable` build; the # build script fails the compile when that flag is missing. Off by default so # ordinary builds neither pay for nor depend on Tokio's unstable API. @@ -58,6 +102,7 @@ required-features = ["dial9"] workspace = true [dependencies] +hotpath.workspace = true rustfs-audit = { workspace = true } rustfs-common = { workspace = true } rustfs-config = { workspace = true, features = ["observability"] } diff --git a/crates/policy/Cargo.toml b/crates/policy/Cargo.toml index b85ded8b5..1732045fb 100644 --- a/crates/policy/Cargo.toml +++ b/crates/policy/Cargo.toml @@ -30,9 +30,27 @@ workspace = true [features] default = [] -hotpath = ["hotpath/hotpath", "hotpath/reqwest-0-13"] -hotpath-alloc = ["hotpath/hotpath-alloc"] -hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] +hotpath = [ + "hotpath/hotpath", + "hotpath/reqwest-0-13", + "rustfs-config/hotpath", + "rustfs-credentials/hotpath", + "rustfs-crypto/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-crypto/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-crypto/hotpath-cpu", +] [dependencies] rustfs-credentials = { workspace = true } diff --git a/crates/protocols/Cargo.toml b/crates/protocols/Cargo.toml index 2d53eca84..91da8efce 100644 --- a/crates/protocols/Cargo.toml +++ b/crates/protocols/Cargo.toml @@ -30,6 +30,55 @@ workspace = true [features] default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "rustfs-config/hotpath", + "rustfs-credentials/hotpath", + "rustfs-ecstore?/hotpath", + "rustfs-iam/hotpath", + "rustfs-keystone?/hotpath", + "rustfs-policy/hotpath", + "rustfs-rio?/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-tls-runtime?/hotpath", + "rustfs-trusted-proxies?/hotpath", + "rustfs-utils/hotpath", + "rustfs-test-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-ecstore?/hotpath-alloc", + "rustfs-iam/hotpath-alloc", + "rustfs-keystone?/hotpath-alloc", + "rustfs-policy/hotpath-alloc", + "rustfs-rio?/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-tls-runtime?/hotpath-alloc", + "rustfs-trusted-proxies?/hotpath-alloc", + "rustfs-utils/hotpath-alloc", + "rustfs-test-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-ecstore?/hotpath-cpu", + "rustfs-iam/hotpath-cpu", + "rustfs-keystone?/hotpath-cpu", + "rustfs-policy/hotpath-cpu", + "rustfs-rio?/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-tls-runtime?/hotpath-cpu", + "rustfs-trusted-proxies?/hotpath-cpu", + "rustfs-utils/hotpath-cpu", + "rustfs-test-utils/hotpath-cpu", +] ftps = ["dep:libunftp", "dep:unftp-core", "dep:rustls", "dep:rustfs-tls-runtime", "dep:subtle"] swift = [ "dep:rustfs-keystone", @@ -62,6 +111,7 @@ webdav = ["dep:dav-server", "dep:hyper", "dep:hyper-util", "dep:http-body-util", sftp = ["dep:russh", "dep:russh-sftp", "dep:uuid", "dep:subtle", "dep:tokio-util", "dep:socket2"] [dependencies] +hotpath.workspace = true # Core RustFS dependencies rustfs-iam = { workspace = true } rustfs-credentials = { workspace = true } diff --git a/crates/protos/Cargo.toml b/crates/protos/Cargo.toml index 3f57bbc6b..27d723a48 100644 --- a/crates/protos/Cargo.toml +++ b/crates/protos/Cargo.toml @@ -32,7 +32,38 @@ workspace = true name = "gproto" path = "src/main.rs" +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "rustfs-common/hotpath", + "rustfs-config/hotpath", + "rustfs-io-metrics/hotpath", + "rustfs-tls-runtime/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-io-metrics/hotpath-alloc", + "rustfs-tls-runtime/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-io-metrics/hotpath-cpu", + "rustfs-tls-runtime/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-common.workspace = true rustfs-io-metrics.workspace = true rustfs-config.workspace = true diff --git a/crates/replication/Cargo.toml b/crates/replication/Cargo.toml index 5dc9a1108..e5554b922 100644 --- a/crates/replication/Cargo.toml +++ b/crates/replication/Cargo.toml @@ -25,7 +25,14 @@ keywords = ["replication", "storage", "rustfs", "Minio"] categories = ["web-programming", "development-tools", "filesystem"] documentation = "https://docs.rs/rustfs-replication/latest/rustfs_replication/" +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true bytes = { workspace = true, features = ["serde"] } byteorder.workspace = true regex.workspace = true diff --git a/crates/rio-v2/Cargo.toml b/crates/rio-v2/Cargo.toml index b03ac2b88..b941ed7fc 100644 --- a/crates/rio-v2/Cargo.toml +++ b/crates/rio-v2/Cargo.toml @@ -28,7 +28,32 @@ documentation = "https://docs.rs/rustfs-rio-v2/latest/rustfs_rio_v2/" [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "rustfs-rio/hotpath", + "rustfs-utils/hotpath", + "rustfs-filemeta/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-rio/hotpath-alloc", + "rustfs-utils/hotpath-alloc", + "rustfs-filemeta/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-rio/hotpath-cpu", + "rustfs-utils/hotpath-cpu", + "rustfs-filemeta/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true aes-gcm = { workspace = true, features = ["rand_core"] } bytes = { workspace = true, features = ["serde"] } chacha20poly1305.workspace = true diff --git a/crates/rio/Cargo.toml b/crates/rio/Cargo.toml index 4d0f3916e..72bb71af0 100644 --- a/crates/rio/Cargo.toml +++ b/crates/rio/Cargo.toml @@ -30,9 +30,32 @@ workspace = true [features] default = [] -hotpath = ["hotpath/hotpath", "hotpath/tokio", "hotpath/futures", "hotpath/reqwest-0-13"] -hotpath-alloc = ["hotpath/hotpath-alloc"] -hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/reqwest-0-13", + "rustfs-config/hotpath", + "rustfs-io-metrics/hotpath", + "rustfs-tls-runtime/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-io-metrics/hotpath-alloc", + "rustfs-tls-runtime/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-io-metrics/hotpath-cpu", + "rustfs-tls-runtime/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] [dependencies] hotpath.workspace = true diff --git a/crates/s3-ops/Cargo.toml b/crates/s3-ops/Cargo.toml index 170f423c3..23dbd432a 100644 --- a/crates/s3-ops/Cargo.toml +++ b/crates/s3-ops/Cargo.toml @@ -27,7 +27,14 @@ categories = ["data-structures"] [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath", "rustfs-s3-types/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-s3-types/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-s3-types/hotpath-cpu"] + [dependencies] +hotpath.workspace = true rustfs-s3-types = { workspace = true } [lib] diff --git a/crates/s3-types/Cargo.toml b/crates/s3-types/Cargo.toml index 0d25caf4b..76c408594 100644 --- a/crates/s3-types/Cargo.toml +++ b/crates/s3-types/Cargo.toml @@ -27,7 +27,14 @@ categories = ["data-structures"] [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true serde = { workspace = true, features = ["derive"] } serde_json = { workspace = true, features = ["raw_value"] } diff --git a/crates/s3select-api/Cargo.toml b/crates/s3select-api/Cargo.toml index 78edb9daf..4e850b62b 100644 --- a/crates/s3select-api/Cargo.toml +++ b/crates/s3select-api/Cargo.toml @@ -28,7 +28,37 @@ documentation = "https://docs.rs/rustfs-s3select-api/latest/rustfs_s3select_api/ [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/parking_lot", + "rustfs-common/hotpath", + "rustfs-ecstore/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-test-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-test-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-test-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true metrics = { workspace = true } async-trait.workspace = true bytes = { workspace = true, features = ["serde"] } diff --git a/crates/s3select-query/Cargo.toml b/crates/s3select-query/Cargo.toml index 5ce13b9fd..619997edb 100644 --- a/crates/s3select-query/Cargo.toml +++ b/crates/s3select-query/Cargo.toml @@ -28,7 +28,20 @@ documentation = "https://docs.rs/rustfs-s3select-query/latest/rustfs_s3select_qu [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/parking_lot", + "rustfs-s3select-api/hotpath", +] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-s3select-api/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-s3select-api/hotpath-cpu"] + [dependencies] +hotpath.workspace = true rustfs-s3select-api = { workspace = true } async-recursion = { workspace = true } async-trait.workspace = true diff --git a/crates/scanner/Cargo.toml b/crates/scanner/Cargo.toml index 33dde7715..127604853 100644 --- a/crates/scanner/Cargo.toml +++ b/crates/scanner/Cargo.toml @@ -29,7 +29,48 @@ documentation = "https://docs.rs/rustfs-scanner/latest/rustfs_scanner/" [lints] workspace = true +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "rustfs-common/hotpath", + "rustfs-config/hotpath", + "rustfs-credentials/hotpath", + "rustfs-data-usage/hotpath", + "rustfs-ecstore/hotpath", + "rustfs-filemeta/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-data-usage/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-filemeta/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-data-usage/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-filemeta/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-config = { workspace = true, features = ["server-config-model"] } rustfs-common = { workspace = true } rustfs-credentials = { workspace = true } diff --git a/crates/scanner/src/scanner.rs b/crates/scanner/src/scanner.rs index 91f9e1597..95c5fb194 100644 --- a/crates/scanner/src/scanner.rs +++ b/crates/scanner/src/scanner.rs @@ -2531,6 +2531,7 @@ where } #[instrument(skip_all)] +#[hotpath::measure] async fn run_data_scanner_cycle( ctx: &CancellationToken, storeapi: &Arc, diff --git a/crates/security-governance/Cargo.toml b/crates/security-governance/Cargo.toml index f71e1b032..da0b48470 100644 --- a/crates/security-governance/Cargo.toml +++ b/crates/security-governance/Cargo.toml @@ -30,5 +30,12 @@ doctest = false [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true thiserror = { workspace = true } diff --git a/crates/signer/Cargo.toml b/crates/signer/Cargo.toml index 24cb667eb..b13c03275 100644 --- a/crates/signer/Cargo.toml +++ b/crates/signer/Cargo.toml @@ -25,7 +25,14 @@ keywords = ["digital-signature", "verification", "integrity", "rustfs", "Minio"] categories = ["web-programming", "development-tools", "cryptography"] documentation = "https://docs.rs/rustfs-signer/latest/rustfs_signer/" +[features] +default = [] +hotpath = ["hotpath/hotpath", "rustfs-utils/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-utils/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-utils/hotpath-cpu"] + [dependencies] +hotpath.workspace = true tracing.workspace = true bytes = { workspace = true, features = ["serde"] } http.workspace = true diff --git a/crates/storage-api/Cargo.toml b/crates/storage-api/Cargo.toml index f67d34331..74b519d3d 100644 --- a/crates/storage-api/Cargo.toml +++ b/crates/storage-api/Cargo.toml @@ -27,7 +27,14 @@ categories = ["web-programming", "development-tools", "filesystem"] [lib] doctest = false +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio", "rustfs-filemeta/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-filemeta/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-filemeta/hotpath-cpu"] + [dependencies] +hotpath.workspace = true async-trait.workspace = true # Storage-facing replication contracts are isolated in src/replication.rs until # the underlying wire types can move without creating a replication/storage-api cycle. diff --git a/crates/targets/Cargo.toml b/crates/targets/Cargo.toml index bd0671a40..ae759eaf7 100644 --- a/crates/targets/Cargo.toml +++ b/crates/targets/Cargo.toml @@ -11,7 +11,41 @@ keywords = ["file-system", "notification", "target", "rustfs", "Minio"] categories = ["web-programming", "development-tools", "filesystem"] documentation = "https://docs.rs/rustfs-targets/latest/rustfs_targets/" +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/futures", + "hotpath/parking_lot", + "hotpath/reqwest-0-13", + "rustfs-config/hotpath", + "rustfs-extension-schema/hotpath", + "rustfs-s3-types/hotpath", + "rustfs-tls-runtime/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-extension-schema/hotpath-alloc", + "rustfs-s3-types/hotpath-alloc", + "rustfs-tls-runtime/hotpath-alloc", + "rustfs-utils/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-extension-schema/hotpath-cpu", + "rustfs-s3-types/hotpath-cpu", + "rustfs-tls-runtime/hotpath-cpu", + "rustfs-utils/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-config = { workspace = true, features = ["notify", "audit", "server-config-model"] } rustfs-extension-schema = { workspace = true } rustfs-tls-runtime = { workspace = true } diff --git a/crates/targets/src/runtime/mod.rs b/crates/targets/src/runtime/mod.rs index 3a235e069..7999a33ce 100644 --- a/crates/targets/src/runtime/mod.rs +++ b/crates/targets/src/runtime/mod.rs @@ -522,6 +522,7 @@ fn snapshot_from_delivery(target_id: TargetID, delivery: TargetDeliverySnapshot) } } +#[hotpath::measure] pub async fn init_target_and_optionally_start_replay( target: Box + Send + Sync>, on_replay_start: F, @@ -557,6 +558,7 @@ where Some((shared, cancel)) } +#[hotpath::measure] pub(crate) async fn prepare_target( target: Box + Send + Sync>, cancellation: Option<&CancellationToken>, @@ -598,6 +600,7 @@ where type ActivatedTarget = (SharedTarget, Option<(mpsc::Sender<()>, JoinHandle<()>)>); +#[hotpath::measure] pub async fn activate_targets_with_replay( targets: Vec + Send + Sync>>, mut activate_one: F, @@ -670,6 +673,7 @@ fn seed_interval_start(now: tokio::time::Instant, interval: Duration) -> tokio:: now.checked_sub(interval).unwrap_or(now) } +#[hotpath::measure] async fn stream_replay_worker( store: &mut (dyn Store + Send), target: SharedTarget, @@ -805,6 +809,7 @@ async fn stream_replay_worker( /// Returns `true` if a cancel signal was observed while processing (e.g. during /// retry backoff), so the caller can stop promptly instead of continuing to /// drain a store that a replacement worker may already own. +#[hotpath::measure] async fn process_replay_batch( store: &(dyn Store + Send), batch_keys: &mut Vec, diff --git a/crates/targets/src/target/webhook.rs b/crates/targets/src/target/webhook.rs index e39dc9d49..b7ac97f0b 100644 --- a/crates/targets/src/target/webhook.rs +++ b/crates/targets/src/target/webhook.rs @@ -85,6 +85,7 @@ fn classify_probe_error(err: &reqwest::Error) -> TargetHealthReason { TargetHealthReason::Unreachable } +#[hotpath::measure] async fn probe_health_url(client: &Client, health_check_url: &Url) -> TargetHealth { match tokio::time::timeout(WEBHOOK_HEALTH_TIMEOUT, client.head(health_check_url.as_str()).send()).await { Ok(Ok(_)) => TargetHealth::online(TargetHealthReason::Reachable), @@ -480,6 +481,7 @@ where build_queued_payload(event) } + #[hotpath::measure] async fn send_body(&self, body: Vec, meta: &QueuedPayloadMeta) -> Result<(), TargetError> { debug!( event = EVENT_WEBHOOK_DELIVERY_STATE, diff --git a/crates/test-utils/Cargo.toml b/crates/test-utils/Cargo.toml index 81d2a53c0..bf56d069a 100644 --- a/crates/test-utils/Cargo.toml +++ b/crates/test-utils/Cargo.toml @@ -25,7 +25,32 @@ keywords = ["testing", "storage", "rustfs", "Minio"] categories = ["development-tools", "filesystem"] documentation = "https://docs.rs/rustfs-test-utils/latest/rustfs_test_utils/" +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "rustfs-ecstore/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-data-usage/hotpath", +] +hotpath-alloc = [ + "hotpath", + "hotpath/hotpath-alloc", + "rustfs-ecstore/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-data-usage/hotpath-alloc", +] +hotpath-cpu = [ + "hotpath", + "hotpath/hotpath-cpu", + "rustfs-ecstore/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-data-usage/hotpath-cpu", +] + [dependencies] +hotpath.workspace = true rustfs-ecstore = { workspace = true } rustfs-storage-api = { workspace = true } tokio = { workspace = true, features = ["fs", "rt-multi-thread"] } diff --git a/crates/tls-runtime/Cargo.toml b/crates/tls-runtime/Cargo.toml index fba5f3fc2..fb8920547 100644 --- a/crates/tls-runtime/Cargo.toml +++ b/crates/tls-runtime/Cargo.toml @@ -27,7 +27,14 @@ categories = ["network-programming", "web-programming", "development-tools"] [lints] workspace = true +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio", "rustfs-common/hotpath", "rustfs-config/hotpath"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-common/hotpath-alloc", "rustfs-config/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-common/hotpath-cpu", "rustfs-config/hotpath-cpu"] + [dependencies] +hotpath.workspace = true rustfs-common.workspace = true rustfs-config.workspace = true arc-swap.workspace = true diff --git a/crates/trusted-proxies/Cargo.toml b/crates/trusted-proxies/Cargo.toml index 0a15e5a7f..1494d8a35 100644 --- a/crates/trusted-proxies/Cargo.toml +++ b/crates/trusted-proxies/Cargo.toml @@ -24,7 +24,20 @@ description = " RustFS Trusted Proxies module provides secure and efficient mana keywords = ["trusted-proxies", "network-security", "rustfs", "proxy-management"] categories = ["network-programming", "security", "web-programming"] +[features] +default = [] +hotpath = [ + "hotpath/hotpath", + "hotpath/tokio", + "hotpath/reqwest-0-13", + "rustfs-config/hotpath", + "rustfs-utils/hotpath", +] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc", "rustfs-config/hotpath-alloc", "rustfs-utils/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-config/hotpath-cpu", "rustfs-utils/hotpath-cpu"] + [dependencies] +hotpath.workspace = true async-trait = { workspace = true } axum = { workspace = true } http = { workspace = true } diff --git a/crates/trusted-proxies/src/cloud/ranges.rs b/crates/trusted-proxies/src/cloud/ranges.rs index cd3b8e4e0..dbf814674 100644 --- a/crates/trusted-proxies/src/cloud/ranges.rs +++ b/crates/trusted-proxies/src/cloud/ranges.rs @@ -227,6 +227,7 @@ pub struct GoogleCloudIpRanges; impl GoogleCloudIpRanges { /// Fetches the latest Google Cloud IP ranges from their official source. + #[hotpath::measure] pub async fn fetch() -> Result, AppError> { let client = Client::builder() .timeout(Duration::from_secs(10)) diff --git a/crates/utils/Cargo.toml b/crates/utils/Cargo.toml index 5ba901965..2e3c7c6d8 100644 --- a/crates/utils/Cargo.toml +++ b/crates/utils/Cargo.toml @@ -25,6 +25,7 @@ keywords = ["utilities", "hashing", "compression", "network", "rustfs"] categories = ["web-programming", "development-tools", "cryptography"] [dependencies] +hotpath.workspace = true base64-simd = { workspace = true, optional = true } blake2 = { workspace = true, optional = true } brotli = { workspace = true, optional = true } @@ -76,6 +77,9 @@ workspace = true [features] default = ["ip"] # features that are enabled by default +hotpath = ["hotpath/hotpath", "hotpath/tokio", "hotpath/futures", "hotpath/reqwest-0-13"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] ip = ["dep:local-ip-address"] # ip characteristics and their dependencies net = ["ip", "dep:url", "dep:netif", "dep:futures", "dep:transform-stream", "dep:bytes", "dep:hyper", "dep:tokio"] # network features with DNS resolver egress = ["ip", "dep:reqwest", "dep:tokio", "dep:url"] diff --git a/crates/zip/Cargo.toml b/crates/zip/Cargo.toml index 8348d3189..54cdd9547 100644 --- a/crates/zip/Cargo.toml +++ b/crates/zip/Cargo.toml @@ -32,7 +32,14 @@ doctest = false name = "zip_benchmark" harness = false +[features] +default = [] +hotpath = ["hotpath/hotpath", "hotpath/tokio"] +hotpath-alloc = ["hotpath", "hotpath/hotpath-alloc"] +hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu"] + [dependencies] +hotpath.workspace = true async-compression = { workspace = true, features = [ "tokio", "bzip2", diff --git a/deploy/observability/README.md b/deploy/observability/README.md index d06deb7b3..9c19a9bd7 100644 --- a/deploy/observability/README.md +++ b/deploy/observability/README.md @@ -9,3 +9,5 @@ Import `grafana/rustfs-node-observability.json` into Grafana and select a Promet The dashboard uses the RustFS `server` metric label introduced with the node-local observability updates. `server` represents the RustFS node identity and is preferred for RustFS node comparisons. Prometheus `instance` still identifies the scrape target and remains useful for scrape/debugging views, but dashboards that compare RustFS nodes should group and filter by `server`. During a rolling upgrade, older nodes may still emit metrics without the `server` label. Complete the rollout before using this dashboard for node-by-node comparisons. + +`grafana/rustfs-kms-observability.json` covers the KMS backend operation metrics emitted at the operation-policy choke point. Unlike the node dashboard, KMS metrics do not carry the `server` label — use `job`/`instance` or promoted OTel resource attributes to split by node. Matching Prometheus alert rules live in `.docker/observability/prometheus-rules/rustfs-kms-alerts.yml`, and the alert response procedures are documented in `docs/operations/kms-observability-runbook.md`. diff --git a/deploy/observability/grafana/rustfs-kms-observability.json b/deploy/observability/grafana/rustfs-kms-observability.json new file mode 100644 index 000000000..0810a3e4a --- /dev/null +++ b/deploy/observability/grafana/rustfs-kms-observability.json @@ -0,0 +1,793 @@ +{ + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "iconColor": "rgba(0, 211, 255, 1)", + "name": "Annotations & Alerts", + "target": { + "limit": 100, + "matchAny": false, + "tags": [], + "type": "dashboard" + }, + "type": "dashboard" + } + ] + }, + "description": "KMS backend operation metrics emitted at the operation-policy choke point (crates/kms/src/policy.rs). All label values are static enum strings; key identifiers, key material, and tokens never appear in labels. These metrics do not carry the RustFS `server` label — use your scrape topology (job/instance or promoted OTel resource attributes) to split by node. Alert response procedures: docs/operations/kms-observability-runbook.md.", + "editable": true, + "fiscalYearStartMonth": 0, + "graphTooltip": 1, + "id": null, + "links": [], + "liveNow": false, + "panels": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "description": "Terminal outcomes of KMS backend operations. `fatal` means a non-retryable failure ended the operation on first observation; `budget_exhausted` and `deadline_exceeded` mean retries ran out; `cancelled` is normal during shutdown.", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 12, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 4, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "never", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + }, + "unit": "ops" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "id": 1, + "options": { + "legend": { + "calcs": [ + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "sum by (outcome) (rate(rustfs_kms_backend_operations_total{operation=~\"$operation\"}[$__rate_interval]))", + "legendFormat": "{{outcome}}", + "range": true, + "refId": "A" + } + ], + "title": "Backend Operation Rate by Outcome", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "description": "Operation names are static per-call-site identifiers (e.g. vault_kv2_read_key_version, vault_transit_encrypt, vault_login). `op_class` distinguishes read_idempotent, mutating_non_idempotent, and auth operations.", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 12, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 4, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "never", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + }, + "unit": "ops" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "id": 2, + "options": { + "legend": { + "calcs": [ + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "sum by (operation, op_class) (rate(rustfs_kms_backend_operations_total{operation=~\"$operation\"}[$__rate_interval]))", + "legendFormat": "{{operation}} ({{op_class}})", + "range": true, + "refId": "A" + } + ], + "title": "Backend Operation Rate by Operation", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "description": "Share of operations that terminated in fatal, budget_exhausted, or deadline_exceeded. The cancelled outcome is plotted separately because shutdown windows legitimately spike it. The ratio is meaningless at near-zero traffic — read it together with the operation rate panels.", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 12, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 4, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "never", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "max": 1, + "min": 0, + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + }, + "unit": "percentunit" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "id": 3, + "options": { + "legend": { + "calcs": [ + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "sum(rate(rustfs_kms_backend_operations_total{outcome!~\"success|cancelled\",operation=~\"$operation\"}[$__rate_interval])) / clamp_min(sum(rate(rustfs_kms_backend_operations_total{operation=~\"$operation\"}[$__rate_interval])), 1e-9)", + "legendFormat": "non-success (excl. cancelled)", + "range": true, + "refId": "A" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "sum(rate(rustfs_kms_backend_operations_total{outcome=\"cancelled\",operation=~\"$operation\"}[$__rate_interval])) / clamp_min(sum(rate(rustfs_kms_backend_operations_total{operation=~\"$operation\"}[$__rate_interval])), 1e-9)", + "legendFormat": "cancelled", + "range": true, + "refId": "B" + } + ], + "title": "Non-Success Outcome Ratio", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "description": "Per-attempt failures by retry classification. retryable_conn covers connection-level failures, retryable_status covers retryable backend status codes, attempt_timeout covers attempts cut off by the per-attempt timeout, and fatal covers non-retryable failures (auth, permission, malformed request). Sustained fatal traffic is always actionable.", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 12, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 4, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "never", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + }, + "unit": "ops" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "id": 4, + "options": { + "legend": { + "calcs": [ + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "sum by (error_class) (rate(rustfs_kms_backend_attempt_failures_total{operation=~\"$operation\"}[$__rate_interval]))", + "legendFormat": "{{error_class}}", + "range": true, + "refId": "A" + } + ], + "title": "Attempt Failure Rate by Error Class", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "description": "Wall-clock duration of whole operations, including retries and backoff sleeps. A rising p99 with a flat p50 usually means a slow retry tail (backend degradation), not a uniform slowdown.", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 12, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 4, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "never", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + }, + "unit": "s" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "id": 5, + "options": { + "legend": { + "calcs": [ + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "histogram_quantile(0.50, sum by (le) (rate(rustfs_kms_backend_operation_duration_seconds_bucket{operation=~\"$operation\"}[$__rate_interval])))", + "legendFormat": "p50", + "range": true, + "refId": "A" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "histogram_quantile(0.99, sum by (le) (rate(rustfs_kms_backend_operation_duration_seconds_bucket{operation=~\"$operation\"}[$__rate_interval])))", + "legendFormat": "p99", + "range": true, + "refId": "B" + } + ], + "title": "Operation Duration p50 / p99", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "description": "p99 duration split by operation. Auth operations (vault_login, vault_token_renew) and mutating writes are expected to sit higher than idempotent reads.", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 12, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 4, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "never", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + }, + "unit": "s" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "id": 6, + "options": { + "legend": { + "calcs": [ + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "histogram_quantile(0.99, sum by (le, operation) (rate(rustfs_kms_backend_operation_duration_seconds_bucket{operation=~\"$operation\"}[$__rate_interval])))", + "legendFormat": "{{operation}}", + "range": true, + "refId": "A" + } + ], + "title": "Operation Duration p99 by Operation", + "type": "timeseries" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "description": "Attempts one operation used before completing. Healthy operations average close to 1. A rising average or p99 means the retry policy is absorbing backend failures — read together with the attempt-failure panel to see the failure class.", + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisCenteredZero": false, + "axisColorMode": "text", + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 12, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 4, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "never", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + }, + "unit": "short" + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "id": 7, + "options": { + "legend": { + "calcs": [ + "lastNotNull" + ], + "displayMode": "table", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "sum by (operation) (rate(rustfs_kms_backend_operation_attempts_sum{operation=~\"$operation\"}[$__rate_interval])) / clamp_min(sum by (operation) (rate(rustfs_kms_backend_operation_attempts_count{operation=~\"$operation\"}[$__rate_interval])), 1e-9)", + "legendFormat": "{{operation}} avg", + "range": true, + "refId": "A" + }, + { + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "editorMode": "code", + "expr": "histogram_quantile(0.99, sum by (le) (rate(rustfs_kms_backend_operation_attempts_bucket{operation=~\"$operation\"}[$__rate_interval])))", + "legendFormat": "p99 (all operations)", + "range": true, + "refId": "B" + } + ], + "title": "Operation Attempts Distribution", + "type": "timeseries" + }, + { + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "id": 8, + "options": { + "code": { + "language": "plaintext", + "showLineNumbers": false, + "showMiniMap": false + }, + "content": "Placeholder for KMS metrics planned by sibling changes of rustfs/backlog#1584 that have **not landed yet**. Do not add panels for these names until the emitting code is merged, or the panels will render empty and mask real gaps.\n\n- TODO: key-cache effectiveness — hit/miss counters, entry gauge, eviction counter (replaces the former hardcoded-zero miss stat).\n- TODO: key lifecycle — pending-deletion/tombstone gauges from the deletion worker sweep, aggregate rotation-age gauge.\n- TODO: Vault credentials — token TTL remaining and fail-closed state gauges.\n- TODO: synthetic backend probe — probe outcome/failure-class metrics feeding the readiness cache.\n\nWhen a family lands, replace one bullet with a real panel and keep this list in sync with docs/operations/kms-observability-runbook.md (Coverage gaps).", + "mode": "markdown" + }, + "title": "Planned Panels (TODO — metrics not landed yet)", + "type": "text" + } + ], + "refresh": "30s", + "schemaVersion": 39, + "style": "dark", + "tags": [ + "rustfs", + "observability", + "kms" + ], + "templating": { + "list": [ + { + "current": {}, + "hide": 0, + "includeAll": false, + "label": "Prometheus", + "multi": false, + "name": "datasource", + "options": [], + "query": "prometheus", + "refresh": 1, + "regex": "", + "skipUrlSync": false, + "type": "datasource" + }, + { + "allValue": ".*", + "current": { + "selected": true, + "text": "All", + "value": "$__all" + }, + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "definition": "label_values(rustfs_kms_backend_operations_total, operation)", + "hide": 0, + "includeAll": true, + "label": "KMS operation", + "multi": true, + "name": "operation", + "options": [], + "query": "label_values(rustfs_kms_backend_operations_total, operation)", + "refresh": 1, + "regex": "", + "skipUrlSync": false, + "sort": 1, + "type": "query" + } + ] + }, + "time": { + "from": "now-6h", + "to": "now" + }, + "timepicker": {}, + "timezone": "", + "title": "RustFS KMS Observability", + "uid": "rustfs-kms-observability", + "version": 1, + "weekStart": "" +} diff --git a/docs/operations/hotpath-warp-abba-runbook.md b/docs/operations/hotpath-warp-abba-runbook.md new file mode 100644 index 000000000..5d73b59d1 --- /dev/null +++ b/docs/operations/hotpath-warp-abba-runbook.md @@ -0,0 +1,277 @@ +# Hotpath warp ABBA validation runbook + +This runbook describes how to collect formal Linux or production-cluster +evidence for hotpath performance changes. Use it when a short local A/B smoke +run is too noisy to decide whether a regression is real. + +The ABBA runner executes each workload and drive-sync cell as: + +```text +A1 baseline -> B1 candidate -> B2 candidate -> A2 baseline +``` + +`B1` and `B2` are compared with `A1` to measure the candidate delta. `A2` is +also compared with `A1` to measure baseline drift. Treat a candidate regression +as actionable only when the `A2` drift is passing or materially smaller than +the `B1` and `B2` delta for the same workload. + +## Scope + +Use this runbook for hotpath profiling and performance validation of RustFS +object I/O changes, especially when CPU, memory allocation, lock/channel wait +time, request throughput, or tail latency is the review question. + +The script validates the same workload matrix as the hotpath warp A/B gate: + +| Workload | mode | size | +| --- | --- | --- | +| `put-4kib` | put | 4KiB | +| `put-4mib` | put | 4MiB | +| `get-4kib` | get | 4KiB | +| `get-4mib` | get | 4MiB | +| `get-10mib` | get | 10MiB | +| `mixed-256k` | mixed | 256KiB | + +Each workload runs with `RUSTFS_DRIVE_SYNC_ENABLE=true` and +`RUSTFS_DRIVE_SYNC_ENABLE=false`, so a full ABBA pass produces 48 measurement +cells: 6 workloads x 2 drive-sync modes x 4 ABBA legs. + +## Prerequisites + +Run the formal pass on Linux, not on a laptop smoke environment. + +Required tools on the bench host: + +- `bash`, `curl`, `git`, and core GNU userland. +- `warp` on `PATH`, or pass `--warp-bin`. +- Two RustFS Linux binaries: one baseline and one candidate. +- Enough isolated disks or directories for the local runner, or an externally + managed RustFS cluster for production-like validation. +- Stable host telemetry collection such as `pidstat`, `mpstat`, `iostat`, + `sar`, `perf`, `heaptrack`, or the platform's equivalent observability stack. + +Cluster-mode requirements: + +- A deploy hook that can replace the RustFS binary on every node. +- The hook must apply `RUSTFS_DRIVE_SYNC_ENABLE` for the current ABBA leg. +- The hook must restart RustFS and return only after the rollout command has + been accepted. The ABBA script performs the HTTP readiness wait. +- The benchmark client should run outside the RustFS nodes when possible. +- Do not run against a production data set unless the workload bucket and test + credentials are isolated and approved for destructive benchmark traffic. + +## Build the binaries + +Build the baseline from the comparison commit, usually `origin/main` or the +previous accepted release: + +```bash +git fetch origin main +git switch --detach origin/main +cargo build --release -p rustfs --bins +cp target/release/rustfs /tmp/rustfs-baseline +``` + +Build the candidate from the PR commit: + +```bash +git switch +cargo build --release -p rustfs --bins +cp target/release/rustfs /tmp/rustfs-candidate +``` + +For cross-compiled cluster binaries, keep both outputs on the bench host and +make the deploy hook copy the selected binary to the cluster. The ABBA runner +passes the selected binary path through `HOTPATH_ABBA_BINARY`. + +## Local Linux runner + +Use local mode for a dedicated Linux runner with disposable data paths. This is +not a substitute for a production-like cluster, but it is useful before spending +cluster time. + +```bash +scripts/run_hotpath_warp_abba.sh \ + --baseline-bin /tmp/rustfs-baseline \ + --candidate-bin /tmp/rustfs-candidate \ + --address 127.0.0.1:9000 \ + --data-root /var/tmp/rustfs-hotpath-abba \ + --disks 4 \ + --duration 120s \ + --rounds 3 \ + --cooldown 30 \ + --concurrency 16 \ + --out-dir target/hotpath-abba/linux-local +``` + +The script starts and stops RustFS for each ABBA leg. The data root is +throwaway and should not contain important data. + +## Production-like cluster runner + +Use external mode when RustFS lifecycle is managed by ansible, systemd, a +cluster scheduler, or a dedicated deployment harness. In this mode the ABBA +script does not start RustFS directly; it calls `--deploy-hook` before each leg +and then waits for `http://`. + +The deploy hook receives: + +| Environment variable | Value | +| --- | --- | +| `HOTPATH_ABBA_LEG` | `A1`, `B1`, `B2`, or `A2` | +| `HOTPATH_ABBA_PHASE` | `baseline` or `candidate` | +| `HOTPATH_ABBA_BINARY` | selected baseline or candidate binary path | +| `HOTPATH_ABBA_DRIVE_SYNC` | `true` or `false` | + +Example ansible-shaped command: + +```bash +scripts/run_hotpath_warp_abba.sh \ + --baseline-bin /srv/rustfs-binaries/rustfs-baseline \ + --candidate-bin /srv/rustfs-binaries/rustfs-candidate \ + --endpoint rustfs-bench.example.internal:9000 \ + --deploy-hook ' + set -euo pipefail + cd /srv/rustfs-ansible + cp "${HOTPATH_ABBA_BINARY:?}" roles/rustfs/files/rustfs + export RUSTFS_DRIVE_SYNC_ENABLE="${HOTPATH_ABBA_DRIVE_SYNC:?}" + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags stop + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags config + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags binary-copy + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags start + ' \ + --duration 180s \ + --rounds 5 \ + --cooldown 45 \ + --concurrency 32 \ + --out-dir target/hotpath-abba/cluster-pr-XXXX +``` + +For formal evidence, prefer `--rounds 5` or higher when the cluster budget +allows it. The script enforces `--rounds >= 3`. + +## CPU and memory evidence + +ABBA warp output answers whether the candidate changed throughput or latency. +Collect host telemetry at the same time to explain why. + +Recommended minimum: + +```bash +mkdir -p target/hotpath-abba/cluster-pr-XXXX/telemetry + +pidstat -durh 5 > target/hotpath-abba/cluster-pr-XXXX/telemetry/pidstat.txt & +PIDSTAT_PID=$! + +mpstat 5 > target/hotpath-abba/cluster-pr-XXXX/telemetry/mpstat.txt & +MPSTAT_PID=$! + +iostat -xz 5 > target/hotpath-abba/cluster-pr-XXXX/telemetry/iostat.txt & +IOSTAT_PID=$! +``` + +Stop the collectors after the ABBA script exits: + +```bash +kill "$PIDSTAT_PID" "$MPSTAT_PID" "$IOSTAT_PID" +``` + +For deeper CPU attribution, run `perf record` around one representative +workload after the ABBA gate identifies a candidate regression or improvement: + +```bash +perf record -F 99 -g -- sleep 180 +perf report --stdio > target/hotpath-abba/cluster-pr-XXXX/telemetry/perf-report.txt +``` + +For allocation profiling, build the candidate with: + +```bash +cargo build --release -p rustfs --bins --features hotpath-alloc +``` + +Then run the same ABBA command with that binary. Compare allocation-heavy +function sections only within the same build mode. Do not compare +`hotpath-alloc` binaries directly with default release binaries for throughput +acceptance, because allocation instrumentation intentionally changes what is +measured. + +For CPU hotpath sections emitted by hotpath, build with: + +```bash +cargo build --release -p rustfs --bins --features hotpath-cpu +``` + +Use the CPU-enabled report to explain hotspots after the default or plain +`hotpath` ABBA gate shows a real effect. + +## Output layout + +The ABBA runner writes: + +```text +/ + manifest.env + abba_schedule.csv + candidate_gate.md + baseline_drift_gate.md + summary.md + ///median_summary.csv + ///baseline_compare.csv +``` + +Attach or link at least these files in the issue or PR: + +- `summary.md` +- `candidate_gate.md` +- `baseline_drift_gate.md` +- `abba_schedule.csv` +- every `median_summary.csv` and `baseline_compare.csv` for a failed or + borderline workload +- host telemetry files used to explain CPU, memory, or disk saturation + +## Interpretation + +Use this decision table: + +| Candidate gate | A2 drift gate | Interpretation | +| --- | --- | --- | +| PASS | PASS | Candidate is acceptable for the measured matrix. | +| WARN | PASS | Candidate has a small measurable signal; inspect telemetry and decide if it is expected. | +| FAIL | PASS | Candidate likely regressed the affected workload; investigate before merge. | +| FAIL | FAIL on the same workload | Environment drift is high; rerun on a quieter runner or increase duration and rounds. | +| PASS | FAIL | Candidate did not exceed the budget, but the rig was unstable; avoid using the numbers as proof of improvement. | + +When `B1` and `B2` disagree, treat the result as inconclusive even if the gate +passes. Increase duration, rounds, cooldown, or runner isolation before drawing +a conclusion. + +## AI execution checklist + +When delegating the run to an AI agent or an automation runner, provide these +inputs explicitly: + +- repository checkout and candidate branch or commit; +- baseline commit or binary path; +- candidate binary path; +- runner type: local Linux or external cluster; +- endpoint, access key, secret key source, and region; +- deploy hook path or exact command for cluster mode; +- output directory; +- required duration, rounds, cooldown, concurrency, and fail/warn budgets; +- where to upload artifacts after the run. + +The AI agent should execute this sequence: + +1. Confirm `uname -a`, RustFS commits, binary SHA256 sums, `warp --version`, + CPU model, memory size, disk layout, and whether the run is local or cluster. +2. Run `scripts/run_hotpath_warp_abba.sh --dry-run` with the final arguments. +3. Run the real ABBA command with `--rounds >= 3`. +4. Preserve the full output directory without editing generated CSV files. +5. Read `summary.md`, `candidate_gate.md`, and `baseline_drift_gate.md`. +6. Summarize only measured facts: candidate deltas, baseline drift, CPU or + memory saturation, and any failed workloads. +7. Post the summary and artifact location to the tracking issue or PR. + +Do not report a performance win or loss when the baseline drift gate failed on +the same workload and no rerun was collected. diff --git a/docs/operations/kms-observability-runbook.md b/docs/operations/kms-observability-runbook.md new file mode 100644 index 000000000..310df6cbe --- /dev/null +++ b/docs/operations/kms-observability-runbook.md @@ -0,0 +1,130 @@ +# KMS observability runbook + +This runbook covers the KMS backend operation metrics, the Grafana dashboard that visualizes them, and the response procedure for each Prometheus alert shipped in `.docker/observability/prometheus-rules/rustfs-kms-alerts.yml`. It is the `runbook_url` target for those alerts. For what each KMS backend protects and how Vault authentication behaves, see the [KMS backend security properties](kms-backend-security.md) and the [Vault KMS authentication runbook](vault-kms-authentication.md). + +## Metric reference + +All four metrics are emitted at the single operation-policy choke point (`crates/kms/src/policy.rs`) that every instrumented KMS backend call flows through. Label values are exclusively static enum strings — key identifiers, key material, ciphertext, paths, and tokens never appear in metric labels, and any change that would add such a label is a regression. + +| Metric | Type | Labels | Meaning | +| --- | --- | --- | --- | +| `rustfs_kms_backend_operations_total` | counter | `operation`, `op_class`, `outcome` | Operations executed under the operation policy, counted once per terminal outcome | +| `rustfs_kms_backend_attempt_failures_total` | counter | `operation`, `error_class` | Individual failed attempts, including attempts the retry policy later absorbed | +| `rustfs_kms_backend_operation_duration_seconds` | histogram | `operation`, `outcome` | Wall-clock duration of a whole operation, including retries and backoff sleeps | +| `rustfs_kms_backend_operation_attempts` | histogram | `operation`, `outcome` | Number of attempts one operation used before completing | + +Label values: + +- `outcome`: `success`, `fatal` (a non-retryable failure ended the operation on first observation), `budget_exhausted` (the attempt budget ran out on retryable failures), `deadline_exceeded` (the operation deadline ran out before another attempt could complete), `cancelled` (shutdown or caller cancellation). +- `op_class`: `read_idempotent` (safe to retry), `mutating_non_idempotent` (never replayed — a retryable failure terminates after a single attempt because the server may have processed the request), `auth` (login and token renewal). +- `error_class`: `retryable_conn` (connection-level failure: dial, TLS, broken connection), `retryable_status` (retryable backend status, e.g. Vault 5xx or a sealed Vault's 503), `attempt_timeout` (the per-attempt timeout cut the attempt off; retried like a connection failure because the server may still have processed the request), `fatal` (non-retryable: authentication, permissions, malformed request, missing key or version). +- `operation`: static per-call-site names, e.g. `vault_kv2_read_key_version`, `vault_kv2_cas_write_key`, `vault_transit_encrypt`, `vault_transit_decrypt`, `vault_login`, `vault_token_renew`. + +Export path: the `metrics` facade feeds the OTel recorder in `crates/obs`, which exports over OTLP to the collector scraped by Prometheus. Histograms therefore appear in Prometheus as `_bucket`/`_sum`/`_count` series. These metrics do not carry the RustFS `server` label used by the node observability dashboard — distinguish nodes through your scrape topology (`job`/`instance` or promoted OTel resource attributes such as `service_instance_id`). + +Instrumentation boundary: the Local and Static backends do not flow through the choke point and emit no operation metrics; bringing them under the same instrumentation is tracked separately (rustfs/backlog#1569). Absence of KMS series on a cluster using those backends is expected, not an outage. + +## Dashboard + +Import `deploy/observability/grafana/rustfs-kms-observability.json` into Grafana and select a Prometheus data source that scrapes RustFS metrics. The dashboard has two variables: `datasource` (Prometheus data source) and `operation` (multi-select over the `operation` label). In the docker-compose observability stack (`.docker/observability/`), dashboards are provisioned from a directory (`grafana/provisioning/dashboards/dashboard.yml` points at `/etc/grafana/dashboards`), so no per-file registration is needed there. + +The dashboard's "Planned Panels (TODO)" text panel lists metric families designed under rustfs/backlog#1584 but not yet landed (key-cache effectiveness, lifecycle gauges, token TTL, synthetic probe). Do not add panels or alerts for those names until the emitting code is merged; keep that panel and the [Coverage gaps](#coverage-gaps-and-planned-metrics) section below in sync as they land. + +## Alert rules + +The rules live in `.docker/observability/prometheus-rules/rustfs-kms-alerts.yml`. The docker-compose Prometheus loads `/etc/prometheus/rules/*.yml`, so the `.yml` extension is load-bearing. Validate edits with `promtool check rules rustfs-kms-alerts.yml`. + +Every threshold in that file is a conservative default chosen without a production baseline; see [Threshold calibration](#threshold-calibration) before treating a firing alert as an SLO breach or a quiet one as health. + +## Alert response procedures + +### KmsBackendFatalErrors + +Meaning: attempts are failing with `error_class="fatal"` — failures the policy never retries. Each one is a KMS backend call that failed permanently (authentication, permissions, malformed request, or a missing key/version), so callers are seeing errors right now. This is the highest-signal KMS alert: fatal failures do not appear as background noise in a healthy system. + +Investigation: + +1. Break the rate down by operation: `sum by (operation) (rate(rustfs_kms_backend_attempt_failures_total{error_class="fatal"}[5m]))`. +2. If the failing operations are `vault_login` or `vault_token_renew` (`op_class="auth"`), the Vault credentials are invalid or expired. Follow the [Vault KMS authentication runbook](vault-kms-authentication.md) — note that credential refresh is fail-closed, so a broken credential eventually takes down all Vault-backed operations, not just auth. Look for the `Vault token renewal failed; falling back to a fresh login` and `Vault credential refresh failed; retrying until the credentials recover` warnings in the RustFS logs. +3. If the failing operations are `vault_kv2_*` or `vault_transit_*`, check for Vault permission denials: compare the token's policy against the minimal policy in [KMS backend security properties](kms-backend-security.md) (a policy that drifted or was re-scoped produces 403s that classify as fatal), and check the Vault audit log for the corresponding denied requests. +4. A fatal `KeyVersionNotFound` on decrypt-path operations means a DEK envelope references a key version whose record is missing. Decryption deliberately fails closed with no fallback — see the rotation retention preconditions in [KMS backend security properties](kms-backend-security.md) and verify nobody destroyed version records under the key subtree. +5. Confirm blast radius with the outcome view: `sum by (operation) (rate(rustfs_kms_backend_operations_total{outcome="fatal"}[5m]))`. + +Related signals: the "Attempt Failure Rate by Error Class" and "Backend Operation Rate by Outcome" dashboard panels; Vault server audit and server logs; S3-level 5xx on encrypted buckets. + +### KmsBackendHighErrorRate + +Meaning: more than 5% of KMS operations are terminating without success (`fatal`, `budget_exhausted`, or `deadline_exceeded`; `cancelled` is excluded because shutdown windows legitimately produce it). A traffic guard suppresses the alert below ~0.02 ops/s so a single failure on a near-idle cluster does not page. + +Investigation: + +1. Break the failures down by outcome: `sum by (outcome) (rate(rustfs_kms_backend_operations_total{outcome!~"success|cancelled"}[5m]))`. +2. If `fatal` dominates, follow [KmsBackendFatalErrors](#kmsbackendfatalerrors). +3. If `budget_exhausted` or `deadline_exceeded` dominates, follow [KmsBackendRetryBudgetExhausted](#kmsbackendretrybudgetexhausted) — the backend is unavailable or too slow for longer than the retry policy can bridge. +4. Correlate with client impact: encrypted-object PUT/GET failures and S3 error rates on buckets with encryption configured. + +Related signals: the "Non-Success Outcome Ratio" dashboard panel; the KMS-related warnings listed under the other alerts in this runbook. + +### KmsBackendP99LatencyHigh + +Meaning: the p99 wall-clock duration of KMS operations is sustained above 2s. The histogram includes retries and backoff sleeps, so a high p99 with a healthy p50 usually means a slow retry tail (a subset of calls failing and being retried), not a uniform slowdown. + +Investigation: + +1. Compare p50 and p99 on the "Operation Duration p50 / p99" panel. Flat p50 with elevated p99 points at retries; both elevated points at the backend or the network path being uniformly slow. +2. Split by operation with `histogram_quantile(0.99, sum by (le, operation) (rate(rustfs_kms_backend_operation_duration_seconds_bucket[5m])))` to see whether one backend call or all of them regressed. +3. Check the attempts histogram: an average meaningfully above 1 confirms the latency is retry-driven; follow [KmsBackendAttemptFailureSpike](#kmsbackendattemptfailurespike) for the failure classes. +4. If latency is not retry-driven, check the network path to Vault (TLS handshakes, DNS, proxies) and Vault's own telemetry (storage backend latency, load). +5. Remember that this latency sits inside S3 request latency for encrypted objects: sustained p99 near the operation deadline will start converting into `deadline_exceeded` outcomes. + +Related signals: the "Operation Duration p99 by Operation" and "Operation Attempts Distribution" panels; `KMS backend attempt failed with a retryable error; backing off before retry` warnings (fields: `operation`, `attempt`, `error_class`, `backoff`). + +### KmsBackendAttemptFailureSpike + +Meaning: individual attempts are failing at a sustained rate across all error classes. The retry policy may still be absorbing these — operations can keep succeeding while this alert fires — but the system is burning retry budget and running degraded, and a small further degradation will surface to callers. + +Investigation: + +1. Break the rate down by class: `sum by (error_class) (rate(rustfs_kms_backend_attempt_failures_total[5m]))`. +2. `retryable_conn`: network-level failures — check connectivity, TLS, DNS, and whether Vault is down or restarting. +3. `retryable_status`: the backend answered with a retryable error — check Vault health and seal status (a sealed Vault returns 503, which lands here), and Vault-side rate limiting. +4. `attempt_timeout`: attempts are being cut off by the per-attempt timeout — either the backend is slow (correlate with [KmsBackendP99LatencyHigh](#kmsbackendp99latencyhigh)) or the configured attempt timeout is too tight for the deployment's network path. +5. `fatal`: follow [KmsBackendFatalErrors](#kmsbackendfatalerrors). +6. Grep RustFS logs for `KMS backend attempt failed with a retryable error; backing off before retry` — the structured fields (`operation`, `attempt`, `error_class`, `backoff`) identify which call sites are cycling. + +Related signals: the "Attempt Failure Rate by Error Class" panel; the attempts histogram average rising above 1. + +### KmsBackendRetryBudgetExhausted + +Meaning: operations are terminating as `budget_exhausted` or `deadline_exceeded` — every individual failure was retryable, but the backend stayed unhealthy for longer than the retry policy could bridge, so callers received hard failures. + +Investigation: + +1. Identify the failing operations: `sum by (operation) (rate(rustfs_kms_backend_operations_total{outcome=~"budget_exhausted|deadline_exceeded"}[5m]))`. +2. Establish how long the underlying failure has persisted from the attempt-failure rate history; follow [KmsBackendAttemptFailureSpike](#kmsbackendattemptfailurespike) for the class-specific diagnosis. +3. Note the by-design case: `mutating_non_idempotent` operations (e.g. `vault_kv2_cas_write_key`, `vault_transit_create_key`) are never replayed, so a single retryable failure terminates them as `budget_exhausted` after one attempt. A spike confined to mutating operations means write-path failures, not an exhausted retry loop. +4. `deadline_exceeded` clustering with duration p99 near the operation deadline means the budget is being spent on slow attempts rather than fast failures — treat as a latency problem first. +5. Confirm client impact and, if the backend outage is confirmed external (Vault down), coordinate recovery there; RustFS will resume without intervention once the backend recovers. + +Related signals: the "Backend Operation Rate by Outcome" panel; retry-backoff warnings in RustFS logs; Vault availability monitoring. + +## Threshold calibration + +Every numeric threshold in `rustfs-kms-alerts.yml` (5% error ratio, 2s p99, 0.5/s attempt failures, 0.05/s budget exhaustion) is a conservative default chosen without a production baseline, biased toward not paging on healthy-but-busy systems. Before relying on these alerts for paging: run the workload in staging for at least a week, record the steady-state values of the expressions above, then tighten thresholds to sit clearly above observed peaks. Once a stable baseline exists, consider converting `KmsBackendAttemptFailureSpike` to a baseline-relative form (`offset 1d` ratio, see `.docker/observability/prometheus-rules/rustfs-get-optimization-alerts.yaml` for the pattern). Formal SLO targets for KMS operations are deliberately out of scope until that baseline exists (rustfs/backlog#1584). + +## Coverage gaps and planned metrics + +The following families are designed under rustfs/backlog#1584 but land in separate changes; they are intentionally absent from the dashboard and alert rules, and referencing their names before the emitting code merges would produce permanently-empty panels and never-firing alerts: + +- Key-cache effectiveness: hit/miss counters, entry gauge, eviction counter (replacing the former hardcoded-zero miss statistic). +- Key lifecycle: pending-deletion/tombstone gauges from the deletion-worker sweep and an aggregate rotation-age gauge. +- Vault credentials: token TTL remaining and fail-closed state gauges (today only the log warnings quoted above exist). +- Synthetic backend probe: probe outcome and failure-class metrics feeding the readiness cache. + +When one of these lands, add its panels, extend the alert rules, replace the corresponding TODO bullet in the dashboard's "Planned Panels" text panel, and update this section. + +## Related documents + +- [KMS backend security properties](kms-backend-security.md) — backend trust boundaries, minimal Vault policies, rotation retention preconditions. +- [Vault KMS authentication runbook](vault-kms-authentication.md) — credential sources, refresh behavior, and the fail-closed window. +- `deploy/observability/README.md` — dashboard import notes for all RustFS dashboards. diff --git a/rustfs/Cargo.toml b/rustfs/Cargo.toml index 53db5ca19..24a18a938 100644 --- a/rustfs/Cargo.toml +++ b/rustfs/Cargo.toml @@ -60,25 +60,137 @@ hotpath = [ "hotpath/crossbeam", "hotpath/parking_lot", "hotpath/reqwest-0-13", + "rustfs-audit/hotpath", + "rustfs-common/hotpath", + "rustfs-concurrency/hotpath", + "rustfs-config/hotpath", + "rustfs-credentials/hotpath", + "rustfs-crypto/hotpath", + "rustfs-data-usage/hotpath", "rustfs-ecstore/hotpath", + "rustfs-extension-schema/hotpath", "rustfs-filemeta/hotpath", + "rustfs-heal/hotpath", + "rustfs-iam/hotpath", + "rustfs-io-core/hotpath", + "rustfs-io-metrics/hotpath", + "rustfs-keystone/hotpath", + "rustfs-kms/hotpath", + "rustfs-lock/hotpath", + "rustfs-log-analyzer/hotpath", + "rustfs-madmin/hotpath", + "rustfs-notify/hotpath", + "rustfs-object-capacity/hotpath", + "rustfs-object-data-cache/hotpath", + "rustfs-obs/hotpath", "rustfs-policy/hotpath", + "rustfs-protocols/hotpath", + "rustfs-protos/hotpath", "rustfs-rio/hotpath", + "rustfs-s3-ops/hotpath", + "rustfs-s3-types/hotpath", + "rustfs-s3select-api/hotpath", + "rustfs-s3select-query/hotpath", + "rustfs-scanner/hotpath", + "rustfs-security-governance/hotpath", + "rustfs-signer/hotpath", + "rustfs-storage-api/hotpath", + "rustfs-targets/hotpath", + "rustfs-tls-runtime/hotpath", + "rustfs-trusted-proxies/hotpath", + "rustfs-utils/hotpath", + "rustfs-zip/hotpath", + "rustfs-test-utils/hotpath", ] hotpath-alloc = [ + "hotpath", "hotpath/hotpath-alloc", + "rustfs-audit/hotpath-alloc", + "rustfs-common/hotpath-alloc", + "rustfs-concurrency/hotpath-alloc", + "rustfs-config/hotpath-alloc", + "rustfs-credentials/hotpath-alloc", + "rustfs-crypto/hotpath-alloc", + "rustfs-data-usage/hotpath-alloc", "rustfs-ecstore/hotpath-alloc", + "rustfs-extension-schema/hotpath-alloc", "rustfs-filemeta/hotpath-alloc", + "rustfs-heal/hotpath-alloc", + "rustfs-iam/hotpath-alloc", + "rustfs-io-core/hotpath-alloc", + "rustfs-io-metrics/hotpath-alloc", + "rustfs-keystone/hotpath-alloc", + "rustfs-kms/hotpath-alloc", + "rustfs-lock/hotpath-alloc", + "rustfs-log-analyzer/hotpath-alloc", + "rustfs-madmin/hotpath-alloc", + "rustfs-notify/hotpath-alloc", + "rustfs-object-capacity/hotpath-alloc", + "rustfs-object-data-cache/hotpath-alloc", + "rustfs-obs/hotpath-alloc", "rustfs-policy/hotpath-alloc", + "rustfs-protocols/hotpath-alloc", + "rustfs-protos/hotpath-alloc", "rustfs-rio/hotpath-alloc", + "rustfs-s3-ops/hotpath-alloc", + "rustfs-s3-types/hotpath-alloc", + "rustfs-s3select-api/hotpath-alloc", + "rustfs-s3select-query/hotpath-alloc", + "rustfs-scanner/hotpath-alloc", + "rustfs-security-governance/hotpath-alloc", + "rustfs-signer/hotpath-alloc", + "rustfs-storage-api/hotpath-alloc", + "rustfs-targets/hotpath-alloc", + "rustfs-tls-runtime/hotpath-alloc", + "rustfs-trusted-proxies/hotpath-alloc", + "rustfs-utils/hotpath-alloc", + "rustfs-zip/hotpath-alloc", + "rustfs-test-utils/hotpath-alloc", ] hotpath-cpu = [ "hotpath", "hotpath/hotpath-cpu", + "rustfs-audit/hotpath-cpu", + "rustfs-common/hotpath-cpu", + "rustfs-concurrency/hotpath-cpu", + "rustfs-config/hotpath-cpu", + "rustfs-credentials/hotpath-cpu", + "rustfs-crypto/hotpath-cpu", + "rustfs-data-usage/hotpath-cpu", "rustfs-ecstore/hotpath-cpu", + "rustfs-extension-schema/hotpath-cpu", "rustfs-filemeta/hotpath-cpu", + "rustfs-heal/hotpath-cpu", + "rustfs-iam/hotpath-cpu", + "rustfs-io-core/hotpath-cpu", + "rustfs-io-metrics/hotpath-cpu", + "rustfs-keystone/hotpath-cpu", + "rustfs-kms/hotpath-cpu", + "rustfs-lock/hotpath-cpu", + "rustfs-log-analyzer/hotpath-cpu", + "rustfs-madmin/hotpath-cpu", + "rustfs-notify/hotpath-cpu", + "rustfs-object-capacity/hotpath-cpu", + "rustfs-object-data-cache/hotpath-cpu", + "rustfs-obs/hotpath-cpu", "rustfs-policy/hotpath-cpu", + "rustfs-protocols/hotpath-cpu", + "rustfs-protos/hotpath-cpu", "rustfs-rio/hotpath-cpu", + "rustfs-s3-ops/hotpath-cpu", + "rustfs-s3-types/hotpath-cpu", + "rustfs-s3select-api/hotpath-cpu", + "rustfs-s3select-query/hotpath-cpu", + "rustfs-scanner/hotpath-cpu", + "rustfs-security-governance/hotpath-cpu", + "rustfs-signer/hotpath-cpu", + "rustfs-storage-api/hotpath-cpu", + "rustfs-targets/hotpath-cpu", + "rustfs-tls-runtime/hotpath-cpu", + "rustfs-trusted-proxies/hotpath-cpu", + "rustfs-utils/hotpath-cpu", + "rustfs-zip/hotpath-cpu", + "rustfs-test-utils/hotpath-cpu", ] [lints] diff --git a/rustfs/src/admin/router.rs b/rustfs/src/admin/router.rs index a72858a8a..7bec6c88a 100644 --- a/rustfs/src/admin/router.rs +++ b/rustfs/src/admin/router.rs @@ -2708,18 +2708,16 @@ impl S3Router { pub fn insert(&mut self, method: Method, path: &str, operation: T) -> std::io::Result<()> { let path = Self::make_route_str(method, path); + #[cfg(test)] + let registered_path = path.clone(); // warn!("set uri {}", &path); - #[cfg(test)] - { - self.router.insert(path.clone(), operation).map_err(std::io::Error::other)?; - self.registered_routes.push(path); - } - - #[cfg(not(test))] self.router.insert(path, operation).map_err(std::io::Error::other)?; + #[cfg(test)] + self.registered_routes.push(registered_path); + Ok(()) } diff --git a/rustfs/src/app/object_usecase.rs b/rustfs/src/app/object_usecase.rs index 8d46feadd..806b07ff9 100644 --- a/rustfs/src/app/object_usecase.rs +++ b/rustfs/src/app/object_usecase.rs @@ -6072,19 +6072,6 @@ impl DefaultObjectUsecase { )); } }; - let has_replacement_metadata = metadata.is_some() - || cache_control.is_some() - || content_disposition.is_some() - || content_encoding.is_some() - || content_language.is_some() - || content_type.is_some() - || expires.is_some(); - if has_replacement_metadata && !replaces_metadata { - return Err(S3Error::with_message( - S3ErrorCode::InvalidRequest, - "Replacement metadata requires the REPLACE metadata directive".to_string(), - )); - } let replacement_metadata = if replaces_metadata { validate_archive_content_encoding(&key, content_type.as_deref(), content_encoding.as_deref())?; let mut replacement_metadata = metadata.unwrap_or_default(); diff --git a/scripts/run_hotpath_warp_abba.sh b/scripts/run_hotpath_warp_abba.sh new file mode 100755 index 000000000..ccaff476d --- /dev/null +++ b/scripts/run_hotpath_warp_abba.sh @@ -0,0 +1,428 @@ +#!/usr/bin/env bash +# Formal Linux / production-cluster ABBA runner for the hotpath warp matrix. +# +# This script is intentionally a thin orchestrator around the existing +# run_object_batch_bench_enhanced.sh load driver and hotpath_warp_ab_gate.sh +# relative-budget gate. It runs each durability/workload cell as: +# +# A1 baseline -> B1 candidate -> B2 candidate -> A2 baseline +# +# Candidate legs are compared against A1. The final A2 leg is also compared +# against A1 to quantify baseline drift separately from candidate deltas. + +set -euo pipefail + +PROJECT_ROOT="$(git rev-parse --show-toplevel)" +ENHANCED_BENCH="${PROJECT_ROOT}/scripts/run_object_batch_bench_enhanced.sh" +GATE="${PROJECT_ROOT}/scripts/hotpath_warp_ab_gate.sh" + +BASELINE_BIN="" +CANDIDATE_BIN="" +ENDPOINT="" +DEPLOY_HOOK="" +HEALTH_PATH="/health" +ADDRESS="127.0.0.1:9000" +DATA_ROOT="/tmp/rustfs-hotpath-abba" +DISKS=4 +ACCESS_KEY="rustfsadmin" +SECRET_KEY="rustfsadmin" +REGION="us-east-1" +WARP_BIN="warp" +CONCURRENCY=8 +DURATION="60s" +ROUNDS=3 +COOLDOWN_SECS=20 +HEALTH_TIMEOUT_SECS=180 +FAIL_PCT=10 +WARN_PCT=5 +ALLOW_REGRESSION=false +EXEMPTION_REASON="deliberate correctness tradeoff" +OUT_DIR="${PROJECT_ROOT}/target/hotpath-abba/$(date -u +%Y%m%dT%H%M%SZ 2>/dev/null || echo run)" +DRY_RUN=false + +WORKLOADS=( + "put-4kib|put|4KiB" + "put-4mib|put|4MiB" + "get-4kib|get|4KiB" + "get-4mib|get|4MiB" + "get-10mib|get|10MiB" + "mixed-256k|mixed|256KiB" +) +DRIVE_SYNC_MATRIX=("sync-on|true" "sync-off|false") + +usage() { + cat <<'USAGE' +Usage: scripts/run_hotpath_warp_abba.sh --baseline-bin --candidate-bin [options] + +Formal ABBA mode for Linux runners or production-like clusters. The schedule is +A1 baseline -> B1 candidate -> B2 candidate -> A2 baseline for every workload +and drive-sync cell. + +Required: + --baseline-bin Baseline RustFS binary. + --candidate-bin Candidate RustFS binary. + +Local Linux runner mode: + --address Local RustFS address (default 127.0.0.1:9000). + --disks Throwaway local disks per node (default 4). + --data-root Local disk root (default /tmp/rustfs-hotpath-abba). + +Production / cluster mode: + --endpoint Existing cluster endpoint. Enables external mode. + --deploy-hook Command run before each ABBA leg. It receives: + HOTPATH_ABBA_LEG=A1|B1|B2|A2 + HOTPATH_ABBA_PHASE=baseline|candidate + HOTPATH_ABBA_BINARY= + HOTPATH_ABBA_DRIVE_SYNC=true|false + --health-path Readiness path (default /health). + +Benchmark: + --duration warp duration per cell (default 60s). + --rounds rounds per cell; must be >= 3 (default 3). + --cooldown cooldown seconds between rounds/sizes (default 20). + --concurrency warp concurrency (default 8). + --warp-bin warp binary (default warp). + +Credentials: + --access-key S3 access key (default rustfsadmin). + --secret-key S3 secret key (default rustfsadmin). + --region S3 region (default us-east-1). + +Gate: + --fail-pct Regression budget that fails gate (default 10). + --warn-pct Regression budget that warns (default 5). + --allow-regression Downgrade candidate gate FAIL to WARN. + --exemption-reason Reason recorded when allow-regression is used. + +Output: + --out-dir Output dir (default target/hotpath-abba/). + --dry-run Print commands without starting servers or warp. + -h, --help + +Outputs: + /abba_schedule.csv + /candidate_gate.md + /baseline_drift_gate.md + /summary.md + ////{median_summary.csv,baseline_compare.csv} +USAGE +} + +die() { + echo "error: $*" >&2 + exit 2 +} + +log() { + printf '[hotpath-abba] %s\n' "$*" >&2 +} + +run() { + if [[ "$DRY_RUN" == "true" ]]; then + { printf 'DRY-RUN:'; printf ' %q' "$@"; printf '\n'; } >&2 + return 0 + fi + "$@" +} + +validate_positive_int() { + local value="$1" name="$2" + [[ "$value" =~ ^[0-9]+$ && "$value" -gt 0 ]] || die "$name must be a positive integer" +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --baseline-bin) BASELINE_BIN="$2"; shift 2 ;; + --candidate-bin) CANDIDATE_BIN="$2"; shift 2 ;; + --endpoint) ENDPOINT="$2"; shift 2 ;; + --deploy-hook) DEPLOY_HOOK="$2"; shift 2 ;; + --health-path) HEALTH_PATH="$2"; shift 2 ;; + --address) ADDRESS="$2"; shift 2 ;; + --data-root) DATA_ROOT="$2"; shift 2 ;; + --disks) DISKS="$2"; shift 2 ;; + --access-key) ACCESS_KEY="$2"; shift 2 ;; + --secret-key) SECRET_KEY="$2"; shift 2 ;; + --region) REGION="$2"; shift 2 ;; + --warp-bin) WARP_BIN="$2"; shift 2 ;; + --concurrency) CONCURRENCY="$2"; shift 2 ;; + --duration) DURATION="$2"; shift 2 ;; + --rounds) ROUNDS="$2"; shift 2 ;; + --cooldown) COOLDOWN_SECS="$2"; shift 2 ;; + --health-timeout) HEALTH_TIMEOUT_SECS="$2"; shift 2 ;; + --fail-pct) FAIL_PCT="$2"; shift 2 ;; + --warn-pct) WARN_PCT="$2"; shift 2 ;; + --allow-regression) ALLOW_REGRESSION=true; shift ;; + --exemption-reason) EXEMPTION_REASON="$2"; shift 2 ;; + --out-dir) OUT_DIR="$2"; shift 2 ;; + --dry-run) DRY_RUN=true; shift ;; + -h|--help) usage; exit 0 ;; + *) die "unknown argument: $1" ;; + esac +done + +validate_positive_int "$DISKS" "--disks" +validate_positive_int "$CONCURRENCY" "--concurrency" +validate_positive_int "$ROUNDS" "--rounds" +validate_positive_int "$COOLDOWN_SECS" "--cooldown" +validate_positive_int "$HEALTH_TIMEOUT_SECS" "--health-timeout" +validate_positive_int "$FAIL_PCT" "--fail-pct" +validate_positive_int "$WARN_PCT" "--warn-pct" +[[ "$ROUNDS" -ge 3 ]] || die "--rounds must be >= 3 for formal ABBA evidence" +[[ -n "$BASELINE_BIN" ]] || die "--baseline-bin is required" +[[ -n "$CANDIDATE_BIN" ]] || die "--candidate-bin is required" +[[ "$DRY_RUN" == "true" || -x "$BASELINE_BIN" ]] || die "baseline binary is not executable: $BASELINE_BIN" +[[ "$DRY_RUN" == "true" || -x "$CANDIDATE_BIN" ]] || die "candidate binary is not executable: $CANDIDATE_BIN" +[[ -x "$ENHANCED_BENCH" ]] || die "missing load driver: $ENHANCED_BENCH" +[[ -x "$GATE" ]] || die "missing gate: $GATE" +if [[ "$DRY_RUN" != "true" ]] && ! command -v "$WARP_BIN" >/dev/null 2>&1; then + die "warp not found on PATH; install warp or pass --warp-bin" +fi + +EXTERNAL=false +if [[ -n "$ENDPOINT" ]]; then + EXTERNAL=true + ADDRESS="$ENDPOINT" + [[ -n "$DEPLOY_HOOK" ]] || log "warning: external mode without --deploy-hook; binaries must be swapped out of band" +fi + +mkdir -p "$OUT_DIR" +SERVER_LOG_DIR="$OUT_DIR/server-logs" +run mkdir -p "$SERVER_LOG_DIR" + +SERVER_PID="" +SERVER_LOG="" +tear_down() { + [[ "$EXTERNAL" == "true" ]] && return 0 + [[ -n "$SERVER_PID" ]] || return 0 + run kill "$SERVER_PID" 2>/dev/null || true + SERVER_PID="" +} +trap tear_down EXIT INT TERM + +dump_server_log() { + [[ -n "$SERVER_LOG" && -f "$SERVER_LOG" ]] || return 0 + echo "----- last 80 lines of $SERVER_LOG -----" >&2 + tail -n 80 "$SERVER_LOG" >&2 || true + echo "----------------------------------------" >&2 +} + +wait_health() { + [[ "$DRY_RUN" == "true" ]] && return 0 + local i + for ((i = 0; i < HEALTH_TIMEOUT_SECS; i++)); do + if [[ "$EXTERNAL" != "true" && -n "$SERVER_PID" ]] && ! kill -0 "$SERVER_PID" 2>/dev/null; then + echo "error: rustfs server (pid $SERVER_PID) exited before becoming healthy after ${i}s" >&2 + dump_server_log + return 1 + fi + if curl -fsS "http://${ADDRESS}${HEALTH_PATH}" >/dev/null 2>&1; then + return 0 + fi + sleep 1 + done + echo "error: endpoint http://${ADDRESS}${HEALTH_PATH} did not become healthy within ${HEALTH_TIMEOUT_SECS}s" >&2 + dump_server_log + return 1 +} + +binary_for_leg() { + case "$1" in + A1|A2) echo "$BASELINE_BIN" ;; + B1|B2) echo "$CANDIDATE_BIN" ;; + *) die "unknown ABBA leg: $1" ;; + esac +} + +phase_for_leg() { + case "$1" in + A1|A2) echo "baseline" ;; + B1|B2) echo "candidate" ;; + *) die "unknown ABBA leg: $1" ;; + esac +} + +bring_up() { + local leg="$1" drive_sync="$2" + local phase bin + phase="$(phase_for_leg "$leg")" + bin="$(binary_for_leg "$leg")" + + if [[ "$EXTERNAL" == "true" ]]; then + if [[ -n "$DEPLOY_HOOK" ]]; then + log "deploy hook: leg=$leg phase=$phase drive_sync=$drive_sync" + HOTPATH_ABBA_LEG="$leg" HOTPATH_ABBA_PHASE="$phase" HOTPATH_ABBA_BINARY="$bin" HOTPATH_ABBA_DRIVE_SYNC="$drive_sync" \ + run bash -c "$DEPLOY_HOOK" + fi + wait_health + return 0 + fi + + local node_dir="$DATA_ROOT/$leg-sync-$drive_sync" + local disks=() d + for ((d = 1; d <= DISKS; d++)); do + disks+=("$node_dir/d$d") + done + run mkdir -p "${disks[@]}" + SERVER_LOG="$SERVER_LOG_DIR/$leg-sync-$drive_sync.log" + + if [[ "$DRY_RUN" == "true" ]]; then + { printf 'DRY-RUN: RUSTFS_DRIVE_SYNC_ENABLE=%s %q server' "$drive_sync" "$bin" + printf ' %q' "${disks[@]}"; printf '\n'; } >&2 + SERVER_PID="dry-run" + return 0 + fi + + cat >"$SERVER_LOG_DIR/$leg-sync-$drive_sync.env" </dev/null || echo unknown) +warp_version=$("$WARP_BIN" --version 2>/dev/null | head -n1 || echo unknown) +EOF + RUSTFS_UNSAFE_BYPASS_DISK_CHECK=true \ + RUSTFS_ADDRESS="$ADDRESS" \ + RUSTFS_ACCESS_KEY="$ACCESS_KEY" \ + RUSTFS_SECRET_KEY="$SECRET_KEY" \ + RUSTFS_REGION="$REGION" \ + RUSTFS_CONSOLE_ENABLE=false \ + RUSTFS_DRIVE_SYNC_ENABLE="$drive_sync" \ + "$bin" server "${disks[@]}" >"$SERVER_LOG" 2>&1 & + SERVER_PID=$! + wait_health +} + +measure() { + local leg="$1" workload="$2" mode="$3" size="$4" sync_label="$5" baseline_csv="${6:-}" + local cell="$OUT_DIR/$workload/$sync_label/$leg" + local args=( + --tool warp --warp-bin "$WARP_BIN" --warp-mode "$mode" + --endpoint "$ADDRESS" --access-key "$ACCESS_KEY" --secret-key "$SECRET_KEY" + --region "$REGION" --sizes "$size" --concurrency "$CONCURRENCY" + --duration "$DURATION" --rounds "$ROUNDS" --cooldown-secs "$COOLDOWN_SECS" + --out-dir "$cell" + ) + [[ -n "$baseline_csv" ]] && args+=(--baseline-csv "$baseline_csv") + run "$ENHANCED_BENCH" "${args[@]}" >&2 + echo "$cell" +} + +write_schedule_header() { + echo "sync_label,drive_sync,workload,mode,size,leg,phase,binary,out_dir" >"$OUT_DIR/abba_schedule.csv" +} + +append_schedule() { + local sync_label="$1" drive_sync="$2" workload="$3" mode="$4" size="$5" leg="$6" + local phase bin + phase="$(phase_for_leg "$leg")" + bin="$(binary_for_leg "$leg")" + echo "$sync_label,$drive_sync,$workload,$mode,$size,$leg,$phase,$bin,$OUT_DIR/$workload/$sync_label/$leg" >>"$OUT_DIR/abba_schedule.csv" +} + +write_manifest() { + cat >"$OUT_DIR/manifest.env" </dev/null || echo unknown) +runner=$(uname -srm 2>/dev/null || echo unknown) +schedule=ABBA +rounds=$ROUNDS +duration=$DURATION +cooldown_secs=$COOLDOWN_SECS +concurrency=$CONCURRENCY +baseline_bin=$BASELINE_BIN +candidate_bin=$CANDIDATE_BIN +external=$EXTERNAL +endpoint=$ADDRESS +warp_version=$("$WARP_BIN" --version 2>/dev/null | head -n1 || echo unknown) +EOF +} + +declare -a CANDIDATE_COMPARE_CSVS=() +declare -a DRIFT_COMPARE_CSVS=() + +write_manifest +write_schedule_header + +for ds_spec in "${DRIVE_SYNC_MATRIX[@]}"; do + IFS='|' read -r sync_label drive_sync <<<"$ds_spec" + + for leg in A1 B1 B2 A2; do + log "=== $sync_label leg $leg ($(phase_for_leg "$leg")) ===" + bring_up "$leg" "$drive_sync" + + for wl_spec in "${WORKLOADS[@]}"; do + IFS='|' read -r workload mode size <<<"$wl_spec" + append_schedule "$sync_label" "$drive_sync" "$workload" "$mode" "$size" "$leg" + + baseline_csv="" + if [[ "$leg" != "A1" ]]; then + baseline_csv="$OUT_DIR/$workload/$sync_label/A1/median_summary.csv" + fi + + cell="$(measure "$leg" "$workload" "$mode" "$size" "$sync_label" "$baseline_csv")" + case "$leg" in + B1|B2) CANDIDATE_COMPARE_CSVS+=("$cell/baseline_compare.csv") ;; + A2) DRIFT_COMPARE_CSVS+=("$cell/baseline_compare.csv") ;; + esac + done + + tear_down + done +done + +gate_args=(--fail-pct "$FAIL_PCT" --warn-pct "$WARN_PCT" --markdown "$OUT_DIR/candidate_gate.md") +for csv in "${CANDIDATE_COMPARE_CSVS[@]}"; do + gate_args+=(--compare-csv "$csv") +done +[[ "$ALLOW_REGRESSION" == "true" ]] && gate_args+=(--allow-regression --exemption-reason "$EXEMPTION_REASON") + +drift_gate_args=(--fail-pct "$FAIL_PCT" --warn-pct "$WARN_PCT" --markdown "$OUT_DIR/baseline_drift_gate.md") +for csv in "${DRIFT_COMPARE_CSVS[@]}"; do + drift_gate_args+=(--compare-csv "$csv") +done + +if [[ "$DRY_RUN" == "true" ]]; then + log "dry-run complete; candidate compare CSVs=${#CANDIDATE_COMPARE_CSVS[@]} baseline drift CSVs=${#DRIFT_COMPARE_CSVS[@]}" + { printf 'DRY-RUN:'; printf ' %q' "$GATE" "${gate_args[@]}"; printf '\n'; } >&2 + { printf 'DRY-RUN:'; printf ' %q' "$GATE" "${drift_gate_args[@]}"; printf '\n'; } >&2 + exit 0 +fi + +log "applying candidate relative-budget gate" +set +e +"$GATE" "${gate_args[@]}" +candidate_status=$? +set -e + +log "applying A2-vs-A1 baseline drift gate" +set +e +"$GATE" "${drift_gate_args[@]}" +drift_status=$? +set -e + +cat >"$OUT_DIR/summary.md" < B1 candidate -> B2 candidate -> A2 baseline +- runner: $(uname -srm 2>/dev/null || echo unknown) +- warp: $("$WARP_BIN" --version 2>/dev/null | head -n1 || echo unknown) +- matrix: duration=$DURATION rounds=$ROUNDS cooldown=$COOLDOWN_SECS disks=$DISKS concurrency=$CONCURRENCY +- endpoint: $ADDRESS +- baseline binary: $BASELINE_BIN +- candidate binary: $CANDIDATE_BIN +- candidate gate: $OUT_DIR/candidate_gate.md (exit $candidate_status) +- baseline drift gate: $OUT_DIR/baseline_drift_gate.md (exit $drift_status) + +Interpretation: + +- Treat candidate gate failures as actionable only when the A2-vs-A1 drift gate is PASS or the affected workload's A2 drift is materially smaller than the B1/B2 candidate delta. +- If both candidate and baseline drift fail on the same workload, rerun with longer duration, more rounds, or a quieter runner before assigning causality. +EOF + +log "summary written to $OUT_DIR/summary.md" +if [[ "$candidate_status" -ne 0 || "$drift_status" -ne 0 ]]; then + exit 1 +fi