Compare commits

..

23 Commits

Author SHA1 Message Date
overtrue 8aa256d4df refactor(e2e/kms): replace fixed startup sleeps with KMS readiness probe
Replace 33 hard-coded sleep(3s) / sleep(2s) startup waits in KMS e2e tests
with an active readiness probe (wait_for_kms_ready) that polls the KMS
status endpoint with exponential backoff (200ms→1s, 5s budget).

This cuts per-test startup latency from a fixed 3s to ~200-500ms while
remaining robust against slow CI machines.

Non-startup sleeps (ILM polling loops, fault-recovery detection delays,
test-runner inter-test pauses) are left untouched.
2026-08-22 01:14:47 +08:00
Zhengchao An bc07cfd115 ci: harden test selection and nightly coverage (#6341) 2026-08-21 15:04:08 +00:00
Zhengchao An bce5922aef feat(connect): answer offline enrolment challenges without a network (#6335) 2026-08-21 13:16:18 +00:00
cxymds adb90fc6e1 fix(scanner): defer usage publication during pool recovery (#6333)
* fix(scanner): defer usage publication during pool recovery

* fix(scanner): preserve metrics when publication is deferred

* fix(scanner): route test types through storage boundary

* fix(scanner): keep cache floor deferred during movement
2026-08-21 17:32:59 +08:00
cxymds cdfac5d7e3 fix(ecstore): avoid decommission walk deadline on backpressure (#6332)
fix(ecstore): bound decommission background walks
2026-08-21 17:32:12 +08:00
houseme ca4adea0c9 perf(server): trim internode REST compat stack (#6330)
Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-21 08:52:03 +00:00
GatewayJ 23a0f6324c fix(iam): preserve MinIO permanent credentials in migration (#6328)
* fix(iam): preserve MinIO permanent credentials in migration

* test(iam): cover MinIO credential migration end to end
2026-08-21 15:34:25 +08:00
Zhengchao An cdd9ab1124 fix(ci): update package checksums safely (#6329) 2026-08-21 15:28:42 +08:00
houseme 122a69df65 feat(ecstore): tune fdatasync group wait budget (#6327)
* feat(ecstore): tune fdatasync group wait budget

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(ecstore): cover fdatasync wait budget contract

Co-Authored-By: heihutu <heihutu@gmail.com>

---------

Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-21 14:08:21 +08:00
cxymds dfeb732ac8 fix: make DeleteObjects idempotent for raw not-found errors (#6323)
* fix: make DeleteObjects idempotent for raw not-found errors

* fix: cover DeleteObjects raw not-found result dispatch
2026-08-21 03:12:37 +00:00
Henry Guo 1aae680373 fix(lifecycle): honor bucket default retention (#6324)
* fix(lifecycle): honor bucket retention during scanner expiry

* fix(lifecycle): reject zero object lock retention

* test(lifecycle): cover malformed retention metadata

---------

Co-authored-by: Henry Guo <marshawcoco@users.noreply.github.com>
2026-08-21 10:11:13 +08:00
cxymds 1b4f62d501 docs(agents): run adversarial review before pre-pr (#6325)
docs(agents): order adversarial review before pre-pr
2026-08-21 10:10:56 +08:00
houseme 4283591838 feat(ecstore): observe PUT commit lock admission (#6319)
* feat(ecstore): observe PUT commit lock admission

Co-Authored-By: heihutu <heihutu@gmail.com>

* update h2 v0.4.18

* test(e2e): box SSE-KMS negative errors

Co-Authored-By: heihutu <heihutu@gmail.com>

---------

Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-21 01:18:53 +00:00
Zhengchao An f2957a680d test: stabilize release-blocking full E2E checks (#6322) 2026-08-21 08:43:12 +08:00
Zhengchao An d22cb5d07a fix: resolve release-blocking integration failures (#6320)
* fix: resolve release-blocking integration failures

* fix: satisfy stable clippy lints

* fix: satisfy Rust 1.98 CI lints
2026-08-21 06:39:29 +08:00
houseme 762919b1ba perf(scanner): reduce per-object allocation churn (#6318)
Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-20 16:47:48 +00:00
houseme cee0d5cf9b perf(io-metrics): cache read version metric handles (#6317)
Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-20 16:39:26 +00:00
houseme 35af688cd9 test(obs): add metric dimension smoke harness (#6316)
test(obs): add metrics dimension smoke harness

Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-20 16:26:19 +00:00
GatewayJ 205337151a fix(webdav): allow bucket-scoped root listings (#6298)
* fix(webdav): allow bucket-scoped root listings

* test(webdav): use public protocol export

* test(webdav): initialize identity inline

---------

Co-authored-by: cxymds <cxymds@gmail.com>
2026-08-21 00:15:33 +08:00
Henry Guo 105b6fbfde fix(scanner): honor explicit cycle cadence (#6313)
Co-authored-by: Henry Guo <marshawcoco@users.noreply.github.com>
Co-authored-by: houseme <housemecn@gmail.com>
2026-08-20 23:35:03 +08:00
houseme b2e573c48b feat(ecstore): bound put commit lock admission (#6315)
Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-20 23:33:17 +08:00
houseme 114bf5148c refactor(heal): prune statistics label helpers (#6312)
Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-20 22:43:59 +08:00
houseme 830e553a3c feat(obs): complete metric dimension coverage (#6314)
Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-20 22:43:40 +08:00
129 changed files with 8021 additions and 824 deletions
+2
View File
@@ -0,0 +1,2 @@
sha256-darwin=b4ae71aa894e5c7795ae3eb8116f1777a7601d0f5db3898be2e48faf3329bd9b
sha256-linux=433debd9d9defa832986269abdf0f1d131597b2d7a417ce930e17c1fd47d85ba
+1
View File
@@ -0,0 +1 @@
sha256=9b9bc336b43b70d0e06e0adb5455bf035bb18945d85d60936eb6fe4d48e0e680
+2
View File
@@ -0,0 +1,2 @@
sha256-darwin=55534a97fbd376f64c8f6c341d319017d11ff77cad6da8629a1a7f6a874e0315
sha256-linux=c06fb8c19aed6f388b9dc61cb8251b7a44f8561a9bf764ad2b9e635598f8dc17
+1
View File
@@ -0,0 +1 @@
sha256=655a3f3c1d042e694339d15caba7580518320322d1bac0f09450b37e6c09e2e7
+1
View File
@@ -0,0 +1 @@
sha256=ec27cde6ce6400723c4b372bfbd2ac61709c744294e4810af765e8a808d8e31d
+5
View File
@@ -75,6 +75,11 @@ embedded-secrets-check: ## Check no private key material or credential literal i
@echo "🔑 Checking embedded secret material guard..."
./scripts/check_embedded_secrets.sh
.PHONY: test-wiring-check
test-wiring-check: ## Check tests stay registered and selected by their intended runners
@echo "🧪 Checking test wiring..."
python3 ./scripts/check_test_wiring.py
.PHONY: log-analyzer-rules-check
log-analyzer-rules-check: core-deps ## Check log-analyzer rule anchors still exist verbatim in source
@echo "🩺 Checking log-analyzer rule anchors..."
+3 -3
View File
@@ -19,13 +19,13 @@ planning-docs-check: ## Check that no planning-type documents are committed
./scripts/check_no_planning_docs.sh
.PHONY: pre-commit
pre-commit: fmt-check unsafe-code-check architecture-migration-check logging-guardrails-check tokio-io-uring-check extension-schema-check body-cache-whitelist-check s3s-footprint-check fips-wording-check embedded-secrets-check doc-paths-check planning-docs-check quick-check ## Run fast pre-commit checks without clippy/full tests
pre-commit: fmt-check unsafe-code-check architecture-migration-check logging-guardrails-check tokio-io-uring-check extension-schema-check body-cache-whitelist-check s3s-footprint-check fips-wording-check embedded-secrets-check test-wiring-check doc-paths-check planning-docs-check quick-check ## Run fast pre-commit checks without clippy/full tests
@echo "✅ All pre-commit checks passed!"
.PHONY: pre-pr
pre-pr: fmt-check unsafe-code-check architecture-migration-check logging-guardrails-check tokio-io-uring-check extension-schema-check body-cache-whitelist-check s3s-footprint-check fips-wording-check embedded-secrets-check doc-paths-check planning-docs-check log-analyzer-rules-check clippy-check test ## Run full pre-PR checks with clippy and tests
pre-pr: fmt-check unsafe-code-check architecture-migration-check logging-guardrails-check tokio-io-uring-check extension-schema-check body-cache-whitelist-check s3s-footprint-check fips-wording-check embedded-secrets-check test-wiring-check doc-paths-check planning-docs-check log-analyzer-rules-check clippy-check test ## Run full pre-PR checks with clippy and tests
@echo "✅ All pre-PR checks passed!"
.PHONY: dev-check
dev-check: fmt-check unsafe-code-check architecture-migration-check logging-guardrails-check tokio-io-uring-check extension-schema-check body-cache-whitelist-check s3s-footprint-check fips-wording-check embedded-secrets-check doc-paths-check planning-docs-check quick-check ## Run fast local development checks
dev-check: fmt-check unsafe-code-check architecture-migration-check logging-guardrails-check tokio-io-uring-check extension-schema-check body-cache-whitelist-check s3s-footprint-check fips-wording-check embedded-secrets-check test-wiring-check doc-paths-check planning-docs-check quick-check ## Run fast local development checks
@echo "✅ Fast development checks passed!"
+2
View File
@@ -35,6 +35,8 @@ script-tests: ## Run shell script tests
./scripts/test_pinned_paired_abba_bench.sh
./scripts/test_manual_transition_runbooks.sh
./scripts/check_embedded_secrets.sh --self-test
python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/s3-tests/test_report_compat.py
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
python3 ./scripts/check_object_data_cache_follower_samples.py --self-test
./scripts/validate_object_data_cache_cold_stampede.sh --self-test
+48 -14
View File
@@ -38,10 +38,11 @@ e2e-vault = { max-threads = 1 }
# replacement_privileged_e2e_test when explicitly run as root on Linux). They
# are correct in isolation but resource-heavy; serialize them under nextest's
# process boundary (serial_test's #[serial] does not cross it) so several 4-disk
# servers never run at once. ci-7's nightly picks these up via the e2e suite;
# servers never run at once. The e2e-full merge/main lane picks these up;
# they are deliberately NOT in the fast PR `e2e-smoke` filter.
e2e-reliability = { max-threads = 1 }
e2e-inline-boundaries = { max-threads = 1 }
e2e-cluster-nightly = { max-threads = 1 }
# --- default profile (local): serialize the flaky groups, never retry --------
[[profile.default.overrides]]
@@ -161,7 +162,7 @@ retries = 2
# Serialize the 4-disk reliability / degraded-read e2e tests under the ci
# profile too (see the e2e-reliability test-group note near the top). Not a
# quarantine: no retries, just single-threaded so several 4-disk servers never
# run concurrently when ci-7's nightly runs the full e2e suite.
# run concurrently when e2e-full runs the suite.
[[profile.ci.overrides]]
filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression|replacement_privileged_e2e)_test::/)'
test-group = 'e2e-reliability'
@@ -230,8 +231,8 @@ test-group = 'ecstore-serial-flaky'
# the nightly profile derives its set as "the replication module MINUS this
# allowlist", so any new replication test lands in nightly by default (never
# silently unrun) until it is explicitly blessed as fast here. Keep the two
# regexes byte-identical. Count invariant: 20 here + 49 nightly = 69 total
# (authority: `cargo nextest list`; docs/testing/e2e-suite-inventory.md).
# regexes byte-identical. The committed profile selection digests make changes
# visible in CI; current counts live in docs/testing/e2e-suite-inventory.md.
# HISTORY (2026-07-11): the 20 fast tests were briefly pulled out of this lane
# (#4724) because they set a loopback (127.0.0.1) replication target that the
# SSRF egress guard rejected on every PR after repl-1 (#4712). That is fixed —
@@ -327,9 +328,8 @@ slow-timeout = { period = "60s", terminate-after = 2, grace-period = "10s" }
# the STS dual-node test actually exercises its path (it skips gracefully with
# a visible log line when awscurl is absent), and routes scheduled failures
# through .github/actions/schedule-failure-issue (ci-8). Explicit division of
# labor with ci-5's future e2e-full merge gate: these tests run ONLY here, not
# double-run there. TODO(ci-7): fold this interim repl-owned lane into the ci
# domain's consolidated scheduled e2e workflow once it exists.
# labor with e2e-full: these tests run only in the consolidated nightly
# workflow, not in the merge/main lane.
[profile.e2e-repl-nightly]
default-filter = """
package(e2e_test)
@@ -343,26 +343,60 @@ fail-fast = false
# workflow as the failure-triage artifact.
path = "junit.xml"
# ---------------------------------------------------------------------------
# e2e-nightly profile — destructive multi-process cluster fault domains
# ---------------------------------------------------------------------------
# These seven modules are deliberately outside e2e-full's merge budget. Each
# starts a real multi-process or multi-disk topology and exercises node/disk
# loss, quorum, cleanup, notification fan-in, or admin-timeout behavior. The
# consolidated nightly workflow runs them serially to avoid resource
# starvation; failures are never retried.
[profile.e2e-nightly]
default-filter = """
package(e2e_test)
& test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
"""
fail-fast = false
[profile.e2e-nightly.junit]
path = "junit.xml"
[[profile.e2e-nightly.overrides]]
filter = 'package(e2e_test)'
test-group = 'e2e-cluster-nightly'
# ---------------------------------------------------------------------------
# e2e-protocols profile — serial protocol lane
# ---------------------------------------------------------------------------
# The suite owns fixed ports, so the nightly workflow runs this exact profile
# with one nextest worker.
[profile.e2e-protocols]
default-filter = 'package(e2e_test) & test(/^protocols::/)'
fail-fast = false
[profile.e2e-protocols.junit]
path = "junit.xml"
# ---------------------------------------------------------------------------
# e2e-full profile — merge-gate full single-node e2e lane (backlog#1149 ci-5)
# ---------------------------------------------------------------------------
# The merge gate (ci.yml `e2e-full` job: push main + merge_group +
# workflow_dispatch). Runs the never-automated user-visible suites — KMS (40),
# object_lock (33), multipart_auth (109), quota, checksum, encryption,
# workflow_dispatch). Runs the user-visible KMS, object-lock, multipart-auth,
# quota, checksum, encryption,
# security-boundary, ... — that the fast PR `e2e-smoke` subset deliberately
# skips. Budget <= 45 min; authority for the suite count is `cargo nextest list
# --profile e2e-full` (see docs/testing/e2e-suite-inventory.md).
#
# The filter is "the whole e2e_test crate MINUS the sets owned by other lanes":
# * protocols:: — FTPS/SFTP/WebDAV, still pinned to --test-threads=1 by fixed
# ports; they join a scheduled lane once ci-6 randomises the ports (ci-7).
# * protocols:: — FTPS/SFTP/WebDAV, run from the dedicated protocol profile
# with one worker because the suite owns fixed ports.
# * the 7 cluster suites that spin up a RustFSTestClusterEnvironment
# (cluster_concurrency, cluster_multidrive_pool, stale_multipart_cleanup_cluster,
# namespace_lock_quorum, heal_erasure_disk_rebuild, admin_timeout_regression,
# object_lambda) — too heavy for the merge budget; they run in ci-7's
# nightly 4-node lane.
# object_lambda) — too heavy for the merge budget; they run in the
# e2e-nightly serial cluster-fault lane.
# * replication_extension_test — repl-1 already splits it into the PR
# `e2e-smoke` (20 fast) and `e2e-repl-nightly` (49 slow) lanes and reserves
# `e2e-smoke` (20 fast) and `e2e-repl-nightly` (55 slow) lanes and reserves
# it for those, so e2e-full does not double-run it.
# * #[ignore]d tests — nextest skips them by default (no --run-ignored); the
# manual-localhost:9000 reliant/policy tests are ci-13's migration.
+3 -4
View File
@@ -46,10 +46,9 @@ lists when upstream changes.
the PR.
- **Weekly + manual**: `.github/workflows/e2e-s3tests.yml` runs the full
upstream suite (`TEST_SCOPE=all`) against a Docker deployment (single node
or a 4-node distributed cluster behind HAProxy). It fails only on
regressions in the implemented whitelist and publishes a classification
report (`compat-report.md`, also shown in the job summary) listing promotion
candidates and unclassified tests.
or a 4-node distributed cluster behind HAProxy). The canonical gate policy
and compatibility-report behavior are documented in
[`scripts/s3-tests/README.md`](../../scripts/s3-tests/README.md).
## Running Tests Locally
+3
View File
@@ -125,6 +125,9 @@ jobs:
- name: Check no embedded secret material
run: ./scripts/check_embedded_secrets.sh
- name: Check test wiring
run: python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed
run: ./scripts/check_no_planning_docs.sh
+18 -5
View File
@@ -160,6 +160,9 @@ jobs:
- name: Check no embedded secret material
run: ./scripts/check_embedded_secrets.sh
- name: Check test wiring
run: python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed
run: ./scripts/check_no_planning_docs.sh
@@ -686,9 +689,9 @@ jobs:
- name: Make binary executable
run: chmod +x ./target/debug/rustfs
# Build the e2e test graph once. The archive is reused by the security
# count-floor check and the smoke run below, avoiding a second compile of
# the same e2e_test target on cold runners (backlog#1645).
# Build the e2e test graph once. The archive is reused by the smoke
# selection guard, security exact-count check, and run below, avoiding a
# second compile of the same e2e_test target on cold runners (backlog#1645).
- name: Archive e2e smoke test binaries
env:
NEXTEST_ARCHIVE: ${{ runner.temp }}/rustfs-e2e-smoke.tar.zst
@@ -696,6 +699,7 @@ jobs:
run: |
cargo nextest archive --profile e2e-smoke -p e2e_test --archive-file "${NEXTEST_ARCHIVE}"
cargo nextest list --profile e2e-smoke --archive-file "${NEXTEST_ARCHIVE}" --message-format json > "${NEXTEST_LISTING}"
python3 ./scripts/check_test_wiring.py --check-profile e2e-smoke "${NEXTEST_LISTING}"
./scripts/check_security_smoke_count.sh check "${NEXTEST_LISTING}"
# PR smoke subset of the in-repo e2e suite (backlog#1149 ci-4). The
@@ -760,7 +764,7 @@ jobs:
# suites — KMS, object_lock, multipart_auth, quota, checksum, encryption,
# security-boundary, ... — via the e2e-full nextest profile. Too heavy for
# every PR, so it is gated to main pushes, the merge queue, and manual
# dispatch. protocols / the 6 cluster suites / replication / #[ignore] are
# dispatch. protocols / the 7 cluster suites / replication / #[ignore] are
# owned by other lanes (see .config/nextest.toml profile.e2e-full).
if: >-
github.event_name == 'workflow_dispatch' ||
@@ -820,6 +824,13 @@ jobs:
- name: Make binary executable
run: chmod +x ./target/debug/rustfs
- name: Verify e2e full membership
env:
NEXTEST_LISTING: ${{ runner.temp }}/rustfs-e2e-full-list.json
run: |
cargo nextest list --profile e2e-full -p e2e_test --message-format json > "${NEXTEST_LISTING}"
python3 ./scripts/check_test_wiring.py --check-profile e2e-full "${NEXTEST_LISTING}"
# Full single-node e2e lane (backlog#1149 ci-5). The e2e-full
# default-filter in .config/nextest.toml is the single wiring mechanism —
# extend that filter, never add ad-hoc e2e jobs here. Reuses the downloaded
@@ -832,7 +843,9 @@ jobs:
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: e2e-full-junit-${{ github.run_number }}
path: target/nextest/e2e-full/junit.xml
path: |
target/nextest/e2e-full/junit.xml
${{ runner.temp }}/rustfs-e2e-full-list.json
retention-days: 7
e2e-tests-rio-v2:
+122 -11
View File
@@ -12,7 +12,7 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# Nightly full replication e2e lane (backlog#1147 repl-1, deps: ci-4).
# Consolidated nightly e2e lane for replication, cluster faults, and protocols.
#
# The per-PR gate (ci.yml `e2e-tests` job, `--profile e2e-smoke`) runs the
# FAST replication tests. This scheduled lane runs the remaining heavier
@@ -28,15 +28,12 @@
# add ad-hoc cargo-test steps here; change the filterset instead. The
# authoritative membership and count come from
# `cargo nextest list -p e2e_test --profile e2e-repl-nightly`; the PR/nightly
# count invariant is maintained next to the filtersets in .config/nextest.toml
# (deliberately not duplicated here).
# selection digest is committed under .config/.
#
# Explicit division of labor: the nightly subset runs ONLY here, never double-run
# in ci-5's future e2e-full merge gate. TODO(ci-7): once the ci domain's
# consolidated scheduled e2e workflow exists, fold this interim repl-owned lane
# into it rather than growing a second scheduled entrypoint.
# Explicit division of labor: these subsets run only here and never double-run
# in the e2e-full merge gate.
name: e2e-replication-nightly
name: e2e-nightly
on:
workflow_dispatch:
@@ -50,6 +47,10 @@ on:
permissions:
contents: read
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: false
jobs:
repl-nightly:
name: Replication e2e (nightly)
@@ -97,9 +98,20 @@ jobs:
# demand otherwise, but a single explicit build avoids several parallel
# nextest test processes racing to build it at once.
- name: Build rustfs binary
run: cargo build -p rustfs --bins
run: |
cargo build -p rustfs --bins
: > target/debug/rustfs.features
- name: Verify replication e2e membership
env:
NEXTEST_LISTING: ${{ runner.temp }}/rustfs-e2e-repl-nightly-list.json
run: |
cargo nextest list --profile e2e-repl-nightly -p e2e_test --message-format json > "${NEXTEST_LISTING}"
python3 ./scripts/check_test_wiring.py --check-profile e2e-repl-nightly "${NEXTEST_LISTING}"
- name: Run replication e2e nightly suite
env:
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-repl-nightly-logs
run: cargo nextest run --profile e2e-repl-nightly -p e2e_test
- name: Upload nextest junit report
@@ -107,13 +119,112 @@ jobs:
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: e2e-replication-nightly-junit-${{ github.run_number }}
path: target/nextest/e2e-repl-nightly/junit.xml
path: |
target/nextest/e2e-repl-nightly/junit.xml
${{ runner.temp }}/rustfs-e2e-repl-nightly-list.json
${{ runner.temp }}/rustfs-e2e-repl-nightly-logs/
retention-days: 7
if-no-files-found: ignore
cluster-nightly:
name: Cluster fault e2e (nightly)
runs-on: sm-standard-4
timeout-minutes: 90
env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
steps:
- name: Checkout repository
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Setup Rust environment
uses: ./.github/actions/setup
with:
rust-version: stable
cache-shared-key: ci-e2e-nightly
cache-save-if: 'false'
install-build-packaging-tools: 'false'
- name: Build rustfs binary
run: |
cargo build -p rustfs --bins --features e2e-test-hooks
: > target/debug/rustfs.features
- name: Verify cluster fault e2e membership
env:
NEXTEST_LISTING: ${{ runner.temp }}/rustfs-e2e-nightly-list.json
run: |
cargo nextest list --profile e2e-nightly -p e2e_test --message-format json > "${NEXTEST_LISTING}"
python3 ./scripts/check_test_wiring.py --check-profile e2e-nightly "${NEXTEST_LISTING}"
- name: Run cluster fault e2e nightly suite
env:
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-nightly-logs
run: cargo nextest run --profile e2e-nightly -p e2e_test
- name: Upload cluster fault diagnostics
if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: e2e-cluster-nightly-${{ github.run_number }}
path: |
target/nextest/e2e-nightly/junit.xml
${{ runner.temp }}/rustfs-e2e-nightly-list.json
${{ runner.temp }}/rustfs-e2e-nightly-logs/
retention-days: 7
if-no-files-found: warn
protocols-nightly:
name: Protocol e2e (nightly)
runs-on: sm-standard-4
timeout-minutes: 90
env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
RUSTFS_BUILD_FEATURES: ftps,webdav,sftp
steps:
- name: Checkout repository
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Setup Rust environment
uses: ./.github/actions/setup
with:
rust-version: stable
cache-shared-key: ci-e2e-protocols
cache-save-if: 'false'
install-build-packaging-tools: 'false'
# The suite owns fixed protocol ports and serializes its internal cases.
- name: Verify protocol e2e membership
env:
NEXTEST_LISTING: ${{ runner.temp }}/rustfs-e2e-protocols-list.json
run: |
cargo nextest list --profile e2e-protocols -p e2e_test --message-format json > "${NEXTEST_LISTING}"
python3 ./scripts/check_test_wiring.py --check-profile e2e-protocols "${NEXTEST_LISTING}"
- name: Run protocol e2e nightly suite
env:
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-protocol-e2e-logs
run: >-
cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
- name: Upload protocol diagnostics
if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: e2e-protocol-nightly-${{ github.run_number }}
path: |
target/nextest/e2e-protocols/junit.xml
${{ runner.temp }}/rustfs-e2e-protocols-list.json
${{ runner.temp }}/rustfs-protocol-e2e-logs/
retention-days: 7
if-no-files-found: warn
alert-on-failure:
name: Alert on scheduled failure
needs: [repl-nightly]
needs: [repl-nightly, cluster-nightly, protocols-nightly]
# Only scheduled runs open/append the tracking issue (backlog#1149 ci-8);
# manual workflow_dispatch runs stay quiet so a debugging run never files a
# spurious alert.
+28 -35
View File
@@ -18,10 +18,9 @@
# runs only the implemented_tests.txt whitelist. This workflow complements it:
#
# - Scheduled weekly full sweep (TEST_SCOPE=all): runs the ENTIRE upstream
# suite and reports promotion candidates (tests that newly pass) and
# unclassified tests. The job fails only on regressions in the implemented
# whitelist or on infrastructure errors — expected failures from
# not-yet-implemented features do not turn the run red.
# suite and reports promotion candidates. Regressions, unclassified tests,
# incomplete execution, and infrastructure errors fail the job; classified
# failures for not-yet-implemented features remain informational.
# - Manual runs (workflow_dispatch): same, with configurable mode/scope.
#
# All test execution is delegated to scripts/s3-tests/run.sh (single source of
@@ -45,13 +44,6 @@
# The PR gate (ci.yml s3-implemented-tests) is unaffected: it avoids Docker
# via DEPLOY_MODE=binary and defers all pip setup to run.sh's self-bootstrap.
# DISABLED. This workflow is switched off in the repository's Actions settings
# (state: disabled_manually) and does not run on any trigger, including its cron
# and workflow_dispatch. That state lives in GitHub's UI and is invisible when
# reading this file, which has already misled at least one audit — hence this
# banner. Re-enabling is a UI action; anyone doing so should first check that the
# workflow still matches the current CI layout. See rustfs/backlog#1603.
#
name: e2e-s3tests
on:
@@ -81,6 +73,19 @@ on:
description: "Stop after N failures. '0' to run everything."
required: false
default: "0"
shard-count:
description: "Exact-node-ID shard count for a targeted manual run"
required: false
default: "1"
type: choice
options:
- "1"
- "2"
- "4"
shard-index:
description: "Zero-based shard index for a targeted manual run"
required: false
default: "0"
markexpr:
description: "Optional pytest -m expression"
required: false
@@ -111,6 +116,8 @@ env:
XDIST: ${{ github.event.inputs.xdist || '4' }}
MAXFAIL: ${{ github.event.inputs.maxfail || '0' }}
MARKEXPR: ${{ github.event.inputs.markexpr || '' }}
S3_SHARD_COUNT: ${{ github.event_name == 'schedule' && '4' || github.event.inputs.shard-count || '1' }}
TEST_TIMEOUT: "300"
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}-${{ github.event.inputs['test-mode'] || 'single' }}
@@ -127,19 +134,22 @@ defaults:
jobs:
s3tests:
name: s3tests (${{ matrix.test-mode }}, shard ${{ matrix.shard-index }})
# GitHub-hosted: reliably provides Docker + docker compose + python3/pip.
# See the header note (ci-1) for why the self-hosted sm-standard-4 label
# was abandoned. TODO(ci-8): scheduled-failure alerting (auto-open issue)
# is added by the ci-8 composite action; do not implement it here.
# was abandoned. Scheduled failures are handled by alert-on-failure below.
runs-on: ubuntu-latest
timeout-minutes: 180
strategy:
fail-fast: false
max-parallel: 2
matrix:
# Scheduled sweeps cover both topologies; manual runs use the input.
test-mode: ${{ github.event_name == 'schedule' && fromJSON('["single", "multi"]') || fromJSON(format('["{0}"]', github.event.inputs.test-mode || 'single')) }}
shard-index: ${{ github.event_name == 'schedule' && fromJSON('[0, 1, 2, 3]') || fromJSON(format('[{0}]', github.event.inputs.shard-index || '0')) }}
env:
TEST_MODE: ${{ matrix.test-mode }}
S3_SHARD_INDEX: ${{ matrix.shard-index }}
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
@@ -181,6 +191,7 @@ jobs:
- name: Start single RustFS
if: env.TEST_MODE == 'single'
run: |
SSE_KEY="$(head -c 32 /dev/zero | base64 -w0)"
docker network inspect rustfs-net >/dev/null 2>&1 || docker network create rustfs-net
docker rm -f rustfs-single >/dev/null 2>&1 || true
# The four disks share one physical device on the runner (a single
@@ -193,6 +204,7 @@ jobs:
-e RUSTFS_ADDRESS=0.0.0.0:9000 \
-e RUSTFS_ACCESS_KEY="${S3_ACCESS_KEY}" \
-e RUSTFS_SECRET_KEY="${S3_SECRET_KEY}" \
-e RUSTFS_SSE_S3_MASTER_KEY="${SSE_KEY}" \
-e RUSTFS_VOLUMES="/data/rustfs{0...3}" \
-e RUSTFS_UNSAFE_BYPASS_DISK_CHECK=true \
-v /tmp/rustfs-single:/data \
@@ -201,6 +213,7 @@ jobs:
- name: Start 4-node distributed cluster
if: env.TEST_MODE == 'multi'
run: |
SSE_KEY="$(head -c 32 /dev/zero | base64 -w0)"
# A real distributed deployment: every node lists all endpoints in
# RUSTFS_VOLUMES so data is erasure-coded ACROSS nodes. Do not use
# node-local volume paths here — that would create four independent
@@ -213,6 +226,7 @@ jobs:
RUSTFS_ADDRESS: "0.0.0.0:9000"
RUSTFS_ACCESS_KEY: ${S3_ACCESS_KEY}
RUSTFS_SECRET_KEY: ${S3_SECRET_KEY}
RUSTFS_SSE_S3_MASTER_KEY: "${SSE_KEY}"
RUSTFS_VOLUMES: "http://rustfs{1...4}:9000/data/rustfs{0...3}"
# Each node's four disks share one physical device inside its
# container, so bypass the local physical-disk-independence guard
@@ -294,7 +308,6 @@ jobs:
- name: Run ceph s3-tests
run: |
set +e
DEPLOY_MODE=existing \
TEST_MODE="${TEST_MODE}" \
TEST_SCOPE="${TEST_SCOPE}" \
@@ -302,26 +315,6 @@ jobs:
MAXFAIL="${MAXFAIL}" \
MARKEXPR="${MARKEXPR}" \
./scripts/s3-tests/run.sh
RC=$?
set -e
if [ "${TEST_SCOPE}" = "implemented" ]; then
# Whitelist run: every failure is a regression.
exit "${RC}"
fi
# Full sweep: failures outside the implemented whitelist are
# inventory (promotion candidates / unimplemented features), not a
# gate. Fail only on whitelist regressions or infrastructure errors.
JUNIT="artifacts/s3tests-${TEST_MODE}/junit.xml"
if [ ! -f "${JUNIT}" ]; then
echo "No junit.xml produced — infrastructure failure (exit ${RC})" >&2
exit "${RC}"
fi
python3 scripts/s3-tests/report_compat.py \
--junit "${JUNIT}" \
--lists-dir scripts/s3-tests \
--fail-on-regression
- name: Publish compatibility report
if: always()
@@ -346,7 +339,7 @@ jobs:
if: always() && env.ACT != 'true'
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: s3tests-${{ env.TEST_MODE }}
name: s3tests-${{ env.TEST_MODE }}-shard-${{ matrix.shard-index }}
path: artifacts/**
alert-on-failure:
+11 -24
View File
@@ -12,27 +12,22 @@
# See the License for the specific language governing permissions and
# limitations under the License.
# DISABLED. This workflow is switched off in the repository's Actions settings
# (state: disabled_manually) and does not run on any trigger, including its cron
# and workflow_dispatch. That state lives in GitHub's UI and is invisible when
# reading this file, which has already misled at least one audit — hence this
# banner. Re-enabling is a UI action; anyone doing so should first check that the
# workflow still matches the current CI layout. See rustfs/backlog#1603.
#
name: Fuzz
on:
pull_request:
types: [ opened, synchronize, reopened, closed ]
# PR trigger is intentionally narrow: only changes to the fuzz harness
# itself gate a PR. Broad crate paths (ecstore/filemeta/utils/policy/…)
# are covered by the nightly `schedule` run below, which fuzzes against
# whatever landed on main. Widening these paths previously queued a
# ~45min fuzz-build on nearly every PR and is why this workflow was
# disabled; do not re-add crate paths here.
# Run when the harness or any directly fuzzed production crate changes.
paths:
- "fuzz/**"
- "scripts/fuzz/**"
- "crates/ecstore/**"
- "crates/filemeta/**"
- "crates/policy/**"
- "crates/security-governance/**"
- "crates/utils/**"
- "Cargo.toml"
- "Cargo.lock"
- ".github/workflows/fuzz.yml"
schedule:
- cron: "0 2 * * *"
@@ -81,7 +76,7 @@ jobs:
github.event_name == 'schedule' ||
github.event_name == 'workflow_dispatch'
runs-on: sm-standard-4
timeout-minutes: 45
timeout-minutes: 60
env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
steps:
@@ -121,12 +116,7 @@ jobs:
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: fuzz-prebuilt-binaries-${{ github.run_number }}
path: |
fuzz/prebuilt/${{ env.CARGO_BUILD_TARGET }}/release/archive_extract
fuzz/prebuilt/${{ env.CARGO_BUILD_TARGET }}/release/bucket_validation
fuzz/prebuilt/${{ env.CARGO_BUILD_TARGET }}/release/local_metadata
fuzz/prebuilt/${{ env.CARGO_BUILD_TARGET }}/release/path_containment
fuzz/prebuilt/${{ env.CARGO_BUILD_TARGET }}/release/policy_ingress
path: fuzz/prebuilt/${{ env.CARGO_BUILD_TARGET }}/release/
if-no-files-found: error
retention-days: 1
compression-level: 0
@@ -192,10 +182,7 @@ jobs:
nightly-fuzz-corpus:
name: "Nightly / ${{ matrix.target }}"
needs: fuzz-build
# TODO(ci-8): when the schedule-failure-issue composite action lands,
# add a step here (or a dependent job) that opens/updates a GitHub issue
# on nightly failure. ci-8 is the single alerting mechanism for all
# scheduled workflows; do not self-roll alerting in this workflow.
# Scheduled failures are handled by alert-on-failure below.
if: >
github.event_name == 'schedule' ||
(github.event_name == 'workflow_dispatch' &&
+4 -4
View File
@@ -189,6 +189,7 @@ jobs:
timeout-minutes: 30
strategy:
fail-fast: false
max-parallel: 1
matrix:
include:
- arch: x86_64
@@ -510,15 +511,13 @@ jobs:
CHECKSUM_DIR="$(mktemp -d)"
gh release download "$TAG" -p 'SHA256SUMS' -p 'SHA512SUMS' \
-D "$CHECKSUM_DIR" --clobber 2>/dev/null || true
-D "$CHECKSUM_DIR" --clobber
for spec in "SHA256SUMS:sha256sum" "SHA512SUMS:sha512sum"; do
asset="${spec%%:*}"
checksum_cmd="${spec##*:}"
checksum_file="${CHECKSUM_DIR}/${asset}"
touch "$checksum_file"
for f in "$DEB_FILE" "$RPM_FILE"; do
if [[ -n "$f" && -f "$f" ]]; then
base="$(basename "$f")"
@@ -531,7 +530,8 @@ jobs:
grep -Fv -- "$base" "$checksum_file" > "${checksum_file}.tmp" || true
grep -Fv -- "$github_base" "${checksum_file}.tmp" > "${checksum_file}.tmp2" || true
mv "${checksum_file}.tmp2" "$checksum_file"
(cd "$(dirname "$f")" && "$checksum_cmd" -- "$github_base") >> "$checksum_file"
digest=$("$checksum_cmd" -- "$f" | awk '{print $1}')
printf '%s %s\n' "$digest" "$github_base" >> "$checksum_file"
fi
done
+9 -6
View File
@@ -127,8 +127,9 @@ the broadest gate. Inspect only the final task-owned diff, classify it by
behavioral impact rather than line count or path alone, and run the smallest
set of checks that provides meaningful coverage. Do not let unrelated
worktree changes or a generic contributor checklist expand the scope.
Non-exempt changes must also pass Adversarial Validation (next section) before
the checks below count as completion.
For non-exempt changes, complete the applicable multi-role adversarial review
before running `make pre-pr` (or an equivalent full gate). Resolve or rebut
every finding first, then run the gate against the reviewed final diff.
### Validation floor
@@ -166,8 +167,9 @@ the checks below count as completion.
dependency set is identifiable, validate those packages and known
dependents instead of the whole workspace. Use `make pre-commit` only when
a repository-wide fast gate adds useful confidence beyond those checks.
4. **Broad or high-risk change:** Run `make pre-pr` only when targeted coverage
cannot bound the impact, including:
4. **Broad or high-risk change:** After the applicable adversarial review has
completed, run `make pre-pr` only when targeted coverage cannot bound the
impact, including:
- dependency, feature, build-script, procedural-macro, code-generation,
toolchain, or CI changes that alter compilation or the test matrix;
- cross-crate public APIs, shared foundational code, or broad refactors with
@@ -287,8 +289,9 @@ High risk: all seven roles.
- Every testable behavior change has a focused regression check. Exceptions
follow the validation floor and state why a check is impractical and what
risk remains.
- The Verification Before PR gates pass — adversarial review supplements
those gates, never replaces them.
- After the applicable adversarial review has completed, the Verification
Before PR gates pass; adversarial review supplements those gates, never
replaces them.
- High risk only: record a one-line verdict per role in the PR description.
## Git and PR Baseline
+6 -4
View File
@@ -91,8 +91,9 @@ A green `make pre-commit` is not enough to open a pull request.
`make pre-pr` is the **full** gate: it runs all of the guard checks above,
then `clippy-check` (`cargo clippy --all-targets --all-features -- -D warnings`)
and `test` (shell script tests, workspace tests excluding `e2e_test`, and doc
tests). Run `make pre-pr` before opening or updating a pull request — this is
what CI enforces.
tests). Complete the applicable multi-role adversarial review described in
`AGENTS.md` before running `make pre-pr`; then run the gate before opening or
updating a pull request. This is what CI enforces.
### 🔒 Git Pre-commit Hooks (optional)
@@ -150,8 +151,9 @@ Example output when formatting fails:
2. **Format your code**: `make fmt` or `cargo fmt --all`
3. **Run the fast gate**: `make pre-commit` (no clippy, no tests)
4. **Commit your changes**: `git commit -m "your message"`
5. **Run the full gate before opening/updating a PR**: `make pre-pr` (clippy + tests)
6. **Push to your branch**: `git push`
5. **Complete the applicable multi-role adversarial review** for non-exempt changes (see `AGENTS.md`)
6. **Run the full gate before opening/updating a PR**: `make pre-pr` (clippy + tests)
7. **Push to your branch**: `git push`
### 🛠️ IDE Integration
Generated
+2 -2
View File
@@ -4757,9 +4757,9 @@ dependencies = [
[[package]]
name = "h2"
version = "0.4.17"
version = "0.4.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9f877e75f39e9827ec50a572dd592684ac28c029578726c85f1b2aa6ab807449"
checksum = "839c0e8a181239723652be9062bb56ca5bf5f64011f73b623f6f4fc59086a228"
dependencies = [
"atomic-waker",
"bytes",
+110 -5
View File
@@ -729,7 +729,7 @@ fn timestamp_elapsed_seconds_since(now: Timestamp, earlier: Timestamp) -> u64 {
return 0;
}
u64::try_from(duration.as_secs()).map_or(u64::MAX, |seconds| seconds)
u64::try_from(duration.as_secs()).unwrap_or(u64::MAX)
}
#[derive(Clone, Copy, Debug, Default)]
@@ -781,6 +781,19 @@ struct ScannerBucketDriveResultValue {
last_seen: u64,
}
#[derive(Clone, Debug, Eq, Hash, PartialEq)]
struct ScannerActiveBucketDriveKey {
source: String,
bucket: String,
drive: String,
}
#[derive(Clone, Copy, Debug)]
struct ScannerActiveBucketDriveValue {
count: u64,
started_at: Timestamp,
}
// ---------------------------------------------------------------------------
// Metrics
// ---------------------------------------------------------------------------
@@ -813,6 +826,7 @@ pub struct Metrics {
scanner_set_scans_active: AtomicU64,
scanner_disk_bucket_scan_states: Mutex<HashMap<ScannerDiskBucketScanKey, ScannerDiskBucketScanState>>,
scanner_bucket_drive_results: Mutex<ScannerBucketDriveResults>,
scanner_active_bucket_drive_scans: Mutex<HashMap<ScannerActiveBucketDriveKey, ScannerActiveBucketDriveValue>>,
scanner_bucket_drive_result_clock: AtomicU64,
current_scan_cycle_bucket_drive_results_start: Mutex<HashMap<ScannerBucketDriveResultKey, u64>>,
last_scan_cycle_bucket_drive_results: Mutex<Vec<ScannerBucketDriveResultSnapshot>>,
@@ -1045,6 +1059,15 @@ pub struct ScannerBucketDriveResultSnapshot {
pub count: u64,
}
#[derive(Clone, Debug, Default, Serialize, Deserialize, PartialEq, Eq)]
pub struct ScannerActiveBucketDriveSnapshot {
pub source: String,
pub bucket: String,
pub drive: String,
pub count: u64,
pub age_seconds: u64,
}
#[derive(Clone, Debug, Default, Serialize, Deserialize, PartialEq, Eq)]
pub struct ScannerReplicationRepairSnapshot {
pub source: String,
@@ -1387,6 +1410,8 @@ pub struct ScannerRuntimeDetailsReport {
pub current_cycle_bucket_drive_results: Vec<ScannerBucketDriveResultSnapshot>,
#[serde(default)]
pub last_cycle_bucket_drive_results: Vec<ScannerBucketDriveResultSnapshot>,
#[serde(default)]
pub active_bucket_drive_scans: Vec<ScannerActiveBucketDriveSnapshot>,
}
impl CurrentCycle {
@@ -1746,7 +1771,7 @@ pub fn emit_scan_cycle_deferred(duration: Duration) {
metrics::counter!(OTEL_SCANNER_CYCLES, "result" => SCAN_CYCLE_RESULT_DEFERRED_LABEL).increment(1);
}
pub fn emit_scan_bucket_drive_complete(success: bool, bucket: &str, disk: &str, duration: Duration) {
pub fn emit_scan_bucket_drive_complete(_source: ScannerWorkSource, success: bool, bucket: &str, disk: &str, duration: Duration) {
let result = if success { "success" } else { "error" };
global_metrics().record_scanner_bucket_drive_result(bucket, disk, result);
metrics::counter!(
@@ -1764,7 +1789,7 @@ pub fn emit_scan_bucket_drive_complete(success: bool, bucket: &str, disk: &str,
.record(duration.as_secs_f64());
}
pub fn emit_scan_bucket_drive_partial(bucket: &str, disk: &str, duration: Duration) {
pub fn emit_scan_bucket_drive_partial(_source: ScannerWorkSource, bucket: &str, disk: &str, duration: Duration) {
global_metrics().record_scanner_bucket_drive_result(bucket, disk, SCAN_CYCLE_RESULT_PARTIAL_LABEL);
metrics::counter!(
OTEL_SCANNER_BUCKETS_SCANNED,
@@ -1817,6 +1842,7 @@ impl Metrics {
scanner_set_scans_active: AtomicU64::new(0),
scanner_disk_bucket_scan_states: Mutex::new(HashMap::new()),
scanner_bucket_drive_results: Mutex::new(ScannerBucketDriveResults::default()),
scanner_active_bucket_drive_scans: Mutex::new(HashMap::new()),
scanner_bucket_drive_result_clock: AtomicU64::new(0),
current_scan_cycle_bucket_drive_results_start: Mutex::new(HashMap::new()),
last_scan_cycle_bucket_drive_results: Mutex::new(Vec::new()),
@@ -2308,8 +2334,45 @@ impl Metrics {
}
}
pub fn record_scan_bucket_drive_start(&self) {
pub fn record_scan_bucket_drive_start(&self, source: ScannerWorkSource, bucket: &str, drive: &str) {
self.operations[Metric::ScanBucketDriveStart as usize].fetch_add(1, Ordering::Relaxed);
if bucket.is_empty() || drive.is_empty() {
return;
}
let key = ScannerActiveBucketDriveKey {
source: source.as_str().to_string(),
bucket: bucket.to_string(),
drive: drive.to_string(),
};
let mut active = self
.scanner_active_bucket_drive_scans
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
active
.entry(key)
.and_modify(|value| value.count = value.count.saturating_add(1))
.or_insert(ScannerActiveBucketDriveValue {
count: 1,
started_at: Timestamp::now(),
});
}
pub fn record_scan_bucket_drive_end(&self, source: ScannerWorkSource, bucket: &str, drive: &str) {
let key = ScannerActiveBucketDriveKey {
source: source.as_str().to_string(),
bucket: bucket.to_string(),
drive: drive.to_string(),
};
let mut active = self
.scanner_active_bucket_drive_scans
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
if let Some(value) = active.get_mut(&key) {
value.count = value.count.saturating_sub(1);
if value.count == 0 {
active.remove(&key);
}
}
}
pub fn record_scan_bucket_drive_failure(&self) {
@@ -2782,6 +2845,26 @@ impl Metrics {
} else {
Vec::new()
};
let now = Timestamp::now();
let mut active_bucket_drive_scans = self
.scanner_active_bucket_drive_scans
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.iter()
.map(|(key, value)| ScannerActiveBucketDriveSnapshot {
source: key.source.clone(),
bucket: key.bucket.clone(),
drive: key.drive.clone(),
count: value.count,
age_seconds: timestamp_elapsed_seconds_since(now, value.started_at),
})
.collect::<Vec<_>>();
active_bucket_drive_scans.sort_by(|left, right| {
left.source
.cmp(&right.source)
.then_with(|| left.bucket.cmp(&right.bucket))
.then_with(|| left.drive.cmp(&right.drive))
});
ScannerRuntimeDetailsReport {
disk_bucket_scan_states: self.scanner_disk_bucket_scan_state_snapshots(),
bucket_drive_results: self.scanner_bucket_drive_result_counter_snapshots(),
@@ -2791,6 +2874,7 @@ impl Metrics {
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner())
.clone(),
active_bucket_drive_scans,
}
}
@@ -4371,7 +4455,7 @@ mod tests {
#[tokio::test]
async fn report_includes_bucket_drive_scan_starts() {
let metrics = Metrics::new();
metrics.record_scan_bucket_drive_start();
metrics.record_scan_bucket_drive_start(ScannerWorkSource::Usage, "bucket-a", "/mnt/data/1");
metrics.record_scan_bucket_drive_failure();
let report = metrics.report().await;
@@ -4380,6 +4464,27 @@ mod tests {
assert_eq!(report.life_time_ops.get("scan_bucket_drive_failure"), Some(&1));
}
#[tokio::test]
async fn active_bucket_drive_snapshot_is_structured_and_retired_on_end() {
let metrics = Metrics::new();
metrics.record_scan_bucket_drive_start(ScannerWorkSource::Usage, "bucket-a", "/mnt/data/1");
metrics.record_scan_bucket_drive_start(ScannerWorkSource::Usage, "bucket-a", "/mnt/data/1");
let active = metrics.scanner_runtime_details_report().active_bucket_drive_scans;
assert_eq!(active.len(), 1);
assert_eq!(active[0].source, ScannerWorkSource::Usage.as_str());
assert_eq!(active[0].bucket, "bucket-a");
assert_eq!(active[0].drive, "/mnt/data/1");
assert_eq!(active[0].count, 2);
metrics.record_scan_bucket_drive_end(ScannerWorkSource::Usage, "bucket-a", "/mnt/data/1");
assert_eq!(metrics.scanner_runtime_details_report().active_bucket_drive_scans[0].count, 1);
metrics.record_scan_bucket_drive_end(ScannerWorkSource::Usage, "bucket-a", "/mnt/data/1");
assert!(metrics.scanner_runtime_details_report().active_bucket_drive_scans.is_empty());
metrics.record_scan_bucket_drive_start(ScannerWorkSource::Usage, "", "/mnt/data/1");
assert!(metrics.scanner_runtime_details_report().active_bucket_drive_scans.is_empty());
}
#[tokio::test]
async fn report_includes_structured_bucket_drive_results() {
let metrics = Metrics::new();
+9
View File
@@ -115,6 +115,15 @@ Current guidance:
- enables KMS readiness enforcement for `/health/ready`.
- default is `false`.
## Object lock admission environment variables
- `RUSTFS_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS`
- experimental same-object PUT commit namespace-lock admission budget.
- default is `0`, which disables this override and keeps `RUSTFS_OBJECT_LOCK_ACQUIRE_TIMEOUT` behavior.
- when set, only `put_object_commit` write-lock acquisition is bounded by this millisecond budget; other namespace lock users keep the global object-lock timeout.
- timeout returns S3 `SlowDown`, so clients should use normal SDK retry handling.
- this is not a fdatasync or group-commit switch. Track fdatasync batching separately with `rustfs_s3_put_object_rename_fdatasync_batch_files`.
## Drive timeout environment variables
- `RUSTFS_DRIVE_METADATA_TIMEOUT_SECS`
+13
View File
@@ -427,6 +427,19 @@ pub const ENV_OBJECT_LOCK_ACQUIRE_TIMEOUT: &str = "RUSTFS_OBJECT_LOCK_ACQUIRE_TI
/// Default lock acquisition timeout: 5 seconds.
pub const DEFAULT_OBJECT_LOCK_ACQUIRE_TIMEOUT: u64 = 5;
/// Environment variable for the experimental PUT commit namespace lock acquire timeout in milliseconds.
///
/// A value of `0` disables the experiment and keeps
/// `RUSTFS_OBJECT_LOCK_ACQUIRE_TIMEOUT` as the timeout. This only bounds the
/// `put_object_commit` namespace write-lock wait and is intended for #925
/// tail-drain admission experiments.
///
/// Default: 0 milliseconds (disabled).
pub const ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS: &str = "RUSTFS_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS";
/// Default: PUT commit namespace lock acquire timeout override is disabled.
pub const DEFAULT_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS: u64 = 0;
/// Environment variable for remote namespace lock RPC transport timeout in milliseconds.
///
/// This timeout bounds the internode RPC call itself. It is intentionally
+26 -21
View File
@@ -48,16 +48,14 @@ cargo nextest run --profile e2e-smoke -p e2e_test
cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \
-E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))'
# Protocols suite — fixed ports, MUST be single-threaded, gated by build features
RUSTFS_BUILD_FEATURES=ftps,webdav,sftp \
cargo test -p e2e_test test_protocol_core_suite -- --test-threads=1 --nocapture
```
The protocols suite has its own contract (fixed bind ports 90229301,
`--test-threads=1`, feature-gated scheduling) documented in
single-worker execution, feature-gated scheduling) documented in
[`src/protocols/README.md`](src/protocols/README.md). `RUSTFS_BUILD_FEATURES`
selects which features the spawned binary is built with; leave it unset to run
every protocol entry.
every protocol entry. Use the exact profile command under
[Troubleshooting](#troubleshooting) for CI-equivalent execution.
### `#[ignore]` semantics
@@ -159,27 +157,26 @@ construction (random port + isolated temp dir) and need no serialization.
## CI map
`e2e_test` is **excluded** from the main `cargo nextest run --profile ci --all`
pass ([`.github/workflows/ci.yml`](../../.github/workflows/ci.yml) line 158,
`--exclude e2e_test`) — the whole crate is too slow to gate every PR. Subsets
join CI through the nextest profile system only (never as ad-hoc jobs):
pass (`--exclude e2e_test`) — the whole crate is too slow to gate every PR.
Subsets join CI through nextest profiles; the fixed-port protocol suite uses
the same profile for membership and execution with one nightly worker.
| Suite | Runs where | Status |
| --- | --- | --- |
| Smoke subset (`e2e-smoke` profile) | `e2e-tests` job, every PR | **Active** (backlog#1149 ci-4) |
| Full single-node suite (`e2e-full` profile) | `e2e-full` job, merge queue + main | **Active** (backlog#1149 ci-5) |
| `s3s-e2e` black-box | `e2e-tests` + `e2e-tests-rio-v2` jobs | **Active** (external conformance tool) |
| ILM / lifecycle (ignored) | `test-ilm-integration-serial` lane, `-j1` | **Active** (backlog#1148 ilm-1) |
| KMS suite | — | Not in CI yet (backlog#1149 ci-5) |
| Protocols (FTPS/WebDAV/SFTP) | — | Not in CI yet (backlog#1149 ci-7) |
| KMS suite | `e2e-full` job, merge queue + main | **Active** |
| Cluster faults (`e2e-nightly` profile) | consolidated nightly workflow | **Active** (backlog#1149 ci-7) |
| Protocols (FTPS/WebDAV/SFTP) | consolidated nightly workflow, serial | **Active** (backlog#1149 ci-7) |
| Replication (fast subset) | `e2e-smoke` profile, `e2e-tests` job, every PR | **Active** (backlog#1147 repl-1) |
| Replication (slow + dual-node) | `e2e-repl-nightly` profile, scheduled workflow | **Active** (backlog#1147 repl-1) |
| `reliant/*` (pre-started server) | — | Manual only |
| Replication (slow + multi-node) | `e2e-repl-nightly` profile, consolidated nightly workflow | **Active** (backlog#1147 repl-1) |
| `reliant/*` | 19 tests in PR smoke; remaining default tests in `e2e-full` | **Active** except `#[ignore]` |
Links: [`ci.yml`](../../.github/workflows/ci.yml) `e2e-tests` (line 347),
`test-ilm-integration-serial` (line 196). The `e2e-smoke` `default-filter` in
[`.config/nextest.toml`](../../.config/nextest.toml) is the **single wiring
mechanism** — extend that filter (or add a sibling profile) to admit more
tests; do not add e2e jobs to `ci.yml`. repl-1 / ilm-3 are landing in parallel
and may add lanes; keep the table above easy to extend.
The profile filters in [`.config/nextest.toml`](../../.config/nextest.toml) are
the wiring source of truth. Committed test-ID digests under
`.config/e2e-*-selection.txt` make every membership change explicit.
## Troubleshooting
@@ -188,9 +185,15 @@ and may add lanes; keep the table above easy to extend.
```bash
# Smoke (e2e-tests job) — includes the 20 fast replication tests
cargo nextest run --profile e2e-smoke -p e2e_test
# Replication nightly lane (16 slow + dual-node tests; install awscurl for the
# STS dual-node test, else it skips gracefully)
# Full single-node merge/main lane
cargo nextest run --profile e2e-full -p e2e_test
# Cluster fault nightly lane
cargo nextest run --profile e2e-nightly -p e2e_test
# Replication nightly lane; install awscurl so STS paths do not skip
cargo nextest run --profile e2e-repl-nightly -p e2e_test
# Fixed-port protocol nightly lane
RUSTFS_BUILD_FEATURES=ftps,webdav,sftp \
cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
# ILM serial lane
cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \
-E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))'
@@ -273,4 +276,6 @@ current subset is.
`docs/testing/e2e-suite-inventory.md` records the per-module test counts as
listed by `cargo nextest list -p e2e_test`. Regenerate it when adding or
moving e2e tests so acceptance numbers in the test-strategy issues
(backlog#1147#1155) stay auditable.
(backlog#1147#1155) stay auditable. When a profile membership change is
intentional, review its JSON listing before updating the matching
`.config/e2e-*-selection.txt` test-ID digest.
+2 -1
View File
@@ -53,7 +53,8 @@ pub(crate) const FAST_DATA_USAGE_SCANNER_ENV: &[(&str, &str)] =
pub const TEST_BUCKET: &str = "e2e-test-bucket";
const RUSTFS_FULL_FEATURE: &str = "full";
const TEST_PORT_MIN: u16 = 20_000;
const TEST_PORT_RANGE: u16 = 40_000;
// Keep allocator ports below the ephemeral range used by bind(..., 0) test helpers.
const TEST_PORT_RANGE: u16 = 10_000;
const TEST_PORT_COUNTER_PATH: &str = "/tmp/rustfs_e2e_next_port";
const TEST_PORT_LOCK_DIR: &str = "/tmp/rustfs_e2e_port_allocator.lock";
const TEST_PORT_LOCK_STALE_AFTER: Duration = Duration::from_secs(30);
@@ -1028,20 +1028,6 @@ impl<'a> ReaderPathExpectation<'a> {
}
}
fn with_size_bucket(
object: ReaderObject<'a>,
expected_path: &'a str,
object_class: &'a str,
expected_size_bucket: &'a str,
) -> Self {
Self {
object,
expected_path,
object_class,
expected_size_bucket: Some(expected_size_bucket),
}
}
fn with_any_size_bucket(object: ReaderObject<'a>, expected_path: &'a str, object_class: &'a str) -> Self {
Self {
object,
@@ -1909,12 +1895,7 @@ async fn four_node_compressed_inline_fallback() -> TestResult {
assert_reader_path(
&collector,
&client,
ReaderPathExpectation::with_size_bucket(
ReaderObject::new(bucket, key, &body, put.e_tag(), None),
LEGACY_DUPLEX,
COMPRESSED,
size_bucket(4 * KIB),
),
ReaderPathExpectation::for_class(ReaderObject::new(bucket, key, &body, put.e_tag(), None), LEGACY_DUPLEX, COMPRESSED),
)
.await?;
@@ -2274,6 +2255,7 @@ async fn four_node_manual_transition_distributed_admission_conflict_reports_stat
hot.set_env("RUSTFS_SCANNER_CYCLE", "3600");
hot.set_env("RUSTFS_MAX_TRANSITION_WORKERS", "1");
hot.set_env("RUSTFS_TRANSITION_QUEUE_CAPACITY", "1");
hot.set_env("RUSTFS_TRANSITION_QUEUE_SEND_TIMEOUT_MS", "1");
hot.start().await?;
let hot_client = hot.create_s3_client(0)?;
@@ -2290,7 +2272,7 @@ async fn four_node_manual_transition_distributed_admission_conflict_reports_stat
.put_object()
.bucket(&bucket)
.key(key)
.body(ByteStream::from(payload(64 * KIB, index)))
.body(ByteStream::from(payload(1024 * KIB, index)))
.send()
.await?;
}
@@ -37,7 +37,7 @@ async fn test_bucket_default_sse_s3_put_object() -> Result<(), Box<dyn std::erro
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -159,7 +159,7 @@ async fn test_bucket_default_sse_kms_put_object() -> Result<(), Box<dyn std::err
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -278,7 +278,7 @@ async fn test_bucket_default_sse_kms_multipart_crc32() -> Result<(), Box<dyn std
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -475,7 +475,7 @@ async fn test_explicit_encryption_overrides_bucket_default() -> Result<(), Box<d
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -570,7 +570,7 @@ async fn test_sse_kms_without_key_id_populates_default() -> Result<(), Box<dyn s
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
+51 -1
View File
@@ -40,7 +40,7 @@ use std::time::Duration;
use tokio::fs;
use tokio::net::TcpStream;
use tokio::time::sleep;
use tracing::{debug, error, info};
use tracing::{debug, error, info, warn};
// KMS-specific constants
pub const TEST_BUCKET: &str = "kms-test-bucket";
@@ -177,6 +177,49 @@ pub async fn get_kms_status(
Ok(status)
}
/// Poll the KMS status endpoint until the backend reports ready or the timeout
/// expires. Replaces hard-coded `sleep(Duration::from_secs(3))` startup waits
/// with an active readiness probe so tests start as soon as KMS is usable
/// (typically < 1 s) instead of always waiting the full 3 s.
///
/// Uses exponential back-off starting at 200 ms (doubling each attempt, capped
/// at 1 s) up to a total wall-clock budget of 5 s.
pub async fn wait_for_kms_ready(
base_url: &str,
access_key: &str,
secret_key: &str,
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
let total_deadline = Duration::from_secs(5);
let start = tokio::time::Instant::now();
let mut backoff = Duration::from_millis(200);
let max_backoff = Duration::from_secs(1);
let mut first_attempt = true;
loop {
if !first_attempt {
if start.elapsed() >= total_deadline {
return Err("KMS failed to become ready within 5 seconds".into());
}
sleep(backoff).await;
backoff = (backoff * 2).min(max_backoff);
}
first_attempt = false;
match get_kms_status(base_url, access_key, secret_key).await {
Ok(status) => {
info!("KMS is ready (status: {})", status);
return Ok(());
}
Err(e) => {
if start.elapsed() >= total_deadline {
return Err(format!("KMS did not become ready within 5 s: last error: {e}").into());
}
warn!(error = %e, elapsed_ms = start.elapsed().as_millis() as u64, "KMS not ready yet, retrying…");
}
}
}
}
/// Create a default KMS key for testing and return the created key ID
pub async fn create_default_key(
base_url: &str,
@@ -861,6 +904,13 @@ impl LocalKMSTestEnvironment {
Ok(default_key_id.to_string())
}
/// Poll the KMS status endpoint until the backend reports ready.
///
/// Prefer this over a fixed `sleep` after calling `start_rustfs_for_local_kms`.
pub async fn wait_for_kms_ready(&self) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
wait_for_kms_ready(&self.base_env.url, &self.base_env.access_key, &self.base_env.secret_key).await
}
/// Configure Local KMS backend with a predefined default key
pub async fn configure_local_kms(&self) -> Result<String, Box<dyn std::error::Error + Send + Sync>> {
// Use a fixed, predictable default key ID
@@ -61,7 +61,7 @@ async fn test_metadata_replace_self_copy_of_sse_object_stays_decryptable() {
)
.await
.expect("failed to start RustFS with local KMS");
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let client = kms_env.base_env.create_s3_client();
// Deliberately an UNVERSIONED bucket: that is the branch where the store layer can service
@@ -160,7 +160,7 @@ async fn test_metadata_replace_self_copy_dropping_sse_rewrites_plaintext() {
)
.await
.expect("failed to start RustFS with local KMS");
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let client = kms_env.base_env.create_s3_client();
// Unversioned, and deliberately WITHOUT a bucket default-encryption rule, so the copy below
@@ -256,7 +256,7 @@ async fn test_metadata_replace_self_copy_under_bucket_default_sse_stays_decrypta
)
.await
.expect("failed to start RustFS with local KMS");
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let client = kms_env.base_env.create_s3_client();
let bucket = "copy-object-self-copy-bucket-default-sse-test";
@@ -56,7 +56,7 @@ async fn test_self_copy_of_historical_sse_s3_version_is_readable() {
)
.await
.expect("failed to start RustFS with local KMS");
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let client = kms_env.base_env.create_s3_client();
let bucket = "copy-object-version-restore-sse-test";
@@ -87,7 +87,7 @@ async fn test_head_reports_managed_metadata_for_sse_s3() -> Result<(), Box<dyn s
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -147,7 +147,7 @@ async fn test_head_reports_managed_metadata_for_sse_kms_and_copy() -> Result<(),
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -250,7 +250,7 @@ async fn test_multipart_upload_writes_encrypted_data() -> Result<(), Box<dyn std
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -39,6 +39,7 @@ use std::time::Duration;
use tracing::info;
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
type S3OperationResult<T> = Result<T, Box<aws_sdk_s3::Error>>;
const ALLOWED_KEY: &str = "kms-matrix-allowed-key";
const OTHER_KEY: &str = "kms-matrix-other-key";
@@ -130,7 +131,7 @@ fn policy_document(statements: Vec<serde_json::Value>) -> String {
serde_json::json!({ "Version": "2012-10-17", "Statement": statements }).to_string()
}
async fn put_sse_kms(client: &Client, key: &str, kms_key_id: &str) -> Result<(), aws_sdk_s3::Error> {
async fn put_sse_kms(client: &Client, key: &str, kms_key_id: &str) -> S3OperationResult<()> {
client
.put_object()
.bucket(BUCKET)
@@ -141,16 +142,23 @@ async fn put_sse_kms(client: &Client, key: &str, kms_key_id: &str) -> Result<(),
.send()
.await
.map(|_| ())
.map_err(aws_sdk_s3::Error::from)
.map_err(|error| Box::new(aws_sdk_s3::Error::from(error)))
}
/// Assert the operation failed with `AccessDenied` rather than any other error.
///
/// A bare `is_err` would also accept `KMSKeyDisabled` or an internal error, which
/// would hide both a leak of key state and an outage masquerading as a denial.
fn assert_access_denied<T: std::fmt::Debug>(result: Result<T, aws_sdk_s3::Error>, what: &str) {
fn assert_access_denied<T: std::fmt::Debug, E: std::fmt::Debug + std::borrow::Borrow<aws_sdk_s3::Error>>(
result: Result<T, E>,
what: &str,
) {
let error = result.expect_err(&format!("{what} must be denied"));
assert_eq!(error.code(), Some("AccessDenied"), "{what} must fail with AccessDenied: {error:?}");
assert_eq!(
error.borrow().code(),
Some("AccessDenied"),
"{what} must fail with AccessDenied: {error:?}"
);
}
/// Retry an SSE-KMS write until the identity's policy has reached the request path.
@@ -296,7 +304,7 @@ async fn sse_kms_per_key_authorization_negative_matrix() -> TestResult {
.send()
.await
.map(|_| ())
.map_err(aws_sdk_s3::Error::from),
.map_err(|err| Box::new(aws_sdk_s3::Error::from(err))),
"SSE-KMS read by an identity holding no kms grant",
);
@@ -310,7 +318,7 @@ async fn sse_kms_per_key_authorization_negative_matrix() -> TestResult {
.send()
.await
.map(|_| ())
.map_err(aws_sdk_s3::Error::from),
.map_err(|err| Box::new(aws_sdk_s3::Error::from(err))),
"SSE-KMS read by an identity holding kms:GenerateDataKey but not kms:Decrypt",
);
@@ -24,7 +24,6 @@ use super::common::{
test_sse_kms_encryption, test_sse_s3_encryption,
};
use crate::common::{TEST_BUCKET, init_logging};
use tokio::time::{Duration, sleep};
use tracing::info;
/// Comprehensive test: Full KMS workflow with all encryption types
@@ -35,7 +34,7 @@ async fn test_comprehensive_kms_full_workflow() -> Result<(), Box<dyn std::error
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -103,7 +102,7 @@ async fn test_comprehensive_stress_test() -> Result<(), Box<dyn std::error::Erro
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -137,7 +136,7 @@ async fn test_comprehensive_key_isolation() -> Result<(), Box<dyn std::error::Er
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -208,7 +207,7 @@ async fn test_comprehensive_concurrent_operations() -> Result<(), Box<dyn std::e
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -253,7 +252,7 @@ async fn test_comprehensive_performance_benchmark() -> Result<(), Box<dyn std::e
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -44,7 +44,7 @@ async fn test_kms_zero_byte_file_encryption() -> Result<(), Box<dyn std::error::
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -117,7 +117,7 @@ async fn test_kms_single_byte_file_encryption() -> Result<(), Box<dyn std::error
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -209,7 +209,7 @@ async fn test_kms_multipart_boundary_conditions() -> Result<(), Box<dyn std::err
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -284,7 +284,7 @@ async fn test_kms_invalid_key_scenarios() -> Result<(), Box<dyn std::error::Erro
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -371,7 +371,7 @@ async fn test_kms_concurrent_encryption() -> Result<(), Box<dyn std::error::Erro
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = Arc::new(kms_env.base_env.create_s3_client());
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -478,7 +478,7 @@ async fn test_kms_key_validation_security() -> Result<(), Box<dyn std::error::Er
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -37,7 +37,7 @@ async fn test_kms_key_directory_unavailable() -> Result<(), Box<dyn std::error::
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -127,7 +127,7 @@ async fn test_kms_corrupted_key_files() -> Result<(), Box<dyn std::error::Error
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -218,7 +218,7 @@ async fn test_kms_multipart_upload_interruption() -> Result<(), Box<dyn std::err
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -401,7 +401,7 @@ async fn test_kms_resource_constraints() -> Result<(), Box<dyn std::error::Error
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
+4 -4
View File
@@ -46,7 +46,7 @@ async fn test_local_kms_end_to_end() -> Result<(), Box<dyn std::error::Error + S
.expect("Failed to start RustFS with Local KMS");
// Wait a moment for RustFS to fully start up and initialize KMS
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
info!("RustFS started with KMS auto-configuration, default_key_id: {}", default_key_id);
@@ -127,7 +127,7 @@ async fn test_local_kms_key_isolation() {
.expect("Failed to start RustFS with Local KMS");
// Wait a moment for RustFS to fully start up and initialize KMS
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
info!("RustFS started with KMS auto-configuration, default_key_id: {}", default_key_id);
@@ -227,7 +227,7 @@ async fn test_local_kms_large_file() {
.expect("Failed to start RustFS with Local KMS");
// Wait a moment for RustFS to fully start up and initialize KMS
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
info!("RustFS started with KMS auto-configuration, default_key_id: {}", default_key_id);
@@ -309,7 +309,7 @@ async fn test_local_kms_multipart_upload() {
.expect("Failed to start RustFS with Local KMS");
// Wait for KMS initialization
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
info!("RustFS started with KMS auto-configuration, default_key_id: {}", default_key_id);
+7 -3
View File
@@ -20,7 +20,6 @@
use crate::common::{TEST_BUCKET, init_logging};
use serial_test::serial;
use tokio::time::{Duration, sleep};
use tracing::{error, info};
use super::common::{
@@ -46,8 +45,13 @@ impl VaultKmsTestContext {
start_kms(&env.base_env.url, &env.base_env.access_key, &env.base_env.secret_key).await?;
// Allow Vault to finish initialising token auth and transit engine.
sleep(Duration::from_secs(2)).await;
// Wait for KMS to finish initialising.
super::common::wait_for_kms_ready(
&env.base_env.url,
&env.base_env.access_key,
&env.base_env.secret_key,
)
.await?;
Ok(Self { env })
}
@@ -33,7 +33,7 @@ async fn test_step1_basic_single_file_encryption() -> Result<(), Box<dyn std::er
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -89,7 +89,7 @@ async fn test_step2_basic_multipart_upload_without_encryption() -> Result<(), Bo
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -187,7 +187,7 @@ async fn test_step3_multipart_upload_with_sse_s3() -> Result<(), Box<dyn std::er
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -310,7 +310,7 @@ async fn test_step4_large_multipart_upload_with_encryption() -> Result<(), Box<d
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
@@ -435,7 +435,7 @@ async fn test_step5_all_encryption_types_multipart() -> Result<(), Box<dyn std::
let mut kms_env = LocalKMSTestEnvironment::new().await?;
let _default_key_id = kms_env.start_rustfs_for_local_kms().await?;
tokio::time::sleep(tokio::time::Duration::from_secs(3)).await;
kms_env.wait_for_kms_ready().await?;
let s3_client = kms_env.base_env.create_s3_client();
kms_env.base_env.create_test_bucket(TEST_BUCKET).await?;
+7 -1
View File
@@ -11,10 +11,17 @@ test process directly.
## Running Tests
Use the canonical CI-equivalent protocol command in the parent
[`e2e_test` README](../../README.md#troubleshooting).
For targeted debugging of the core suite only:
```bash
RUSTFS_BUILD_FEATURES=ftps,webdav,sftp cargo test --package e2e_test test_protocol_core_suite -- --test-threads=1 --nocapture
```
This targeted command does not cover the full `e2e-protocols` profile.
`RUSTFS_BUILD_FEATURES` controls which features the test rustfs binary is
built with. When this variable is set, the protocol test runner schedules
only entries whose protocol is present in the requested feature list. Leave
@@ -133,4 +140,3 @@ property without consulting any external doc.
Bind ports 9023 (SFTP) and 9100 (S3). Spawns rustfs with
`RUSTFS_SFTP_IDLE_TIMEOUT=5`, sleeps 10 s past the timeout, then issues an
SFTP request and asserts the server has closed the session.
@@ -233,6 +233,111 @@ pub async fn test_webdav_core_operations() -> Result<()> {
);
info!("PASS: PUT file '{}' successful", filename);
// Regression for #6260: a bucket-scoped policy must be able to discover its bucket at the
// WebDAV root without the unrelated global ListAllMyBuckets permission.
let scoped_bucket = "webdav-scoped-bucket";
let scoped_file = "visible.txt";
let scoped_user = "webdav-scoped-user";
let scoped_secret = "webdav-scoped-secret";
let scoped_policy_name = "webdav-scoped-policy";
let resp = client
.request(reqwest::Method::from_bytes(b"MKCOL").unwrap(), format!("{}/{}", base_url, scoped_bucket))
.header("Authorization", &auth_header)
.send()
.await?;
assert_eq!(resp.status().as_u16(), 201, "scoped test bucket should be created");
let resp = client
.put(format!("{}/{}/{}", base_url, scoped_bucket, scoped_file))
.header("Authorization", &auth_header)
.body("visible to the scoped principal")
.send()
.await?;
assert_eq!(resp.status().as_u16(), 201, "scoped test object should be created");
admin_create_user(&admin_base_url, scoped_user, scoped_secret).await?;
admin_add_canned_policy(
&admin_base_url,
scoped_policy_name,
&serde_json::json!({
"Version": "2012-10-17",
"Statement": [
{
"Effect": "Allow",
"Action": ["s3:*"],
"Resource": [
format!("arn:aws:s3:::{}", scoped_bucket),
format!("arn:aws:s3:::{}/*", scoped_bucket)
]
},
{
"Effect": "Deny",
"Action": ["s3:*"],
"Resource": [
format!("arn:aws:s3:::{}", scoped_bucket),
format!("arn:aws:s3:::{}/*", scoped_bucket)
],
"Condition": { "Bool": { "aws:SecureTransport": "true" } }
},
{
"Effect": "Deny",
"Action": ["s3:*"],
"Resource": [
format!("arn:aws:s3:::{}", scoped_bucket),
format!("arn:aws:s3:::{}/*", scoped_bucket)
],
"Condition": { "StringEquals": { "s3:signatureversion": "AWS4-HMAC-SHA256" } }
}
]
}),
)
.await?;
admin_attach_policy_to_user(&admin_base_url, scoped_policy_name, scoped_user).await?;
let scoped_auth = basic_auth_header_for(scoped_user, scoped_secret);
let resp = client
.request(reqwest::Method::from_bytes(b"PROPFIND").unwrap(), &base_url)
.header("Authorization", &scoped_auth)
.header("Depth", "1")
.header("x-amz-content-sha256", "STREAMING-AWS4-HMAC-SHA256-PAYLOAD")
.send()
.await?;
assert_eq!(resp.status().as_u16(), 207, "bucket-scoped root PROPFIND should succeed");
let root_listing = resp.text().await?;
assert!(root_listing.contains(scoped_bucket), "the authorized bucket should be listed");
assert!(!root_listing.contains(bucket_name), "an unauthorized bucket must not be listed");
let resp = client
.request(
reqwest::Method::from_bytes(b"PROPFIND").unwrap(),
format!("{}/{}", base_url, scoped_bucket),
)
.header("Authorization", &scoped_auth)
.header("Depth", "1")
.send()
.await?;
assert_eq!(resp.status().as_u16(), 207, "authorized bucket PROPFIND should succeed");
assert!(resp.text().await?.contains(scoped_file), "the authorized object should be listed");
let denied_user = "webdav-no-buckets-user";
let denied_secret = "webdav-no-buckets-secret";
admin_create_user(&admin_base_url, denied_user, denied_secret).await?;
let resp = client
.request(reqwest::Method::from_bytes(b"PROPFIND").unwrap(), &base_url)
.header("Authorization", basic_auth_header_for(denied_user, denied_secret))
.header("Depth", "1")
.send()
.await?;
assert_eq!(
resp.status().as_u16(),
207,
"PROPFIND keeps the root resource visible when the directory listing is forbidden"
);
let denied_body = resp.text().await?;
assert!(!denied_body.contains(scoped_bucket), "a denied response must not leak the scoped bucket");
assert!(!denied_body.contains(bucket_name), "a denied response must not leak the admin bucket");
// Test GET (download file)
info!("Testing WebDAV: GET (download file '{}')", filename);
let resp = client
+40 -24
View File
@@ -169,6 +169,42 @@ impl QuotaTestEnv {
bucket: &str,
quota_bytes: u64,
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
self.wait_for_quota_usage_for(bucket).await?;
let quota_path = format!("/rustfs/admin/v3/quota/{bucket}");
let quota_config = serde_json::json!({
"quota": quota_bytes,
"quota_type": "HARD"
})
.to_string();
let readiness = async {
loop {
let (status, response) = admin_request(
&self.env.url,
Method::PUT,
&quota_path,
Some(quota_config.clone()),
&self.env.access_key,
&self.env.secret_key,
)
.await?;
if status.is_success() {
return Ok::<(), Box<dyn std::error::Error + Send + Sync>>(());
}
if status != StatusCode::SERVICE_UNAVAILABLE {
return Err(format!("failed to set quota for {bucket}: {status} {response}").into());
}
sleep(Duration::from_secs(1)).await;
}
};
match timeout(Duration::from_secs(30), readiness).await {
Ok(result) => result,
Err(_) => Err(format!("quota readiness did not converge for {bucket} within 30 seconds").into()),
}
}
pub async fn wait_for_quota_usage_for(&self, bucket: &str) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
let stats_path = format!("/rustfs/admin/v3/quota-stats/{bucket}");
let readiness = async {
loop {
@@ -181,28 +217,12 @@ impl QuotaTestEnv {
if status != StatusCode::SERVICE_UNAVAILABLE {
return Err(format!("quota usage readiness failed for {bucket}: {status} {response}").into());
}
sleep(Duration::from_secs(1)).await;
}
};
match timeout(Duration::from_secs(30), readiness).await {
Ok(result) => result?,
Err(_) => {
return Err(format!("quota usage did not become authoritative for {bucket} within 30 seconds").into());
}
}
let url = format!("{}/rustfs/admin/v3/quota/{}", self.env.url, bucket);
let quota_config = serde_json::json!({
"quota": quota_bytes,
"quota_type": "HARD"
});
let response = awscurl_put(&url, &quota_config.to_string(), &self.env.access_key, &self.env.secret_key).await?;
if response.contains("error") {
Err(format!("Failed to set quota: {}", response).into())
} else {
Ok(())
Ok(result) => result,
Err(_) => Err(format!("quota usage did not become authoritative for {bucket} within 30 seconds").into()),
}
}
@@ -614,6 +634,7 @@ mod integration_tests {
let env = QuotaTestEnv::new().await?;
env.create_bucket().await?;
env.wait_for_quota_usage_for(&env.bucket_name).await?;
// Test 1: GET quota for bucket without quota config
let url = format!("{}/rustfs/admin/v3/quota/{}", env.env.url, env.bucket_name);
@@ -621,12 +642,7 @@ mod integration_tests {
assert!(response.contains("quota") && response.contains("null"));
// Test 2: PUT quota - valid config
let quota_config = serde_json::json!({
"quota": 1048576,
"quota_type": "HARD"
});
let response = awscurl_put(&url, &quota_config.to_string(), &env.env.access_key, &env.env.secret_key).await?;
assert!(response.contains("success") || !response.contains("error"));
env.set_bucket_quota(1048576).await?;
// Test 3: GET quota after setting
let response = awscurl_get(&url, &env.env.access_key, &env.env.secret_key).await?;
+37 -14
View File
@@ -110,6 +110,7 @@ const USER_META_KEY: &str = "ilm7-origin";
const USER_META_VAL: &str = "hermetic-transition";
const HDR_SOURCE_REPLICATION_REQUEST: &str = "x-rustfs-source-replication-request";
const HDR_SOURCE_MTIME: &str = "x-rustfs-source-mtime";
const TIER_MUTATION_RECOVERY_CHANGED: &str = "Remote tier mutation recovery changed before publish";
/// 5 MiB — the S3 minimum size for a non-final multipart part; the object's only
/// internal part boundary sits at this offset.
@@ -183,19 +184,39 @@ async fn add_rustfs_tier(hot: &RustFSTestEnvironment, cold: &RustFSTestEnvironme
})
.to_string();
let (status, resp) = signed_admin_request(
&hot.url,
Method::PUT,
"/rustfs/admin/v3/tier",
Some(&body),
&hot.access_key,
&hot.secret_key,
)
.await?;
if !status.is_success() {
return Err(format!("AddTier(RustFS) failed: status={status}, body={resp}").into());
let verify_path = format!("/rustfs/admin/v3/tier/{TIER_NAME}");
let deadline = Instant::now() + StdDuration::from_secs(30);
let mut recovery_changed = false;
loop {
if recovery_changed {
let (status, _) =
signed_admin_request(&hot.url, Method::GET, &verify_path, None, &hot.access_key, &hot.secret_key).await?;
if status.is_success() {
return Ok(());
}
}
let (status, resp) = signed_admin_request(
&hot.url,
Method::PUT,
"/rustfs/admin/v3/tier",
Some(&body),
&hot.access_key,
&hot.secret_key,
)
.await?;
if status.is_success() {
return Ok(());
}
if resp.contains(TIER_MUTATION_RECOVERY_CHANGED) {
recovery_changed = true;
} else if !recovery_changed || !resp.contains("TierNameAlreadyExist") {
return Err(format!("AddTier(RustFS) failed: status={status}, body={resp}").into());
}
if Instant::now() >= deadline {
return Err(format!("AddTier(RustFS) failed: status={status}, body={resp}").into());
}
tokio::time::sleep(StdDuration::from_millis(100)).await;
}
Ok(())
}
async fn remove_rustfs_tier_force(hot: &RustFSTestEnvironment) -> TestResult {
@@ -207,10 +228,12 @@ async fn remove_rustfs_tier_force(hot: &RustFSTestEnvironment) -> TestResult {
if status.is_success() {
return Ok(());
}
if !resp.contains("TierNameBackendInUse") || Instant::now() >= deadline {
if (!resp.contains("TierNameBackendInUse") && !resp.contains(TIER_MUTATION_RECOVERY_CHANGED))
|| Instant::now() >= deadline
{
return Err(format!("RemoveTier(RustFS) failed: status={status}, body={resp}").into());
}
// AddTier cleanup is asynchronous; wait until its committed mutation fence clears.
// Tier mutation cleanup and startup recovery are asynchronous.
tokio::time::sleep(StdDuration::from_millis(100)).await;
}
}
+19 -12
View File
@@ -90,6 +90,12 @@ use uuid::Uuid;
const MAX_CONCURRENT_TARGET_HEALTH_CHECKS: usize = 16;
const REDACTED_CREDENTIAL: &str = "<redacted>";
pub type HeadObjectSdkError = Box<SdkError<HeadObjectError>>;
pub type GetObjectSdkError = Box<SdkError<GetObjectError>>;
pub type GetObjectTaggingSdkError = Box<SdkError<GetObjectTaggingError>>;
pub type PutObjectTaggingSdkError = Box<SdkError<PutObjectTaggingError>>;
pub type DeleteObjectTaggingSdkError = Box<SdkError<DeleteObjectTaggingError>>;
pub static GLOBAL_BUCKET_TARGET_SYS: OnceLock<BucketTargetSys> = OnceLock::new();
fn replication_target_versioning_enabled(versioning: Option<&BucketVersioningStatus>) -> bool {
@@ -1968,7 +1974,7 @@ impl TargetClient {
bucket: &str,
object: &str,
version_id: Option<String>,
) -> Result<HeadObjectOutput, SdkError<HeadObjectError>> {
) -> Result<HeadObjectOutput, HeadObjectSdkError> {
// Announce the replication check so a RustFS target returns SSE-C
// object metadata (etag/size) without the customer key the replication
// worker cannot hold; otherwise SSE-C replicas never converge on HEAD.
@@ -1981,8 +1987,7 @@ impl TargetClient {
// object with an identical ETag, and the worker concludes the object
// already converged — so it never actually replicates it.
insert_header(&mut headers, SUFFIX_SOURCE_PROXY_REQUEST, "false");
match self
.client
self.client
.head_object()
.bucket(bucket)
.key(object)
@@ -1999,10 +2004,7 @@ impl TargetClient {
})
.send()
.await
{
Ok(res) => Ok(res),
Err(e) => Err(e),
}
.map_err(Box::new)
}
/// HEAD used by the read-proxy path (GET/HEAD of an object not yet
@@ -2023,7 +2025,7 @@ impl TargetClient {
range: Option<String>,
part_number: Option<i32>,
extra_headers: HeaderMap,
) -> Result<HeadObjectOutput, SdkError<HeadObjectError>> {
) -> Result<HeadObjectOutput, HeadObjectSdkError> {
let headers = proxy_outbound_headers(extra_headers);
self.client
.head_object()
@@ -2036,6 +2038,7 @@ impl TargetClient {
.map_request(move |req| apply_extra_headers(req, &headers))
.send()
.await
.map_err(Box::new)
}
/// GET used by the read-proxy path (MinIO `proxyGetToReplicationTarget`).
@@ -2051,7 +2054,7 @@ impl TargetClient {
range: Option<String>,
part_number: Option<i32>,
extra_headers: HeaderMap,
) -> Result<GetObjectOutput, SdkError<GetObjectError>> {
) -> Result<GetObjectOutput, GetObjectSdkError> {
let headers = proxy_outbound_headers(extra_headers);
self.client
.get_object()
@@ -2064,6 +2067,7 @@ impl TargetClient {
.map_request(move |req| apply_extra_headers(req, &headers))
.send()
.await
.map_err(Box::new)
}
/// GetObjectTagging for the tagging read-proxy path
@@ -2073,7 +2077,7 @@ impl TargetClient {
bucket: &str,
object: &str,
version_id: Option<String>,
) -> Result<GetObjectTaggingOutput, SdkError<GetObjectTaggingError>> {
) -> Result<GetObjectTaggingOutput, GetObjectTaggingSdkError> {
let headers = proxy_outbound_headers(HeaderMap::new());
self.client
.get_object_tagging()
@@ -2084,6 +2088,7 @@ impl TargetClient {
.map_request(move |req| apply_extra_headers(req, &headers))
.send()
.await
.map_err(Box::new)
}
/// PutObjectTagging for the tagging proxy path
@@ -2094,7 +2099,7 @@ impl TargetClient {
object: &str,
version_id: Option<String>,
tagging: SdkTagging,
) -> Result<PutObjectTaggingOutput, SdkError<PutObjectTaggingError>> {
) -> Result<PutObjectTaggingOutput, PutObjectTaggingSdkError> {
let headers = proxy_outbound_headers(HeaderMap::new());
self.client
.put_object_tagging()
@@ -2106,6 +2111,7 @@ impl TargetClient {
.map_request(move |req| apply_extra_headers(req, &headers))
.send()
.await
.map_err(Box::new)
}
/// DeleteObjectTagging for the tagging proxy path
@@ -2115,7 +2121,7 @@ impl TargetClient {
bucket: &str,
object: &str,
version_id: Option<String>,
) -> Result<DeleteObjectTaggingOutput, SdkError<DeleteObjectTaggingError>> {
) -> Result<DeleteObjectTaggingOutput, DeleteObjectTaggingSdkError> {
let headers = proxy_outbound_headers(HeaderMap::new());
self.client
.delete_object_tagging()
@@ -2126,6 +2132,7 @@ impl TargetClient {
.map_request(move |req| apply_extra_headers(req, &headers))
.send()
.await
.map_err(Box::new)
}
/// On success returns the version id the target assigned (from
@@ -2180,7 +2180,7 @@ pub async fn recover_manual_transition_jobs_once(
if limit == 0 {
return Err(Error::other("manual transition job recovery limit must be greater than zero"));
}
let list_limit = i32::try_from(limit).map_or(i32::MAX, |value| value);
let list_limit = i32::try_from(limit).unwrap_or(i32::MAX);
let page = api
.clone()
.list_objects_v2(
@@ -2386,7 +2386,7 @@ async fn replay_manual_transition_pending_tasks(
version_id: task.version_id,
etag: task.etag,
mod_time,
size: task.size.map_or(0, |size| size),
size: task.size.unwrap_or(0),
is_latest: task.is_latest.unwrap_or(false),
..Default::default()
};
@@ -1016,7 +1016,7 @@ pub async fn recover_transition_transaction_records(
return Err(Error::other("transition transaction recovery limit must be greater than zero"));
}
let list_limit = i32::try_from(limit).map_or(i32::MAX, |value| value);
let list_limit = i32::try_from(limit).unwrap_or(i32::MAX);
let list = api
.clone()
.list_objects_v2(
+61
View File
@@ -41,6 +41,7 @@ const IAM_FORMAT_FILE_PATH: &str = "config/iam/format.json";
const IAM_USERS_PREFIX: &str = "config/iam/users/";
const IAM_SERVICE_ACCOUNTS_PREFIX: &str = "config/iam/service-accounts/";
const IAM_STS_PREFIX: &str = "config/iam/sts/";
const MINIO_GO_ZERO_TIME: OffsetDateTime = time::macros::datetime!(0001-01-01 00:00 UTC);
const IAM_GROUPS_PREFIX: &str = "config/iam/groups/";
const IAM_POLICIES_PREFIX: &str = "config/iam/policies/";
const IAM_POLICY_DB_PREFIX: &str = "config/iam/policydb/";
@@ -120,6 +121,15 @@ fn normalize_iam_config_blob(path: &str, data: &[u8]) -> std::result::Result<Opt
if is_identity_path(path) {
let mut identity: UserIdentity =
serde_json::from_slice(data).map_err(|err| format!("parse IAM identity failed: {err}"))?;
if (path.starts_with(IAM_USERS_PREFIX) || path.starts_with(IAM_SERVICE_ACCOUNTS_PREFIX))
&& identity
.credentials
.expiration
.as_ref()
.is_some_and(|expiration| *expiration == MINIO_GO_ZERO_TIME || *expiration == OffsetDateTime::UNIX_EPOCH)
{
identity.credentials.expiration = None;
}
if identity.update_at.is_none() {
identity.update_at = Some(OffsetDateTime::now_utc());
}
@@ -441,7 +451,10 @@ mod tests {
use crate::bucket::replication::{
BucketReplicationResyncStatus, ReplicationMigrationBridge, ResyncStatusType, TargetReplicationResyncStatus,
};
use rustfs_policy::auth::UserIdentity;
use std::collections::HashMap;
use time::OffsetDateTime;
use time::format_description::well_known::Rfc3339;
#[test]
fn test_normalize_policy_mapping_legacy_timestamp_and_fields() {
@@ -493,6 +506,54 @@ mod tests {
assert!(v.get("updatedAt").is_some(), "normalize should backfill updatedAt");
}
#[test]
fn test_normalize_minio_permanent_credential_expiration() {
let cases = [
("config/iam/users/alice/identity.json", "0001-01-01T00:00:00Z", true),
("config/iam/users/alice/identity.json", "1970-01-01T00:00:00Z", true),
("config/iam/service-accounts/svc/identity.json", "0001-01-01T00:00:00Z", true),
("config/iam/service-accounts/svc/identity.json", "1970-01-01T00:00:00Z", true),
("config/iam/service-accounts/svc/identity.json", "1970-01-01T00:00:00.000000001Z", false),
("config/iam/sts/temp/identity.json", "0001-01-01T00:00:00Z", false),
("config/iam/sts/temp/identity.json", "1970-01-01T00:00:00Z", false),
("config/iam/users/alice/identity.json", "1969-12-31T23:59:59Z", false),
("config/iam/users/alice/identity.json", "1970-01-01T00:00:00.000000001Z", false),
("config/iam/users/alice/identity.json", "0001-01-01T00:00:00.000000001Z", false),
("config/iam/users/alice/identity.json", "2030-01-01T00:00:00Z", false),
];
for (path, expiration, should_clear) in cases {
let input = serde_json::json!({
"version": 1,
"credentials": {
"accessKey": "test-access",
"secretKey": "test-secret",
"sessionToken": "test-session-token",
"parentUser": "test-parent",
"expiration": expiration,
}
});
let output = normalize_iam_config_blob(path, &serde_json::to_vec(&input).expect("serialize identity fixture"))
.expect("normalize should succeed")
.expect("identity path should be supported");
let identity: UserIdentity = serde_json::from_slice(&output).expect("deserialize normalized identity");
assert_eq!(identity.credentials.access_key, "test-access");
assert_eq!(identity.credentials.secret_key, "test-secret");
assert_eq!(identity.credentials.session_token, "test-session-token");
assert_eq!(identity.credentials.parent_user, "test-parent");
if should_clear {
assert_eq!(identity.credentials.expiration, None, "path: {path}, expiration: {expiration}");
} else {
assert_eq!(
identity.credentials.expiration,
Some(OffsetDateTime::parse(expiration, &Rfc3339).expect("parse expected expiration")),
"path: {path}, expiration: {expiration}"
);
}
}
}
#[test]
fn test_normalize_bucket_meta_blob_resync_reencode() {
let path = ".buckets/test/.replication/resync.bin";
+6 -1
View File
@@ -76,7 +76,12 @@ impl QuotaChecker {
let current_usage = self.get_real_time_usage(bucket).await?;
let admission_size = if uses_durable_reservations { 0 } else { operation_size };
// The reporting path projects this operation; storage mutations reserve it at commit.
let admission_size = if uses_durable_reservations && !force_usage_calculation {
0
} else {
operation_size
};
let expected_usage = match operation {
QuotaOperation::PutObject | QuotaOperation::PostObject | QuotaOperation::CopyObject => {
current_usage.saturating_add(admission_size)
@@ -52,8 +52,8 @@ use super::replication_storage_boundary::{
ReplicationObjectIO, ReplicationStorage, StatObjectOptions, StorageObjectInfoOrErr, WalkOptions,
};
use super::replication_target_boundary::{
ERR_REPLICATION_SSEC_PASSTHROUGH_UNSUPPORTED, PutObjectOptions, PutObjectPartOptions, ReplicationTargetStore,
SsecPassthroughCapability, SsecPassthroughGate, TargetClient, is_replication_target_offline_error,
ERR_REPLICATION_SSEC_PASSTHROUGH_UNSUPPORTED, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions,
ReplicationTargetStore, SsecPassthroughCapability, SsecPassthroughGate, TargetClient, is_replication_target_offline_error,
replication_action_for_target_head, replication_complete_multipart_options, replication_delete_marker_purge_remove_options,
replication_delete_remove_options, replication_force_delete_remove_options, replication_object_is_ssec_encrypted,
replication_put_object_header_size, replication_put_object_options, replication_target_head_is_newer_null_version,
@@ -214,7 +214,7 @@ async fn head_object_for_worker(
target_bucket: &str,
object: &str,
version_id: Option<String>,
) -> std::result::Result<HeadObjectOutput, SdkError<HeadObjectError>> {
) -> std::result::Result<HeadObjectOutput, HeadObjectSdkError> {
target_client.head_object(target_bucket, object, version_id).await
}
@@ -233,7 +233,7 @@ async fn mark_replication_target_offline_if_needed(target_client: &Arc<TargetCli
async fn head_object_fallback(
tgt_client: &TargetClient,
object: &str,
) -> std::result::Result<Option<HeadObjectOutput>, SdkError<HeadObjectError>> {
) -> std::result::Result<Option<HeadObjectOutput>, HeadObjectSdkError> {
match head_object_for_worker(tgt_client, &tgt_client.bucket, object, None).await {
Ok(oi) => Ok(Some(oi)),
Err(e) if e.as_service_error().is_some_and(|se| se.is_not_found()) || has_raw_status(&e, 404) => Ok(None),
@@ -1152,11 +1152,11 @@ fn spawn_resync_walk_task<S: ReplicationStorage>(
/// updating the per-object status counters and returning the accounted size
/// together with any verification error.
async fn verify_resync_head_result(
head_result: std::result::Result<HeadObjectOutput, SdkError<HeadObjectError>>,
head_result: std::result::Result<HeadObjectOutput, HeadObjectSdkError>,
roi: &ReplicateObjectInfo,
st: &mut TargetReplicationResyncStatus,
target_client: &Arc<TargetClient>,
) -> (i64, Option<SdkError<HeadObjectError>>) {
) -> (i64, Option<HeadObjectSdkError>) {
match head_result {
Ok(_) => {
st.replicated_count += 1;
@@ -1275,7 +1275,7 @@ async fn resync_worker_process_object<S: ReplicationStorage>(
"Processed resync object"
);
}
st.error = err.as_ref().and_then(resync_target_error_detail);
st.error = err.as_ref().and_then(|err| resync_target_error_detail(err.as_ref()));
st
}
@@ -2467,7 +2467,7 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli
Ok(_) => {}
Err(e) => {
let non_retryable = matches!(
&e,
e.as_ref(),
SdkError::ServiceError(service_err)
if is_retryable_delete_replication_head_error(
service_err.err().is_not_found(),
@@ -36,7 +36,8 @@ use time::OffsetDateTime;
use time::format_description::well_known::Rfc3339;
pub(crate) use crate::bucket::bucket_target_sys::{
AdvancedPutOptions, PutObjectOptions, PutObjectPartOptions, RemoveObjectOptions, TargetClient, resolve_read_api_version_id,
AdvancedPutOptions, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, RemoveObjectOptions, TargetClient,
resolve_read_api_version_id,
};
#[cfg(test)]
pub(crate) use crate::bucket::target::BucketTarget;
+5
View File
@@ -94,6 +94,9 @@ const DECOMMISSION_BUCKET_CONCURRENCY_DEFAULT_CAP: usize = 4;
const DECOMMISSION_TARGET_CAPACITY_OVERHEAD_PERCENT: usize = 30;
const DECOMMISSION_LISTING_MAX_ATTEMPTS: usize = 3;
const DECOMMISSION_LISTING_RETRY_DELAY: std::time::Duration = std::time::Duration::from_secs(5);
/// Background decommission walks must tolerate slow object migrations; the
/// stall timeout is the drive-health bound, not the total listing duration.
const DECOMMISSION_BACKGROUND_WALKDIR_STALL_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(60);
pub const POOL_META_NAME: &str = "pool.bin";
pub const POOL_META_FORMAT: u16 = 1;
@@ -5047,6 +5050,8 @@ impl SetDisks {
path: bucket_info.prefix.clone(),
recursive: true,
min_disks: listing_quorum,
skip_walkdir_total_timeout: true,
walkdir_stall_timeout: Some(DECOMMISSION_BACKGROUND_WALKDIR_STALL_TIMEOUT),
agreed: Some(Box::new(move |entry: MetaCacheEntry| Box::pin(cb1(entry)))),
partial: Some(Box::new(move |entries: MetaCacheEntries, _: &[Option<DiskError>]| {
let resolver = resolver.clone();
+136 -1
View File
@@ -23,7 +23,7 @@ use std::{
io,
path::{Component, Path, PathBuf},
sync::{Arc, LazyLock, Weak},
time::Instant,
time::{Duration, Instant},
};
use tokio::fs;
use tokio::sync::{
@@ -328,6 +328,9 @@ const ENV_DST_DIR_FSYNC_GROUP_COMMIT_ENABLE: &str = "RUSTFS_EXPERIMENTAL_DST_DIR
const DEFAULT_DST_DIR_FSYNC_GROUP_COMMIT_ENABLE: bool = false;
const ENV_FILE_FDATASYNC_GROUP_COMMIT_ENABLE: &str = "RUSTFS_EXPERIMENTAL_FILE_FDATASYNC_GROUP_COMMIT_ENABLE";
const DEFAULT_FILE_FDATASYNC_GROUP_COMMIT_ENABLE: bool = false;
const ENV_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS: &str = "RUSTFS_EXPERIMENTAL_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS";
const DEFAULT_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS: u64 = 0;
const MAX_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS: u64 = 1_000;
#[cfg(not(test))]
const MAX_DST_DIR_FSYNC_GROUPS: usize = 1024;
#[cfg(test)]
@@ -354,6 +357,16 @@ static DST_DIR_FSYNC_GROUP_COMMIT_ENABLED: LazyLock<bool> = LazyLock::new(|| {
static FILE_FDATASYNC_GROUP_COMMIT_ENABLED: LazyLock<bool> = LazyLock::new(|| {
rustfs_utils::get_env_bool(ENV_FILE_FDATASYNC_GROUP_COMMIT_ENABLE, DEFAULT_FILE_FDATASYNC_GROUP_COMMIT_ENABLE)
});
fn file_fdatasync_group_commit_wait_duration(wait_micros: u64) -> Duration {
Duration::from_micros(wait_micros.min(MAX_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS))
}
static FILE_FDATASYNC_GROUP_COMMIT_WAIT: LazyLock<Duration> = LazyLock::new(|| {
file_fdatasync_group_commit_wait_duration(rustfs_utils::get_env_u64(
ENV_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS,
DEFAULT_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS,
))
});
#[cfg(test)]
mod dst_dir_fsync_group_commit_override {
@@ -402,6 +415,7 @@ mod file_fdatasync_group_commit_override {
use std::sync::{Mutex, MutexGuard, PoisonError, RwLock};
static OVERRIDE: RwLock<Option<bool>> = RwLock::new(None);
static WAIT_OVERRIDE_MICROS: RwLock<Option<u64>> = RwLock::new(None);
static SERIAL: Mutex<()> = Mutex::new(());
pub(crate) fn get() -> Option<bool> {
@@ -415,6 +429,7 @@ mod file_fdatasync_group_commit_override {
impl Drop for OverrideGuard {
fn drop(&mut self) {
*OVERRIDE.write().unwrap_or_else(PoisonError::into_inner) = None;
*WAIT_OVERRIDE_MICROS.write().unwrap_or_else(PoisonError::into_inner) = None;
}
}
@@ -423,6 +438,14 @@ mod file_fdatasync_group_commit_override {
*OVERRIDE.write().unwrap_or_else(PoisonError::into_inner) = Some(enabled);
OverrideGuard { _serial: serial }
}
pub(crate) fn set_wait_micros(wait_micros: u64) {
*WAIT_OVERRIDE_MICROS.write().unwrap_or_else(PoisonError::into_inner) = Some(wait_micros);
}
pub(crate) fn wait_micros() -> Option<u64> {
*WAIT_OVERRIDE_MICROS.read().unwrap_or_else(PoisonError::into_inner)
}
}
#[cfg(test)]
@@ -430,6 +453,11 @@ pub(crate) fn set_file_fdatasync_group_commit_for_test(enabled: bool) -> file_fd
file_fdatasync_group_commit_override::set(enabled)
}
#[cfg(test)]
fn set_file_fdatasync_group_commit_wait_for_test(wait_micros: u64) {
file_fdatasync_group_commit_override::set_wait_micros(wait_micros);
}
fn file_fdatasync_group_commit_enabled() -> bool {
#[cfg(test)]
if let Some(enabled) = file_fdatasync_group_commit_override::get() {
@@ -439,6 +467,15 @@ fn file_fdatasync_group_commit_enabled() -> bool {
*FILE_FDATASYNC_GROUP_COMMIT_ENABLED
}
fn file_fdatasync_group_commit_wait() -> Duration {
#[cfg(test)]
if let Some(wait_micros) = file_fdatasync_group_commit_override::wait_micros() {
return file_fdatasync_group_commit_wait_duration(wait_micros);
}
*FILE_FDATASYNC_GROUP_COMMIT_WAIT
}
#[derive(Clone, Eq, Hash, PartialEq)]
struct DstDirFsyncGroupKey {
canonical_path: PathBuf,
@@ -934,6 +971,10 @@ async fn run_file_fdatasync_group_worker(group: Arc<FileFdatasyncGroup>) {
#[cfg(test)]
file_sync_probe::run_before_group_batch();
tokio::task::yield_now().await;
let wait = file_fdatasync_group_commit_wait();
if !wait.is_zero() {
tokio::time::sleep(wait).await;
}
let (batch, batch_file_count): (Vec<FileFdatasyncWaiter>, usize) = {
let mut group_state = group.inner.lock();
let batch_file_count = group_state.pending_files;
@@ -6075,6 +6116,7 @@ mod tests {
use std::sync::mpsc;
let _group_commit = set_file_fdatasync_group_commit_for_test(true);
set_file_fdatasync_group_commit_wait_for_test(0);
clear_file_fdatasync_group_commit_for_test();
let temp_dir = tempdir().expect("create temp dir");
let first_dir = temp_dir.path().join("first");
@@ -6141,12 +6183,105 @@ mod tests {
assert_eq!(file_fdatasync_group_commit_counts_for_test(), (0, 0, 0));
}
#[test]
fn file_fdatasync_group_commit_wait_duration_uses_default_and_cap() {
assert_eq!(DEFAULT_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS, 0);
assert_eq!(
file_fdatasync_group_commit_wait_duration(DEFAULT_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS),
Duration::ZERO
);
assert_eq!(file_fdatasync_group_commit_wait_duration(250), Duration::from_micros(250));
assert_eq!(
file_fdatasync_group_commit_wait_duration(MAX_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS),
Duration::from_micros(MAX_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS)
);
assert_eq!(
file_fdatasync_group_commit_wait_duration(u64::MAX),
Duration::from_micros(MAX_FILE_FDATASYNC_GROUP_COMMIT_WAIT_MICROS)
);
}
#[tokio::test(flavor = "current_thread", start_paused = true)]
#[serial_test::serial(file_sync_probe)]
async fn file_fdatasync_group_commit_wait_budget_batches_late_follower() {
use std::sync::mpsc;
let _group_commit = set_file_fdatasync_group_commit_for_test(true);
let wait_budget_micros = 1_000;
let wait_budget = file_fdatasync_group_commit_wait_duration(wait_budget_micros);
set_file_fdatasync_group_commit_wait_for_test(wait_budget_micros);
clear_file_fdatasync_group_commit_for_test();
let temp_dir = tempdir().expect("create temp dir");
let first_dir = temp_dir.path().join("first");
let second_dir = temp_dir.path().join("second");
std::fs::create_dir(&first_dir).expect("create first dir");
std::fs::create_dir(&second_dir).expect("create second dir");
std::fs::write(first_dir.join("part.1"), b"first").expect("write first part");
std::fs::write(second_dir.join("part.1"), b"second").expect("write second part");
let _probe = file_sync_probe::set_blocking(temp_dir.path());
let (entered_tx, entered_rx) = mpsc::channel();
file_sync_probe::set_before_group_batch(move || {
entered_tx.send(()).expect("signal first file fdatasync group worker");
});
let limiter = file_sync_limiter();
let first_limiter = limiter.clone();
let first_path = first_dir.clone();
let first = tokio::spawn(async move { sync_dir_files_with_limiter(first_path, first_limiter).await });
tokio::task::spawn_blocking(move || entered_rx.recv_timeout(Duration::from_secs(30)))
.await
.expect("group worker hook waiter should run")
.expect("first file fdatasync group worker should start");
let second_limiter = limiter.clone();
let second_path = second_dir.clone();
let second = tokio::spawn(async move { sync_dir_files_with_limiter(second_path, second_limiter).await });
tokio::time::timeout(Duration::from_secs(30), async {
loop {
if file_fdatasync_group_commit_counts_for_test().1 == 2 {
return;
}
tokio::task::yield_now().await;
}
})
.await
.expect("second waiter should enqueue during the configured wait budget");
tokio::time::advance(wait_budget).await;
tokio::task::yield_now().await;
file_sync_probe::wait_for_active(1).await;
assert_eq!(
file_sync_probe::group_batches(),
vec![2],
"configured wait budget should let a follower join the leader's batch"
);
file_sync_probe::release();
first
.await
.expect("join first wait-budget file sync")
.expect("first wait-budget file sync must succeed");
second
.await
.expect("join second wait-budget file sync")
.expect("second wait-budget file sync must succeed");
assert!(
fsync_dir_recorder::was_fsynced(&first_dir),
"first source directory must still be fsynced"
);
assert!(
fsync_dir_recorder::was_fsynced(&second_dir),
"second source directory must still be fsynced"
);
assert_eq!(file_fdatasync_group_commit_counts_for_test(), (0, 0, 0));
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
#[serial_test::serial(file_sync_probe)]
async fn file_fdatasync_group_commit_failure_fails_all_waiters_before_dir_fsync() {
use std::sync::mpsc;
let _group_commit = set_file_fdatasync_group_commit_for_test(true);
set_file_fdatasync_group_commit_wait_for_test(0);
clear_file_fdatasync_group_commit_for_test();
let temp_dir = tempdir().expect("create temp dir");
let first_dir = temp_dir.path().join("first");
+23 -2
View File
@@ -719,14 +719,23 @@ impl ObjectInfo {
}
pub fn from_file_info(fi: &FileInfo, bucket: &str, object: &str, versioned: bool) -> ObjectInfo {
let name = decode_dir_object(object);
let mut version_id = fi.version_id;
if versioned && version_id.is_none() {
version_id = Some(Uuid::nil())
}
Self::from_file_info_with_version_id(fi, bucket, object, version_id)
}
pub(crate) fn from_file_info_with_version_id(
fi: &FileInfo,
bucket: &str,
object: &str,
version_id: Option<Uuid>,
) -> ObjectInfo {
let name = decode_dir_object(object);
// etag
let (content_type, content_encoding, etag) = {
let content_type = fi.metadata.get("content-type").cloned();
@@ -1640,6 +1649,18 @@ mod tests {
assert_eq!(info.replication_decision, "arn=true;false;arn:replication::1:dest;rule-id");
}
#[test]
fn from_file_info_with_version_id_keeps_normalized_absent_version() {
let fi = FileInfo {
version_id: Some(Uuid::new_v4()),
..Default::default()
};
let info = ObjectInfo::from_file_info_with_version_id(&fi, "bucket", "object", None);
assert_eq!(info.version_id, None, "a normalized absent version must not be rewritten to nil");
}
#[test]
fn from_file_info_reports_effective_storage_class_for_legacy_metadata() {
for legacy_label in [
@@ -657,7 +657,7 @@ where
prefix,
marker,
None,
i32::try_from(limit).map_or(i32::MAX, |value| value),
i32::try_from(limit).unwrap_or(i32::MAX),
false,
None,
false,
+485 -25
View File
@@ -922,14 +922,10 @@ mod prepared_get_object_metadata_tests {
.expect("test should find an object whose initial fanout covers both data shards")
}
#[allow(
dead_code,
reason = "test fixture no assertion in this module uses today; the live namesake lives in io_primitives tests (backlog#1823)"
)]
fn bounded_spare_disk_index(bucket: &str, object: &str) -> usize {
fn bounded_initial_parity_disk_index(bucket: &str, object: &str) -> usize {
*bounded_metadata_fanout_order(bucket, object, 4, 2)
.get(3)
.expect("4-disk test geometry should leave one bounded spare disk")
.get(2)
.expect("4-disk test geometry should schedule one parity disk initially")
}
#[tokio::test]
@@ -1087,7 +1083,7 @@ mod prepared_get_object_metadata_tests {
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", None::<&str>),
],
async {
let slow_parity_disk = bounded_spare_disk_index(bucket, &object);
let slow_parity_disk = bounded_initial_parity_disk_index(bucket, &object);
let barrier =
rename_fanout_barrier::arm(&object, slow_parity_disk, rename_fanout_barrier::PHASE_READ_VERSION);
let calls = disk_call_counters::observe(&object);
@@ -1501,6 +1497,102 @@ pub fn get_lock_acquire_timeout() -> Duration {
}
}
fn get_put_object_commit_lock_acquire_timeout_override_ms() -> u64 {
#[cfg(test)]
{
rustfs_utils::get_env_u64(
rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS,
rustfs_config::DEFAULT_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS,
)
}
#[cfg(not(test))]
{
static CACHED: OnceLock<u64> = OnceLock::new();
*CACHED.get_or_init(|| {
rustfs_utils::get_env_u64(
rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS,
rustfs_config::DEFAULT_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS,
)
})
}
}
fn get_put_object_commit_lock_acquire_timeout(op: &'static str) -> Duration {
let default_timeout = get_lock_acquire_timeout();
if op != "put_object_commit" {
return default_timeout;
}
let timeout_ms = get_put_object_commit_lock_acquire_timeout_override_ms();
if timeout_ms == 0 {
default_timeout
} else {
Duration::from_millis(timeout_ms)
}
}
fn put_object_commit_lock_timeout_override_enabled(op: &'static str) -> bool {
op == "put_object_commit" && get_put_object_commit_lock_acquire_timeout_override_ms() != 0
}
fn put_object_commit_lock_admission_budget_label() -> &'static str {
match get_put_object_commit_lock_acquire_timeout_override_ms() {
0 => rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_DISABLED,
1..=250 => rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
251..=500 => rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS,
501..=1000 => rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_1000MS,
_ => rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_GT_1000MS,
}
}
fn record_put_object_commit_lock_admission(op: &'static str, outcome: &'static str) {
if op != "put_object_commit" || !rustfs_io_metrics::put_stage_metrics_enabled() {
return;
}
rustfs_io_metrics::record_put_object_commit_lock_admission(put_object_commit_lock_admission_budget_label(), outcome);
}
fn put_object_commit_lock_acquire_error_outcome(op: &'static str, err: &rustfs_lock::error::LockError) -> &'static str {
if put_object_commit_lock_timeout_override_enabled(op) && matches!(err, rustfs_lock::error::LockError::Timeout { .. }) {
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN
} else {
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_LOCK_ERROR
}
}
fn resolve_put_object_commit_lock_acquire_result(
set: &SetDisks,
op: &'static str,
bucket: &str,
object: &str,
result: std::result::Result<rustfs_lock::namespace::NamespaceLockGuard, rustfs_lock::error::LockError>,
) -> Result<rustfs_lock::namespace::NamespaceLockGuard> {
match result {
Ok(guard) => {
record_put_object_commit_lock_admission(op, rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED);
Ok(guard)
}
Err(err) => {
record_put_object_commit_lock_admission(op, put_object_commit_lock_acquire_error_outcome(op, &err));
Err(map_put_object_commit_lock_acquire_error(set, op, bucket, object, err))
}
}
}
fn map_put_object_commit_lock_acquire_error(
set: &SetDisks,
op: &'static str,
bucket: &str,
object: &str,
err: rustfs_lock::error::LockError,
) -> StorageError {
if put_object_commit_lock_timeout_override_enabled(op) && matches!(err, rustfs_lock::error::LockError::Timeout { .. }) {
StorageError::SlowDown
} else {
set.map_namespace_lock_error(bucket, object, "write", err)
}
}
pub fn is_object_lock_diag_enabled() -> bool {
*OBJECT_LOCK_DIAG_ENABLED.get_or_init(|| {
let enabled = rustfs_utils::get_env_bool(
@@ -3302,10 +3394,14 @@ impl SetDisks {
let diag_enabled = is_object_lock_diag_enabled();
let ns_lock = self.new_ns_lock(bucket, object).await?;
let acquire_start = Instant::now();
let guard = ns_lock
.get_write_lock(get_lock_acquire_timeout())
.await
.map_err(|e| self.map_namespace_lock_error(bucket, object, "write", e))?;
let acquire_timeout = get_put_object_commit_lock_acquire_timeout(op);
let guard = resolve_put_object_commit_lock_acquire_result(
self,
op,
bucket,
object,
ns_lock.get_write_lock(acquire_timeout).await,
)?;
Self::record_put_object_commit_namespace_lock_wait(op, acquire_start);
let owner = diag_enabled.then(|| ns_lock.owner().to_string());
self.log_object_lock_acquire_if_slow(
@@ -3340,20 +3436,26 @@ impl SetDisks {
let diag_enabled = is_object_lock_diag_enabled();
let ns_lock = self.new_ns_lock(bucket, object).await?;
let acquire_start = Instant::now();
let acquire = ns_lock.get_write_lock(get_lock_acquire_timeout());
let acquire_timeout = get_put_object_commit_lock_acquire_timeout(op);
let acquire = ns_lock.get_write_lock(acquire_timeout);
tokio::pin!(acquire);
let mut on_pending = Some(on_pending);
let guard = futures::future::poll_fn(|cx| match std::future::Future::poll(acquire.as_mut(), cx) {
std::task::Poll::Pending => {
if let Some(on_pending) = on_pending.take() {
on_pending();
let guard = resolve_put_object_commit_lock_acquire_result(
self,
op,
bucket,
object,
futures::future::poll_fn(|cx| match std::future::Future::poll(acquire.as_mut(), cx) {
std::task::Poll::Pending => {
if let Some(on_pending) = on_pending.take() {
on_pending();
}
std::task::Poll::Pending
}
std::task::Poll::Pending
}
std::task::Poll::Ready(result) => std::task::Poll::Ready(result),
})
.await
.map_err(|e| self.map_namespace_lock_error(bucket, object, "write", e))?;
std::task::Poll::Ready(result) => std::task::Poll::Ready(result),
})
.await,
)?;
Self::record_put_object_commit_namespace_lock_wait(op, acquire_start);
let owner = diag_enabled.then(|| ns_lock.owner().to_string());
self.log_object_lock_acquire_if_slow(
@@ -5717,8 +5819,8 @@ mod tests {
.filter(|(composite, _, _, _)| {
composite.key().name() == "rustfs_s3_put_object_stage_duration_ms"
&& composite.key().labels().any(|label| {
label.key().to_string() == "stage"
&& label.value().to_string() == rustfs_io_metrics::PUT_STAGE_PUT_OBJECT_COMMIT_NAMESPACE_LOCK_WAIT
label.key() == "stage"
&& label.value() == rustfs_io_metrics::PUT_STAGE_PUT_OBJECT_COMMIT_NAMESPACE_LOCK_WAIT
})
})
.map(|(_, _, _, value)| match value {
@@ -5728,6 +5830,81 @@ mod tests {
.sum()
}
fn put_object_commit_lock_admission_count(
rows: &[(
metrics_util::CompositeKey,
Option<metrics::Unit>,
Option<metrics::SharedString>,
DebugValue,
)],
budget: &'static str,
outcome: &'static str,
) -> u64 {
rows.iter()
.filter(|(composite, _, _, _)| {
composite.key().name() == "rustfs_s3_put_object_commit_namespace_lock_admission_total"
&& composite
.key()
.labels()
.any(|label| label.key() == "budget" && label.value() == budget)
&& composite
.key()
.labels()
.any(|label| label.key() == "outcome" && label.value() == outcome)
})
.map(|(_, _, _, value)| match value {
DebugValue::Counter(count) => *count,
_ => 0,
})
.sum()
}
#[test]
#[serial]
fn put_object_commit_lock_admission_budget_labels_are_bounded() {
let cases = [
("0", rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_DISABLED),
("250", rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS),
("251", rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS),
("500", rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS),
("501", rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_1000MS),
("1000", rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_1000MS),
("1001", rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_GT_1000MS),
];
for (timeout_ms, expected) in cases {
temp_env::with_vars(
[(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some(timeout_ms))],
|| {
assert_eq!(put_object_commit_lock_admission_budget_label(), expected);
},
);
}
}
#[test]
#[serial]
fn put_object_commit_lock_admission_error_outcomes_are_bounded() {
let timeout = LockError::timeout("bucket/object", Duration::from_millis(1));
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("1"))], || {
assert_eq!(
put_object_commit_lock_acquire_error_outcome("put_object_commit", &timeout),
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN
);
assert_eq!(
put_object_commit_lock_acquire_error_outcome("complete_multipart_upload_commit", &timeout),
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_LOCK_ERROR
);
});
let internal = LockError::internal("simulated lock manager error");
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("1"))], || {
assert_eq!(
put_object_commit_lock_acquire_error_outcome("put_object_commit", &internal),
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_LOCK_ERROR
);
});
}
#[test]
#[serial]
fn put_object_commit_namespace_lock_wait_metric_is_wired_to_both_write_lock_paths() {
@@ -5793,6 +5970,289 @@ mod tests {
});
}
#[test]
#[serial]
fn put_object_commit_lock_timeout_override_only_applies_to_put_commit() {
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("17"))], || {
assert_eq!(get_put_object_commit_lock_acquire_timeout("put_object_commit"), Duration::from_millis(17));
assert_eq!(
get_put_object_commit_lock_acquire_timeout("complete_multipart_upload_commit"),
get_lock_acquire_timeout()
);
});
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("0"))], || {
assert_eq!(
get_put_object_commit_lock_acquire_timeout("put_object_commit"),
get_lock_acquire_timeout()
);
});
}
#[test]
#[serial]
fn put_object_commit_lock_timeout_override_bounds_contention_wait() {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("test runtime should start");
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("1"))], || {
runtime.block_on(async {
let ctx = Arc::new(InstanceContext::new());
ctx.update_erasure_type(SetupType::Erasure).await;
let set = make_test_set_disks_with_ctx(Vec::new(), ctx).await;
let bucket = "bucket";
let object = "object";
let held_guard = set
.acquire_write_lock_diag("put_object_commit", bucket, object)
.await
.expect("holder acquire should succeed");
let started = Instant::now();
let err = match set.acquire_write_lock_diag("put_object_commit", bucket, object).await {
Ok(_) => panic!("contended PUT commit lock should honor the short timeout"),
Err(err) => err,
};
assert!(
started.elapsed() < Duration::from_secs(1),
"short PUT commit lock timeout should not wait for the global timeout"
);
assert!(matches!(err, StorageError::SlowDown));
drop(held_guard);
set.acquire_write_lock_diag("put_object_commit", bucket, object)
.await
.expect("permit should not leak after timeout");
});
});
}
#[test]
#[serial]
fn put_object_commit_lock_admission_records_acquired_and_timeout() {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("test runtime should start");
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("1"))], || {
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
metrics::with_local_recorder(&recorder, || {
rustfs_io_metrics::set_put_stage_metrics_enabled(true);
runtime.block_on(async {
let ctx = Arc::new(InstanceContext::new());
ctx.update_erasure_type(SetupType::Erasure).await;
let set = make_test_set_disks_with_ctx(Vec::new(), ctx).await;
let held_guard = set
.acquire_write_lock_diag("put_object_commit", "bucket", "object")
.await
.expect("holder acquire should succeed");
let err = match set.acquire_write_lock_diag("put_object_commit", "bucket", "object").await {
Ok(_) => panic!("contended PUT commit acquire should return SlowDown"),
Err(err) => err,
};
assert!(matches!(err, StorageError::SlowDown));
drop(held_guard);
rustfs_io_metrics::set_put_stage_metrics_enabled(false);
});
});
let rows = snapshotter.snapshot().into_vec();
assert_eq!(
put_object_commit_lock_admission_count(
&rows,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED,
),
1
);
assert_eq!(
put_object_commit_lock_admission_count(
&rows,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN,
),
1
);
});
}
#[test]
#[serial]
fn put_object_commit_lock_admission_records_disabled_budget_acquired() {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("test runtime should start");
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("0"))], || {
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
metrics::with_local_recorder(&recorder, || {
rustfs_io_metrics::set_put_stage_metrics_enabled(true);
runtime.block_on(async {
let ctx = Arc::new(InstanceContext::new());
ctx.update_erasure_type(SetupType::Erasure).await;
let set = make_test_set_disks_with_ctx(Vec::new(), ctx).await;
let guard = set
.acquire_write_lock_diag("put_object_commit", "bucket", "object")
.await
.expect("PUT commit acquire should succeed with default timeout");
drop(guard);
rustfs_io_metrics::set_put_stage_metrics_enabled(false);
});
});
let rows = snapshotter.snapshot().into_vec();
assert_eq!(
put_object_commit_lock_admission_count(
&rows,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_DISABLED,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED,
),
1
);
});
}
#[test]
#[serial]
fn put_object_commit_lock_admission_skips_non_put_commit_ops() {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("test runtime should start");
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("250"))], || {
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
metrics::with_local_recorder(&recorder, || {
rustfs_io_metrics::set_put_stage_metrics_enabled(true);
runtime.block_on(async {
let ctx = Arc::new(InstanceContext::new());
ctx.update_erasure_type(SetupType::Erasure).await;
let set = make_test_set_disks_with_ctx(Vec::new(), ctx).await;
let guard = set
.acquire_write_lock_diag("complete_multipart_upload_commit", "bucket", "object")
.await
.expect("non-PUT commit acquire should succeed");
drop(guard);
rustfs_io_metrics::set_put_stage_metrics_enabled(false);
});
});
let rows = snapshotter.snapshot().into_vec();
assert_eq!(
rows.iter()
.filter(|(composite, _, _, _)| {
composite.key().name() == "rustfs_s3_put_object_commit_namespace_lock_admission_total"
})
.count(),
0
);
});
}
#[test]
#[serial]
fn put_object_commit_lock_admission_records_lock_error() {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("test runtime should start");
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("250"))], || {
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
metrics::with_local_recorder(&recorder, || {
rustfs_io_metrics::set_put_stage_metrics_enabled(true);
runtime.block_on(async {
let healthy: Arc<dyn LockClient> =
Arc::new(LocalClient::with_manager(Arc::new(rustfs_lock::GlobalLockManager::new())));
let failing: Arc<dyn LockClient> = Arc::new(FailingClient);
let ctx = Arc::new(InstanceContext::new());
ctx.update_erasure_type(SetupType::DistErasure).await;
let set = make_test_set_disks_with_ctx(vec![healthy, failing], ctx).await;
assert!(
set.acquire_write_lock_diag("put_object_commit", "bucket", "object")
.await
.is_err(),
"one healthy locker must not satisfy the PUT commit write quorum"
);
rustfs_io_metrics::set_put_stage_metrics_enabled(false);
});
});
let rows = snapshotter.snapshot().into_vec();
assert_eq!(
put_object_commit_lock_admission_count(
&rows,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_LOCK_ERROR,
),
1
);
assert_eq!(
put_object_commit_lock_admission_count(
&rows,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN,
),
0
);
});
}
#[test]
#[serial]
fn put_object_commit_lock_admission_records_pending_hook_acquired() {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.expect("test runtime should start");
temp_env::with_vars([(rustfs_config::ENV_PUT_COMMIT_NAMESPACE_LOCK_ACQUIRE_TIMEOUT_MS, Some("500"))], || {
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
metrics::with_local_recorder(&recorder, || {
rustfs_io_metrics::set_put_stage_metrics_enabled(true);
runtime.block_on(async {
let ctx = Arc::new(InstanceContext::new());
ctx.update_erasure_type(SetupType::Erasure).await;
let set = make_test_set_disks_with_ctx(Vec::new(), ctx).await;
let held_guard = set
.acquire_write_lock_diag("put_object_commit", "bucket", "object")
.await
.expect("holder acquire should succeed");
let (pending_tx, pending_rx) = tokio::sync::oneshot::channel();
let pending_acquire =
set.acquire_write_lock_diag_with_pending_hook("put_object_commit", "bucket", "object", move || {
let _ = pending_tx.send(());
});
let release_holder = async {
pending_rx.await.expect("pending hook should fire");
drop(held_guard);
};
let (pending_guard, ()) = tokio::join!(pending_acquire, release_holder);
drop(pending_guard.expect("pending-hook PUT commit acquire should succeed"));
rustfs_io_metrics::set_put_stage_metrics_enabled(false);
});
});
let rows = snapshotter.snapshot().into_vec();
assert_eq!(
put_object_commit_lock_admission_count(
&rows,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS,
rustfs_io_metrics::PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED,
),
2
);
});
}
#[tokio::test]
async fn new_ns_lock_shares_clients_without_changing_quorum() {
let healthy: Arc<dyn LockClient> = Arc::new(LocalClient::with_manager(Arc::new(rustfs_lock::GlobalLockManager::new())));
+2 -16
View File
@@ -124,14 +124,7 @@ impl HealWalkCollector {
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
let version_uuid = fi.version_id.filter(|version_id| !version_id.is_nil());
let lifecycle_object_info = if self.include_lifecycle_object_info {
let mut lifecycle_fi = fi.clone();
lifecycle_fi.version_id = version_uuid;
Some(ObjectInfo::from_file_info(
&lifecycle_fi,
&self.bucket,
&entry.name,
version_uuid.is_some(),
))
Some(ObjectInfo::from_file_info_with_version_id(fi, &self.bucket, &entry.name, version_uuid))
} else {
None
};
@@ -198,14 +191,7 @@ impl HealWalkCollector {
let vid = version_uuid.map(|u| u.to_string());
if seen.insert(vid.clone()) {
let lifecycle_object_info = if self.include_lifecycle_object_info {
let mut lifecycle_fi = fi.clone();
lifecycle_fi.version_id = version_uuid;
Some(ObjectInfo::from_file_info(
&lifecycle_fi,
&self.bucket,
&entry.name,
version_uuid.is_some(),
))
Some(ObjectInfo::from_file_info_with_version_id(fi, &self.bucket, &entry.name, version_uuid))
} else {
None
};
+242 -14
View File
@@ -1322,7 +1322,12 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
let object_info = prepared_object_info
.unwrap_or_else(|| build_get_object_info(fi, bucket, object, opts.versioned || opts.version_suspended));
let object_class = classify_get_codec_streaming_object_class(&range, &object_info, fi);
let size_bucket = rustfs_io_metrics::get_object_size_bucket(object_info.size);
let metrics_size = if stage_metrics_enabled {
object_info.get_actual_size().unwrap_or(object_info.size)
} else {
object_info.size
};
let size_bucket = rustfs_io_metrics::get_object_size_bucket(metrics_size);
record_get_stage_duration_if_enabled(GET_OBJECT_PATH_SET_DISK, GET_STAGE_OBJECT_INFO, object_info_stage_start);
let metadata_elapsed = metadata_stage_start.elapsed().as_secs_f64();
rustfs_io_metrics::record_get_object_metadata_phase_duration(metadata_elapsed);
@@ -3766,7 +3771,7 @@ pub(crate) async fn complete_transition_upload<Remote, Producer>(
producer: Producer,
expected_size: u64,
consumed: Arc<AtomicU64>,
) -> std::result::Result<TransitionUploadCompletion, TransitionUploadFailure>
) -> std::result::Result<TransitionUploadCompletion, Box<TransitionUploadFailure>>
where
Remote: Future<Output = std::result::Result<String, std::io::Error>>,
Producer: Future<Output = Result<u64>>,
@@ -3784,23 +3789,23 @@ where
Err(_) => StorageError::Unexpected,
Ok(Ok(_)) => StorageError::Io(remote_error),
};
return Err(TransitionUploadFailure { error, candidate: None });
return Err(Box::new(TransitionUploadFailure { error, candidate: None }));
}
};
let candidate = TransitionUploadCandidate::from_put_response(remote_version);
let produced = match producer_result {
Ok(Ok(produced)) => produced,
Ok(Err(error)) => {
return Err(TransitionUploadFailure {
return Err(Box::new(TransitionUploadFailure {
error,
candidate: Some(candidate),
});
}));
}
Err(_) => {
return Err(TransitionUploadFailure {
return Err(Box::new(TransitionUploadFailure {
error: StorageError::Unexpected,
candidate: Some(candidate),
});
}));
}
};
let consumed = consumed.load(Ordering::Acquire);
@@ -3810,10 +3815,10 @@ where
} else {
StorageError::MoreData
};
return Err(TransitionUploadFailure {
return Err(Box::new(TransitionUploadFailure {
error,
candidate: Some(candidate),
});
}));
}
Ok(TransitionUploadCompletion {
candidate,
@@ -7284,7 +7289,7 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks {
}
let gr = gr?;
let reader = BufReader::new(gr.stream);
let hash_reader = HashReader::from_stream(reader, gr.object_info.size, gr.object_info.size, None, None, false)?;
let hash_reader = HashReader::from_stream(reader, gr.object_info.size, oi.get_actual_size()?, None, None, false)?;
let mut p_reader = PutObjReader::new(hash_reader);
return match self_.clone().put_object(bucket, object, &mut p_reader, &ropts).await {
Ok(restored_info) => {
@@ -8826,7 +8831,7 @@ mod transition_commit_failure_tests {
use s3s::dto::RestoreRequest;
use tokio::io::{AsyncReadExt, AsyncWriteExt};
fn restore_operation_id_metadata(operation_id: Uuid) -> HashMap<String, String> {
pub(super) fn restore_operation_id_metadata(operation_id: Uuid) -> HashMap<String, String> {
let mut metadata = HashMap::new();
rustfs_utils::http::metadata_compat::insert_str(
&mut metadata,
@@ -8836,7 +8841,7 @@ mod transition_commit_failure_tests {
metadata
}
fn restore_metadata(operation_id: Uuid, ongoing: bool) -> HashMap<String, String> {
pub(super) fn restore_metadata(operation_id: Uuid, ongoing: bool) -> HashMap<String, String> {
let mut metadata = restore_operation_id_metadata(operation_id);
metadata.insert(s3s::header::X_AMZ_RESTORE.as_str().to_string(), format!("ongoing-request=\"{ongoing}\""));
metadata
@@ -10097,6 +10102,51 @@ mod transition_commit_failure_tests {
.await
.expect("operation B should replace operation A before final commit");
let mismatch = set_disks
.finalize_restore_metadata(
bucket,
object,
&set_disks
.get_object_info(bucket, object, &ObjectOptions::default())
.await
.expect("operation B metadata should be readable"),
&ObjectOptions {
user_defined: restore_operation_id_metadata(operation_a),
..Default::default()
},
)
.await
.expect_err("operation A must not finalize operation B metadata");
assert!(matches!(
mismatch,
Error::Io(ref error)
if error.kind() == std::io::ErrorKind::Other
&& error.to_string() == "restore operation id changed before metadata finalization"
));
let current = set_disks
.get_object_info(bucket, object, &ObjectOptions::default())
.await
.expect("operation B metadata should remain after mismatched finalization");
assert_eq!(
rustfs_utils::http::metadata_compat::get_consistent_str(
current.user_defined.as_ref(),
rustfs_utils::http::metadata_compat::SUFFIX_RESTORE_OPERATION_ID,
),
Some(operation_b.to_string().as_str()),
"mismatched finalization must not remove operation B"
);
assert!(
parse_restore_obj_status(
current
.user_defined
.get(s3s::header::X_AMZ_RESTORE.as_str())
.expect("operation B restore header should remain pending"),
)
.expect("operation B restore header should parse")
.on_going(),
"mismatched finalization must not publish restore completion"
);
let mut stale_restore_reader = PutObjReader::from_vec(b"stale A restored body".repeat(1024));
let result = set_disks
.put_object(
@@ -10126,18 +10176,37 @@ mod transition_commit_failure_tests {
let mut matching_restore_reader = PutObjReader::from_vec(b"matching B restored body".repeat(1024));
let operation_b_restore_metadata = restore_metadata(operation_b, false);
set_disks
let restored = set_disks
.put_object(
bucket,
object,
&mut matching_restore_reader,
&ObjectOptions {
user_defined: operation_b_restore_metadata,
user_defined: operation_b_restore_metadata.clone(),
..Default::default()
},
)
.await
.expect("matching operation B should be allowed to commit");
set_disks
.finalize_restore_metadata(
bucket,
object,
&restored,
&ObjectOptions {
user_defined: restore_operation_id_metadata(operation_b),
transition: TransitionOptions {
restore_request: RestoreRequest {
days: Some(1),
..Default::default()
},
..Default::default()
},
..Default::default()
},
)
.await
.expect("matching operation B should finalize after its commit consumes the operation id");
let restored = set_disks
.get_object_info(bucket, object, &ObjectOptions::default())
.await
@@ -10503,13 +10572,16 @@ mod transition_commit_failure_tests {
#[cfg(all(test, feature = "test-util"))]
mod transition_upload_integrity_tests {
use super::hermetic_set_disks_support::{hermetic_set_disks, hermetic_set_disks_with_lockers};
use super::transition_commit_failure_tests::{restore_metadata, restore_operation_id_metadata};
use super::*;
use crate::bucket::lifecycle::lifecycle::{TRANSITION_PENDING, TransitionOptions};
use crate::disk::DiskAPI as _;
use crate::layout::endpoints::SetupType;
use crate::services::tier::test_util::register_mock_tier;
use crate::set_disk::replication::RestoreFinalizeBarrier;
use crate::storage_api_contracts::object::{ObjectIO as _, ObjectOperations as _};
use http::HeaderMap;
use rustfs_filemeta::RestoreStatusOps as _;
use rustfs_lock::client::local::LocalClient;
use rustfs_lock::{LockClient, LockError, LockId, LockInfo, LockRequest, LockResponse, LockStats};
use std::collections::HashSet;
@@ -10655,6 +10727,162 @@ mod transition_upload_integrity_tests {
}
}
async fn write_committed_restore(
set_disks: &Arc<SetDisks>,
disk_stores: &[DiskStore],
bucket: &str,
object: &str,
operation_id: Uuid,
) -> ObjectInfo {
for disk in disk_stores {
disk.make_volume(bucket).await.expect("bucket volume should be created");
}
let mut source = PutObjReader::from_vec(b"restore source body".repeat(1024));
set_disks
.put_object(bucket, object, &mut source, &ObjectOptions::default())
.await
.expect("source object should be written");
set_disks
.put_object_metadata(
bucket,
object,
&ObjectOptions {
eval_metadata: Some(restore_metadata(operation_id, true)),
..Default::default()
},
)
.await
.expect("pending restore metadata should be installed");
let mut restored_reader = PutObjReader::from_vec(b"restored body".repeat(1024));
set_disks
.put_object(
bucket,
object,
&mut restored_reader,
&ObjectOptions {
user_defined: restore_metadata(operation_id, true),
..Default::default()
},
)
.await
.expect("matching restore commit should consume its operation id")
}
fn restore_finalize_options(operation_id: Uuid) -> ObjectOptions {
ObjectOptions {
user_defined: restore_operation_id_metadata(operation_id),
transition: TransitionOptions {
restore_request: s3s::dto::RestoreRequest {
days: Some(1),
..Default::default()
},
..Default::default()
},
..Default::default()
}
}
async fn assert_committed_restore_remains_pending(set_disks: &Arc<SetDisks>, bucket: &str, object: &str) {
let current = set_disks
.get_object_info(bucket, object, &ObjectOptions::default())
.await
.expect("pending restore metadata should remain readable");
assert!(
restore_operation_id_from_metadata(current.user_defined.as_ref())
.expect("operation id metadata should parse")
.is_none(),
"successful restore commit must have consumed the operation id"
);
assert!(
rustfs_filemeta::parse_restore_obj_status(
current
.user_defined
.get(s3s::header::X_AMZ_RESTORE.as_str())
.expect("pending restore header should remain"),
)
.expect("restore header should parse")
.on_going(),
"failed finalization must not publish completion metadata"
);
}
#[tokio::test(flavor = "current_thread", start_paused = true)]
#[serial_test::serial]
async fn restore_finalize_rejects_acquired_lock_loss_after_commit() {
let refresh_calls = Arc::new(AtomicUsize::new(0));
let lockers: Vec<Arc<dyn LockClient>> = (0..4)
.map(|_| Arc::new(LockLostRefreshClient::new(Arc::clone(&refresh_calls))) as Arc<dyn LockClient>)
.collect();
let (_temp_dirs, disk_stores, set_disks) = hermetic_set_disks_with_lockers(4, 0, 2, lockers).await;
let bucket = "restore-finalize-acquired-lock-lost-bucket";
let object = "object.bin";
let operation_id = Uuid::new_v4();
let restored = write_committed_restore(&set_disks, &disk_stores, bucket, object, operation_id).await;
let _setup_type_guard = SetupTypeGuard::switch_to(SetupType::DistErasure).await;
let barrier = RestoreFinalizeBarrier::install(bucket, object);
let finalize_set = Arc::clone(&set_disks);
let finalize = tokio::spawn(async move {
finalize_set
.finalize_restore_metadata(bucket, object, &restored, &restore_finalize_options(operation_id))
.await
});
barrier.wait_until_paused().await;
tokio::time::advance(Duration::from_secs(11)).await;
tokio::task::yield_now().await;
assert!(refresh_calls.load(Ordering::SeqCst) > 0, "restore finalization lock must attempt renewal");
barrier.release();
let error = finalize
.await
.expect("restore finalization task should join")
.expect_err("lost acquired lock must reject restore finalization");
assert!(matches!(
error,
Error::Io(ref error)
if error.kind() == std::io::ErrorKind::Other
&& error.to_string() == "restore finalization lock lost before metadata update"
));
assert_committed_restore_remains_pending(&set_disks, bucket, object).await;
}
#[tokio::test]
#[serial_test::serial]
async fn restore_finalize_rejects_outer_fence_loss_after_metadata_read() {
let (_temp_dirs, disk_stores, set_disks) = hermetic_set_disks(4).await;
let bucket = "restore-finalize-outer-fence-lost-bucket";
let object = "object.bin";
let operation_id = Uuid::new_v4();
let restored = write_committed_restore(&set_disks, &disk_stores, bucket, object, operation_id).await;
let (fence, loss_handle) = NamespaceLockFence::loss_handle_for_test();
let barrier = RestoreFinalizeBarrier::install(bucket, object);
let finalize_set = Arc::clone(&set_disks);
let finalize = tokio::spawn(async move {
let mut opts = restore_finalize_options(operation_id);
opts.no_lock = true;
opts.namespace_lock_fence = Some(fence);
finalize_set.finalize_restore_metadata(bucket, object, &restored, &opts).await
});
barrier.wait_until_paused().await;
loss_handle.store(true, std::sync::atomic::Ordering::Release);
barrier.release();
let error = finalize
.await
.expect("restore finalization task should join")
.expect_err("lost outer fence must reject restore finalization");
assert!(matches!(
error,
Error::NamespaceLockQuorumUnavailable {
mode: "restore_finalize_metadata",
required: 1,
achieved: 0,
..
}
));
assert_committed_restore_remains_pending(&set_disks, bucket, object).await;
}
async fn assert_local_source_intact(set_disks: &Arc<SetDisks>, bucket: &str, object: &str, payload: &[u8]) {
let mut restored = Vec::new();
set_disks
+83 -4
View File
@@ -18,6 +18,78 @@ use rustfs_filemeta::RestoreStatusOps;
use rustfs_utils::http::headers::{AMZ_RESTORE_EXPIRY_DAYS, AMZ_RESTORE_REQUEST_DATE};
use s3s::dto::{RestoreStatus, Timestamp};
#[cfg(all(test, feature = "test-util"))]
struct RestoreFinalizeBarrierState {
bucket: String,
object: String,
arrived: tokio::sync::Notify,
release: tokio::sync::Notify,
}
#[cfg(all(test, feature = "test-util"))]
static RESTORE_FINALIZE_BARRIER: std::sync::OnceLock<std::sync::Mutex<Option<Arc<RestoreFinalizeBarrierState>>>> =
std::sync::OnceLock::new();
#[cfg(all(test, feature = "test-util"))]
pub(in crate::set_disk) struct RestoreFinalizeBarrier {
state: Arc<RestoreFinalizeBarrierState>,
}
#[cfg(all(test, feature = "test-util"))]
impl RestoreFinalizeBarrier {
pub(in crate::set_disk) fn install(bucket: &str, object: &str) -> Self {
let state = Arc::new(RestoreFinalizeBarrierState {
bucket: bucket.to_string(),
object: object.to_string(),
arrived: tokio::sync::Notify::new(),
release: tokio::sync::Notify::new(),
});
let mut slot = RESTORE_FINALIZE_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("restore finalize barrier mutex should not poison");
assert!(slot.is_none(), "restore finalize barrier must be installed by one test at a time");
*slot = Some(Arc::clone(&state));
Self { state }
}
pub(in crate::set_disk) async fn wait_until_paused(&self) {
self.state.arrived.notified().await;
}
pub(in crate::set_disk) fn release(&self) {
self.state.release.notify_one();
}
}
#[cfg(all(test, feature = "test-util"))]
impl Drop for RestoreFinalizeBarrier {
fn drop(&mut self) {
let mut slot = RESTORE_FINALIZE_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("restore finalize barrier mutex should not poison");
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
*slot = None;
}
}
}
#[cfg(all(test, feature = "test-util"))]
async fn maybe_pause_restore_finalize(bucket: &str, object: &str) {
let barrier = RESTORE_FINALIZE_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("restore finalize barrier mutex should not poison")
.as_ref()
.filter(|barrier| barrier.bucket == bucket && barrier.object == object)
.cloned();
if let Some(barrier) = barrier {
barrier.arrived.notify_one();
barrier.release.notified().await;
}
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
struct RestoreCleanupIdentity {
version_id: Option<Uuid>,
@@ -80,7 +152,7 @@ impl SetDisks {
.clone()
.unwrap_or_else(|| get_raw_etag(obj_info.user_defined.as_ref()));
let version_id = expected.version_id.map(|v| v.to_string());
let _lock_guard = if !opts.no_lock {
let lock_guard = if !opts.no_lock {
Some(
self.acquire_write_lock_diag("restore_finalize_metadata", bucket, object)
.await?,
@@ -99,13 +171,16 @@ impl SetDisks {
.get_object_fileinfo_gated(bucket, object, &read_opts, false, false)
.await?
.into_owned();
if let Some(expected_operation_id) = expected_operation_id {
require_restore_operation_id(&fi.metadata, expected_operation_id)?;
if let Some(expected_operation_id) = expected_operation_id
&& restore_operation_id_from_metadata(&fi.metadata)?.is_some_and(|actual| actual != expected_operation_id)
{
return Err(Error::other("restore operation id changed before metadata finalization"));
}
if !expected.matches_file_info(&fi, &expected_etag) {
return Err(Error::other("restored object changed before restore metadata finalization"));
}
ensure_restore_metadata_lock_held(bucket, object, opts, "restore_finalize_metadata")?;
#[cfg(all(test, feature = "test-util"))]
maybe_pause_restore_finalize(bucket, object).await;
let restore_expiry =
lifecycle::expected_expiry_time(OffsetDateTime::now_utc(), opts.transition.restore_request.days.unwrap_or(1));
fi.metadata.insert(
@@ -117,6 +192,10 @@ impl SetDisks {
.to_string(),
);
self.invalidate_get_object_metadata_cache(bucket, object).await;
ensure_restore_metadata_lock_held(bucket, object, opts, "restore_finalize_metadata")?;
if lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
return Err(Error::other("restore finalization lock lost before metadata update"));
}
self.update_object_meta_with_opts(
bucket,
object,
+84
View File
@@ -343,6 +343,23 @@ impl ECStore {
let (decommission, rebalance) = tokio::join!(self.is_decommission_running(), self.is_rebalance_started());
decommission || rebalance
}
/// Returns whether scanner metadata may still be hidden by a local
/// data-movement state. Terminal failed/canceled decommission entries
/// remain suspended until an operator clears or retries them, so they are
/// a publication barrier even after the worker has stopped.
pub async fn scanner_data_usage_publication_blocked(&self) -> bool {
if self.scanner_data_movement_active().await {
return true;
}
let pool_meta = self.pool_meta.read().await;
pool_meta.pools.iter().any(|pool| {
pool.decommission
.as_ref()
.is_some_and(|info| !info.queued && (info.failed || info.canceled))
})
}
}
// impl Clone for ECStore {
@@ -875,6 +892,7 @@ impl crate::storage_api_contracts::admin::StorageAdminApi for ECStore {
#[cfg(test)]
mod tests {
use super::*;
use crate::core::pools::{PoolDecommissionInfo, PoolStatus};
use crate::layout::endpoints::{Endpoints, PoolEndpoints, SetupType};
use crate::runtime::global::reset_local_disk_test_state;
use crate::runtime::sources::{clear_local_disk_id_map_for_test, local_disk_path_by_id};
@@ -911,6 +929,72 @@ mod tests {
})
}
#[tokio::test]
async fn scanner_data_usage_publication_blocks_active_and_unqueued_terminal_decommission() {
let store = build_store_with_ctx(Arc::new(InstanceContext::new()));
let cases = [
(
"active",
PoolDecommissionInfo {
start_time: Some(OffsetDateTime::now_utc()),
..Default::default()
},
true,
),
(
"failed",
PoolDecommissionInfo {
failed: true,
..Default::default()
},
true,
),
(
"canceled",
PoolDecommissionInfo {
canceled: true,
..Default::default()
},
true,
),
(
"queued_failed",
PoolDecommissionInfo {
failed: true,
queued: true,
..Default::default()
},
false,
),
(
"complete",
PoolDecommissionInfo {
complete: true,
..Default::default()
},
false,
),
("idle", PoolDecommissionInfo::default(), false),
];
for (name, decommission, expected) in cases {
*store.pool_meta.write().await = PoolMeta {
pools: vec![PoolStatus {
id: 0,
cmd_line: format!("scanner-publication-{name}"),
last_update: OffsetDateTime::now_utc(),
decommission: Some(decommission),
}],
..Default::default()
};
assert_eq!(
store.scanner_data_usage_publication_blocked().await,
expected,
"unexpected scanner publication barrier state for {name}"
);
}
}
// The object graph is the isolation carrier: two ECStore instances holding
// distinct contexts report independent erasure state through their real
// `&self` accessors — no cross-contamination.
+4 -22
View File
@@ -567,35 +567,17 @@ impl HealManager {
pub(super) fn heal_request_set_key(request: &HealRequest) -> Option<String> {
match &request.heal_type {
HealType::ErasureSet { set_disk_id, .. } => Some(set_disk_id.clone()),
HealType::Object { .. } => heal_options_set_key(&request.options),
_ => None,
}
}
pub(super) fn heal_options_set_key(options: &HealOptions) -> Option<String> {
match (options.pool_index, options.set_index) {
(Some(pool), Some(set)) => Some(format!("pool_{pool}_set_{set}")),
HealType::Object { .. } => request.options.set_key(),
_ => None,
}
}
pub(super) fn heal_request_type_label(request: &HealRequest) -> &'static str {
match &request.heal_type {
HealType::Cluster => "cluster",
HealType::Object { .. } => "object",
HealType::Bucket { .. } => "bucket",
HealType::Prefix { .. } => "prefix",
HealType::ErasureSet { .. } => "erasure_set",
HealType::Metadata { .. } => "metadata",
HealType::ECDecode { .. } => "ec_decode",
}
request.heal_type.kind_label()
}
pub(super) fn heal_request_set_metric_label(request: &HealRequest) -> String {
heal_request_set_key(request).unwrap_or_else(|| match (request.options.pool_index, request.options.set_index) {
(Some(pool), Some(set)) => format!("pool_{pool}_set_{set}"),
_ => "global".to_string(),
})
heal_request_set_key(request).unwrap_or_else(|| request.options.set_metric_label())
}
pub(super) fn record_scheduler_skip(set_label: &str) {
@@ -673,7 +655,7 @@ fn emit_mrf_repaired_events(targets: Vec<MrfRepairNoticeTarget>) {
pub(super) fn heal_request_set_key_for_task(task: &HealTask) -> Option<String> {
match &task.heal_type {
HealType::ErasureSet { set_disk_id, .. } => Some(set_disk_id.clone()),
HealType::Object { .. } => heal_options_set_key(&task.options),
HealType::Object { .. } => task.options.set_key(),
_ => None,
}
}
+31 -12
View File
@@ -744,10 +744,8 @@ fn test_priority_queue_pop_runnable_skips_blocked_erasure_set() {
let mut running = HashMap::new();
running.insert("pool_0_set_1".to_string(), 1);
let (popped, skipped_sets) = queue.pop_runnable_with_skips(
|request| can_schedule_request(request, &running, 1),
|request| heal_request_set_key(request),
);
let (popped, skipped_sets) =
queue.pop_runnable_with_skips(|request| can_schedule_request(request, &running, 1), heal_request_set_key);
let popped = popped.expect("should find runnable request");
assert_eq!(skipped_sets, vec!["pool_0_set_1".to_string()]);
@@ -788,10 +786,8 @@ fn test_priority_queue_pop_runnable_restores_all_blocked_items() {
running.insert("pool_0_set_2".to_string(), 1);
running.insert("pool_0_set_3".to_string(), 1);
let (popped, skipped_sets) = queue.pop_runnable_with_skips(
|request| can_schedule_request(request, &running, 1),
|request| heal_request_set_key(request),
);
let (popped, skipped_sets) =
queue.pop_runnable_with_skips(|request| can_schedule_request(request, &running, 1), heal_request_set_key);
assert!(popped.is_none());
assert_eq!(
@@ -843,10 +839,8 @@ fn test_priority_queue_pop_runnable_restores_deferred_with_tail() {
running.insert("pool_0_set_1".to_string(), 1);
running.insert("pool_0_set_2".to_string(), 1);
let (popped, skipped_sets) = queue.pop_runnable_with_skips(
|request| can_schedule_request(request, &running, 1),
|request| heal_request_set_key(request),
);
let (popped, skipped_sets) =
queue.pop_runnable_with_skips(|request| can_schedule_request(request, &running, 1), heal_request_set_key);
assert_eq!(skipped_sets, vec!["pool_0_set_1".to_string(), "pool_0_set_2".to_string()]);
assert!(matches!(
@@ -904,6 +898,31 @@ fn test_can_schedule_scoped_object_request_respects_per_set_limit() {
assert!(can_schedule_request(&request, &running, 2));
}
#[test]
fn test_heal_request_and_task_metric_labels_match() {
let request = HealRequest::new(
HealType::Object {
bucket: "bucket".to_string(),
object: "object".to_string(),
version_id: None,
},
HealOptions {
pool_index: Some(0),
set_index: Some(1),
..Default::default()
},
HealPriority::Normal,
);
assert_eq!(heal_request_type_label(&request), "object");
assert_eq!(heal_request_set_key(&request), Some("pool_0_set_1".to_string()));
assert_eq!(heal_request_set_metric_label(&request), "pool_0_set_1");
let task = HealTask::from_request(request, Arc::new(MockStorage));
assert_eq!(task.metric_type_label(), "object");
assert_eq!(task.metric_set_label(), "pool_0_set_1");
}
#[tokio::test]
async fn test_submit_heal_request_returns_merged_for_duplicate() {
let storage: Arc<dyn HealStorageAPI> = Arc::new(MockStorage);
-43
View File
@@ -218,15 +218,6 @@ impl HealStatistics {
self.total_bytes_healed += bytes;
self.last_update_time = SystemTime::now();
}
pub fn get_success_rate(&self) -> f64 {
let total = self.successful_tasks + self.failed_tasks;
if total > 0 {
(self.successful_tasks as f64 / total as f64) * 100.0
} else {
0.0
}
}
}
#[cfg(test)]
@@ -539,38 +530,4 @@ mod tests {
assert_eq!(stats.total_objects_healed, 8);
assert_eq!(stats.total_bytes_healed, 8192);
}
#[test]
fn test_heal_statistics_get_success_rate() {
let mut stats = HealStatistics::new();
stats.successful_tasks = 8;
stats.failed_tasks = 2;
// success_rate = 8 / (8 + 2) * 100 = 80%
assert!((stats.get_success_rate() - 80.0).abs() < 0.001);
}
#[test]
fn test_heal_statistics_get_success_rate_zero_total() {
let stats = HealStatistics::new();
assert_eq!(stats.get_success_rate(), 0.0);
}
#[test]
fn test_heal_statistics_get_success_rate_all_success() {
let mut stats = HealStatistics::new();
stats.successful_tasks = 10;
stats.failed_tasks = 0;
assert!((stats.get_success_rate() - 100.0).abs() < 0.001);
}
#[test]
fn test_heal_statistics_get_success_rate_all_failure() {
let mut stats = HealStatistics::new();
stats.successful_tasks = 0;
stats.failed_tasks = 5;
assert_eq!(stats.get_success_rate(), 0.0);
}
}
+16 -7
View File
@@ -1202,13 +1202,22 @@ impl HealStorageAPI for ECStoreHealStorage {
let version_id = obj.version_id.map(|u| u.to_string());
let mod_time_unix_nanos = obj.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos());
let is_delete_marker = obj.delete_marker;
let lifecycle_object_info = include_lifecycle_object_info.then(|| obj.clone());
HealListItem {
name: obj.name,
version_id,
mod_time_unix_nanos,
lifecycle_object_info,
is_delete_marker,
if include_lifecycle_object_info {
HealListItem {
name: obj.name.clone(),
version_id,
mod_time_unix_nanos,
lifecycle_object_info: Some(obj),
is_delete_marker,
}
} else {
HealListItem {
name: obj.name,
version_id,
mod_time_unix_nanos,
lifecycle_object_info: None,
is_delete_marker,
}
}
})
.collect();
+23 -21
View File
@@ -109,7 +109,7 @@ pub enum HealType {
}
impl HealType {
fn log_kind(&self) -> &'static str {
pub(crate) fn kind_label(&self) -> &'static str {
match self {
Self::Cluster => "cluster",
Self::Object { .. } => "object",
@@ -227,6 +227,19 @@ impl Default for HealOptions {
}
}
impl HealOptions {
pub(crate) fn set_key(&self) -> Option<String> {
match (self.pool_index, self.set_index) {
(Some(pool), Some(set)) => Some(format!("pool_{pool}_set_{set}")),
_ => None,
}
}
pub(crate) fn set_metric_label(&self) -> String {
self.set_key().unwrap_or_else(|| "global".to_string())
}
}
/// Heal task status
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
pub enum HealTaskStatus {
@@ -491,15 +504,7 @@ impl HealTask {
}
pub fn metric_type_label(&self) -> &'static str {
match &self.heal_type {
HealType::Cluster => "cluster",
HealType::Object { .. } => "object",
HealType::Bucket { .. } => "bucket",
HealType::Prefix { .. } => "prefix",
HealType::ErasureSet { .. } => "erasure_set",
HealType::Metadata { .. } => "metadata",
HealType::ECDecode { .. } => "ec_decode",
}
self.heal_type.kind_label()
}
pub(crate) fn has_batch_failure(&self) -> bool {
@@ -520,10 +525,7 @@ impl HealTask {
pub fn metric_set_label(&self) -> String {
match &self.heal_type {
HealType::ErasureSet { set_disk_id, .. } => set_disk_id.clone(),
_ => match (self.options.pool_index, self.options.set_index) {
(Some(pool), Some(set)) => format!("pool_{pool}_set_{set}"),
_ => "global".to_string(),
},
_ => self.options.set_metric_label(),
}
}
@@ -532,7 +534,7 @@ impl HealTask {
let mut event = TraceEvent::new(TraceKind::Heal, TraceFunc::HealTask)
.with_duration(duration)
.with_attr("task_id", self.id.as_str())
.with_attr("heal_type", self.heal_type.log_kind())
.with_attr("heal_type", self.heal_type.kind_label())
.with_attr("state", state)
.with_attr("source", self.source.as_str())
.with_attr("priority", self.priority.as_str())
@@ -795,7 +797,7 @@ impl HealTask {
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
heal_type = self.heal_type.log_kind(),
heal_type = self.heal_type.kind_label(),
state = "started",
queue_delay = ?queue_delay,
"Heal task started"
@@ -836,7 +838,7 @@ impl HealTask {
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
heal_type = self.heal_type.log_kind(),
heal_type = self.heal_type.kind_label(),
state = "completed",
"Heal task completed"
});
@@ -850,7 +852,7 @@ impl HealTask {
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
heal_type = self.heal_type.log_kind(),
heal_type = self.heal_type.kind_label(),
state = "cancelled",
"Heal task cancelled"
);
@@ -863,7 +865,7 @@ impl HealTask {
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
heal_type = self.heal_type.log_kind(),
heal_type = self.heal_type.kind_label(),
state = "timed_out",
"Heal task timed out"
});
@@ -880,7 +882,7 @@ impl HealTask {
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
heal_type = self.heal_type.log_kind(),
heal_type = self.heal_type.kind_label(),
state = "failed",
error = %e,
"Heal task failed"
@@ -909,7 +911,7 @@ impl HealTask {
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_TASK,
task_id = %self.id,
heal_type = self.heal_type.log_kind(),
heal_type = self.heal_type.kind_label(),
state = "cancelled",
source = "manual",
"Heal task cancellation requested"
+1 -1
View File
@@ -12,7 +12,7 @@
// See the License for the specific language governing permissions and
// limitations under the License.
use super::super::{DiskOption, DiskStore, Endpoint, HealDiskExt as _, new_disk};
use super::super::{DiskOption, DiskStore, Endpoint, new_disk};
use super::*;
use crate::heal::storage::{HealListItem, HealObjectInfo};
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, TraceSubscription, TraceVal, subscribe_trace_events};
+2 -2
View File
@@ -17,11 +17,11 @@
//! All direct `rustfs_ecstore` facade imports used by tests in this crate
//! must go through this module (architecture migration rule:
//! `check_architecture_migration_rules.sh`). Keep the surface minimal —
//! only what the tests actually need to build a temp-disk ECStore fixture
//! and to flip the erasure setup type for lock-quorum fault injection.
//! only what the tests actually need to run storage-backed IAM scenarios.
#[allow(unused_imports)]
pub(crate) mod fixture {
pub(crate) use rustfs_ecstore::api::bucket::migration::try_migrate_iam_config;
pub(crate) use rustfs_ecstore::api::layout::SetupType;
// `update_erasure_type` is a write-side global facade entry. Its use is
@@ -0,0 +1,182 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
mod ecstore_test_compat;
use ecstore_test_compat::fixture::try_migrate_iam_config;
use rustfs_credentials::{get_global_action_cred, init_global_action_credentials};
use rustfs_iam::store::object::{
IAM_CONFIG_POLICY_DB_SERVICE_ACCOUNTS_PREFIX, IAM_CONFIG_POLICY_DB_USERS_PREFIX, IAM_CONFIG_SERVICE_ACCOUNTS_PREFIX,
IAM_CONFIG_USERS_PREFIX, ObjectStore,
};
use rustfs_iam::store::{Store, UserType};
use rustfs_iam::utils::generate_jwt;
use rustfs_policy::auth::UserIdentity;
use serde_json::{Value, json};
use std::collections::HashMap;
const LEGACY_META_BUCKET: &str = ".minio.sys";
const REGULAR_USER: &str = "minio-user";
const SERVICE_ACCOUNT: &str = "minio-service-account";
async fn seed_legacy_iam_object(env: &rustfs_test_utils::TestECStoreEnv, path: &str, value: &Value) {
env.put_object_bytes(
LEGACY_META_BUCKET,
path,
serde_json::to_vec(value).expect("legacy IAM object must serialize"),
)
.await;
}
fn assert_identity_fields(actual: &UserIdentity, expected: &Value) {
assert_eq!(
serde_json::to_value(actual).expect("loaded identity must serialize"),
*expected,
"migration must preserve every credential field except expiration",
);
}
async fn assert_identity_survives(
store: &ObjectStore,
identity_path: &str,
name: &str,
user_type: UserType,
source: &Value,
expected_policy: &Value,
) {
let mut expected = source.clone();
expected["credentials"]["expiration"] = Value::Null;
let persisted: UserIdentity = store
.load_iam_config(identity_path)
.await
.expect("migrated identity must be persisted");
assert_identity_fields(&persisted, &expected);
for _ in 0..2 {
let actual = store
.load_user_identity(name, user_type)
.await
.expect("migrated permanent identity must remain loadable");
assert_identity_fields(&actual, &expected);
}
let mut mappings = HashMap::new();
store
.load_mapped_policy(name, user_type, false, &mut mappings)
.await
.expect("loading the identity must not delete its policy mapping");
let actual_policy = mappings.get(name).expect("migrated policy mapping must exist");
assert_eq!(
serde_json::to_value(actual_policy).expect("loaded policy mapping must serialize"),
*expected_policy,
);
}
#[tokio::test(flavor = "multi_thread")]
async fn minio_permanent_identities_survive_migration_and_repeated_iam_loads() {
if get_global_action_cred().is_none() {
init_global_action_credentials(Some("MINIOMIGRATIONROOT".to_string()), Some("minio-migration-root-secret".to_string()))
.expect("root credentials must initialize for JWT validation");
}
let temp_dir = tempfile::TempDir::with_prefix("rustfs_minio_iam_migration_").expect("temp directory must be created");
let env = rustfs_test_utils::TestECStoreEnv::builder()
.base_dir(temp_dir.path())
.init_bucket_metadata(false)
.build()
.await;
for disk_path in &env.disk_paths {
tokio::fs::create_dir_all(disk_path.join(LEGACY_META_BUCKET))
.await
.expect("legacy metadata volume must be created");
}
let regular_source = json!({
"version": 1,
"credentials": {
"accessKey": REGULAR_USER,
"secretKey": "regular-user-secret",
"sessionToken": "",
"expiration": "0001-01-01T00:00:00Z",
"status": "on",
"parentUser": "regular-parent",
"groups": ["engineering", "operations"],
"claims": {"tenant": "alpha"},
"name": "MinIO regular user",
"description": "migrated regular identity"
},
"updatedAt": "2025-03-07T12:00:00Z"
});
let service_claims = json!({"sa-policy": "inherited-policy", "tenant": "alpha"});
let service_secret = "service-account-secret";
let service_source = json!({
"version": 1,
"credentials": {
"accessKey": SERVICE_ACCOUNT,
"secretKey": service_secret,
"sessionToken": generate_jwt(&service_claims, service_secret).expect("service-account JWT must be generated"),
"expiration": "1970-01-01T00:00:00Z",
"status": "on",
"parentUser": REGULAR_USER,
"groups": ["service-accounts"],
"claims": service_claims,
"name": "MinIO service account",
"description": "migrated service identity"
},
"updatedAt": "2025-03-07T12:00:00Z"
});
let regular_policy_source = json!({"version": 1, "policy": "readwrite", "updatedAt": "2025-03-07T12:00:00Z"});
let service_policy_source = json!({"version": 1, "policy": "readonly", "updatedAt": "2025-03-07T12:00:00Z"});
let regular_identity_path = format!("{}{REGULAR_USER}/identity.json", IAM_CONFIG_USERS_PREFIX.as_str());
let service_identity_path = format!("{}{SERVICE_ACCOUNT}/identity.json", IAM_CONFIG_SERVICE_ACCOUNTS_PREFIX.as_str());
seed_legacy_iam_object(&env, &regular_identity_path, &regular_source).await;
seed_legacy_iam_object(&env, &service_identity_path, &service_source).await;
seed_legacy_iam_object(
&env,
&format!("{}{REGULAR_USER}.json", IAM_CONFIG_POLICY_DB_USERS_PREFIX.as_str()),
&regular_policy_source,
)
.await;
seed_legacy_iam_object(
&env,
&format!("{}{SERVICE_ACCOUNT}.json", IAM_CONFIG_POLICY_DB_SERVICE_ACCOUNTS_PREFIX.as_str()),
&service_policy_source,
)
.await;
try_migrate_iam_config(env.ecstore.clone(), None).await;
let store = ObjectStore::new(env.ecstore);
assert_identity_survives(
&store,
&regular_identity_path,
REGULAR_USER,
UserType::Reg,
&regular_source,
&regular_policy_source,
)
.await;
assert_identity_survives(
&store,
&service_identity_path,
SERVICE_ACCOUNT,
UserType::Svc,
&service_source,
&service_policy_source,
)
.await;
}
+202
View File
@@ -211,6 +211,146 @@ pub const INTERNODE_OPERATION_METRICS: &[InternodeOperationMetricDescriptor] = &
static STABLE_SERVER_LABEL: OnceLock<String> = OnceLock::new();
#[cfg(not(test))]
struct InternodeServerMetricHandles {
sent_bytes: metrics::Counter,
recv_bytes: metrics::Counter,
outgoing_requests: metrics::Counter,
incoming_requests: metrics::Counter,
errors: metrics::Counter,
}
#[cfg(not(test))]
impl InternodeServerMetricHandles {
fn new(server: &'static str) -> Self {
Self {
sent_bytes: counter!("rustfs_system_network_internode_sent_bytes_total", SERVER_LABEL => server),
recv_bytes: counter!("rustfs_system_network_internode_recv_bytes_total", SERVER_LABEL => server),
outgoing_requests: counter!("rustfs_system_network_internode_requests_outgoing_total", SERVER_LABEL => server),
incoming_requests: counter!("rustfs_system_network_internode_requests_incoming_total", SERVER_LABEL => server),
errors: counter!("rustfs_system_network_internode_errors_total", SERVER_LABEL => server),
}
}
}
#[cfg(not(test))]
static INTERNODE_SERVER_METRIC_HANDLES: LazyLock<InternodeServerMetricHandles> =
LazyLock::new(|| InternodeServerMetricHandles::new(current_server_label()));
#[cfg(not(test))]
struct GrpcReadVersionMetricHandles {
sent_bytes: metrics::Counter,
recv_bytes: metrics::Counter,
outgoing_requests: metrics::Counter,
incoming_requests: metrics::Counter,
errors: metrics::Counter,
duration: metrics::Histogram,
request_encode: metrics::Histogram,
request_decode: metrics::Histogram,
disk_read: metrics::Histogram,
response_json_encode: metrics::Histogram,
response_msgpack_encode: metrics::Histogram,
rpc_roundtrip: metrics::Histogram,
response_decode: metrics::Histogram,
}
#[cfg(not(test))]
impl GrpcReadVersionMetricHandles {
fn new(server: &'static str) -> Self {
Self {
sent_bytes: counter!(
INTERNODE_OPERATION_SENT_BYTES_TOTAL,
SERVER_LABEL => server,
OPERATION_LABEL => INTERNODE_OPERATION_GRPC_READ_VERSION,
BACKEND_LABEL => INTERNODE_TRANSPORT_BACKEND_GRPC
),
recv_bytes: counter!(
INTERNODE_OPERATION_RECV_BYTES_TOTAL,
SERVER_LABEL => server,
OPERATION_LABEL => INTERNODE_OPERATION_GRPC_READ_VERSION,
BACKEND_LABEL => INTERNODE_TRANSPORT_BACKEND_GRPC
),
outgoing_requests: counter!(
INTERNODE_OPERATION_REQUESTS_OUTGOING_TOTAL,
SERVER_LABEL => server,
OPERATION_LABEL => INTERNODE_OPERATION_GRPC_READ_VERSION,
BACKEND_LABEL => INTERNODE_TRANSPORT_BACKEND_GRPC
),
incoming_requests: counter!(
INTERNODE_OPERATION_REQUESTS_INCOMING_TOTAL,
SERVER_LABEL => server,
OPERATION_LABEL => INTERNODE_OPERATION_GRPC_READ_VERSION,
BACKEND_LABEL => INTERNODE_TRANSPORT_BACKEND_GRPC
),
errors: counter!(
INTERNODE_OPERATION_ERRORS_TOTAL,
SERVER_LABEL => server,
OPERATION_LABEL => INTERNODE_OPERATION_GRPC_READ_VERSION,
BACKEND_LABEL => INTERNODE_TRANSPORT_BACKEND_GRPC
),
duration: metrics::histogram!(
INTERNODE_OPERATION_DURATION_MS,
SERVER_LABEL => server,
OPERATION_LABEL => INTERNODE_OPERATION_GRPC_READ_VERSION,
BACKEND_LABEL => INTERNODE_TRANSPORT_BACKEND_GRPC
),
request_encode: Self::stage_duration(server, INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE),
request_decode: Self::stage_duration(server, INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE),
disk_read: Self::stage_duration(server, INTERNODE_STAGE_READ_VERSION_DISK_READ),
response_json_encode: Self::stage_duration(server, INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE),
response_msgpack_encode: Self::stage_duration(server, INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE),
rpc_roundtrip: Self::stage_duration(server, INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP),
response_decode: Self::stage_duration(server, INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE),
}
}
fn stage_duration(server: &'static str, stage: &'static str) -> metrics::Histogram {
metrics::histogram!(
INTERNODE_OPERATION_STAGE_DURATION_MS,
SERVER_LABEL => server,
OPERATION_LABEL => INTERNODE_OPERATION_GRPC_READ_VERSION,
BACKEND_LABEL => INTERNODE_TRANSPORT_BACKEND_GRPC,
STAGE_LABEL => stage
)
}
fn stage_duration_for(&self, stage: &'static str) -> Option<&metrics::Histogram> {
match stage {
INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE => Some(&self.request_encode),
INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE => Some(&self.request_decode),
INTERNODE_STAGE_READ_VERSION_DISK_READ => Some(&self.disk_read),
INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE => Some(&self.response_json_encode),
INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE => Some(&self.response_msgpack_encode),
INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP => Some(&self.rpc_roundtrip),
INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE => Some(&self.response_decode),
_ => None,
}
}
}
#[cfg(not(test))]
static GRPC_READ_VERSION_METRIC_HANDLES: LazyLock<GrpcReadVersionMetricHandles> =
LazyLock::new(|| GrpcReadVersionMetricHandles::new(current_server_label()));
#[cfg(not(test))]
fn server_metric_handles_if_ready() -> Option<&'static InternodeServerMetricHandles> {
STABLE_SERVER_LABEL.get()?;
Some(&INTERNODE_SERVER_METRIC_HANDLES)
}
#[cfg(not(test))]
fn grpc_read_version_metric_handles_if_ready(
operation: &'static str,
backend: &'static str,
) -> Option<&'static GrpcReadVersionMetricHandles> {
STABLE_SERVER_LABEL.get()?;
if operation == INTERNODE_OPERATION_GRPC_READ_VERSION && backend == INTERNODE_TRANSPORT_BACKEND_GRPC {
Some(&GRPC_READ_VERSION_METRIC_HANDLES)
} else {
None
}
}
/// Injects the stable server label (node name or address) stamped on
/// internode metrics. The runtime calls this when the local node name is
/// published (see ecstore's `set_local_node_name`); the first write wins.
@@ -284,6 +424,11 @@ impl InternodeMetrics {
return;
}
self.sent_bytes_total.fetch_add(bytes, Ordering::Relaxed);
#[cfg(not(test))]
if let Some(handles) = server_metric_handles_if_ready() {
handles.sent_bytes.increment(bytes);
return;
}
counter!("rustfs_system_network_internode_sent_bytes_total", SERVER_LABEL => current_server_label()).increment(bytes);
}
@@ -298,6 +443,11 @@ impl InternodeMetrics {
if bytes == 0 {
return;
}
#[cfg(not(test))]
if let Some(handles) = grpc_read_version_metric_handles_if_ready(operation, backend) {
handles.sent_bytes.increment(bytes);
return;
}
counter!(
INTERNODE_OPERATION_SENT_BYTES_TOTAL,
SERVER_LABEL => current_server_label(),
@@ -313,6 +463,11 @@ impl InternodeMetrics {
return;
}
self.recv_bytes_total.fetch_add(bytes, Ordering::Relaxed);
#[cfg(not(test))]
if let Some(handles) = server_metric_handles_if_ready() {
handles.recv_bytes.increment(bytes);
return;
}
counter!("rustfs_system_network_internode_recv_bytes_total", SERVER_LABEL => current_server_label()).increment(bytes);
}
@@ -327,6 +482,11 @@ impl InternodeMetrics {
if bytes == 0 {
return;
}
#[cfg(not(test))]
if let Some(handles) = grpc_read_version_metric_handles_if_ready(operation, backend) {
handles.recv_bytes.increment(bytes);
return;
}
counter!(
INTERNODE_OPERATION_RECV_BYTES_TOTAL,
SERVER_LABEL => current_server_label(),
@@ -338,6 +498,11 @@ impl InternodeMetrics {
pub fn record_outgoing_request(&self) {
self.outgoing_requests_total.fetch_add(1, Ordering::Relaxed);
#[cfg(not(test))]
if let Some(handles) = server_metric_handles_if_ready() {
handles.outgoing_requests.increment(1);
return;
}
counter!("rustfs_system_network_internode_requests_outgoing_total", SERVER_LABEL => current_server_label()).increment(1);
}
@@ -347,6 +512,11 @@ impl InternodeMetrics {
pub fn record_outgoing_request_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
self.record_outgoing_request();
#[cfg(not(test))]
if let Some(handles) = grpc_read_version_metric_handles_if_ready(operation, backend) {
handles.outgoing_requests.increment(1);
return;
}
counter!(
INTERNODE_OPERATION_REQUESTS_OUTGOING_TOTAL,
SERVER_LABEL => current_server_label(),
@@ -358,6 +528,11 @@ impl InternodeMetrics {
pub fn record_incoming_request(&self) {
self.incoming_requests_total.fetch_add(1, Ordering::Relaxed);
#[cfg(not(test))]
if let Some(handles) = server_metric_handles_if_ready() {
handles.incoming_requests.increment(1);
return;
}
counter!("rustfs_system_network_internode_requests_incoming_total", SERVER_LABEL => current_server_label()).increment(1);
}
@@ -367,6 +542,11 @@ impl InternodeMetrics {
pub fn record_incoming_request_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
self.record_incoming_request();
#[cfg(not(test))]
if let Some(handles) = grpc_read_version_metric_handles_if_ready(operation, backend) {
handles.incoming_requests.increment(1);
return;
}
counter!(
INTERNODE_OPERATION_REQUESTS_INCOMING_TOTAL,
SERVER_LABEL => current_server_label(),
@@ -378,6 +558,11 @@ impl InternodeMetrics {
pub fn record_error(&self) {
self.errors_total.fetch_add(1, Ordering::Relaxed);
#[cfg(not(test))]
if let Some(handles) = server_metric_handles_if_ready() {
handles.errors.increment(1);
return;
}
counter!("rustfs_system_network_internode_errors_total", SERVER_LABEL => current_server_label()).increment(1);
}
@@ -387,6 +572,11 @@ impl InternodeMetrics {
pub fn record_error_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
self.record_error();
#[cfg(not(test))]
if let Some(handles) = grpc_read_version_metric_handles_if_ready(operation, backend) {
handles.errors.increment(1);
return;
}
counter!(
INTERNODE_OPERATION_ERRORS_TOTAL,
SERVER_LABEL => current_server_label(),
@@ -398,6 +588,11 @@ impl InternodeMetrics {
pub fn record_duration_for_operation_and_backend(&self, operation: &'static str, backend: &'static str, duration: Duration) {
let duration_ms = duration.as_secs_f64() * 1000.0;
#[cfg(not(test))]
if let Some(handles) = grpc_read_version_metric_handles_if_ready(operation, backend) {
handles.duration.record(duration_ms);
return;
}
metrics::histogram!(
INTERNODE_OPERATION_DURATION_MS,
SERVER_LABEL => current_server_label(),
@@ -415,6 +610,13 @@ impl InternodeMetrics {
duration: Duration,
) {
let duration_ms = duration.as_secs_f64() * 1000.0;
#[cfg(not(test))]
if let Some(handles) = grpc_read_version_metric_handles_if_ready(operation, backend)
&& let Some(histogram) = handles.stage_duration_for(stage)
{
histogram.record(duration_ms);
return;
}
metrics::histogram!(
INTERNODE_OPERATION_STAGE_DURATION_MS,
SERVER_LABEL => current_server_label(),
+95
View File
@@ -121,6 +121,16 @@ pub const PUT_STAGE_SET_DISK_RENAME_BACKUP_DIR_FSYNC: &str = "set_disk_rename_ba
pub const PUT_STAGE_SET_DISK_RENAME_ANCESTOR_DIR_FSYNC: &str = "set_disk_rename_ancestor_dir_fsync";
pub const PUT_STAGE_SET_DISK_RENAME_RENAME_SYSCALL: &str = "set_disk_rename_rename_syscall";
pub const PUT_COMMIT_LOCK_ADMISSION_BUDGET_DISABLED: &str = "disabled";
pub const PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS: &str = "le_250ms";
pub const PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS: &str = "le_500ms";
pub const PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_1000MS: &str = "le_1000ms";
pub const PUT_COMMIT_LOCK_ADMISSION_BUDGET_GT_1000MS: &str = "gt_1000ms";
pub const PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED: &str = "acquired";
pub const PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN: &str = "timeout_slowdown";
pub const PUT_COMMIT_LOCK_ADMISSION_OUTCOME_LOCK_ERROR: &str = "lock_error";
pub const PUT_RENAME_FDATASYNC_BATCH_MODE_SERIAL: &str = "serial";
pub const PUT_RENAME_FDATASYNC_BATCH_MODE_PARALLEL: &str = "parallel";
pub const PUT_RENAME_FDATASYNC_GROUP_WAIT_ROLE_LEADER: &str = "leader";
@@ -2060,6 +2070,14 @@ pub fn record_put_object_stage_duration_from(stage: &'static str, started_at: Op
}
}
#[inline(always)]
pub fn record_put_object_commit_lock_admission(budget: &'static str, outcome: &'static str) {
if !put_stage_metrics_enabled() {
return;
}
counter!("rustfs_s3_put_object_commit_namespace_lock_admission_total", "budget" => budget, "outcome" => outcome).increment(1);
}
#[inline(always)]
fn put_stage_count_value(value: usize) -> f64 {
match u32::try_from(value) {
@@ -3204,6 +3222,83 @@ mod tests {
assert!(stages.iter().all(|stage| recorded.contains(*stage)));
}
#[test]
fn put_commit_lock_admission_labels_are_static_and_gated() {
let _guard = METRICS_FLAG_LOCK.lock().unwrap_or_else(|e| e.into_inner());
let budgets = [
PUT_COMMIT_LOCK_ADMISSION_BUDGET_DISABLED,
PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS,
PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_1000MS,
PUT_COMMIT_LOCK_ADMISSION_BUDGET_GT_1000MS,
];
let outcomes = [
PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED,
PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN,
PUT_COMMIT_LOCK_ADMISSION_OUTCOME_LOCK_ERROR,
];
assert_eq!(budgets.iter().copied().collect::<HashSet<_>>().len(), budgets.len());
assert_eq!(outcomes.iter().copied().collect::<HashSet<_>>().len(), outcomes.len());
assert!(budgets.iter().chain(outcomes.iter()).all(|label| {
!label.contains('/')
&& !label.contains('{')
&& !label.contains('}')
&& !label.contains(' ')
&& label
.chars()
.all(|ch| ch.is_ascii_lowercase() || ch.is_ascii_digit() || ch == '_')
}));
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
metrics::with_local_recorder(&recorder, || {
set_put_stage_metrics_enabled(false);
record_put_object_commit_lock_admission(
PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN,
);
set_put_stage_metrics_enabled(true);
record_put_object_commit_lock_admission(
PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS,
PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN,
);
record_put_object_commit_lock_admission(
PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS,
PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED,
);
set_put_stage_metrics_enabled(false);
});
let rows = snapshotter.snapshot().into_vec();
assert_eq!(
counter_total(&rows, "rustfs_s3_put_object_commit_namespace_lock_admission_total"),
Some(2)
);
let label_sets = rows
.iter()
.filter(|(composite, _, _, _)| {
composite.kind() == MetricKind::Counter
&& composite.key().name() == "rustfs_s3_put_object_commit_namespace_lock_admission_total"
})
.map(|(composite, _, _, _)| {
composite
.key()
.labels()
.map(|label| (label.key().to_string(), label.value().to_string()))
.collect::<HashSet<_>>()
})
.collect::<Vec<_>>();
assert!(label_sets.contains(&HashSet::from([
("budget".to_string(), PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_250MS.to_string()),
("outcome".to_string(), PUT_COMMIT_LOCK_ADMISSION_OUTCOME_TIMEOUT_SLOWDOWN.to_string(),),
])));
assert!(label_sets.contains(&HashSet::from([
("budget".to_string(), PUT_COMMIT_LOCK_ADMISSION_BUDGET_LE_500MS.to_string()),
("outcome".to_string(), PUT_COMMIT_LOCK_ADMISSION_OUTCOME_ACQUIRED.to_string()),
])));
}
#[test]
fn put_rename_code_level_metrics_are_static_and_gated() {
let _guard = METRICS_FLAG_LOCK.lock().unwrap_or_else(|e| e.into_inner());
@@ -0,0 +1,216 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use metrics::with_local_recorder;
use metrics_util::debugging::{DebugValue, DebuggingRecorder};
use rustfs_io_metrics::internode_metrics::{
INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_STAGE_READ_VERSION_DISK_READ, INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP,
INTERNODE_TRANSPORT_BACKEND_GRPC, InternodeMetrics, set_internode_server_label,
};
use std::time::Duration;
type MetricRow = (
metrics_util::CompositeKey,
Option<metrics::Unit>,
Option<metrics::SharedString>,
DebugValue,
);
const SERVER_LABEL: &str = "server";
const OPERATION_LABEL: &str = "operation";
const BACKEND_LABEL: &str = "backend";
const STAGE_LABEL: &str = "stage";
const SENT_BYTES_TOTAL: &str = "rustfs_system_network_internode_sent_bytes_total";
const RECV_BYTES_TOTAL: &str = "rustfs_system_network_internode_recv_bytes_total";
const REQUESTS_OUTGOING_TOTAL: &str = "rustfs_system_network_internode_requests_outgoing_total";
const REQUESTS_INCOMING_TOTAL: &str = "rustfs_system_network_internode_requests_incoming_total";
const ERRORS_TOTAL: &str = "rustfs_system_network_internode_errors_total";
const OPERATION_SENT_BYTES_TOTAL: &str = "rustfs_system_network_internode_operation_sent_bytes_total";
const OPERATION_RECV_BYTES_TOTAL: &str = "rustfs_system_network_internode_operation_recv_bytes_total";
const OPERATION_REQUESTS_OUTGOING_TOTAL: &str = "rustfs_system_network_internode_operation_requests_outgoing_total";
const OPERATION_REQUESTS_INCOMING_TOTAL: &str = "rustfs_system_network_internode_operation_requests_incoming_total";
const OPERATION_ERRORS_TOTAL: &str = "rustfs_system_network_internode_operation_errors_total";
const OPERATION_DURATION_MS: &str = "rustfs_system_network_internode_operation_duration_ms";
const OPERATION_STAGE_DURATION_MS: &str = "rustfs_system_network_internode_operation_stage_duration_ms";
#[test]
fn cached_grpc_read_version_metric_handles_preserve_labels_and_values() {
set_internode_server_label("cached-grpc-read-version-test");
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
let metrics = InternodeMetrics::default();
with_local_recorder(&recorder, || {
metrics.record_sent_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
17,
);
metrics.record_recv_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
23,
);
metrics.record_outgoing_request_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
metrics.record_incoming_request_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
metrics.record_error_for_operation_and_backend(INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_TRANSPORT_BACKEND_GRPC);
metrics.record_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
Duration::from_micros(250),
);
metrics.record_stage_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP,
Duration::from_micros(125),
);
metrics.record_stage_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
INTERNODE_STAGE_READ_VERSION_DISK_READ,
Duration::from_micros(75),
);
});
let rows = snapshotter.snapshot().into_vec();
assert_counter(&rows, SENT_BYTES_TOTAL, &[(SERVER_LABEL, "cached-grpc-read-version-test")], 17);
assert_counter(&rows, RECV_BYTES_TOTAL, &[(SERVER_LABEL, "cached-grpc-read-version-test")], 23);
assert_counter(&rows, REQUESTS_OUTGOING_TOTAL, &[(SERVER_LABEL, "cached-grpc-read-version-test")], 1);
assert_counter(&rows, REQUESTS_INCOMING_TOTAL, &[(SERVER_LABEL, "cached-grpc-read-version-test")], 1);
assert_counter(&rows, ERRORS_TOTAL, &[(SERVER_LABEL, "cached-grpc-read-version-test")], 1);
assert_counter(
&rows,
OPERATION_SENT_BYTES_TOTAL,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
],
17,
);
assert_counter(
&rows,
OPERATION_RECV_BYTES_TOTAL,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
],
23,
);
assert_counter(
&rows,
OPERATION_REQUESTS_OUTGOING_TOTAL,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
],
1,
);
assert_counter(
&rows,
OPERATION_REQUESTS_INCOMING_TOTAL,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
],
1,
);
assert_counter(
&rows,
OPERATION_ERRORS_TOTAL,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
],
1,
);
assert_histogram(
&rows,
OPERATION_DURATION_MS,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
],
&[0.25],
);
assert_histogram(
&rows,
OPERATION_STAGE_DURATION_MS,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
(STAGE_LABEL, INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP),
],
&[0.125],
);
assert_histogram(
&rows,
OPERATION_STAGE_DURATION_MS,
&[
(SERVER_LABEL, "cached-grpc-read-version-test"),
(OPERATION_LABEL, INTERNODE_OPERATION_GRPC_READ_VERSION),
(BACKEND_LABEL, INTERNODE_TRANSPORT_BACKEND_GRPC),
(STAGE_LABEL, INTERNODE_STAGE_READ_VERSION_DISK_READ),
],
&[0.075],
);
}
fn assert_counter(rows: &[MetricRow], name: &str, labels: &[(&str, &str)], expected: u64) {
match metric_value(rows, name, labels) {
DebugValue::Counter(value) => assert_eq!(*value, expected),
other => panic!("{name} should be a counter, got {other:?}"),
}
}
fn assert_histogram(rows: &[MetricRow], name: &str, labels: &[(&str, &str)], expected: &[f64]) {
match metric_value(rows, name, labels) {
DebugValue::Histogram(samples) => {
let actual: Vec<_> = samples.iter().map(|sample| sample.0).collect();
assert_eq!(actual, expected);
}
other => panic!("{name} should be a histogram, got {other:?}"),
}
}
fn metric_value<'a>(rows: &'a [MetricRow], name: &str, labels: &[(&str, &str)]) -> &'a DebugValue {
let mut matches = rows.iter().filter(|(composite, _, _, _)| {
composite.key().name() == name
&& labels.iter().all(|(key, value)| {
composite
.key()
.labels()
.any(|label| label.key() == *key && label.value() == *value)
})
});
let Some((_, _, _, value)) = matches.next() else {
panic!("{name} with labels {labels:?} was not recorded; rows={rows:?}");
};
assert!(matches.next().is_none(), "{name} with labels {labels:?} must be unique; rows={rows:?}");
value
}
+86 -4
View File
@@ -66,9 +66,10 @@ impl Evaluator {
}
/// IsObjectLocked checks if it is appropriate to remove an
/// object according to its persisted object-lock metadata.
/// object according to its persisted object-lock metadata and the bucket
/// default retention.
pub fn is_object_locked(&self, obj: &ObjectOpts) -> bool {
object_lock::is_object_locked_by_metadata(&obj.user_defined, obj.delete_marker)
object_lock::is_object_locked(&obj.user_defined, obj.delete_marker, self.lock_retention.as_deref(), obj.mod_time)
}
/// eval will return a lifecycle event for each object in objs for a given time.
@@ -198,8 +199,9 @@ mod tests {
use rustfs_common::metrics::IlmAction;
use s3s::dto::{
BucketLifecycleConfiguration, ExpirationStatus, LifecycleExpiration, LifecycleRule, ObjectLockConfiguration,
ObjectLockEnabled, Transition, TransitionStorageClass,
BucketLifecycleConfiguration, DefaultRetention, ExpirationStatus, LifecycleExpiration, LifecycleRule,
NoncurrentVersionExpiration, ObjectLockConfiguration, ObjectLockEnabled, ObjectLockRetentionMode, ObjectLockRule,
Transition, TransitionStorageClass,
};
use s3s::header::{X_AMZ_OBJECT_LOCK_LEGAL_HOLD, X_AMZ_OBJECT_LOCK_MODE, X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE};
use time::OffsetDateTime;
@@ -300,6 +302,40 @@ mod tests {
})
}
fn lock_enabled_with_default_retention(days: i32) -> Arc<ObjectLockConfiguration> {
Arc::new(ObjectLockConfiguration {
object_lock_enabled: Some(ObjectLockEnabled::from_static(ObjectLockEnabled::ENABLED)),
rule: Some(ObjectLockRule {
default_retention: Some(DefaultRetention {
days: Some(days),
mode: Some(ObjectLockRetentionMode::from_static(ObjectLockRetentionMode::GOVERNANCE)),
years: None,
}),
}),
})
}
fn noncurrent_expiration_lifecycle() -> Arc<BucketLifecycleConfiguration> {
Arc::new(BucketLifecycleConfiguration {
expiry_updated_at: None,
rules: vec![LifecycleRule {
status: ExpirationStatus::from_static(ExpirationStatus::ENABLED),
expiration: None,
abort_incomplete_multipart_upload: None,
del_marker_expiration: None,
filter: None,
id: Some("expire-noncurrent".to_string()),
noncurrent_version_expiration: Some(NoncurrentVersionExpiration {
noncurrent_days: Some(1),
newer_noncurrent_versions: None,
}),
noncurrent_version_transitions: None,
prefix: None,
transitions: None,
}],
})
}
fn object_opts(replication_status: ReplicationStatusType, version_purge_status: VersionPurgeStatusType) -> ObjectOpts {
ObjectOpts {
name: "logs/object".to_string(),
@@ -459,6 +495,52 @@ mod tests {
assert_eq!(events[0].action, IlmAction::NoneAction);
}
#[tokio::test]
async fn evaluator_skips_noncurrent_expiration_during_default_retention() {
let evaluator =
Evaluator::new(noncurrent_expiration_lifecycle()).with_lock_retention(Some(lock_enabled_with_default_retention(30)));
let successor_time = OffsetDateTime::now_utc() - time::Duration::days(2);
let noncurrent = ObjectOpts {
name: "logs/object".to_string(),
mod_time: Some(successor_time - time::Duration::days(1)),
successor_mod_time: Some(successor_time),
version_id: Some(Uuid::new_v4()),
is_latest: false,
num_versions: 1,
..Default::default()
};
let events = evaluator
.eval(&[noncurrent])
.await
.expect("lifecycle evaluation should succeed");
assert_eq!(events[0].action, IlmAction::NoneAction);
}
#[tokio::test]
async fn evaluator_allows_noncurrent_expiration_after_default_retention() {
let evaluator =
Evaluator::new(noncurrent_expiration_lifecycle()).with_lock_retention(Some(lock_enabled_with_default_retention(1)));
let successor_time = OffsetDateTime::now_utc() - time::Duration::days(2);
let noncurrent = ObjectOpts {
name: "logs/object".to_string(),
mod_time: Some(successor_time - time::Duration::days(1)),
successor_mod_time: Some(successor_time),
version_id: Some(Uuid::new_v4()),
is_latest: false,
num_versions: 1,
..Default::default()
};
let events = evaluator
.eval(&[noncurrent])
.await
.expect("lifecycle evaluation should succeed");
assert_eq!(events[0].action, IlmAction::DeleteVersionAction);
}
#[tokio::test]
async fn evaluator_skips_transition_while_replication_pending() {
let evaluator = Evaluator::new(latest_transition_lifecycle());
+204 -1
View File
@@ -14,7 +14,7 @@
use std::collections::HashMap;
use s3s::dto::ObjectLockRetentionMode;
use s3s::dto::{ObjectLockConfiguration, ObjectLockRetentionMode};
use s3s::header::{X_AMZ_OBJECT_LOCK_LEGAL_HOLD, X_AMZ_OBJECT_LOCK_MODE, X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE};
use time::{OffsetDateTime, format_description};
@@ -43,6 +43,90 @@ pub fn is_object_locked_by_metadata(user_defined: &HashMap<String, String>, is_d
.is_some_and(|retain_until| retain_until.unix_timestamp() > OffsetDateTime::now_utc().unix_timestamp())
}
/// Check persisted object-lock metadata and the bucket default retention.
///
/// A configured default retention with missing or malformed input is treated
/// as locked so a lifecycle worker cannot turn incomplete metadata into an
/// unsafe delete.
pub fn is_object_locked(
user_defined: &HashMap<String, String>,
is_delete_marker: bool,
config: Option<&ObjectLockConfiguration>,
mod_time: Option<OffsetDateTime>,
) -> bool {
if is_delete_marker {
return false;
}
if is_object_locked_by_metadata(user_defined, false) {
return true;
}
if has_explicit_lock_metadata(user_defined) {
return !explicit_lock_metadata_is_well_formed(user_defined);
}
let Some(default_retention) = config.and_then(|config| config.rule.as_ref()?.default_retention.as_ref()) else {
return false;
};
let Some(mode) = default_retention.mode.as_ref() else {
return true;
};
if !is_retention_mode(mode.as_str()) {
return true;
}
let Some(mod_time) = mod_time else {
return true;
};
let Some(retain_until) = default_retention_until(mod_time, default_retention) else {
return true;
};
retain_until.unix_timestamp() > OffsetDateTime::now_utc().unix_timestamp()
}
fn has_explicit_lock_metadata(user_defined: &HashMap<String, String>) -> bool {
user_defined.contains_key(X_AMZ_OBJECT_LOCK_LEGAL_HOLD.as_str())
|| user_defined.contains_key(X_AMZ_OBJECT_LOCK_MODE.as_str())
|| user_defined.contains_key(X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE.as_str())
}
fn explicit_lock_metadata_is_well_formed(user_defined: &HashMap<String, String>) -> bool {
if user_defined
.get(X_AMZ_OBJECT_LOCK_LEGAL_HOLD.as_str())
.is_some_and(|value| !value.eq_ignore_ascii_case("ON") && !value.eq_ignore_ascii_case("OFF"))
{
return false;
}
match (
user_defined.get(X_AMZ_OBJECT_LOCK_MODE.as_str()),
user_defined.get(X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE.as_str()),
) {
(None, None) => true,
(Some(mode), Some(retain_until)) => {
is_retention_mode(mode)
&& OffsetDateTime::parse(retain_until, &format_description::well_known::Iso8601::DEFAULT).is_ok()
}
_ => false,
}
}
fn default_retention_until(mod_time: OffsetDateTime, retention: &s3s::dto::DefaultRetention) -> Option<OffsetDateTime> {
match (retention.days, retention.years) {
(Some(days), None) if days > 0 => Some(mod_time.saturating_add(time::Duration::days(i64::from(days)))),
(None, Some(years)) if years > 0 => add_years(mod_time, years),
_ => None,
}
}
fn add_years(mod_time: OffsetDateTime, years: i32) -> Option<OffsetDateTime> {
let target_year = mod_time.year().checked_add(years)?;
mod_time
.replace_year(target_year)
.or_else(|_| mod_time.replace_day(28).and_then(|date| date.replace_year(target_year)))
.ok()
}
fn is_retention_mode(mode: &str) -> bool {
mode.eq_ignore_ascii_case(ObjectLockRetentionMode::COMPLIANCE)
|| mode.eq_ignore_ascii_case(ObjectLockRetentionMode::GOVERNANCE)
@@ -52,6 +136,9 @@ fn is_retention_mode(mode: &str) -> bool {
mod tests {
use super::*;
use s3s::dto::{DefaultRetention, ObjectLockEnabled, ObjectLockRule};
use time::Duration;
#[test]
fn is_object_locked_by_metadata_preserves_object_lock_parser_behavior() {
let mut user_defined = HashMap::new();
@@ -60,4 +147,120 @@ mod tests {
assert!(is_object_locked_by_metadata(&user_defined, false));
assert!(!is_object_locked_by_metadata(&user_defined, true));
}
fn default_retention_config(days: i32) -> ObjectLockConfiguration {
ObjectLockConfiguration {
object_lock_enabled: Some(ObjectLockEnabled::from_static(ObjectLockEnabled::ENABLED)),
rule: Some(ObjectLockRule {
default_retention: Some(DefaultRetention {
days: Some(days),
mode: Some(ObjectLockRetentionMode::from_static(ObjectLockRetentionMode::GOVERNANCE)),
years: None,
}),
}),
}
}
#[test]
fn default_retention_blocks_lifecycle_delete_until_expired() {
let config = default_retention_config(30);
let created = OffsetDateTime::now_utc() - Duration::days(1);
assert!(is_object_locked(&HashMap::new(), false, Some(&config), Some(created)));
}
#[test]
fn expired_default_retention_allows_lifecycle_delete() {
let config = default_retention_config(1);
let created = OffsetDateTime::now_utc() - Duration::days(2);
assert!(!is_object_locked(&HashMap::new(), false, Some(&config), Some(created)));
}
#[test]
fn missing_mod_time_blocks_default_retention_delete() {
let config = default_retention_config(30);
assert!(is_object_locked(&HashMap::new(), false, Some(&config), None));
}
#[test]
fn zero_default_retention_days_fail_closed() {
let config = default_retention_config(0);
let created = OffsetDateTime::now_utc() - Duration::days(2);
assert!(is_object_locked(&HashMap::new(), false, Some(&config), Some(created)));
}
#[test]
fn zero_default_retention_years_fail_closed() {
let config = ObjectLockConfiguration {
object_lock_enabled: Some(ObjectLockEnabled::from_static(ObjectLockEnabled::ENABLED)),
rule: Some(ObjectLockRule {
default_retention: Some(DefaultRetention {
days: None,
mode: Some(ObjectLockRetentionMode::from_static(ObjectLockRetentionMode::GOVERNANCE)),
years: Some(0),
}),
}),
};
let created = OffsetDateTime::now_utc() - Duration::days(2);
assert!(is_object_locked(&HashMap::new(), false, Some(&config), Some(created)));
}
#[test]
fn default_retention_years_block_lifecycle_delete_until_expired() {
let config = ObjectLockConfiguration {
object_lock_enabled: Some(ObjectLockEnabled::from_static(ObjectLockEnabled::ENABLED)),
rule: Some(ObjectLockRule {
default_retention: Some(DefaultRetention {
days: None,
mode: Some(ObjectLockRetentionMode::from_static(ObjectLockRetentionMode::COMPLIANCE)),
years: Some(1),
}),
}),
};
let created = OffsetDateTime::now_utc() - Duration::days(1);
assert!(is_object_locked(&HashMap::new(), false, Some(&config), Some(created)));
}
#[test]
fn expired_explicit_retention_does_not_reapply_default_retention() {
let config = default_retention_config(30);
let mut user_defined = HashMap::new();
user_defined.insert(
X_AMZ_OBJECT_LOCK_MODE.as_str().to_string(),
ObjectLockRetentionMode::GOVERNANCE.to_string(),
);
user_defined.insert(
X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE.as_str().to_string(),
(OffsetDateTime::now_utc() - Duration::days(1))
.format(&format_description::well_known::Iso8601::DEFAULT)
.expect("expired retention date should format"),
);
assert!(!is_object_locked(&user_defined, false, Some(&config), Some(OffsetDateTime::now_utc())));
}
#[test]
fn malformed_explicit_retention_fails_closed() {
let config = default_retention_config(1);
let mut user_defined = HashMap::new();
user_defined.insert(
X_AMZ_OBJECT_LOCK_MODE.as_str().to_string(),
ObjectLockRetentionMode::GOVERNANCE.to_string(),
);
let created = OffsetDateTime::now_utc() - Duration::days(2);
assert!(is_object_locked(&user_defined, false, Some(&config), Some(created)));
}
#[test]
fn delete_markers_are_not_locked_by_default_retention() {
let config = default_retention_config(30);
assert!(!is_object_locked(&HashMap::new(), true, Some(&config), None));
}
}
+104 -1
View File
@@ -54,12 +54,37 @@ pub(crate) struct IlmActionTaskStats {
pub(crate) value: u64,
}
#[derive(Debug, Clone, Default)]
pub(crate) struct IlmQueueTaskStats {
pub(crate) action: String,
pub(crate) state: String,
pub(crate) value: u64,
}
#[derive(Debug, Clone, Default)]
pub(crate) struct IlmTaskEventStats {
pub(crate) action: String,
pub(crate) result: String,
pub(crate) value: u64,
}
#[derive(Debug, Clone, Default)]
pub(crate) struct IlmBackpressureStats {
pub(crate) action: String,
pub(crate) reason: String,
pub(crate) value: u64,
}
/// ILM statistics with runtime-local node identity and bounded action/state details.
#[derive(Debug, Clone, Default)]
pub(crate) struct IlmRuntimeStats {
pub(crate) server: String,
pub(crate) stats: IlmStats,
pub(crate) action_tasks: Vec<IlmActionTaskStats>,
pub(crate) queue_tasks: Vec<IlmQueueTaskStats>,
pub(crate) task_events: Vec<IlmTaskEventStats>,
pub(crate) backpressure: Vec<IlmBackpressureStats>,
pub(crate) versions_scanned: u64,
}
fn is_live_action_task_state(state: &str) -> bool {
@@ -112,6 +137,30 @@ pub(crate) fn collect_ilm_runtime_metrics(stats: &IlmRuntimeStats) -> Vec<Promet
}),
);
metrics.extend(stats.queue_tasks.iter().map(|task| {
PrometheusMetric::from_descriptor(&ILM_TASKS_MD, task.value as f64)
.with_label_owned(SERVER_LABEL, stats.server.clone())
.with_label_owned(ACTION_LABEL, task.action.clone())
.with_label_owned(QUEUE_STATE_LABEL, task.state.clone())
}));
metrics.extend(stats.task_events.iter().map(|event| {
PrometheusMetric::from_descriptor(&ILM_TASK_EVENTS_MD, event.value as f64)
.with_label_owned(SERVER_LABEL, stats.server.clone())
.with_label_owned(ACTION_LABEL, event.action.clone())
.with_label_owned(RESULT_LABEL, event.result.clone())
}));
metrics.extend(stats.backpressure.iter().map(|event| {
PrometheusMetric::from_descriptor(&ILM_QUEUE_BACKPRESSURE_MD, event.value as f64)
.with_label_owned(SERVER_LABEL, stats.server.clone())
.with_label_owned(ACTION_LABEL, event.action.clone())
.with_label_owned(REASON_LABEL, event.reason.clone())
}));
metrics.push(
PrometheusMetric::from_descriptor(&ILM_VERSIONS_SCANNED_BY_SERVER_MD, stats.versions_scanned as f64)
.with_label_owned(SERVER_LABEL, stats.server.clone())
.with_label_owned(SOURCE_LABEL, "lifecycle".to_string()),
);
metrics
}
@@ -135,6 +184,22 @@ mod tests {
let runtime_stats = IlmRuntimeStats {
server: "node1:9000".to_string(),
stats,
queue_tasks: vec![IlmQueueTaskStats {
action: "transition".to_string(),
state: "pending".to_string(),
value: 8,
}],
task_events: vec![IlmTaskEventStats {
action: "transition".to_string(),
result: "completed".to_string(),
value: 7,
}],
backpressure: vec![IlmBackpressureStats {
action: "transition".to_string(),
reason: "queue_full".to_string(),
value: 2,
}],
versions_scanned: 1000000,
action_tasks: vec![
IlmActionTaskStats {
action: "expiry".to_string(),
@@ -156,7 +221,7 @@ mod tests {
let metrics = collect_ilm_runtime_metrics(&runtime_stats);
assert_eq!(metrics.len(), 11);
assert_eq!(metrics.len(), 15);
let pending = metrics.iter().find(|m| m.value == 100.0);
assert!(pending.is_some());
@@ -178,6 +243,44 @@ mod tests {
});
assert!(transition_timeout.is_none());
let transition_queue = metrics.iter().find(|m| {
m.name == ILM_TASKS_MD.get_full_metric_name()
&& m.labels
.iter()
.any(|(name, value)| *name == ACTION_LABEL && value.as_ref() == "transition")
&& m.labels
.iter()
.any(|(name, value)| *name == QUEUE_STATE_LABEL && value.as_ref() == "pending")
});
assert_eq!(transition_queue.map(|metric| metric.value), Some(8.0));
let completed = metrics.iter().find(|m| {
m.name == ILM_TASK_EVENTS_MD.get_full_metric_name()
&& m.labels
.iter()
.any(|(name, value)| *name == RESULT_LABEL && value.as_ref() == "completed")
});
assert_eq!(completed.map(|metric| metric.value), Some(7.0));
let backpressure = metrics.iter().find(|m| {
m.name == ILM_QUEUE_BACKPRESSURE_MD.get_full_metric_name()
&& m.labels
.iter()
.any(|(name, value)| *name == REASON_LABEL && value.as_ref() == "queue_full")
});
assert_eq!(backpressure.map(|metric| metric.value), Some(2.0));
let version_detail = metrics.iter().find(|m| {
m.name == ILM_VERSIONS_SCANNED_BY_SERVER_MD.get_full_metric_name()
&& m.labels
.iter()
.any(|(name, value)| *name == SERVER_LABEL && value.as_ref() == "node1:9000")
&& m.labels
.iter()
.any(|(name, value)| *name == SOURCE_LABEL && value.as_ref() == "lifecycle")
});
assert_eq!(version_detail.map(|metric| metric.value), Some(1000000.0));
let transition_active = metrics.iter().find(|m| {
m.name == ILM_ACTION_TASKS_MD.get_full_metric_name()
&& m.labels
+4 -1
View File
@@ -59,9 +59,12 @@ pub use cluster_iam::{IamStats, collect_iam_metrics};
pub use cluster_usage::{BucketUsageStats, ClusterUsageStats, collect_bucket_usage_metrics, collect_cluster_usage_metrics};
pub use compression::{CompressionClusterStats, collect_compression_cluster_metrics};
pub use dial9::{Dial9Stats, collect_current_dial9_metrics, collect_dial9_metrics, is_dial9_enabled};
pub(crate) use ilm::{IlmActionTaskStats, IlmRuntimeStats, collect_ilm_runtime_metrics};
pub(crate) use ilm::{
IlmActionTaskStats, IlmBackpressureStats, IlmQueueTaskStats, IlmRuntimeStats, IlmTaskEventStats, collect_ilm_runtime_metrics,
};
pub use ilm::{IlmStats, collect_ilm_metrics};
pub use node::{DiskStats, collect_node_metrics};
pub(crate) use notification::collect_notification_runtime_metrics;
pub use notification::{NotificationStats, collect_notification_metrics};
pub(crate) use notification_target::{NotificationTargetRuntimeStats, collect_notification_target_runtime_metrics};
pub use notification_target::{NotificationTargetStats, collect_notification_target_metrics};
@@ -19,9 +19,12 @@
use crate::metrics::report::PrometheusMetric;
use crate::metrics::schema::cluster_notification::{
NOTIFICATION_CURRENT_SEND_IN_PROGRESS_MD, NOTIFICATION_EVENTS_ERRORS_TOTAL_MD, NOTIFICATION_EVENTS_SENT_TOTAL_MD,
NOTIFICATION_EVENTS_SKIPPED_TOTAL_MD,
NOTIFICATION_CURRENT_SEND_IN_PROGRESS_BY_SERVER_MD, NOTIFICATION_CURRENT_SEND_IN_PROGRESS_MD,
NOTIFICATION_EVENTS_ERRORS_TOTAL_BY_SERVER_MD, NOTIFICATION_EVENTS_ERRORS_TOTAL_MD,
NOTIFICATION_EVENTS_SENT_TOTAL_BY_SERVER_MD, NOTIFICATION_EVENTS_SENT_TOTAL_MD,
NOTIFICATION_EVENTS_SKIPPED_TOTAL_BY_SERVER_MD, NOTIFICATION_EVENTS_SKIPPED_TOTAL_MD, SERVER,
};
use std::borrow::Cow;
/// Notification statistics.
#[derive(Debug, Clone, Default)]
@@ -49,6 +52,30 @@ pub fn collect_notification_metrics(stats: &NotificationStats) -> Vec<Prometheus
]
}
/// Collects the legacy aggregate metrics and node-local runtime siblings.
pub(crate) fn collect_notification_runtime_metrics(stats: &NotificationStats, server: &str) -> Vec<PrometheusMetric> {
let mut metrics = collect_notification_metrics(stats);
if server.is_empty() {
return metrics;
}
let server_label: Cow<'static, str> = Cow::Owned(server.to_string());
metrics.extend([
PrometheusMetric::from_descriptor(
&NOTIFICATION_CURRENT_SEND_IN_PROGRESS_BY_SERVER_MD,
stats.current_send_in_progress as f64,
)
.with_label(SERVER, server_label.clone()),
PrometheusMetric::from_descriptor(&NOTIFICATION_EVENTS_ERRORS_TOTAL_BY_SERVER_MD, stats.events_errors_total as f64)
.with_label(SERVER, server_label.clone()),
PrometheusMetric::from_descriptor(&NOTIFICATION_EVENTS_SENT_TOTAL_BY_SERVER_MD, stats.events_sent_total as f64)
.with_label(SERVER, server_label.clone()),
PrometheusMetric::from_descriptor(&NOTIFICATION_EVENTS_SKIPPED_TOTAL_BY_SERVER_MD, stats.events_skipped_total as f64)
.with_label(SERVER, server_label),
]);
metrics
}
#[cfg(test)]
mod tests {
use super::*;
@@ -86,4 +113,32 @@ mod tests {
assert!(metric.labels.is_empty());
}
}
#[test]
fn runtime_metrics_keep_aggregate_and_add_server_siblings() {
let stats = NotificationStats {
current_send_in_progress: 5,
events_errors_total: 10,
events_sent_total: 100,
events_skipped_total: 2,
};
let metrics = collect_notification_runtime_metrics(&stats, "node1:9000");
assert_eq!(metrics.len(), 8);
assert_eq!(metrics.iter().filter(|metric| metric.labels.is_empty()).count(), 4);
assert_eq!(metrics.iter().filter(|metric| metric.labels.len() == 1).count(), 4);
assert!(metrics.iter().filter(|metric| metric.labels.len() == 1).all(|metric| {
metric
.labels
.iter()
.any(|(name, value)| *name == SERVER && value == "node1:9000")
}));
}
#[test]
fn runtime_metrics_do_not_publish_empty_server_series() {
let metrics = collect_notification_runtime_metrics(&NotificationStats::default(), "");
assert_eq!(metrics.len(), 4);
assert!(metrics.iter().all(|metric| metric.labels.is_empty()));
}
}
+64 -1
View File
@@ -184,6 +184,15 @@ pub struct ScannerBucketDriveResultStats {
pub count: u64,
}
#[derive(Debug, Clone, Default)]
pub struct ScannerActiveBucketDriveStats {
pub source: String,
pub bucket: String,
pub drive: String,
pub count: u64,
pub age_seconds: u64,
}
/// Scanner statistics with runtime-local node identity and bounded source/result details.
#[derive(Debug, Clone, Default)]
pub(crate) struct ScannerRuntimeStats {
@@ -195,6 +204,7 @@ pub(crate) struct ScannerRuntimeStats {
pub(crate) bucket_drive_results: Vec<ScannerBucketDriveResultStats>,
pub(crate) current_cycle_bucket_drive_results: Vec<ScannerBucketDriveResultStats>,
pub(crate) last_cycle_bucket_drive_results: Vec<ScannerBucketDriveResultStats>,
pub(crate) active_bucket_drive_scans: Vec<ScannerActiveBucketDriveStats>,
}
/// Collects scanner metrics from the given stats.
@@ -452,6 +462,23 @@ fn collect_scanner_metrics_with_runtime(stats: &ScannerStats, runtime: Option<&S
&runtime.last_cycle_bucket_drive_results,
Some("last"),
);
for active in &runtime.active_bucket_drive_scans {
let labels = |metric: PrometheusMetric| {
metric
.with_label_owned(SERVER_LABEL, runtime.server.clone())
.with_label_owned(SOURCE_LABEL, active.source.clone())
.with_label_owned(BUCKET_LABEL, active.bucket.clone())
.with_label_owned(DRIVE_LABEL, active.drive.clone())
};
metrics.push(labels(PrometheusMetric::from_descriptor(
&SCANNER_ACTIVE_BUCKET_DRIVE_SCANS_MD,
active.count as f64,
)));
metrics.push(labels(PrometheusMetric::from_descriptor(
&SCANNER_ACTIVE_BUCKET_DRIVE_SCAN_AGE_SECONDS_MD,
active.age_seconds as f64,
)));
}
}
metrics
@@ -566,6 +593,13 @@ mod tests {
result: "error".to_string(),
count: 2,
}],
active_bucket_drive_scans: vec![ScannerActiveBucketDriveStats {
source: "usage".to_string(),
bucket: "photos".to_string(),
drive: "/data1".to_string(),
count: 2,
age_seconds: 7,
}],
stats: ScannerStats {
bucket_scans_finished: 100,
bucket_scans_started: 100,
@@ -642,7 +676,7 @@ mod tests {
let metrics = collect_scanner_runtime_metrics(&stats);
report_metrics(&metrics);
assert_eq!(metrics.len(), 90);
assert_eq!(metrics.len(), 92);
let objects = metrics.iter().find(|m| m.value == 1000000.0);
assert!(objects.is_some());
@@ -656,6 +690,35 @@ mod tests {
assert_eq!(active_paths.map(|m| m.value), Some(4.0));
assert_eq!(active_paths.map(|m| m.labels.len()), Some(0));
let active_bucket_drive = metrics
.iter()
.find(|m| m.name == SCANNER_ACTIVE_BUCKET_DRIVE_SCANS_MD.get_full_metric_name())
.expect("active bucket-drive metric");
assert_eq!(active_bucket_drive.value, 2.0);
assert!(
active_bucket_drive
.labels
.iter()
.any(|(name, value)| *name == SOURCE_LABEL && value == "usage")
);
assert!(
active_bucket_drive
.labels
.iter()
.any(|(name, value)| *name == BUCKET_LABEL && value == "photos")
);
assert!(
active_bucket_drive
.labels
.iter()
.any(|(name, value)| *name == DRIVE_LABEL && value == "/data1")
);
let active_age = metrics
.iter()
.find(|m| m.name == SCANNER_ACTIVE_BUCKET_DRIVE_SCAN_AGE_SECONDS_MD.get_full_metric_name())
.expect("active bucket-drive age metric");
assert_eq!(active_age.value, 7.0);
let bucket_drive_result = metrics
.iter()
.find(|m| m.name == SCANNER_BUCKET_DRIVE_RESULT_TOTAL_MD.get_full_metric_name());
@@ -60,6 +60,10 @@ pub struct DriveDetailedStats {
pub api_latency_micros: Option<u64>,
/// Health status (1=healthy, 0=unhealthy)
pub health: u8,
/// Total successful write operations when backed by a real disk metric.
pub writes_total: Option<u64>,
/// Total successful delete operations when backed by a real disk metric.
pub deletes_total: Option<u64>,
/// Reads per second when backed by a real iostat sample
pub reads_per_sec: Option<f64>,
/// Kilobytes read per second when backed by a real iostat sample
@@ -282,6 +286,12 @@ pub(crate) fn collect_drive_runtime_detailed_metrics(stats: &[DriveRuntimeDetail
if let Some(value) = stat.stats.perc_util {
push_drive_metric(&mut metrics, &DRIVE_PERC_UTIL_MD, value, server_label, drive_label);
}
if let Some(value) = stat.stats.writes_total {
push_drive_metric(&mut metrics, &DRIVE_WRITES_TOTAL_MD, value as f64, server_label, drive_label);
}
if let Some(value) = stat.stats.deletes_total {
push_drive_metric(&mut metrics, &DRIVE_DELETES_TOTAL_MD, value as f64, server_label, drive_label);
}
if let Some(labels) = &topology_labels {
if let Some(disk_id) = stat.disk_id.as_ref().filter(|disk_id| !disk_id.is_empty()) {
metrics.push(
@@ -449,6 +459,8 @@ mod tests {
waiting_io: Some(3),
api_latency_micros: Some(1500),
health: 1,
writes_total: Some(11),
deletes_total: Some(4),
reads_per_sec: Some(100.0),
reads_kb_per_sec: Some(1024.0),
reads_await: Some(5.5),
@@ -462,7 +474,7 @@ mod tests {
let metrics = collect_drive_runtime_detailed_metrics(&stats);
report_metrics(&metrics);
assert_eq!(metrics.len(), 34);
assert_eq!(metrics.len(), 36);
// Verify total bytes metric
let total_bytes_name = DRIVE_TOTAL_BYTES_MD.get_full_metric_name();
@@ -503,6 +515,8 @@ mod tests {
API_LABEL,
],
);
assert_metric_label_keys(&metrics, &DRIVE_WRITES_TOTAL_MD, 11.0, &[SERVER_LABEL, DRIVE_LABEL]);
assert_metric_label_keys(&metrics, &DRIVE_DELETES_TOTAL_MD, 4.0, &[SERVER_LABEL, DRIVE_LABEL]);
}
#[test]
@@ -524,6 +538,8 @@ mod tests {
waiting_io: None,
api_latency_micros: None,
health: 1,
writes_total: None,
deletes_total: None,
reads_per_sec: None,
reads_kb_per_sec: None,
reads_await: None,
+93 -8
View File
@@ -62,7 +62,7 @@ use crate::metrics::collectors::{
collect_memory_metrics,
collect_network_metrics,
collect_node_metrics,
collect_notification_metrics,
collect_notification_runtime_metrics,
collect_notification_target_runtime_metrics,
collect_process_attributes,
collect_process_cpu_metrics,
@@ -120,12 +120,13 @@ use crate::metrics::schema::notification_target::{
};
use crate::metrics::schema::scanner::{
BUCKET_LABEL as SCANNER_BUCKET_LABEL, CYCLE_SCOPE_LABEL as SCANNER_CYCLE_SCOPE_LABEL, DRIVE_LABEL as SCANNER_DRIVE_LABEL,
RESULT_LABEL as SCANNER_RESULT_LABEL, SCANNER_BUCKET_DRIVE_RESULT_TOTAL_MD, SCANNER_CYCLE_BUCKET_DRIVE_RESULT_MD,
RESULT_LABEL as SCANNER_RESULT_LABEL, SCANNER_ACTIVE_BUCKET_DRIVE_SCAN_AGE_SECONDS_MD, SCANNER_ACTIVE_BUCKET_DRIVE_SCANS_MD,
SCANNER_BUCKET_DRIVE_RESULT_TOTAL_MD, SCANNER_CYCLE_BUCKET_DRIVE_RESULT_MD, SOURCE_LABEL as SCANNER_SOURCE_LABEL,
};
use crate::metrics::schema::system_drive::{
API_LABEL as DRIVE_API_LABEL, DISK_ID_LABEL, DRIVE_API_CALLS_MD, DRIVE_API_LATENCY_BY_API_MD, DRIVE_HEALING_MD,
DRIVE_INDEX_LABEL, DRIVE_INFO_MD, DRIVE_LABEL, DRIVE_OFFLINE_DURATION_SECONDS_MD, DRIVE_RUNTIME_STATE_MD, DRIVE_SCANNING_MD,
POOL_INDEX_LABEL, SET_INDEX_LABEL, STATE_LABEL as DRIVE_STATE_LABEL,
API_LABEL as DRIVE_API_LABEL, DISK_ID_LABEL, DRIVE_API_CALLS_MD, DRIVE_API_LATENCY_BY_API_MD, DRIVE_DELETES_TOTAL_MD,
DRIVE_HEALING_MD, DRIVE_INDEX_LABEL, DRIVE_INFO_MD, DRIVE_LABEL, DRIVE_OFFLINE_DURATION_SECONDS_MD, DRIVE_RUNTIME_STATE_MD,
DRIVE_SCANNING_MD, DRIVE_WRITES_TOTAL_MD, POOL_INDEX_LABEL, SET_INDEX_LABEL, STATE_LABEL as DRIVE_STATE_LABEL,
};
use crate::metrics::schema::system_process::{PROCESS_EXECUTABLE_NAME_LABEL, PROCESS_PID_LABEL};
use crate::metrics::stats_collector::{
@@ -303,15 +304,33 @@ type AuditTargetKey = (String, String); // (server, target_id)
type NotificationLegacyTargetKey = (String, String); // (target_id, target_type)
type NotificationTargetKey = (String, String, String); // (server, target_id, target_type)
type DriveTopologyKey = (String, String, String, String, String); // (server, drive, pool, set, drive_index)
type DriveBasicKey = (String, String); // (server, drive)
type DriveTopologyApiKey = (String, String, String, String, String, String); // (server, drive, pool, set, drive_index, api)
type DriveInfoKey = (String, String, String, String, String, String); // (server, drive, pool, set, drive_index, disk_id)
type ScannerCycleBucketDriveResultKey = (String, String, String, String, String); // (server, cycle_scope, bucket, drive, result)
type ScannerBucketDriveResultKey = (String, String, String, String); // (server, bucket, drive, result)
type ScannerActiveBucketDriveKey = (String, String, String, String); // (server, source, bucket, drive)
fn drive_info_live_keys(stats: &[DriveRuntimeDetailedStats]) -> HashSet<DriveInfoKey> {
stats.iter().filter_map(drive_info_key).collect()
}
fn drive_basic_live_keys(stats: &[DriveRuntimeDetailedStats]) -> HashSet<DriveBasicKey> {
stats
.iter()
.map(|stat| (stat.stats.server.clone(), stat.stats.drive.clone()))
.collect()
}
fn retire_drive_basic_metric_series(key: &DriveBasicKey) -> usize {
let labels = [
(SERVER_LABEL, Cow::Owned(key.0.clone())),
(DRIVE_LABEL, Cow::Owned(key.1.clone())),
];
retire_metric_series(&DRIVE_WRITES_TOTAL_MD.get_full_metric_name(), &labels)
+ retire_metric_series(&DRIVE_DELETES_TOTAL_MD.get_full_metric_name(), &labels)
}
fn drive_topology_live_keys(stats: &[DriveRuntimeDetailedStats]) -> HashSet<DriveTopologyKey> {
stats.iter().filter_map(drive_topology_key).collect()
}
@@ -469,6 +488,25 @@ fn retire_scanner_bucket_drive_result_metric_series(key: &ScannerBucketDriveResu
retire_metric_series(&SCANNER_BUCKET_DRIVE_RESULT_TOTAL_MD.get_full_metric_name(), &labels)
}
fn scanner_active_bucket_drive_live_keys(stats: &ScannerRuntimeStats) -> HashSet<ScannerActiveBucketDriveKey> {
stats
.active_bucket_drive_scans
.iter()
.map(|active| (stats.server.clone(), active.source.clone(), active.bucket.clone(), active.drive.clone()))
.collect()
}
fn retire_scanner_active_bucket_drive_metric_series(key: &ScannerActiveBucketDriveKey) -> usize {
let labels = [
(SERVER_LABEL, Cow::Owned(key.0.clone())),
(SCANNER_SOURCE_LABEL, Cow::Owned(key.1.clone())),
(SCANNER_BUCKET_LABEL, Cow::Owned(key.2.clone())),
(SCANNER_DRIVE_LABEL, Cow::Owned(key.3.clone())),
];
retire_metric_series(&SCANNER_ACTIVE_BUCKET_DRIVE_SCANS_MD.get_full_metric_name(), &labels)
+ retire_metric_series(&SCANNER_ACTIVE_BUCKET_DRIVE_SCAN_AGE_SECONDS_MD.get_full_metric_name(), &labels)
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Default)]
pub struct MetricsRuntimeCollectorHealthSnapshot {
pub healthy_collectors: u8,
@@ -1841,6 +1879,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
let token_clone = token.clone();
tokio::spawn(async move {
let mut interval = metrics_interval(node_interval, Duration::ZERO);
let mut prev_drive_basic_keys: HashSet<DriveBasicKey> = HashSet::new();
let mut prev_drive_info_keys: HashSet<DriveInfoKey> = HashSet::new();
let mut prev_drive_topology_keys: HashSet<DriveTopologyKey> = HashSet::new();
let mut prev_drive_topology_api_keys: HashSet<DriveTopologyApiKey> = HashSet::new();
@@ -1851,6 +1890,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
run_metrics_collector_tick(health, MetricsCollectorTaskId::NodeDiskStats, "node_disk_stats", async {
let (disk_stats, drive_stats, drive_counts) = collect_disk_and_system_drive_runtime_stats().await;
let current_drive_info_keys = drive_info_live_keys(&drive_stats);
let current_drive_basic_keys = drive_basic_live_keys(&drive_stats);
let current_drive_topology_keys = drive_topology_live_keys(&drive_stats);
let current_drive_topology_api_keys = drive_topology_api_live_keys(&drive_stats);
let retire_drive_info_keys = if has_seen_drive_info_snapshot {
@@ -1858,6 +1898,11 @@ pub fn init_metrics_runtime(token: CancellationToken) {
} else {
Vec::new()
};
let retire_drive_basic_keys = if has_seen_drive_info_snapshot {
prev_drive_basic_keys.difference(&current_drive_basic_keys).cloned().collect::<Vec<_>>()
} else {
Vec::new()
};
let retire_drive_topology_keys = if has_seen_drive_info_snapshot {
prev_drive_topology_keys.difference(&current_drive_topology_keys).cloned().collect::<Vec<_>>()
} else {
@@ -1872,6 +1917,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
Vec::new()
};
prev_drive_info_keys = current_drive_info_keys;
prev_drive_basic_keys = current_drive_basic_keys;
prev_drive_topology_keys = current_drive_topology_keys;
prev_drive_topology_api_keys = current_drive_topology_api_keys;
has_seen_drive_info_snapshot = true;
@@ -1882,6 +1928,9 @@ pub fn init_metrics_runtime(token: CancellationToken) {
for key in retire_drive_info_keys {
let _ = retire_drive_info_metric_series(&key);
}
for key in retire_drive_basic_keys {
let _ = retire_drive_basic_metric_series(&key);
}
for key in retire_drive_topology_keys {
let _ = retire_drive_topology_metric_series(&key);
}
@@ -2106,14 +2155,14 @@ pub fn init_metrics_runtime(token: CancellationToken) {
_ = interval.tick() => {
run_metrics_collector_tick(health, MetricsCollectorTaskId::NotificationStats, "notification_stats", async {
let snapshot = notification_metrics_snapshot();
let mut metrics = collect_notification_metrics(&NotificationStats {
let server = current_local_node_identity();
let mut metrics = collect_notification_runtime_metrics(&NotificationStats {
current_send_in_progress: snapshot.current_send_in_progress,
events_errors_total: snapshot.events_errors_total,
events_sent_total: snapshot.events_sent_total,
events_skipped_total: snapshot.events_skipped_total,
});
}, &server);
let server = current_local_node_identity();
let target_stats = notification_target_metrics().await
.into_iter()
.map(|snapshot| NotificationTargetRuntimeStats {
@@ -2173,6 +2222,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
let mut has_seen_scanner_snapshot = false;
let mut prev_scanner_cycle_bucket_drive_result_keys: HashSet<ScannerCycleBucketDriveResultKey> = HashSet::new();
let mut prev_scanner_bucket_drive_result_keys: HashSet<ScannerBucketDriveResultKey> = HashSet::new();
let mut prev_scanner_active_bucket_drive_keys: HashSet<ScannerActiveBucketDriveKey> = HashSet::new();
loop {
tokio::select! {
_ = interval.tick() => {
@@ -2189,9 +2239,11 @@ pub fn init_metrics_runtime(token: CancellationToken) {
let mut retire_scanner_cycle_bucket_drive_result_keys = Vec::new();
let mut retire_scanner_bucket_drive_result_keys = Vec::new();
let mut retire_scanner_active_bucket_drive_keys = Vec::new();
if let Some(stats) = collect_scanner_runtime_metric_stats().await {
let current_cycle_keys = scanner_cycle_bucket_drive_result_live_keys(&stats);
let current_keys = scanner_bucket_drive_result_live_keys(&stats);
let current_active_keys = scanner_active_bucket_drive_live_keys(&stats);
if has_seen_scanner_snapshot {
retire_scanner_cycle_bucket_drive_result_keys = prev_scanner_cycle_bucket_drive_result_keys
.difference(&current_cycle_keys)
@@ -2201,9 +2253,14 @@ pub fn init_metrics_runtime(token: CancellationToken) {
.difference(&current_keys)
.cloned()
.collect();
retire_scanner_active_bucket_drive_keys = prev_scanner_active_bucket_drive_keys
.difference(&current_active_keys)
.cloned()
.collect();
}
prev_scanner_cycle_bucket_drive_result_keys = current_cycle_keys;
prev_scanner_bucket_drive_result_keys = current_keys;
prev_scanner_active_bucket_drive_keys = current_active_keys;
has_seen_scanner_snapshot = true;
metrics.extend(collect_scanner_runtime_metrics(&stats));
}
@@ -2217,6 +2274,9 @@ pub fn init_metrics_runtime(token: CancellationToken) {
for key in retire_scanner_bucket_drive_result_keys {
let _ = retire_scanner_bucket_drive_result_metric_series(&key);
}
for key in retire_scanner_active_bucket_drive_keys {
let _ = retire_scanner_active_bucket_drive_metric_series(&key);
}
},
).await;
}
@@ -2495,6 +2555,7 @@ fn collect_system_monitoring_metrics(
#[cfg(test)]
mod tests {
use super::*;
use crate::metrics::collectors::scanner::ScannerActiveBucketDriveStats;
use std::collections::{HashMap, HashSet};
use std::time::Duration;
use tokio::time::Instant;
@@ -2723,6 +2784,30 @@ mod tests {
assert!(current.contains(&("server-a".to_string(), "logs".to_string(), "/data1".to_string(), "success".to_string(),)));
}
#[test]
fn scanner_active_bucket_drive_keys_detect_completed_scans() {
let previous = scanner_active_bucket_drive_live_keys(&ScannerRuntimeStats {
server: "server-a".to_string(),
active_bucket_drive_scans: vec![ScannerActiveBucketDriveStats {
source: "usage".to_string(),
bucket: "photos".to_string(),
drive: "/data1".to_string(),
count: 1,
age_seconds: 3,
}],
..Default::default()
});
let current = scanner_active_bucket_drive_live_keys(&ScannerRuntimeStats {
server: "server-a".to_string(),
..Default::default()
});
assert!(
previous
.difference(&current)
.any(|key| key == &("server-a".to_string(), "usage".to_string(), "photos".to_string(), "/data1".to_string()))
);
}
#[test]
fn replication_proxy_bucket_keys_detect_removed_buckets() {
let previous = repl_proxy_bucket_live_keys(&[BucketReplicationRuntimeStats {
@@ -15,6 +15,10 @@
use crate::{MetricDescriptor, MetricName, new_counter_md, new_gauge_md, subsystems};
use std::sync::LazyLock;
pub const SERVER: &str = "server";
const SERVER_LABELS: [&str; 1] = [SERVER];
pub static NOTIFICATION_CURRENT_SEND_IN_PROGRESS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_gauge_md(
MetricName::NotificationCurrentSendInProgress,
@@ -24,6 +28,15 @@ pub static NOTIFICATION_CURRENT_SEND_IN_PROGRESS_MD: LazyLock<MetricDescriptor>
)
});
pub static NOTIFICATION_CURRENT_SEND_IN_PROGRESS_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_gauge_md(
MetricName::Custom("current_send_in_progress_by_server".to_string()),
"Number of concurrent async Send calls active to all targets by server",
&SERVER_LABELS,
subsystems::NOTIFICATION,
)
});
pub static NOTIFICATION_EVENTS_ERRORS_TOTAL_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::NotificationEventsErrorsTotal,
@@ -33,6 +46,15 @@ pub static NOTIFICATION_EVENTS_ERRORS_TOTAL_MD: LazyLock<MetricDescriptor> = Laz
)
});
pub static NOTIFICATION_EVENTS_ERRORS_TOTAL_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::Custom("events_errors_total_by_server".to_string()),
"Events that failed to be sent to the targets by server",
&SERVER_LABELS,
subsystems::NOTIFICATION,
)
});
pub static NOTIFICATION_EVENTS_SENT_TOTAL_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::NotificationEventsSentTotal,
@@ -42,6 +64,15 @@ pub static NOTIFICATION_EVENTS_SENT_TOTAL_MD: LazyLock<MetricDescriptor> = LazyL
)
});
pub static NOTIFICATION_EVENTS_SENT_TOTAL_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::Custom("events_sent_total_by_server".to_string()),
"Total number of events sent to the targets by server",
&SERVER_LABELS,
subsystems::NOTIFICATION,
)
});
pub static NOTIFICATION_EVENTS_SKIPPED_TOTAL_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::NotificationEventsSkippedTotal,
@@ -50,3 +81,12 @@ pub static NOTIFICATION_EVENTS_SKIPPED_TOTAL_MD: LazyLock<MetricDescriptor> = La
subsystems::NOTIFICATION,
)
});
pub static NOTIFICATION_EVENTS_SKIPPED_TOTAL_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::Custom("events_skipped_total_by_server".to_string()),
"Notification dispatch attempts skipped before delivery by server",
&SERVER_LABELS,
subsystems::NOTIFICATION,
)
});
@@ -371,6 +371,8 @@ pub enum MetricName {
DriveWaitingIO,
DriveAPILatencyMicros,
DriveHealth,
DriveWritesTotal,
DriveDeletesTotal,
DriveOfflineCount,
DriveOnlineCount,
@@ -780,6 +782,8 @@ impl MetricName {
Self::DriveWaitingIO => "waiting_io".to_string(),
Self::DriveAPILatencyMicros => "api_latency_micros".to_string(),
Self::DriveHealth => "health".to_string(),
Self::DriveWritesTotal => "writes_total".to_string(),
Self::DriveDeletesTotal => "deletes_total".to_string(),
Self::DriveOfflineCount => "offline_count".to_string(),
Self::DriveOnlineCount => "online_count".to_string(),
+40
View File
@@ -18,6 +18,10 @@ use std::sync::LazyLock;
pub const SERVER_LABEL: &str = "server";
pub const ACTION_LABEL: &str = "action";
pub const STATE_LABEL: &str = "state";
pub const QUEUE_STATE_LABEL: &str = "queue_state";
pub const RESULT_LABEL: &str = "result";
pub const REASON_LABEL: &str = "reason";
pub const SOURCE_LABEL: &str = "source";
pub static ILM_ACTION_TASKS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_gauge_md(
@@ -28,6 +32,33 @@ pub static ILM_ACTION_TASKS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
)
});
pub static ILM_TASKS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_gauge_md(
MetricName::Custom("tasks".to_string()),
"Current ILM task counts by server, action, and queue state",
&[SERVER_LABEL, ACTION_LABEL, QUEUE_STATE_LABEL],
subsystems::ILM,
)
});
pub static ILM_TASK_EVENTS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::Custom("task_events_total".to_string()),
"ILM task events by server, action, and result",
&[SERVER_LABEL, ACTION_LABEL, RESULT_LABEL],
subsystems::ILM,
)
});
pub static ILM_QUEUE_BACKPRESSURE_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::Custom("queue_backpressure_total".to_string()),
"ILM queue backpressure events by server, action, and reason",
&[SERVER_LABEL, ACTION_LABEL, REASON_LABEL],
subsystems::ILM,
)
});
pub static ILM_EXPIRY_PENDING_TASKS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_gauge_md(
MetricName::IlmExpiryPendingTasks,
@@ -108,3 +139,12 @@ pub static ILM_VERSIONS_SCANNED_MD: LazyLock<MetricDescriptor> = LazyLock::new(|
subsystems::ILM,
)
});
pub static ILM_VERSIONS_SCANNED_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::Custom("versions_scanned_by_server".to_string()),
"ILM lifecycle-checked object versions by server and source",
&[SERVER_LABEL, SOURCE_LABEL],
subsystems::ILM,
)
});
+18
View File
@@ -59,6 +59,24 @@ pub static SCANNER_CYCLE_BUCKET_DRIVE_RESULT_MD: LazyLock<MetricDescriptor> = La
)
});
pub static SCANNER_ACTIVE_BUCKET_DRIVE_SCANS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_gauge_md(
MetricName::Custom("active_bucket_drive_scans".to_string()),
"Current active scanner bucket-drive scans by server, source, bucket, and drive",
&[SERVER_LABEL, SOURCE_LABEL, BUCKET_LABEL, DRIVE_LABEL],
subsystems::SCANNER,
)
});
pub static SCANNER_ACTIVE_BUCKET_DRIVE_SCAN_AGE_SECONDS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_gauge_md(
MetricName::Custom("active_bucket_drive_scan_age_seconds".to_string()),
"Age of the oldest active scanner bucket-drive scan by server, source, bucket, and drive",
&[SERVER_LABEL, SOURCE_LABEL, BUCKET_LABEL, DRIVE_LABEL],
subsystems::SCANNER,
)
});
pub static SCANNER_BUCKET_SCANS_FINISHED_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::ScannerBucketScansFinished,
@@ -259,6 +259,24 @@ pub static DRIVE_HEALTH_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
)
});
pub static DRIVE_WRITES_TOTAL_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::DriveWritesTotal,
"Total successful write operations on a drive",
&ALL_DRIVE_LABELS[..],
subsystems::SYSTEM_DRIVE,
)
});
pub static DRIVE_DELETES_TOTAL_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
new_counter_md(
MetricName::DriveDeletesTotal,
"Total successful delete operations on a drive",
&ALL_DRIVE_LABELS[..],
subsystems::SYSTEM_DRIVE,
)
});
pub static DRIVE_OFFLINE_COUNT_MD: LazyLock<MetricDescriptor> =
LazyLock::new(|| new_gauge_md(MetricName::DriveOfflineCount, "Count of offline drives", &[], subsystems::SYSTEM_DRIVE));
+207 -6
View File
@@ -18,15 +18,15 @@
//! RustFS internal sources (storage layer, bucket monitor, system info)
//! and convert them to the Stats structs used by collectors.
use crate::metrics::collectors::scanner::{ScannerBucketDriveResultStats, ScannerSourceWorkStats};
use crate::metrics::collectors::scanner::{ScannerActiveBucketDriveStats, ScannerBucketDriveResultStats, ScannerSourceWorkStats};
use crate::metrics::collectors::{
ApiRequestMetricSupport, ApiRequestStats, BucketReplicationBacklogStats, BucketReplicationBandwidthStats,
BucketReplicationRuntimeStats, BucketReplicationStats, BucketReplicationTargetBacklogStats, BucketReplicationTargetFlowStats,
BucketReplicationTargetStats, BucketStats, BucketUsageStats, ClusterConfigStats, ClusterHealthStats, ClusterStats,
ClusterUsageStats, CompressionClusterStats, CpuStats, DiskStats, DriveCountStats, DriveDetailedStats,
DriveRuntimeDetailedStats, ErasureSetStats, HostNetworkStats, IamStats, IlmActionTaskStats, IlmRuntimeStats, IlmStats,
MemoryStats, NetworkStats, ProcessStats, ProcessStatusType, ReplicationStats, ResourceStats, ScannerRuntimeStats,
ScannerStats,
DriveRuntimeDetailedStats, ErasureSetStats, HostNetworkStats, IamStats, IlmActionTaskStats, IlmBackpressureStats,
IlmQueueTaskStats, IlmRuntimeStats, IlmStats, IlmTaskEventStats, MemoryStats, NetworkStats, ProcessStats, ProcessStatusType,
ReplicationStats, ResourceStats, ScannerRuntimeStats, ScannerStats,
};
use crate::metrics::runtime_sources::{ObsIlmRuntimeSnapshot, bucket_monitor_handle, iam_metrics_snapshot, ilm_runtime_snapshot};
use crate::metrics::{
@@ -38,7 +38,10 @@ use crate::metrics::{
use crate::node_identity::current_local_node_identity;
use jiff::Timestamp;
use rustfs_common::heal_channel::HealScanMode;
use rustfs_common::metrics::{ScannerBucketDriveResultSnapshot, ScannerMetricsReport, ScannerSourceWorkSnapshot, global_metrics};
use rustfs_common::metrics::{
ScannerActiveBucketDriveSnapshot, ScannerBucketDriveResultSnapshot, ScannerMetricsReport, ScannerSourceWorkSnapshot,
global_metrics,
};
use rustfs_io_metrics::internode_metrics::global_internode_metrics;
use rustfs_io_metrics::{
ProcessResourceSnapshot, ProcessSampler, ProcessStatusSnapshot, ProcessSystemSnapshot, s3_op_metrics_snapshot,
@@ -334,7 +337,7 @@ fn timestamp_elapsed_seconds_since(now: Timestamp, earlier: Timestamp) -> u64 {
return 0;
}
u64::try_from(duration.as_secs()).map_or(u64::MAX, |seconds| seconds)
u64::try_from(duration.as_secs()).unwrap_or(u64::MAX)
}
fn scanner_scan_mode_code(scan_mode: &str) -> u64 {
@@ -835,6 +838,8 @@ pub(crate) async fn collect_disk_and_system_drive_runtime_stats()
drive_api_latency_micros(metrics.last_minute.values().map(|action| (action.count, action.acc_time)))
}),
health: if is_online { 1 } else { 0 },
writes_total: disk.metrics.as_ref().map(|metrics| metrics.total_writes),
deletes_total: disk.metrics.as_ref().map(|metrics| metrics.total_deletes),
reads_per_sec: None,
reads_kb_per_sec: None,
reads_await: None,
@@ -1275,6 +1280,110 @@ fn ilm_action_task_stats(ilm: &ObsIlmRuntimeSnapshot) -> Vec<IlmActionTaskStats>
]
}
fn ilm_queue_task_stats(metrics: &ScannerMetricsReport) -> Vec<IlmQueueTaskStats> {
let expiry = &metrics.lifecycle_expiry;
let transition = &metrics.lifecycle_transition;
vec![
IlmQueueTaskStats {
action: "expiry".to_string(),
state: "pending".to_string(),
value: expiry.current_queued,
},
IlmQueueTaskStats {
action: "expiry".to_string(),
state: "active".to_string(),
value: expiry.current_active,
},
IlmQueueTaskStats {
action: "transition".to_string(),
state: "pending".to_string(),
value: transition.current_queued,
},
IlmQueueTaskStats {
action: "transition".to_string(),
state: "active".to_string(),
value: transition.current_active,
},
IlmQueueTaskStats {
action: "transition".to_string(),
state: "compensation_running".to_string(),
value: transition.compensation_running,
},
]
}
fn ilm_task_event_stats(metrics: &ScannerMetricsReport) -> Vec<IlmTaskEventStats> {
let expiry = &metrics.lifecycle_expiry;
let transition = &metrics.lifecycle_transition;
vec![
IlmTaskEventStats {
action: "expiry".to_string(),
result: "queued".to_string(),
value: expiry.scanner_queued,
},
IlmTaskEventStats {
action: "expiry".to_string(),
result: "missed".to_string(),
value: expiry.scanner_missed,
},
IlmTaskEventStats {
action: "expiry".to_string(),
result: "blocked".to_string(),
value: expiry.scanner_blocked,
},
IlmTaskEventStats {
action: "expiry".to_string(),
result: "not_enqueued".to_string(),
value: expiry.scanner_not_enqueued,
},
IlmTaskEventStats {
action: "expiry".to_string(),
result: "failed".to_string(),
value: expiry.delete_failed,
},
IlmTaskEventStats {
action: "transition".to_string(),
result: "queued".to_string(),
value: transition.scanner_queued,
},
IlmTaskEventStats {
action: "transition".to_string(),
result: "missed".to_string(),
value: transition.scanner_missed,
},
IlmTaskEventStats {
action: "transition".to_string(),
result: "completed".to_string(),
value: transition.completed,
},
IlmTaskEventStats {
action: "transition".to_string(),
result: "failed".to_string(),
value: transition.failed,
},
]
}
fn ilm_backpressure_stats(metrics: &ScannerMetricsReport) -> Vec<IlmBackpressureStats> {
vec![
IlmBackpressureStats {
action: "expiry".to_string(),
reason: "queue_missed".to_string(),
value: metrics.lifecycle_expiry.queue_missed,
},
IlmBackpressureStats {
action: "transition".to_string(),
reason: "queue_full".to_string(),
value: metrics.lifecycle_transition.queue_full,
},
IlmBackpressureStats {
action: "transition".to_string(),
reason: "send_timeout".to_string(),
value: metrics.lifecycle_transition.queue_send_timeout,
},
]
}
/// Collect ILM metrics from the current lifecycle runtime state.
pub async fn collect_ilm_metric_stats() -> Option<IlmStats> {
collect_ilm_runtime_metric_stats().await.map(|stats| stats.stats)
@@ -1288,6 +1397,10 @@ pub(crate) async fn collect_ilm_runtime_metric_stats() -> Option<IlmRuntimeStats
Some(IlmRuntimeStats {
server: current_local_node_identity(),
action_tasks: ilm_action_task_stats(&ilm),
queue_tasks: ilm_queue_task_stats(&metrics),
task_events: ilm_task_event_stats(&metrics),
backpressure: ilm_backpressure_stats(&metrics),
versions_scanned,
stats: IlmStats {
expiry_pending_tasks: ilm.expiry_pending_tasks,
transition_active_tasks: ilm.transition_active_tasks,
@@ -1377,6 +1490,27 @@ fn scanner_bucket_drive_result_stats(results: &[ScannerBucketDriveResultSnapshot
stats
}
fn scanner_active_bucket_drive_stats(results: &[ScannerActiveBucketDriveSnapshot]) -> Vec<ScannerActiveBucketDriveStats> {
let mut stats = results
.iter()
.filter(|result| !result.source.is_empty() && !result.bucket.is_empty() && !result.drive.is_empty() && result.count > 0)
.map(|result| ScannerActiveBucketDriveStats {
source: result.source.clone(),
bucket: result.bucket.clone(),
drive: result.drive.clone(),
count: result.count,
age_seconds: result.age_seconds,
})
.collect::<Vec<_>>();
stats.sort_by(|left, right| {
left.source
.cmp(&right.source)
.then_with(|| left.bucket.cmp(&right.bucket))
.then_with(|| left.drive.cmp(&right.drive))
});
stats
}
pub async fn collect_scanner_metric_stats() -> Option<ScannerStats> {
collect_scanner_runtime_metric_stats().await.map(|stats| stats.stats)
}
@@ -1418,6 +1552,7 @@ pub(crate) async fn collect_scanner_runtime_metric_stats() -> Option<ScannerRunt
&runtime_details.current_cycle_bucket_drive_results,
),
last_cycle_bucket_drive_results: scanner_bucket_drive_result_stats(&runtime_details.last_cycle_bucket_drive_results),
active_bucket_drive_scans: scanner_active_bucket_drive_stats(&runtime_details.active_bucket_drive_scans),
stats: ScannerStats {
bucket_scans_finished,
bucket_scans_started,
@@ -1984,6 +2119,72 @@ mod tests {
assert_eq!(scanner_lifecycle_checked_versions(&report), 37);
}
#[test]
fn ilm_detail_stats_keep_expiry_and_transition_results_separate() {
let report = ScannerMetricsReport {
lifecycle_expiry: rustfs_common::metrics::ScannerLifecycleExpirySnapshot {
current_queued: 2,
current_active: 1,
scanner_queued: 10,
scanner_missed: 3,
delete_failed: 4,
..Default::default()
},
lifecycle_transition: rustfs_common::metrics::ScannerLifecycleTransitionSnapshot {
current_queued: 5,
current_active: 6,
queue_full: 7,
queue_send_timeout: 8,
scanner_queued: 11,
completed: 12,
failed: 13,
..Default::default()
},
..Default::default()
};
let queues = ilm_queue_task_stats(&report);
assert!(
queues
.iter()
.any(|task| task.action == "expiry" && task.state == "pending" && task.value == 2)
);
assert!(
queues
.iter()
.any(|task| task.action == "transition" && task.state == "active" && task.value == 6)
);
let events = ilm_task_event_stats(&report);
assert!(
events
.iter()
.any(|event| event.action == "expiry" && event.result == "failed" && event.value == 4)
);
assert!(
events
.iter()
.any(|event| event.action == "transition" && event.result == "completed" && event.value == 12)
);
assert!(
events
.iter()
.any(|event| event.action == "transition" && event.result == "failed" && event.value == 13)
);
let backpressure = ilm_backpressure_stats(&report);
assert!(
backpressure
.iter()
.any(|event| event.action == "transition" && event.reason == "queue_full" && event.value == 7)
);
assert!(
backpressure
.iter()
.any(|event| event.action == "transition" && event.reason == "send_timeout" && event.value == 8)
);
}
#[test]
fn scanner_source_work_stats_sorts_and_skips_empty_source() {
let stats = scanner_source_work_stats(&[
+14 -1
View File
@@ -31,7 +31,7 @@ impl DateFunc {
return false;
};
if !op(&inner.values.0, &rv) {
if !op(&rv, &inner.values.0) {
return false;
}
}
@@ -95,6 +95,7 @@ mod tests {
key_name::KeyName::{self, *},
key_name::S3KeyName::*,
};
use std::collections::HashMap;
use test_case::test_case;
use time::{OffsetDateTime, format_description::well_known::Rfc3339};
@@ -122,4 +123,16 @@ mod tests {
assert_eq!(v, expect);
Ok(())
}
#[test]
fn evaluate_compares_request_date_to_policy_date() {
let function = new_func(S3(S3ObjectLockRetainUntilDate), None, "2030-01-01T00:00:00Z");
let later = HashMap::from([("object-lock-retain-until-date".to_string(), vec!["2099-01-01T00:00:00Z".to_string()])]);
let earlier = HashMap::from([("object-lock-retain-until-date".to_string(), vec!["2029-01-01T00:00:00Z".to_string()])]);
assert!(function.evaluate(OffsetDateTime::gt, &later));
assert!(!function.evaluate(OffsetDateTime::gt, &earlier));
assert!(function.evaluate(OffsetDateTime::lt, &earlier));
assert!(!function.evaluate(OffsetDateTime::lt, &later));
}
}
+1 -1
View File
@@ -106,7 +106,7 @@ swift = [
"dep:base64",
"dep:async-compression",
]
webdav = ["dep:dav-server", "dep:hyper", "dep:hyper-util", "dep:http-body-util", "dep:tokio-rustls", "dep:base64", "dep:rustls", "dep:percent-encoding", "dep:rustfs-tls-runtime", "dep:subtle"]
webdav = ["dep:dav-server", "dep:hyper", "dep:hyper-util", "dep:http", "dep:http-body-util", "dep:tokio-rustls", "dep:base64", "dep:rustls", "dep:percent-encoding", "dep:rustfs-tls-runtime", "dep:subtle"]
sftp = ["dep:russh", "dep:russh-sftp", "dep:uuid", "dep:subtle", "dep:tokio-util", "dep:socket2"]
[dependencies]
+20 -1
View File
@@ -15,6 +15,9 @@
use async_trait::async_trait;
use s3s::dto::*;
#[cfg(feature = "webdav")]
use crate::common::session::SessionContext;
#[async_trait]
pub trait StorageBackend: Send + Sync {
/// Error type for this storage backend
@@ -65,8 +68,24 @@ pub trait StorageBackend: Send + Sync {
access_key: &str,
secret_key: &str,
) -> Result<ListObjectsV2Output, Self::Error>;
/// List all buckets (requires authentication)
/// List all buckets (requires authentication).
async fn list_buckets(&self, access_key: &str, secret_key: &str) -> Result<ListBucketsOutput, Self::Error>;
/// List buckets visible to the authenticated session.
///
/// Backends that implement this must apply per-bucket authorization. The default denies the
/// request so existing backends cannot expose unfiltered bucket names.
#[cfg(feature = "webdav")]
async fn list_buckets_for_session(
&self,
_session_context: &SessionContext,
_request_headers: &http::HeaderMap,
_secure_transport: bool,
) -> s3s::S3Result<ListBucketsOutput> {
Err(s3s::S3Error::with_message(
s3s::S3ErrorCode::AccessDenied,
"Session-aware bucket listing is not supported",
))
}
/// Create a new bucket
async fn create_bucket(&self, bucket: &str, access_key: &str, secret_key: &str) -> Result<CreateBucketOutput, Self::Error>;
/// Delete a bucket (must be empty)
+48 -1
View File
@@ -30,6 +30,8 @@
//! SessionContext type in common::session.
use crate::common::client::s3::StorageBackend;
#[cfg(feature = "webdav")]
use crate::common::session::SessionContext;
use async_trait::async_trait;
use bytes::Bytes;
use futures_util::stream::{self, StreamExt};
@@ -140,6 +142,8 @@ struct Inner {
head_bucket: VecDeque<Result<HeadBucketOutput, DummyError>>,
list_objects_v2: VecDeque<Result<ListObjectsV2Output, DummyError>>,
list_buckets: VecDeque<Result<ListBucketsOutput, DummyError>>,
session_list_buckets: VecDeque<s3s::S3Result<ListBucketsOutput>>,
last_session_list_context: Option<(http::HeaderMap, bool)>,
create_bucket: VecDeque<Result<CreateBucketOutput, DummyError>>,
delete_bucket: VecDeque<Result<DeleteBucketOutput, DummyError>>,
copy_object: VecDeque<Result<CopyObjectOutput, DummyError>>,
@@ -193,6 +197,8 @@ impl Inner {
head_bucket: VecDeque::new(),
list_objects_v2: VecDeque::new(),
list_buckets: VecDeque::new(),
session_list_buckets: VecDeque::new(),
last_session_list_context: None,
create_bucket: VecDeque::new(),
delete_bucket: VecDeque::new(),
copy_object: VecDeque::new(),
@@ -301,6 +307,31 @@ impl DummyBackend {
.push_back(Ok(CreateBucketOutput::default()));
}
/// Queue a legacy list_buckets response.
pub fn queue_list_buckets_ok(&self, output: ListBucketsOutput) {
self.inner.lock().expect("lock").list_buckets.push_back(Ok(output));
}
/// Queue a legacy list_buckets error.
pub fn queue_list_buckets_err(&self, error: DummyError) {
self.inner.lock().expect("lock").list_buckets.push_back(Err(error));
}
/// Queue a session-aware list_buckets response.
pub fn queue_session_list_buckets_ok(&self, output: ListBucketsOutput) {
self.inner.lock().expect("lock").session_list_buckets.push_back(Ok(output));
}
/// Queue a session-aware list_buckets error.
pub fn queue_session_list_buckets_err(&self, error: s3s::S3Error) {
self.inner.lock().expect("lock").session_list_buckets.push_back(Err(error));
}
/// Return the context from the last session-aware list_buckets request.
pub fn last_session_list_context(&self) -> Option<(http::HeaderMap, bool)> {
self.inner.lock().expect("lock").last_session_list_context.clone()
}
/// Queue a put_object error. Used by the commit_write retry tests
/// to script SlowDown / AccessDenied sequences against the
/// rustfs_utils::retry::is_s3code_in_message_retryable predicate.
@@ -690,13 +721,29 @@ impl StorageBackend for DummyBackend {
}
}
async fn list_buckets(&self, _ak: &str, _sk: &str) -> Result<ListBucketsOutput, Self::Error> {
async fn list_buckets(&self, _access_key: &str, _secret_key: &str) -> Result<ListBucketsOutput, Self::Error> {
match self.inner.lock().expect("lock").list_buckets.pop_front() {
Some(r) => r,
None => Ok(ListBucketsOutput::default()),
}
}
#[cfg(feature = "webdav")]
async fn list_buckets_for_session(
&self,
session_context: &SessionContext,
request_headers: &http::HeaderMap,
secure_transport: bool,
) -> s3s::S3Result<ListBucketsOutput> {
let _ = session_context;
let mut inner = self.inner.lock().expect("lock");
inner.last_session_list_context = Some((request_headers.clone(), secure_transport));
inner
.session_list_buckets
.pop_front()
.unwrap_or_else(|| Ok(ListBucketsOutput::default()))
}
async fn create_bucket(&self, _bucket: &str, _ak: &str, _sk: &str) -> Result<CreateBucketOutput, Self::Error> {
match self.inner.lock().expect("lock").create_bucket.pop_front() {
Some(r) => r,
+206 -41
View File
@@ -13,7 +13,7 @@
// limitations under the License.
use crate::common::client::s3::StorageBackend as S3StorageBackend;
use crate::common::gateway::{S3Action, authorize_operation};
use crate::common::gateway::{AuthorizationError, S3Action, authorize_operation};
use crate::common::session::SessionContext;
use bytes::Bytes;
use dav_server::davpath::DavPath;
@@ -24,6 +24,7 @@ use futures_util::{FutureExt, StreamExt, stream};
use percent_encoding::percent_decode_str;
use rustfs_utils::MaskedAccessKey;
use rustfs_utils::path;
use s3s::S3ErrorCode;
use s3s::dto::*;
use std::fmt::Debug;
use std::io::SeekFrom;
@@ -457,6 +458,10 @@ where
storage: S,
/// Session context for authorization
session_context: Arc<SessionContext>,
/// Policy-safe WebDAV request headers used by IAM conditions.
request_headers: Option<http::HeaderMap>,
/// Whether the WebDAV connection uses TLS.
secure_transport: bool,
}
enum ResolvedPath {
@@ -490,6 +495,8 @@ where
Self {
storage: self.storage.clone(),
session_context: self.session_context.clone(),
request_headers: self.request_headers.clone(),
secure_transport: self.secure_transport,
}
}
}
@@ -503,9 +510,18 @@ where
Self {
storage,
session_context,
request_headers: None,
secure_transport: false,
}
}
/// Attach the request context used by IAM policy conditions.
pub fn with_request_context(mut self, request_headers: http::HeaderMap, secure_transport: bool) -> Self {
self.request_headers = Some(request_headers);
self.secure_transport = secure_transport;
self
}
fn credentials(&self) -> (&str, &str) {
(
&self.session_context.principal.user_identity.credentials.access_key,
@@ -799,50 +815,41 @@ where
/// List all buckets (for root path)
async fn list_buckets(&self) -> FsResult<Vec<WebDavDirEntry>> {
match authorize_operation(&self.session_context, &S3Action::ListBuckets, "", None).await {
Ok(_) => {}
Err(_e) => {
return Err(FsError::Forbidden);
Ok(()) => {
let (access_key, secret_key) = self.credentials();
return match self.storage.list_buckets(access_key, secret_key).await {
Ok(output) => Ok(Self::bucket_entries(output)),
Err(error) => {
error!(
event = EVENT_WEBDAV_BUCKET_LIST_FAILED,
component = LOG_COMPONENT_PROTOCOLS,
subsystem = LOG_SUBSYSTEM_WEBDAV_DRIVER,
error = %error,
access_key = %MaskedAccessKey(access_key),
"webdav bucket list failed"
);
Err(FsError::GeneralFailure)
}
};
}
Err(AuthorizationError::AccessDenied) => {}
Err(AuthorizationError::IamUnavailable) => return Err(FsError::GeneralFailure),
}
match self
let Some(request_headers) = self.request_headers.as_ref() else {
return Err(FsError::Forbidden);
};
let result = self
.storage
.list_buckets(
&self.session_context.principal.user_identity.credentials.access_key,
&self.session_context.principal.user_identity.credentials.secret_key,
)
.await
{
Ok(output) => {
let mut entries = Vec::new();
if let Some(buckets) = output.buckets {
for bucket in buckets {
if let Some(ref bucket_name) = bucket.name {
let modified = bucket
.creation_date
.map(|dt| {
let offset_dt: time::OffsetDateTime = dt.into();
SystemTime::from(offset_dt)
})
.unwrap_or_else(SystemTime::now);
.list_buckets_for_session(&self.session_context, request_headers, self.secure_transport)
.await;
entries.push(WebDavDirEntry {
name: bucket_name.clone(),
metadata: WebDavMetaData {
size: 0,
modified,
created: modified,
is_dir: true,
etag: None,
content_type: None,
},
});
}
}
}
Ok(entries)
}
match result {
Ok(output) => Ok(Self::bucket_entries(output)),
Err(e) => {
if matches!(e.code(), S3ErrorCode::AccessDenied) {
return Err(FsError::Forbidden);
}
error!(
event = EVENT_WEBDAV_BUCKET_LIST_FAILED,
component = LOG_COMPONENT_PROTOCOLS,
@@ -856,6 +863,35 @@ where
}
}
fn bucket_entries(output: ListBucketsOutput) -> Vec<WebDavDirEntry> {
output
.buckets
.unwrap_or_default()
.into_iter()
.filter_map(|bucket| {
let name = bucket.name?;
let modified = bucket
.creation_date
.map(|date| {
let date: time::OffsetDateTime = date.into();
SystemTime::from(date)
})
.unwrap_or_else(SystemTime::now);
Some(WebDavDirEntry {
name,
metadata: WebDavMetaData {
size: 0,
modified,
created: modified,
is_dir: true,
etag: None,
content_type: None,
},
})
})
.collect()
}
/// List objects in a bucket
async fn list_objects(&self, bucket: &str, prefix: Option<&str>) -> FsResult<Vec<WebDavDirEntry>> {
// Authorize the operation
@@ -1715,8 +1751,9 @@ where
mod tests {
use super::WebDavDriver;
use crate::common::client::s3::StorageBackend as S3StorageBackend;
use crate::common::gateway::{S3Action, with_test_auth_override};
use crate::common::session::{Protocol, ProtocolPrincipal, SessionContext};
use crate::common::dummy_storage::DummyBackend;
use crate::common::gateway::{S3Action, with_test_auth_override, with_test_iam_unavailable};
use crate::common::session::{Protocol, ProtocolPrincipal, SessionContext, test_session};
use async_trait::async_trait;
use bytes::Bytes;
use dav_server::davpath::DavPath;
@@ -1906,6 +1943,134 @@ mod tests {
WebDavDriver::new(DummyStorage, Arc::new(session_context))
}
#[tokio::test]
async fn session_bucket_listing_does_not_require_global_list_permission() {
let storage = DummyBackend::new();
storage.queue_session_list_buckets_ok(ListBucketsOutput {
buckets: Some(vec![Bucket {
name: Some("allowed-bucket".to_string()),
..Default::default()
}]),
..Default::default()
});
let driver = WebDavDriver::new(storage.clone(), Arc::new(test_session(Protocol::WebDav))).with_request_context(
http::HeaderMap::from_iter([(http::header::USER_AGENT, http::HeaderValue::from_static("webdav-test"))]),
false,
);
let entries = with_test_auth_override(|_, _, _| false, driver.list_buckets())
.await
.expect("session-aware backend should own bucket filtering");
assert_eq!(entries.len(), 1);
assert_eq!(entries[0].name, "allowed-bucket");
let (headers, secure_transport) = storage
.last_session_list_context()
.expect("request context should be forwarded");
assert_eq!(headers.get("user-agent").expect("user agent"), "webdav-test");
assert!(!secure_transport);
}
#[tokio::test]
async fn list_buckets_maps_typed_access_denied_to_forbidden() {
let storage = DummyBackend::new();
storage.queue_session_list_buckets_err(s3s::S3Error::with_message(s3s::S3ErrorCode::AccessDenied, "policy denied"));
let driver = WebDavDriver::new(storage, Arc::new(test_session(Protocol::WebDav)))
.with_request_context(http::HeaderMap::new(), false);
let error = with_test_auth_override(|_, _, _| false, driver.list_buckets())
.await
.expect_err("bucket listing should be denied");
assert!(matches!(error, FsError::Forbidden));
}
#[tokio::test]
async fn list_buckets_does_not_classify_error_text_as_access_denied() {
let storage = DummyBackend::new();
storage.queue_session_list_buckets_err(s3s::S3Error::with_message(
s3s::S3ErrorCode::InternalError,
"AccessDenied appears only in the message",
));
let driver = WebDavDriver::new(storage, Arc::new(test_session(Protocol::WebDav)))
.with_request_context(http::HeaderMap::new(), false);
let error = with_test_auth_override(|_, _, _| false, driver.list_buckets())
.await
.expect_err("bucket listing should fail");
assert!(matches!(error, FsError::GeneralFailure));
}
#[tokio::test]
async fn global_list_permission_keeps_the_legacy_backend_path() {
let storage = DummyBackend::new();
storage.queue_list_buckets_ok(ListBucketsOutput {
buckets: Some(vec![Bucket {
name: Some("legacy-bucket".to_string()),
..Default::default()
}]),
..Default::default()
});
let driver = WebDavDriver::new(storage.clone(), Arc::new(test_session(Protocol::WebDav)));
let entries = with_test_auth_override(|_, _, _| true, driver.list_buckets())
.await
.expect("globally authorized legacy backend should keep working");
assert_eq!(entries[0].name, "legacy-bucket");
assert!(storage.last_session_list_context().is_none());
}
#[tokio::test]
async fn legacy_bucket_list_error_is_a_general_failure() {
let storage = DummyBackend::new();
storage.queue_list_buckets_err(crate::common::dummy_storage::DummyError::Injected("backend failed".to_string()));
let driver = WebDavDriver::new(storage.clone(), Arc::new(test_session(Protocol::WebDav)));
let error = with_test_auth_override(|_, _, _| true, driver.list_buckets())
.await
.expect_err("legacy backend error should fail the listing");
assert!(matches!(error, FsError::GeneralFailure));
assert!(storage.last_session_list_context().is_none());
}
#[tokio::test]
async fn iam_unavailable_does_not_enter_the_session_fallback() {
let storage = DummyBackend::new();
storage.queue_session_list_buckets_ok(ListBucketsOutput::default());
let driver = WebDavDriver::new(storage.clone(), Arc::new(test_session(Protocol::WebDav)))
.with_request_context(http::HeaderMap::new(), false);
let error = with_test_iam_unavailable(driver.list_buckets())
.await
.expect_err("IAM outage must fail closed");
assert!(matches!(error, FsError::GeneralFailure));
assert!(storage.last_session_list_context().is_none());
}
#[tokio::test]
async fn missing_request_context_fails_closed() {
let error = with_test_auth_override(|_, _, _| false, driver().list_buckets())
.await
.expect_err("bucket listing should require the original request context");
assert!(matches!(error, FsError::Forbidden));
}
#[tokio::test]
async fn backend_without_session_listing_fails_closed() {
let driver = driver().with_request_context(http::HeaderMap::new(), false);
let error = with_test_auth_override(|_, _, _| false, driver.list_buckets())
.await
.expect_err("default session listing must deny the request");
assert!(matches!(error, FsError::Forbidden));
}
#[derive(Default)]
struct RecordingStorageState {
objects: HashMap<(String, String), Vec<u8>>,
+48 -4
View File
@@ -19,6 +19,8 @@ use crate::common::session::{Protocol, ProtocolPrincipal, SessionContext, is_tem
use bytes::Bytes;
use dav_server::DavHandler;
use dav_server::fakels::FakeLs;
use http::header::{AUTHORIZATION, REFERER, USER_AGENT};
use http::{HeaderMap, HeaderValue};
use http_body_util::{BodyExt, Full, LengthLimitError, Limited};
use hyper::body::Body as HttpBody;
use hyper::server::conn::http1;
@@ -59,6 +61,20 @@ const EVENT_WEBDAV_CONNECTION_CAP_STATE: &str = "webdav_connection_cap_state";
/// materialise a whole object in memory for every GET.
type WebDavBody = Pin<Box<dyn HttpBody<Data = Bytes, Error = io::Error> + Send>>;
fn policy_request_headers(headers: &HeaderMap) -> HeaderMap {
let mut policy_headers = HeaderMap::new();
for name in [USER_AGENT, REFERER] {
if let Some(value) = headers.get(&name) {
policy_headers.insert(name, value.clone());
}
}
let mut authorization = HeaderValue::from_static("Basic");
authorization.set_sensitive(true);
policy_headers.insert(AUTHORIZATION, authorization);
policy_headers
}
/// WebDAV server implementation
pub struct WebDavServer<S>
where
@@ -216,7 +232,7 @@ where
match timeout(request_timeout, acceptor.accept(stream)).await {
Ok(Ok(tls_stream)) => {
let io = TokioIo::new(tls_stream);
if let Err(e) = Self::handle_connection_impl(io, storage, source_ip, max_body_size, request_timeout).await {
if let Err(e) = Self::handle_connection_impl(io, storage, source_ip, true, max_body_size, request_timeout).await {
debug!(
event = EVENT_WEBDAV_CONNECTION_STATE,
component = LOG_COMPONENT_PROTOCOLS,
@@ -254,7 +270,7 @@ where
}
} else {
let io = TokioIo::new(stream);
if let Err(e) = Self::handle_connection_impl(io, storage, source_ip, max_body_size, request_timeout).await {
if let Err(e) = Self::handle_connection_impl(io, storage, source_ip, false, max_body_size, request_timeout).await {
debug!(
event = EVENT_WEBDAV_CONNECTION_STATE,
component = LOG_COMPONENT_PROTOCOLS,
@@ -313,6 +329,7 @@ where
io: TokioIo<I>,
storage: S,
source_ip: IpAddr,
secure_transport: bool,
max_body_size: u64,
request_timeout: Duration,
) -> Result<(), Box<dyn std::error::Error + Send + Sync>>
@@ -321,7 +338,7 @@ where
{
let service = service_fn(move |req: Request<hyper::body::Incoming>| {
let storage = storage.clone();
async move { Self::handle_request(req, storage, source_ip, max_body_size, request_timeout).await }
async move { Self::handle_request(req, storage, source_ip, secure_transport, max_body_size, request_timeout).await }
});
// A peer that opens a connection and dribbles (or never finishes)
@@ -341,6 +358,7 @@ where
req: Request<hyper::body::Incoming>,
storage: S,
source_ip: IpAddr,
secure_transport: bool,
max_body_size: u64,
request_timeout: Duration,
) -> Result<Response<WebDavBody>, Infallible> {
@@ -398,7 +416,8 @@ where
};
// Create WebDAV driver with session context
let driver = WebDavDriver::new(storage, Arc::new(session_context));
let driver = WebDavDriver::new(storage, Arc::new(session_context))
.with_request_context(policy_request_headers(req.headers()), secure_transport);
// Build DAV handler with boxed filesystem
let dav_handler = DavHandler::builder()
@@ -883,6 +902,30 @@ mod tests {
.expect("build get request")
}
#[test]
fn policy_headers_drop_credentials_and_s3_auth_spoofing() {
let mut headers = HeaderMap::new();
headers.insert(AUTHORIZATION, HeaderValue::from_static("Basic dXNlcjpwYXNzd29yZA=="));
headers.insert(USER_AGENT, HeaderValue::from_static("webdav-client"));
headers.insert(REFERER, HeaderValue::from_static("https://example.test/"));
headers.insert("x-amz-content-sha256", HeaderValue::from_static("STREAMING-AWS4-HMAC-SHA256-PAYLOAD"));
headers.insert("x-amz-signature-age", HeaderValue::from_static("0"));
let policy_headers = policy_request_headers(&headers);
assert_eq!(policy_headers.get(AUTHORIZATION).expect("authorization marker"), "Basic");
assert!(
policy_headers
.get(AUTHORIZATION)
.expect("authorization marker")
.is_sensitive()
);
assert_eq!(policy_headers.get(USER_AGENT).expect("user agent"), "webdav-client");
assert_eq!(policy_headers.get(REFERER).expect("referer"), "https://example.test/");
assert!(!policy_headers.contains_key("x-amz-content-sha256"));
assert!(!policy_headers.contains_key("x-amz-signature-age"));
}
/// R03-CAN-051 / R03-CAN-067 / R05-CAN-094: a chunked upload declares no
/// Content-Length, so the limit has to hold on the bytes actually read.
#[tokio::test]
@@ -959,6 +1002,7 @@ mod tests {
TokioIo::new(server),
StubStorage,
TEST_IP,
false,
1024,
Duration::from_secs(30),
));
+2
View File
@@ -12,6 +12,8 @@
// See the License for the specific language governing permissions and
// limitations under the License.
#![recursion_limit = "256"]
pub mod data_source;
pub mod dispatcher;
pub mod execution;
+1 -1
View File
@@ -22,7 +22,7 @@ use crate::scanner_io::{
use crate::storage_api::owner::NS_SCANNER_PROTOCOL_VERSION;
use crate::{
DATA_USAGE_CACHE_NAME, DataUsageCache, DataUsageCachePrepareOutcome, DataUsageCacheSource, DataUsageEntryInfo,
DataUsageScanPlanDigest, Disk, ScannerDiskExt as _, ScannerError, StorageError, resolve_scanner_object_store_handle,
DataUsageScanPlanDigest, Disk, ScannerError, StorageError, resolve_scanner_object_store_handle,
};
use hmac::{Hmac, KeyInit, Mac};
use rustfs_common::heal_channel::HealScanMode;
+173 -86
View File
@@ -1081,18 +1081,6 @@ async fn run_data_scanner_cycle(
}
};
let (sender, receiver) = mpsc::channel::<DataUsageInfo>(1);
let storeapi_clone = storeapi.clone();
let ctx_clone = ctx.clone();
let mut usage_persist_task = AbortOnDropHandle::new(tokio::spawn(async move {
store_data_usage_in_backend_with_outcome_for_epoch_and_baseline(
ctx_clone,
storeapi_clone,
receiver,
Some(leader_epoch),
Some(usage_persist_baseline),
)
.await
}));
let done_cycle = Metrics::time(Metric::ScanCycle);
let cycle_budget = ScannerCycleBudget::new(ctx, cycle_budget_config);
@@ -1107,47 +1095,78 @@ async fn run_data_scanner_cycle(
scan_mode,
)
.await;
let publication_defer_reason = match &scan_result {
Ok(result) => final_data_usage_publication_defer_reason(storeapi.as_ref(), result.status).await,
Err(_) => Some(ScannerCycleDeferReason::ActivityBaselineUnavailable),
};
let budget_elapsed = cycle_budget.budget_elapsed() && !ctx.is_cancelled();
let usage_persist_outcome = match wait_for_data_usage_persist_task(ctx, &mut usage_persist_task, usage_persist_timeout).await
{
DataUsagePersistTaskResult::Completed(outcome) => outcome,
DataUsagePersistTaskResult::JoinFailed(err) => {
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
state = "usage_persist_task_failed",
error = %err,
"Scanner data usage persistence task failed"
);
DataUsagePersistOutcome::Failed
let usage_persist_outcome = match publication_defer_reason {
Some(reason) => {
drop(receiver);
DataUsagePersistOutcome::Deferred(reason)
}
DataUsagePersistTaskResult::Cancelled => {
debug!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
state = "usage_persist_task_cancelled",
"Scanner data usage persistence task cancelled"
);
DataUsagePersistOutcome::Failed
}
DataUsagePersistTaskResult::TimedOut => {
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
timeout = ?usage_persist_timeout,
state = "usage_persist_task_timed_out",
"Scanner data usage persistence task timed out"
);
DataUsagePersistOutcome::Failed
None => {
// ScannerIO emits its complete or observational update only after
// all set workers finish. Persist after the final activity fence;
// this also avoids blocking the scanner on a denied publication.
let storeapi_clone = storeapi.clone();
let ctx_clone = ctx.clone();
let route_probe_store = storeapi.clone();
let mut usage_persist_task = AbortOnDropHandle::new(tokio::spawn(async move {
store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe(
ctx_clone,
storeapi_clone,
receiver,
Some(leader_epoch),
Some(usage_persist_baseline),
move || {
let storeapi = route_probe_store.clone();
async move { storeapi.scanner_data_usage_publication_blocked().await }
},
)
.await
}));
match wait_for_data_usage_persist_task(ctx, &mut usage_persist_task, usage_persist_timeout).await {
DataUsagePersistTaskResult::Completed(outcome) => outcome,
DataUsagePersistTaskResult::JoinFailed(err) => {
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
state = "usage_persist_task_failed",
error = %err,
"Scanner data usage persistence task failed"
);
DataUsagePersistOutcome::Failed
}
DataUsagePersistTaskResult::Cancelled => {
debug!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
state = "usage_persist_task_cancelled",
"Scanner data usage persistence task cancelled"
);
DataUsagePersistOutcome::Failed
}
DataUsagePersistTaskResult::TimedOut => {
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
timeout = ?usage_persist_timeout,
state = "usage_persist_task_timed_out",
"Scanner data usage persistence task timed out"
);
DataUsagePersistOutcome::Failed
}
}
}
};
let unresolved_heal_work = global_metrics().current_scan_cycle_has_unresolved_heal_work();
@@ -1191,33 +1210,51 @@ async fn run_data_scanner_cycle(
mark_scan_cycle_idle(cycle_info, &mut cycle_metrics_guard).await;
return ScannerCycleOutcome::Failed;
}
if let Some(required_cycle) = scan_cycle_result.required_cycle_floor() {
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_CYCLE_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
required_cycle,
state = "cache_cycle_ahead",
"Scanner cycle is recovering to a newer durable cache generation"
);
emit_scan_cycle_partial_with_source(cycle_start.elapsed(), ScanCyclePartialReason::Unknown, None);
return if persist_required_scanner_cycle_floor(
ctx,
storeapi.clone(),
cycle_info,
cycle_revision,
leader_epoch,
required_cycle,
&mut cycle_metrics_guard,
)
.await
{
ScannerCycleOutcome::Partial
} else {
ScannerCycleOutcome::Failed
};
match scanner_cycle_pre_commit_outcome(scan_cycle_result.required_cycle_floor(), &usage_persist_outcome) {
Some(ScannerCyclePreCommitOutcome::RecoverCacheCycle(required_cycle)) => {
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_CYCLE_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
required_cycle,
state = "cache_cycle_ahead",
"Scanner cycle is recovering to a newer durable cache generation"
);
emit_scan_cycle_partial_with_source(cycle_start.elapsed(), ScanCyclePartialReason::Unknown, None);
return if persist_required_scanner_cycle_floor(
ctx,
storeapi.clone(),
cycle_info,
cycle_revision,
leader_epoch,
required_cycle,
&mut cycle_metrics_guard,
)
.await
{
ScannerCycleOutcome::Partial
} else {
ScannerCycleOutcome::Failed
};
}
Some(ScannerCyclePreCommitOutcome::Deferred(reason)) => {
info!(
target: "rustfs::scanner",
event = EVENT_SCANNER_CYCLE_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
cycle = cycle_info.current,
reason = reason.as_str(),
state = "deferred",
"Scanner cycle deferred before data usage publication"
);
emit_scan_cycle_deferred(cycle_start.elapsed());
mark_scan_cycle_idle(cycle_info, &mut cycle_metrics_guard).await;
return ScannerCycleOutcome::Deferred(reason);
}
None => {}
}
if usage_persist_outcome == DataUsagePersistOutcome::Failed {
error!(
@@ -1804,14 +1841,13 @@ async fn run_data_scanner_with_maintenance_state(
wait_plan.delay,
activity_poll_interval,
&mut scanner_activity_seen,
ScannerCycleObservedGenerations {
// A non-converged cycle holds further activity notifications
// until its bounded retry timer to avoid an unbroken scan loop.
dirty_usage: convergence_retry_interval.is_none().then_some(dirty_usage_generation_seen),
runtime_config: runtime_config_generation_seen,
maintenance: maintenance_generation_before_wait,
defer_cluster_activity: convergence_retry_interval.is_some(),
},
ScannerCycleObservedGenerations::for_wait(
&runtime_config,
convergence_retry_interval,
dirty_usage_generation_seen,
runtime_config_generation_seen,
maintenance_generation_before_wait,
),
|| guard.is_lock_lost(),
|| probe_scanner_activity(storeapi.as_ref(), distributed),
)
@@ -2001,6 +2037,56 @@ impl Drop for ScannerScanModeGuard {
}
}
async fn final_data_usage_publication_defer_reason(
storeapi: &ECStore,
status: ScannerCycleStatus,
) -> Option<ScannerCycleDeferReason> {
match status {
ScannerCycleStatus::Complete | ScannerCycleStatus::Superseded => {
if storeapi.scanner_data_usage_publication_blocked().await {
return Some(ScannerCycleDeferReason::DataMovement);
}
if status == ScannerCycleStatus::Complete {
let distributed = storeapi.setup_is_dist_erasure().await;
match probe_scanner_activity(storeapi, distributed).await {
Ok(snapshot) if scanner_activity_allows_usage_publication(&snapshot) => None,
Ok(_) => Some(ScannerCycleDeferReason::DataMovement),
Err(_) => Some(ScannerCycleDeferReason::ActivityBaselineUnavailable),
}
} else {
// A superseded cycle is explicitly observational and cannot
// replace the authoritative snapshot. It may still be
// persisted as a convergence baseline for the next cycle.
None
}
}
ScannerCycleStatus::Deferred(reason) => Some(reason),
// Incomplete cycles do not publish a usage snapshot. Keep the
// decision permissive so existing partial-cycle handling remains
// unchanged if a future scanner path emits a bookkeeping update.
ScannerCycleStatus::Incomplete => None,
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum ScannerCyclePreCommitOutcome {
RecoverCacheCycle(u64),
Deferred(ScannerCycleDeferReason),
}
fn scanner_cycle_pre_commit_outcome(
required_cycle_floor: Option<u64>,
usage_persist_outcome: &DataUsagePersistOutcome,
) -> Option<ScannerCyclePreCommitOutcome> {
// Keep the publication barrier fail-closed: `.bloomcycle.bin` uses the
// same routed writer and its floor must remain pending while data movement
// hides the source pool.
match usage_persist_outcome {
DataUsagePersistOutcome::Deferred(reason) => Some(ScannerCyclePreCommitOutcome::Deferred(*reason)),
_ => required_cycle_floor.map(ScannerCyclePreCommitOutcome::RecoverCacheCycle),
}
}
fn scanner_cycle_completion_outcome(
scan_status: ScannerCycleStatus,
usage_persist_outcome: DataUsagePersistOutcome,
@@ -2008,6 +2094,7 @@ fn scanner_cycle_completion_outcome(
has_failed_dirty_usage: bool,
) -> ScannerCycleOutcome {
match (scan_status, usage_persist_outcome) {
(_, DataUsagePersistOutcome::Deferred(reason)) => ScannerCycleOutcome::Deferred(reason),
(_, DataUsagePersistOutcome::Failed) => ScannerCycleOutcome::Failed,
(ScannerCycleStatus::Deferred(reason), DataUsagePersistOutcome::NoUpdate)
if !has_dirty_usage && !has_failed_dirty_usage =>
+21
View File
@@ -229,6 +229,27 @@ pub(super) struct ScannerCycleObservedGenerations {
pub(super) defer_cluster_activity: bool,
}
impl ScannerCycleObservedGenerations {
pub(super) fn for_wait(
runtime_config: &ScannerRuntimeConfig,
convergence_retry_interval: Option<Duration>,
dirty_usage_generation_seen: u64,
runtime_config_generation: u64,
maintenance_generation: u64,
) -> Self {
Self {
// An explicit cycle override is a duty-cycle policy; dirty usage
// wakes stay on the default adaptive path so the interval holds.
dirty_usage: (convergence_retry_interval.is_none()
&& runtime_config.cycle_interval_source == ScannerRuntimeConfigSource::Default)
.then_some(dirty_usage_generation_seen),
runtime_config: runtime_config_generation,
maintenance: maintenance_generation,
defer_cluster_activity: convergence_retry_interval.is_some(),
}
}
}
pub(super) const LOCAL_SCANNER_ACTIVITY_NODE: &str = "<local>";
#[derive(Clone, Debug, PartialEq, Eq)]
+246
View File
@@ -153,6 +153,7 @@ struct MemoryConfigStore {
objects: Mutex<HashMap<String, Vec<u8>>>,
revisions: Mutex<HashMap<String, u64>>,
fail_put_number: Mutex<HashMap<String, usize>>,
object_not_found_put_number: Mutex<HashMap<String, usize>>,
error_after_commit_put_number: Mutex<HashMap<String, usize>>,
interleaving_puts: Mutex<HashMap<String, (usize, Vec<u8>)>>,
cancel_after_interleaving_puts: Mutex<HashMap<String, CancellationToken>>,
@@ -224,6 +225,9 @@ impl crate::storage_api::scanner_io::ObjectIO for MemoryConfigStore {
if self.fail_put_number.lock().await.get(&key) == Some(&put_count) {
return Err(EcstoreError::other("injected put failure"));
}
if self.object_not_found_put_number.lock().await.get(&key) == Some(&put_count) {
return Err(EcstoreError::ObjectNotFound(bucket.to_string(), object.to_string()));
}
let interleaving_data = {
let mut interleaving_puts = self.interleaving_puts.lock().await;
@@ -1431,6 +1435,170 @@ async fn test_store_data_usage_in_backend_preserves_newer_snapshot() {
assert_eq!(outcome, DataUsagePersistOutcome::Current);
}
#[tokio::test]
async fn test_usage_save_object_not_found_defers_only_with_a_fresh_route_barrier() {
for (route_blocked, expected) in [
(true, DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement)),
(false, DataUsagePersistOutcome::Failed),
] {
let store = Arc::new(MemoryConfigStore::default());
let key = memory_config_key(RUSTFS_META_BUCKET, DATA_USAGE_OBJ_NAME_PATH.as_str());
let baseline = complete_usage_with_bucket_count(Some(std::time::SystemTime::UNIX_EPOCH + Duration::from_secs(10)), 1);
let baseline_data = serde_json::to_vec(&baseline).expect("baseline usage snapshot should encode");
store.objects.lock().await.insert(key.clone(), baseline_data.clone());
store.revisions.lock().await.insert(key.clone(), 1);
store.object_not_found_put_number.lock().await.insert(key.clone(), 1);
let (sender, receiver) = mpsc::channel(1);
sender
.send(complete_usage_with_bucket_count(
Some(std::time::SystemTime::UNIX_EPOCH + Duration::from_secs(20)),
2,
))
.await
.expect("new usage snapshot should enqueue");
drop(sender);
let probe_calls = Arc::new(std::sync::atomic::AtomicUsize::new(0));
let route_probe_calls = probe_calls.clone();
let outcome = store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe(
CancellationToken::new(),
store.clone(),
receiver,
None,
Some(DataUsagePersistBaseline {
data: Some(Bytes::from(baseline_data.clone())),
revision: DataUsageCacheRevision::Etag("memory-1".to_string()),
}),
move || {
let probe_calls = route_probe_calls.clone();
async move {
let call = probe_calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst);
route_blocked && call > 1
}
},
)
.await;
assert_eq!(outcome, expected);
assert_eq!(
probe_calls.load(std::sync::atomic::Ordering::SeqCst),
3,
"ObjectNotFound must be followed by a fresh route-barrier probe"
);
assert_eq!(
store.objects.lock().await.get(&key),
Some(&baseline_data),
"a route failure must not replace the authoritative baseline"
);
}
}
#[tokio::test]
async fn test_usage_save_route_barrier_prevents_missing_snapshot_creation() {
for observational in [false, true] {
let store = Arc::new(MemoryConfigStore::default());
let target_path = if observational {
DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str()
} else {
DATA_USAGE_OBJ_NAME_PATH.as_str()
};
let target_key = memory_config_key(RUSTFS_META_BUCKET, target_path);
let mut incoming = complete_usage_with_bucket_count(Some(std::time::SystemTime::UNIX_EPOCH + Duration::from_secs(20)), 1);
incoming.usage_snapshot_converged = Some(!observational);
let (sender, receiver) = mpsc::channel(1);
sender.send(incoming).await.expect("usage snapshot should enqueue");
drop(sender);
let outcome = store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe(
CancellationToken::new(),
store.clone(),
receiver,
None,
Some(DataUsagePersistBaseline {
data: None,
revision: DataUsageCacheRevision::Missing,
}),
|| async { true },
)
.await;
assert_eq!(outcome, DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert!(!store.objects.lock().await.contains_key(&target_key));
assert_eq!(
store.put_counts.lock().await.get(&target_key),
None,
"the final pool-state fence must run before the first PUT"
);
}
}
#[tokio::test]
#[serial]
async fn test_usage_route_barrier_precedes_durable_reconciliation() {
let store = Arc::new(MemoryConfigStore::default());
let key = memory_config_key(RUSTFS_META_BUCKET, DATA_USAGE_OBJ_NAME_PATH.as_str());
let snapshot = complete_usage_with_bucket_count(Some(std::time::SystemTime::UNIX_EPOCH + Duration::from_secs(20)), 1);
let snapshot_data = serde_json::to_vec(&snapshot).expect("usage snapshot should encode");
let (sender, receiver) = mpsc::channel(1);
sender.send(snapshot).await.expect("usage snapshot should enqueue");
drop(sender);
let outcome = store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe(
CancellationToken::new(),
store.clone(),
receiver,
None,
Some(DataUsagePersistBaseline {
data: Some(Bytes::from(snapshot_data)),
revision: DataUsageCacheRevision::Etag("memory-1".to_string()),
}),
|| async { true },
)
.await;
assert_eq!(outcome, DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert_eq!(store.put_counts.lock().await.get(&key), None);
}
#[tokio::test]
#[serial]
async fn test_deferred_usage_save_keeps_last_real_save_metric() {
let metrics = global_metrics();
metrics.record_scanner_usage_save_result(ScannerUsageSaveResult::Success);
let before = metrics.report().await.usage_freshness;
let store = Arc::new(MemoryConfigStore::default());
let (sender, receiver) = mpsc::channel(1);
sender
.send(complete_usage_with_bucket_count(
Some(std::time::SystemTime::UNIX_EPOCH + Duration::from_secs(20)),
1,
))
.await
.expect("usage snapshot should enqueue");
drop(sender);
let outcome = store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe(
CancellationToken::new(),
store,
receiver,
None,
Some(DataUsagePersistBaseline {
data: None,
revision: DataUsageCacheRevision::Missing,
}),
|| async { true },
)
.await;
assert_eq!(outcome, DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
let after = metrics.report().await.usage_freshness;
assert_eq!(after.last_usage_save_result, before.last_usage_save_result);
assert_eq!(after.last_usage_save_result_code, before.last_usage_save_result_code);
assert_eq!(after.last_usage_save_unix_secs, before.last_usage_save_unix_secs);
}
#[tokio::test]
async fn test_store_data_usage_in_backend_fences_interleaving_newer_writer() {
let store = Arc::new(MemoryConfigStore::default());
@@ -2325,6 +2493,15 @@ async fn test_store_data_usage_in_backend_reports_missing_snapshot() {
#[test]
fn test_scanner_cycle_completion_prioritizes_persist_failure() {
assert_eq!(
scanner_cycle_completion_outcome(
ScannerCycleStatus::Complete,
DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement),
true,
false,
),
ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement)
);
assert_eq!(
scanner_cycle_completion_outcome(
ScannerCycleStatus::Deferred(ScannerCycleDeferReason::ActivityBaselineUnavailable),
@@ -2421,6 +2598,33 @@ fn test_scanner_cycle_completion_prioritizes_persist_failure() {
);
}
#[test]
fn scanner_cycle_cache_floor_stays_pending_during_deferred_usage_publication() {
for reason in [
ScannerCycleDeferReason::DataMovement,
ScannerCycleDeferReason::ActivityBaselineUnavailable,
] {
let deferred = DataUsagePersistOutcome::Deferred(reason);
assert_eq!(
scanner_cycle_pre_commit_outcome(Some(19), &deferred),
Some(ScannerCyclePreCommitOutcome::Deferred(reason)),
"a blocked publication must not persist the routed scanner cycle floor"
);
assert_eq!(
scanner_cycle_pre_commit_outcome(None, &deferred),
Some(ScannerCyclePreCommitOutcome::Deferred(reason))
);
}
assert_eq!(
scanner_cycle_pre_commit_outcome(Some(19), &DataUsagePersistOutcome::Saved),
Some(ScannerCyclePreCommitOutcome::RecoverCacheCycle(19))
);
assert_eq!(
scanner_cycle_pre_commit_outcome(Some(19), &DataUsagePersistOutcome::Failed),
Some(ScannerCyclePreCommitOutcome::RecoverCacheCycle(19))
);
}
#[test]
#[serial]
fn finalizing_a_saved_cycle_acknowledges_its_exact_dirty_snapshot() {
@@ -2448,6 +2652,23 @@ fn finalizing_a_saved_cycle_acknowledges_its_exact_dirty_snapshot() {
assert!(!crate::scanner_io::dirty_usage_buckets_pending());
}
#[test]
#[serial]
fn finalizing_a_deferred_usage_save_keeps_dirty_work_pending() {
crate::scanner_io::clear_dirty_usage_bucket("photos");
crate::scanner_io::record_dirty_usage_bucket("photos");
let dirty_snapshot = crate::scanner_io::dirty_usage_buckets_for_tests();
let deferred = crate::scanner_io::ScannerCycleResult::new(ScannerCycleStatus::Complete, Some(dirty_snapshot));
let (outcome, _, acknowledgements) =
finalize_scanner_cycle_result(deferred, DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert_eq!(outcome, ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert!(acknowledgements.is_empty());
assert!(crate::scanner_io::dirty_usage_buckets_pending());
crate::scanner_io::clear_dirty_usage_bucket("photos");
}
#[tokio::test]
async fn scanner_cycle_keeps_remote_pending_acknowledgement() {
let pending = remote_dirty_usage_acknowledgement_pending(7, 1, std::future::ready(Ok::<bool, std::io::Error>(true))).await;
@@ -3107,6 +3328,31 @@ fn clean_idle_backoff_requires_activity_probes() {
assert!(!scanner_activity_probe_required(true, false, lifecycle, &default_config));
}
#[test]
fn dirty_usage_wakes_are_disabled_for_explicit_cycle_policy() {
let default_config = ScannerRuntimeConfig::default();
let default_observed = ScannerCycleObservedGenerations::for_wait(&default_config, None, 7, 11, 13);
assert_eq!(default_observed.dirty_usage, Some(7));
assert_eq!(default_observed.runtime_config, 11);
assert_eq!(default_observed.maintenance, 13);
assert!(!default_observed.defer_cluster_activity);
let retry_observed = ScannerCycleObservedGenerations::for_wait(&default_config, Some(Duration::from_secs(11)), 7, 11, 13);
assert_eq!(retry_observed.dirty_usage, None);
assert!(retry_observed.defer_cluster_activity);
for source in [ScannerRuntimeConfigSource::Env, ScannerRuntimeConfigSource::Config] {
let explicit_cycle = ScannerRuntimeConfig {
cycle_interval_source: source,
..default_config.clone()
};
let explicit_observed = ScannerCycleObservedGenerations::for_wait(&explicit_cycle, None, 7, 11, 13);
assert_eq!(explicit_observed.dirty_usage, None);
assert!(!explicit_observed.defer_cluster_activity);
}
}
#[test]
#[serial]
fn clean_idle_cap_preserves_default_bitrot_coverage_window() {
+87 -1
View File
@@ -22,6 +22,10 @@ pub(super) enum DataUsagePersistOutcome {
AlreadyDurable,
PriorCycleDurable,
Saved,
/// The metadata route is temporarily unavailable (for example while a
/// terminal decommission state keeps the source pool suspended). The
/// caller must retry without acknowledging dirty usage.
Deferred(ScannerCycleDeferReason),
Failed,
}
@@ -92,10 +96,33 @@ pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch(
pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_baseline(
ctx: CancellationToken,
storeapi: Arc<impl ScannerObjectIO + ScannerConfigObjectDelete>,
mut receiver: mpsc::Receiver<DataUsageInfo>,
receiver: mpsc::Receiver<DataUsageInfo>,
leader_epoch: Option<u64>,
initial_baseline: Option<DataUsagePersistBaseline>,
) -> DataUsagePersistOutcome {
store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe(
ctx,
storeapi,
receiver,
leader_epoch,
initial_baseline,
|| async { false },
)
.await
}
pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_baseline_and_route_probe<F, Fut>(
ctx: CancellationToken,
storeapi: Arc<impl ScannerObjectIO + ScannerConfigObjectDelete>,
mut receiver: mpsc::Receiver<DataUsageInfo>,
leader_epoch: Option<u64>,
initial_baseline: Option<DataUsagePersistBaseline>,
route_probe: F,
) -> DataUsagePersistOutcome
where
F: Fn() -> Fut + Send + Sync,
Fut: Future<Output = bool> + Send,
{
let mut outcome = DataUsagePersistOutcome::NoUpdate;
let mut next_baseline = initial_baseline;
@@ -113,6 +140,19 @@ pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_basel
} else {
DATA_USAGE_OBJ_NAME_PATH.as_str()
};
if route_probe().await {
debug!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
path = %target_path,
state = "publication_blocked_before_reconcile",
"Scanner data usage publication deferred by the pool-state fence"
);
outcome = DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement);
break;
}
if observational && data_usage_info.usage_snapshot_authoritative_baseline.is_none() {
let authoritative_data = match next_baseline.as_ref() {
@@ -275,6 +315,18 @@ pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_basel
if ctx.is_cancelled() {
break 'updates;
}
if route_probe().await {
debug!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
path = %target_path,
state = "publication_blocked_before_save",
"Scanner data usage publication deferred by the final pool-state fence"
);
break DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement);
}
let done_save = Metrics::time(Metric::SaveUsage);
let save_result = save_config_shared_with_preconditions(
@@ -313,6 +365,33 @@ pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_basel
"Scanner data usage CAS conflict will be reconciled"
);
}
Err(e @ EcstoreError::ObjectNotFound(_, _)) => {
let route_blocked = route_probe().await;
if route_blocked {
warn!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
path = %target_path,
state = "publication_deferred",
error = %e,
"Scanner data usage route is blocked by data movement; retrying later"
);
break DataUsagePersistOutcome::Deferred(ScannerCycleDeferReason::DataMovement);
}
error!(
target: "rustfs::scanner",
event = EVENT_SCANNER_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_RUNTIME,
path = %target_path,
state = "save_failed",
error = %e,
"Scanner data usage save failed"
);
break DataUsagePersistOutcome::Failed;
}
Err(e) => {
error!(
target: "rustfs::scanner",
@@ -370,6 +449,13 @@ pub(super) async fn store_data_usage_in_backend_with_outcome_for_epoch_and_basel
outcome = DataUsagePersistOutcome::Failed;
continue;
}
DataUsagePersistOutcome::Deferred(reason) => {
// A deferred publication is an intentional retryable state, not a
// failed save. Keep the last real save result so admin freshness
// reporting does not turn a pool-recovery fence into a false error.
outcome = DataUsagePersistOutcome::Deferred(reason);
break 'updates;
}
DataUsagePersistOutcome::Saved => {
if observational {
invalidate_admin_data_usage_snapshot_cache().await;
@@ -274,18 +274,13 @@ impl ScannerItem {
/// Transform meta directory by splitting prefix and extracting object name
/// This converts a directory path like "bucket/dir1/dir2/file" to prefix="bucket/dir1/dir2" and object_name="file"
pub fn transform_meta_dir(&mut self) {
let prefix = self.prefix.clone(); // Clone to avoid borrow checker issues
let split: Vec<&str> = prefix.split(SLASH_SEPARATOR).collect();
if split.len() > 1 {
let prefix_parts: Vec<&str> = split[..split.len() - 1].to_vec();
self.prefix = path_join_buf(&prefix_parts);
let prefix = std::mem::take(&mut self.prefix);
if let Some((parent, object_name)) = prefix.rsplit_once(SLASH_SEPARATOR) {
self.prefix = path_join_buf(&[parent]);
self.object_name = object_name.to_string();
} else {
self.prefix = String::new();
self.object_name = prefix;
}
// Object name is the last element
self.object_name = split.last().unwrap_or(&"").to_string();
}
pub(super) fn metadata_object_path(&self) -> String {
@@ -301,13 +296,14 @@ impl ScannerItem {
versioning_config: VersioningConfiguration,
size_summary: &mut SizeSummary,
) {
let object_path = self.object_path();
if object_infos.is_empty() {
debug!(
target: "rustfs::scanner::folder",
event = EVENT_SCANNER_LIFECYCLE_ACTION,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
object_path = %self.object_path(),
object_path = %object_path,
state = "no_object_versions",
"Scanner lifecycle action skipped"
);
@@ -318,7 +314,7 @@ impl ScannerItem {
event = EVENT_SCANNER_LIFECYCLE_ACTION,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
object_path = %self.object_path(),
object_path = %object_path,
state = "started",
"Scanner lifecycle evaluation started"
);
@@ -360,7 +356,7 @@ impl ScannerItem {
event = EVENT_SCANNER_LIFECYCLE_ACTION,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
object_path = %self.object_path(),
object_path = %object_path,
state = "no_lifecycle_config",
"Scanner lifecycle action finished without lifecycle rules"
);
@@ -385,7 +381,7 @@ impl ScannerItem {
event = EVENT_SCANNER_LIFECYCLE_ACTION,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
object_path = %self.object_path(),
object_path = %object_path,
state = "evaluate_failed",
error = %e,
"Scanner lifecycle action evaluation failed"
@@ -502,7 +498,7 @@ impl ScannerItem {
emit_scanner_ilm_action_trace(&self.bucket, &oi.name, event.action, 1, queued, trace_started_at);
if record_scanner_ilm_action_if_queued(global_metrics(), event.action, 1, queued) {
done_ilm(1)();
if !versioning_config.prefix_enabled(&self.object_path()) && event.action == IlmAction::DeleteAction {
if !versioning_config.prefix_enabled(&object_path) && event.action == IlmAction::DeleteAction {
remaining_versions -= 1;
size = 0;
}
@@ -570,7 +566,7 @@ impl ScannerItem {
trace_emit(|| {
TraceEvent::new(TraceKind::Scanner, TraceFunc::ScannerIlmAction)
.with_bucket(self.bucket.as_str())
.with_object(self.object_path())
.with_object(object_path.as_str())
.with_duration(trace_started_at.elapsed())
.with_attr("state", state)
.with_attr("action", action.as_str())
@@ -889,3 +885,48 @@ pub(super) async fn contains_erasure_part_file(path: &str) -> Result<bool, Scann
Ok(false)
}
#[cfg(test)]
mod tests {
use super::*;
fn scanner_item_with_prefix(prefix: &str) -> ScannerItem {
ScannerItem {
path: String::new(),
bucket: "bucket".to_string(),
prefix: prefix.to_string(),
object_name: String::new(),
file_type: std::fs::metadata(std::env::temp_dir())
.expect("temp dir metadata should be readable")
.file_type(),
lifecycle: None,
object_lock: None,
replication: None,
heal_enabled: false,
heal_bitrot: false,
debug: false,
}
}
#[test]
fn transform_meta_dir_splits_parent_and_object_without_extra_components() {
let mut item = scanner_item_with_prefix("bucket/prefix/object");
item.transform_meta_dir();
assert_eq!(item.prefix, "bucket/prefix");
assert_eq!(item.object_name, "object");
assert_eq!(item.object_path(), "bucket/prefix/object");
}
#[test]
fn transform_meta_dir_moves_single_component_into_object_name() {
let mut item = scanner_item_with_prefix("object");
item.transform_meta_dir();
assert_eq!(item.prefix, "");
assert_eq!(item.object_name, "object");
assert_eq!(item.object_path(), "object");
}
}
+36 -2
View File
@@ -147,11 +147,19 @@ impl Drop for DiskBucketScanActiveGuard {
pub(super) struct BucketDriveFailureGuard {
failed: bool,
source: rustfs_common::metrics::ScannerWorkSource,
bucket: String,
drive: String,
}
impl BucketDriveFailureGuard {
pub(super) fn new() -> Self {
Self { failed: true }
pub(super) fn new(source: rustfs_common::metrics::ScannerWorkSource, bucket: &str, drive: &str) -> Self {
Self {
failed: true,
source,
bucket: bucket.to_string(),
drive: drive.to_string(),
}
}
pub(super) fn mark_not_failed(&mut self) {
@@ -161,6 +169,7 @@ impl BucketDriveFailureGuard {
impl Drop for BucketDriveFailureGuard {
fn drop(&mut self) {
global_metrics().record_scan_bucket_drive_end(self.source, &self.bucket, &self.drive);
if self.failed {
global_metrics().record_scan_bucket_drive_failure();
}
@@ -272,3 +281,28 @@ pub(super) fn record_set_scan_failure(first_err: &mut Option<Error>, err: Error)
pub(super) fn scanner_task_join_error(stage: &str, err: tokio::task::JoinError) -> Error {
Error::other(format!("{stage} task join failed: {err}"))
}
#[cfg(test)]
mod tests {
use super::*;
use rustfs_common::metrics::{ScannerWorkSource, global_metrics};
#[test]
fn bucket_drive_failure_guard_retires_active_scan_on_drop() {
let source = ScannerWorkSource::Usage;
let bucket = "__guard_active_lifecycle_test__";
let drive = "/__guard_active_lifecycle_test__";
global_metrics().record_scan_bucket_drive_start(source, bucket, drive);
{
let mut guard = BucketDriveFailureGuard::new(source, bucket, drive);
guard.mark_not_failed();
}
assert!(
!global_metrics()
.scanner_runtime_details_report()
.active_bucket_drive_scans
.iter()
.any(|active| active.source == source.as_str() && active.bucket == bucket && active.drive == drive)
);
}
}
+19
View File
@@ -49,6 +49,25 @@ impl ScannerIOCycle for ECStore {
) -> Result<ScannerCycleResult> {
let child_token = ctx.child_token();
// Check the local pool metadata before listing buckets. A failed or
// canceled decommission remains suspended after its worker exits, so
// starting a scan in that state could build a snapshot that cannot be
// routed to the authoritative metadata object.
if self.scanner_data_usage_publication_blocked().await {
debug!(
target: "rustfs::scanner::io",
event = EVENT_SCANNER_SET_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_IO,
state = "cycle_data_usage_route_blocked",
"Scanner cycle deferred while data usage metadata remains hidden by data movement"
);
return Ok(ScannerCycleResult::new(
ScannerCycleStatus::Deferred(ScannerCycleDeferReason::DataMovement),
None,
));
}
let distributed = self.setup_is_dist_erasure().await;
let activity_before = match scanner_activity_preflight(crate::scanner::probe_scanner_activity(self, distributed).await) {
ScannerActivityPreflight::Ready(snapshot) => snapshot,
+22 -17
View File
@@ -42,7 +42,8 @@ impl ScannerIODisk for Disk {
return Err(StorageError::other(SCANNER_SKIP_FILE_ERROR.to_string()));
}
let data = match self.read_metadata(&item.bucket, &item.object_path()).await {
let metadata_object_path = item.object_path();
let data = match self.read_metadata(&item.bucket, &metadata_object_path).await {
Ok(data) => data,
Err(e) if DiskError::is_err_object_not_found(&e) || DiskError::is_err_version_not_found(&e) => {
return Err(StorageError::other(SCANNER_SKIP_FILE_ERROR.to_string()));
@@ -51,23 +52,23 @@ impl ScannerIODisk for Disk {
return Err(scanner_metadata_transient_error(
format!("failed to read metadata: {e}"),
&item.bucket,
&item.object_path(),
&metadata_object_path,
));
}
};
item.transform_meta_dir();
let object_path = item.object_path();
let meta = FileMeta::load(&data).map_err(|e| {
scanner_metadata_corrupt_error(format!("failed to load metadata: {e}"), &item.bucket, &item.object_path())
})?;
let fivs = match meta.get_file_info_versions(item.bucket.as_str(), item.object_path().as_str(), false) {
let meta = FileMeta::load(&data)
.map_err(|e| scanner_metadata_corrupt_error(format!("failed to load metadata: {e}"), &item.bucket, &object_path))?;
let fivs = match meta.get_file_info_versions(item.bucket.as_str(), object_path.as_str(), false) {
Ok(versions) => versions,
Err(e) => {
return Err(scanner_metadata_corrupt_error(
format!("failed to resolve file info versions: {e}"),
&item.bucket,
&item.object_path(),
&object_path,
));
}
};
@@ -91,17 +92,17 @@ impl ScannerIODisk for Disk {
VersioningConfiguration::default()
}
};
let versioned = versioning_config.versioned(&item.object_path());
let versioned = versioning_config.versioned(&object_path);
let object_infos = fivs
.versions
.iter()
.map(|v| ObjectInfo::from_file_info(v, item.bucket.as_str(), item.object_path().as_str(), versioned))
.map(|v| ObjectInfo::from_file_info(v, item.bucket.as_str(), object_path.as_str(), versioned))
.collect::<Vec<ObjectInfo>>();
let free_version_infos = fivs
.free_versions
.iter()
.map(|v| ObjectInfo::from_file_info(v, item.bucket.as_str(), item.object_path().as_str(), versioned))
.map(|v| ObjectInfo::from_file_info(v, item.bucket.as_str(), object_path.as_str(), versioned))
.collect::<Vec<ObjectInfo>>();
let mut size_summary = SizeSummary::default();
@@ -147,8 +148,12 @@ impl ScannerIODisk for Disk {
let drive_start = std::time::Instant::now();
let bucket = cache.info.name.clone();
let disk_path = self.path().to_string_lossy().to_string();
global_metrics().record_scan_bucket_drive_start();
let mut failure_guard = BucketDriveFailureGuard::new();
let source = match scan_mode {
HealScanMode::Deep => rustfs_common::metrics::ScannerWorkSource::Bitrot,
HealScanMode::Normal | HealScanMode::Unknown => rustfs_common::metrics::ScannerWorkSource::Usage,
};
global_metrics().record_scan_bucket_drive_start(source, &bucket, &disk_path);
let mut failure_guard = BucketDriveFailureGuard::new(source, &bucket, &disk_path);
let _guard = self.start_scan();
let mut cache = cache;
@@ -196,32 +201,32 @@ impl ScannerIODisk for Disk {
match result {
Ok(mut data_usage_info) => {
done_drive();
emit_scan_bucket_drive_complete(true, &bucket, &disk_path, drive_start.elapsed());
emit_scan_bucket_drive_complete(source, true, &bucket, &disk_path, drive_start.elapsed());
data_usage_info.info.last_update = Some(SystemTime::now());
failure_guard.mark_not_failed();
Ok(ScannerDiskScanOutcome::Complete(data_usage_info))
}
Err(ScannerError::PartialCache(mut partial_cache)) => {
done_drive();
emit_scan_bucket_drive_partial(&bucket, &disk_path, drive_start.elapsed());
emit_scan_bucket_drive_partial(source, &bucket, &disk_path, drive_start.elapsed());
partial_cache.info.last_update.get_or_insert_with(SystemTime::now);
failure_guard.mark_not_failed();
Ok(ScannerDiskScanOutcome::Partial(*partial_cache))
}
Err(ScannerError::NamespaceNotFoundCache(mut partial_cache)) => {
done_drive();
emit_scan_bucket_drive_partial(&bucket, &disk_path, drive_start.elapsed());
emit_scan_bucket_drive_partial(source, &bucket, &disk_path, drive_start.elapsed());
partial_cache.info.last_update.get_or_insert_with(SystemTime::now);
failure_guard.mark_not_failed();
Ok(ScannerDiskScanOutcome::NamespaceNotFound(*partial_cache))
}
Err(e) => {
if ctx.is_cancelled() {
emit_scan_bucket_drive_partial(&bucket, &disk_path, drive_start.elapsed());
emit_scan_bucket_drive_partial(source, &bucket, &disk_path, drive_start.elapsed());
failure_guard.mark_not_failed();
} else {
done_drive();
emit_scan_bucket_drive_complete(false, &bucket, &disk_path, drive_start.elapsed());
emit_scan_bucket_drive_complete(source, false, &bucket, &disk_path, drive_start.elapsed());
}
Err(StorageError::other(format!("Failed to scan data folder: {e}")))
}
+40 -1
View File
@@ -17,7 +17,9 @@ use super::io_disk::tier_stats_template;
use super::*;
use crate::scanner_budget::ScannerCycleBudgetConfig;
use crate::scanner_folder::ScannerItem;
use crate::storage_api::owner::{EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta, EcstoreRebalanceStats};
use crate::storage_api::owner::{
EcstorePoolDecommissionInfo, EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta, EcstoreRebalanceStats,
};
use crate::storage_api::scan::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions, ObjectIO as _};
use crate::{
DiskOption, ECStore, Endpoint, EndpointServerPools, Endpoints, InstanceContext, PoolEndpoints, ScannerObjectOptions,
@@ -182,6 +184,39 @@ async fn scanner_cycle_is_deferred_while_rebalance_is_active() {
assert!(receiver.recv().await.is_none(), "rebalance-deferred cycle must not publish usage");
}
#[tokio::test]
#[serial]
async fn scanner_cycle_is_deferred_while_terminal_decommission_is_blocked() {
let (_temp_dir, store) = setup_two_pool_scanner_store().await;
for decommission in [
EcstorePoolDecommissionInfo {
failed: true,
..Default::default()
},
EcstorePoolDecommissionInfo {
canceled: true,
..Default::default()
},
] {
store.pool_meta.write().await.pools[0].decommission = Some(decommission);
assert!(store.scanner_data_usage_publication_blocked().await);
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default());
let (updates, mut receiver) = mpsc::channel(1);
let result = tokio::time::timeout(
Duration::from_secs(30),
ScannerIOCycle::nsscanner_with_status(store.as_ref(), ctx, budget, updates, 1, 1, HealScanMode::Normal),
)
.await
.expect("terminal-decommission-deferred scanner cycle should finish")
.expect("terminal-decommission-deferred scanner cycle should succeed");
assert_eq!(result.status, ScannerCycleStatus::Deferred(ScannerCycleDeferReason::DataMovement));
assert!(receiver.recv().await.is_none(), "blocked cycle must not publish usage");
}
}
#[tokio::test]
async fn data_usage_publish_fails_when_receiver_is_closed() {
let (updates, receiver) = mpsc::channel(1);
@@ -236,6 +271,10 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
assert_eq!(bucket_usage.size, 11);
assert_eq!(usage.objects_total_count, 2);
assert_eq!(usage.objects_total_size, 11);
assert!(
receiver.recv().await.is_none(),
"a scanner cycle must publish at most one terminal usage snapshot"
);
}
#[tokio::test]
+5 -3
View File
@@ -47,6 +47,8 @@ pub(crate) use rustfs_ecstore::api::bucket::versioning_sys::BucketVersioningSys
pub(crate) use rustfs_ecstore::api::cache::{
ListPathRawOptions as EcstoreListPathRawOptions, list_path_raw as ecstore_list_path_raw,
};
#[cfg(test)]
pub(crate) use rustfs_ecstore::api::capacity::PoolDecommissionInfo as EcstorePoolDecommissionInfo;
pub(crate) use rustfs_ecstore::api::capacity::{
is_reserved_or_invalid_bucket as ecstore_is_reserved_or_invalid_bucket, path2_bucket_object as ecstore_path2_bucket_object,
path2_bucket_object_with_base_path as ecstore_path2_bucket_object_with_base_path,
@@ -127,9 +129,9 @@ pub(crate) mod owner {
#[cfg(test)]
pub(crate) use super::{
EcstoreDiskOption, EcstoreDiskStore, EcstoreEndpoint, EcstoreEndpointServerPools, EcstoreEndpoints,
EcstoreInstanceContext, EcstorePoolEndpoints, EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta,
EcstoreRebalanceStats, ecstore_config_init, ecstore_init_bucket_metadata_sys, ecstore_init_local_disks_with_instance_ctx,
ecstore_new_disk,
EcstoreInstanceContext, EcstorePoolDecommissionInfo, EcstorePoolEndpoints, EcstoreRebalStatus, EcstoreRebalanceInfo,
EcstoreRebalanceMeta, EcstoreRebalanceStats, ecstore_config_init, ecstore_init_bucket_metadata_sys,
ecstore_init_local_disks_with_instance_ctx, ecstore_new_disk,
};
}

Some files were not shown because too many files have changed in this diff Show More