mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-16 09:58:21 +00:00
Compare commits
157 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a118d7e4fd | |||
| ed1bedf1fb | |||
| 4392f94e1a | |||
| 81d7b7d07a | |||
| e26668e62c | |||
| 8d3511c1b3 | |||
| d172d05e86 | |||
| 0d86c50760 | |||
| 526d6f667e | |||
| dcf3e4b9e8 | |||
| 04b9c8fd36 | |||
| c1f66969d7 | |||
| cfa9276fad | |||
| db8f55cb97 | |||
| 7f23a1ba91 | |||
| 1619c4be60 | |||
| 72fd7339c9 | |||
| 71e83aeec4 | |||
| 9138c24571 | |||
| e9f5318027 | |||
| 69e8ef9af5 | |||
| 0b2a46b36f | |||
| 56509ead1f | |||
| ffe889ad59 | |||
| e11ce2f132 | |||
| eca6bc1600 | |||
| 85be26b3c1 | |||
| ebbcfa3ac2 | |||
| ebd0531124 | |||
| 4421d4829f | |||
| d6c62b9601 | |||
| d91086d094 | |||
| 69719c257e | |||
| 67a19021b5 | |||
| 0ff3d4cbf4 | |||
| 6f29431a65 | |||
| 5a4c063d16 | |||
| e2be34cade | |||
| d60a77b750 | |||
| 307f50ee1b | |||
| c48a6330d0 | |||
| 8ac2ff5c61 | |||
| eb41f45175 | |||
| 122d200675 | |||
| 161e515c72 | |||
| 83cf063b45 | |||
| f8bbfcbeb1 | |||
| 710dcb4865 | |||
| f5cced910a | |||
| 8c9249054f | |||
| 7c2b513613 | |||
| 00844721ff | |||
| 068a0c2b8c | |||
| 5b54c4303d | |||
| 6178083985 | |||
| e16c07b9cd | |||
| 1ac28d6459 | |||
| 9b66040a02 | |||
| 7710f70fda | |||
| aa4d3317ed | |||
| f704d015d6 | |||
| 6b86d44cac | |||
| e3c15f012c | |||
| d2b1003612 | |||
| 36deab8670 | |||
| e11fcfbd08 | |||
| 11eecdc888 | |||
| 80eb4244a3 | |||
| e4da9bd718 | |||
| e28430ab3d | |||
| db4707f187 | |||
| 3a0dbccc2e | |||
| 846517625b | |||
| f21e88b112 | |||
| a5594c3d89 | |||
| fc927caadd | |||
| b7e6334c13 | |||
| 65091aa6a8 | |||
| 3c78a56ab0 | |||
| bdd7ecd205 | |||
| a70a3787d8 | |||
| b2ae430805 | |||
| 299eb0d965 | |||
| f5a780099b | |||
| 5cfafcf39b | |||
| a49243c671 | |||
| c2a15f5214 | |||
| ee54f1e618 | |||
| e6b85b60a8 | |||
| 8c1e3c09ff | |||
| ca06c7ec2c | |||
| 4a41325d1a | |||
| 45e2bd0c28 | |||
| e2fb0427f9 | |||
| a825326ede | |||
| e313276e49 | |||
| 66af487978 | |||
| d668a9293f | |||
| ace28c1f85 | |||
| 2ad8ab534e | |||
| f7df4fa62a | |||
| ca4e66daab | |||
| 5b9c5289c2 | |||
| 3f9b84ec70 | |||
| 73bd5d9d95 | |||
| 398d2d87c8 | |||
| 59d8d93832 | |||
| 59494d5089 | |||
| 019e80a218 | |||
| 9546baf1ab | |||
| 60d8e8a20b | |||
| 24ca61eb6e | |||
| d92c563b9e | |||
| e087044658 | |||
| 16d381fc0e | |||
| 1021d7228a | |||
| 0a246e3736 | |||
| 380ed40b47 | |||
| 679ea238de | |||
| baadaccc30 | |||
| 87d47a6e5d | |||
| fba0b34f19 | |||
| c9eeb2fa8a | |||
| 2f83d6789b | |||
| 7a4a3d27c6 | |||
| 4c5e73b2f2 | |||
| 4c44bc649a | |||
| 3b49842df0 | |||
| 848b330825 | |||
| 270a003c55 | |||
| 8b57076194 | |||
| d31bd3cd10 | |||
| 698ebdfb3f | |||
| c7233d6624 | |||
| 493a2cc1ba | |||
| 3ebb426abe | |||
| 5aac224a97 | |||
| 1b6ae33ce0 | |||
| 537d34b8cd | |||
| 3fdf2964c8 | |||
| 2e5874f839 | |||
| 5d05897ae0 | |||
| 6850482247 | |||
| b00b7ab8f1 | |||
| 924958bab5 | |||
| 968ec4a8be | |||
| 8d34b4d101 | |||
| 882ad4c113 | |||
| e9728192e2 | |||
| a206a0779e | |||
| 6cce3d60bb | |||
| 42433584ab | |||
| ba6a0f25d9 | |||
| 5e3010c6b5 | |||
| bc888931fd | |||
| 4ac7c56c89 | |||
| ddacce6e75 |
+12
-5
@@ -34,7 +34,8 @@ e2e-vault = { max-threads = 1 }
|
|||||||
|
|
||||||
# Reliability / fault-injection e2e tests each spawn a single-node 4-disk RustFS
|
# Reliability / fault-injection e2e tests each spawn a single-node 4-disk RustFS
|
||||||
# server and manipulate its disk directories at runtime (crates/e2e_test:
|
# server and manipulate its disk directories at runtime (crates/e2e_test:
|
||||||
# reliability_disk_fault_test, degraded_read_eof_regression_test / dist-13). They
|
# reliability_disk_fault_test, degraded_read_eof_regression_test / dist-13, and
|
||||||
|
# replacement_privileged_e2e_test when explicitly run as root on Linux). They
|
||||||
# are correct in isolation but resource-heavy; serialize them under nextest's
|
# are correct in isolation but resource-heavy; serialize them under nextest's
|
||||||
# process boundary (serial_test's #[serial] does not cross it) so several 4-disk
|
# process boundary (serial_test's #[serial] does not cross it) so several 4-disk
|
||||||
# servers never run at once. ci-7's nightly picks these up via the e2e suite;
|
# servers never run at once. ci-7's nightly picks these up via the e2e suite;
|
||||||
@@ -90,7 +91,7 @@ test-group = 'ecstore-serial-flaky'
|
|||||||
# e2e-reliability test-group note above). The matching ci-profile override is at
|
# e2e-reliability test-group note above). The matching ci-profile override is at
|
||||||
# the end of the file, after [profile.ci] is declared.
|
# the end of the file, after [profile.ci] is declared.
|
||||||
[[profile.default.overrides]]
|
[[profile.default.overrides]]
|
||||||
filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression)_test::/)'
|
filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression|replacement_privileged_e2e)_test::/)'
|
||||||
test-group = 'e2e-reliability'
|
test-group = 'e2e-reliability'
|
||||||
|
|
||||||
[[profile.default.overrides]]
|
[[profile.default.overrides]]
|
||||||
@@ -155,7 +156,7 @@ retries = 2
|
|||||||
# quarantine: no retries, just single-threaded so several 4-disk servers never
|
# quarantine: no retries, just single-threaded so several 4-disk servers never
|
||||||
# run concurrently when ci-7's nightly runs the full e2e suite.
|
# run concurrently when ci-7's nightly runs the full e2e suite.
|
||||||
[[profile.ci.overrides]]
|
[[profile.ci.overrides]]
|
||||||
filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression)_test::/)'
|
filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression|replacement_privileged_e2e)_test::/)'
|
||||||
test-group = 'e2e-reliability'
|
test-group = 'e2e-reliability'
|
||||||
|
|
||||||
# Serialize the multipart crash-consistency scenarios under the ci profile too
|
# Serialize the multipart crash-consistency scenarios under the ci profile too
|
||||||
@@ -251,10 +252,16 @@ test-group = 'ecstore-serial-flaky'
|
|||||||
# cluster, so it keeps the lane's parallel-safe / no-external-dependency
|
# cluster, so it keeps the lane's parallel-safe / no-external-dependency
|
||||||
# properties. The RustFS warm backend has no loopback guard (that guard is
|
# properties. The RustFS warm backend has no loopback guard (that guard is
|
||||||
# replication-only), so it needs no opt-in env for its 127.0.0.1 tier target.
|
# replication-only), so it needs no opt-in env for its 127.0.0.1 tier target.
|
||||||
|
#
|
||||||
|
# Disk compression (backlog#1848): the `compression` module joins the smoke
|
||||||
|
# lane so the multipart disk-compression roundtrips (restored after
|
||||||
|
# rustfs/rustfs#5169 disabled them) have PR-lane signal, not just merge-gate.
|
||||||
|
# Single-node servers on random ports with isolated temp dirs — meets the
|
||||||
|
# admission criteria unchanged.
|
||||||
[profile.e2e-smoke]
|
[profile.e2e-smoke]
|
||||||
default-filter = """
|
default-filter = """
|
||||||
package(e2e_test) & (
|
package(e2e_test) & (
|
||||||
test(/^(delete_marker_migration_semantics|version_id_regression|list_objects_v2_pagination|list_object_versions_regression|list_objects_duplicates|list_buckets_double_slash|list_buckets_auth|list_buckets_iam_filter|leading_slash_key|special_chars|create_bucket_region|delete_objects_versioning|head_object_consistency|head_object_range|copy_object_metadata|copy_object_tagging|copy_source_invalid_date|content_encoding|multipart_storage_class|storage_class_capability|ssec_copy|anonymous_access|bucket_policy_check|presigned_negative|negative_sigv4|admin_auth|notification_webhook|tls_hot_reload|console_smoke|admin_iam_crud|admin_pools|sts_query_compat)_test::|^fake_s3_target::/)
|
test(/^(delete_marker_migration_semantics|version_id_regression|list_objects_v2_pagination|list_object_versions_regression|list_objects_duplicates|list_buckets_double_slash|list_buckets_auth|list_buckets_iam_filter|leading_slash_key|special_chars|create_bucket_region|delete_objects_versioning|head_object_consistency|head_object_range|copy_object_metadata|copy_object_tagging|copy_source_invalid_date|content_encoding|compression|multipart_storage_class|storage_class_capability|ssec_copy|anonymous_access|bucket_policy_check|presigned_negative|negative_sigv4|admin_auth|notification_webhook|tls_hot_reload|console_smoke|admin_iam_crud|admin_pools|sts_query_compat)_test::|^fake_s3_target::/)
|
||||||
| test(/^replication_extension_test::(test_replication_check_succeeds_with_remote_target|test_replication_check_rejects_target_without_object_lock|test_set_remote_target_rejects_unversioned_source_bucket|test_replication_check_rejects_unversioned_source_bucket|test_replication_check_rejects_missing_replication_config|test_replication_check_rejects_invalid_bucket|test_set_remote_target_rejects_same_bucket_on_same_deployment|test_set_remote_target_rejects_unversioned_target_bucket|test_set_remote_target_update_requires_arn|test_set_remote_target_update_rejects_missing_target|test_set_remote_target_rejects_invalid_target_url|test_set_remote_target_rejects_self_signed_https_target_without_skip_tls_verify|test_set_remote_target_rejects_private_ca_https_target_without_ca_cert_pem|test_list_remote_targets_rejects_empty_bucket|test_list_remote_targets_rejects_invalid_bucket|test_remove_remote_target_rejects_missing_target|test_remove_remote_target_rejects_missing_arn|test_remove_remote_target_rejects_invalid_bucket|test_remove_remote_target_rejects_target_used_by_replication|test_delete_bucket_replication_removes_remote_target)$/)
|
| test(/^replication_extension_test::(test_replication_check_succeeds_with_remote_target|test_replication_check_rejects_target_without_object_lock|test_set_remote_target_rejects_unversioned_source_bucket|test_replication_check_rejects_unversioned_source_bucket|test_replication_check_rejects_missing_replication_config|test_replication_check_rejects_invalid_bucket|test_set_remote_target_rejects_same_bucket_on_same_deployment|test_set_remote_target_rejects_unversioned_target_bucket|test_set_remote_target_update_requires_arn|test_set_remote_target_update_rejects_missing_target|test_set_remote_target_rejects_invalid_target_url|test_set_remote_target_rejects_self_signed_https_target_without_skip_tls_verify|test_set_remote_target_rejects_private_ca_https_target_without_ca_cert_pem|test_list_remote_targets_rejects_empty_bucket|test_list_remote_targets_rejects_invalid_bucket|test_remove_remote_target_rejects_missing_target|test_remove_remote_target_rejects_missing_arn|test_remove_remote_target_rejects_invalid_bucket|test_remove_remote_target_rejects_target_used_by_replication|test_delete_bucket_replication_removes_remote_target)$/)
|
||||||
| test(/^reliant::lifecycle::/)
|
| test(/^reliant::lifecycle::/)
|
||||||
| test(/^reliant::tiering::/)
|
| test(/^reliant::tiering::/)
|
||||||
@@ -383,7 +390,7 @@ path = "junit.xml"
|
|||||||
# quarantine: no retries, just single-threaded so several 4-disk servers never
|
# quarantine: no retries, just single-threaded so several 4-disk servers never
|
||||||
# run concurrently.
|
# run concurrently.
|
||||||
[[profile.e2e-full.overrides]]
|
[[profile.e2e-full.overrides]]
|
||||||
filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression)_test::/)'
|
filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression|replacement_privileged_e2e)_test::/)'
|
||||||
test-group = 'e2e-reliability'
|
test-group = 'e2e-reliability'
|
||||||
|
|
||||||
[[profile.e2e-full.overrides]]
|
[[profile.e2e-full.overrides]]
|
||||||
|
|||||||
@@ -17,9 +17,11 @@
|
|||||||
# =============================================================================
|
# =============================================================================
|
||||||
#
|
#
|
||||||
# Metric source: the KMS operation-policy choke point in
|
# Metric source: the KMS operation-policy choke point in
|
||||||
# crates/kms/src/policy.rs. All label values are bounded static strings
|
# crates/kms/src/policy.rs, except KmsKeyRotationOverdue, which reads the
|
||||||
# (operation, op_class, outcome, error_class, backend, scope); key identifiers,
|
# label-less key-lifecycle gauge published by the deletion worker's sweep
|
||||||
# key material, and tokens never appear in labels.
|
# (crates/kms/src/deletion_worker.rs). All label values are bounded static
|
||||||
|
# strings (operation, op_class, outcome, error_class, backend, scope); key
|
||||||
|
# identifiers, key material, and tokens never appear in labels.
|
||||||
#
|
#
|
||||||
# Response procedures: docs/operations/kms-observability-runbook.md
|
# Response procedures: docs/operations/kms-observability-runbook.md
|
||||||
#
|
#
|
||||||
@@ -212,3 +214,38 @@ groups:
|
|||||||
circuit_open until the half-open probe succeeds or returns
|
circuit_open until the half-open probe succeeds or returns
|
||||||
a non-retryable failure.
|
a non-retryable failure.
|
||||||
runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmsbackendcircuitopen"
|
runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmsbackendcircuitopen"
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# 7. KmsKeyRotationOverdue
|
||||||
|
# The least recently rotated usable key has gone more than 400
|
||||||
|
# days without a rotation (measured from creation for keys with
|
||||||
|
# no recorded rotation). Direct gauge state published by the
|
||||||
|
# deletion worker's sweep, so no traffic guard applies; the
|
||||||
|
# one-hour hold only bridges scrape gaps. The worker runs only
|
||||||
|
# on backends with the schedule_deletion capability, so on the
|
||||||
|
# Static backend the series never exists and this alert cannot
|
||||||
|
# fire — that backend cannot rotate either; see the rotation
|
||||||
|
# driver matrix in docs/operations/kms-backend-security.md.
|
||||||
|
# Threshold: 400 days — conservative default sitting above a
|
||||||
|
# one-year rotation policy. Align it with the rotation period
|
||||||
|
# your compliance policy requires, and with
|
||||||
|
# RUSTFS_KMS_ROTATION_MAX_AGE_SECS so the per-key rotation_due
|
||||||
|
# verdict and this aggregate alert agree.
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
- alert: KmsKeyRotationOverdue
|
||||||
|
expr: |
|
||||||
|
rustfs_kms_oldest_key_rotation_age_seconds > (400 * 86400)
|
||||||
|
for: 1h
|
||||||
|
labels:
|
||||||
|
severity: warning
|
||||||
|
component: kms
|
||||||
|
annotations:
|
||||||
|
summary: "Oldest KMS key unrotated for more than 400 days"
|
||||||
|
description: >-
|
||||||
|
The least recently rotated usable KMS key was last rotated
|
||||||
|
{{ $value | humanizeDuration }} ago (measured from creation
|
||||||
|
for keys with no recorded rotation). List keys through the
|
||||||
|
admin API and read rotation_due / rotation_due_reason for
|
||||||
|
the per-key verdict; an "unsupported" reason means the
|
||||||
|
backend cannot rotate at all.
|
||||||
|
runbook_url: "https://github.com/rustfs/rustfs/blob/main/docs/operations/kms-observability-runbook.md#kmskeyrotationoverdue"
|
||||||
|
|||||||
@@ -85,7 +85,7 @@ runs:
|
|||||||
repo-token: ${{ github.token }}
|
repo-token: ${{ github.token }}
|
||||||
|
|
||||||
- name: Install flatc
|
- name: Install flatc
|
||||||
uses: Nugine/setup-flatc@e7855e994773ce90094a3f1626d4afc9080c23ae # v1
|
uses: Nugine/setup-flatc@698800de72a96bfb22cf60431dc21a2ff9a7e07b # v1
|
||||||
with:
|
with:
|
||||||
version: "25.12.19"
|
version: "25.12.19"
|
||||||
|
|
||||||
|
|||||||
@@ -182,7 +182,12 @@ jobs:
|
|||||||
echo '```'
|
echo '```'
|
||||||
} >> "$GITHUB_STEP_SUMMARY"
|
} >> "$GITHUB_STEP_SUMMARY"
|
||||||
|
|
||||||
# Readers: test-and-lint-rio-v2, build-rustfs-debug-binary-rio-v2.
|
# Readers: test-and-lint-rio-v2 (per-PR), build-rustfs-debug-binary-rio-v2
|
||||||
|
# (weekly schedule / manual dispatch only — dormant rio-v2 variant, see
|
||||||
|
# rustfs/backlog#1835 and docs/architecture/minio-file-format-compat.md).
|
||||||
|
# The second build below stays despite the reduced cadence: it warms the
|
||||||
|
# rio-v2,e2e-test-hooks feature resolution the scheduled build restores,
|
||||||
|
# which keeps that lane inside its 30-minute timeout.
|
||||||
warm-ci-feat-rio:
|
warm-ci-feat-rio:
|
||||||
name: Warm ci-feat-rio
|
name: Warm ci-feat-rio
|
||||||
runs-on: sm-standard-4
|
runs-on: sm-standard-4
|
||||||
|
|||||||
@@ -533,7 +533,12 @@ jobs:
|
|||||||
|
|
||||||
build-rustfs-debug-binary-rio-v2:
|
build-rustfs-debug-binary-rio-v2:
|
||||||
name: Build RustFS Debug Binary (rio-v2)
|
name: Build RustFS Debug Binary (rio-v2)
|
||||||
if: github.event_name != 'pull_request' || github.event.action != 'closed'
|
# Dormant rio-v2 variant (rustfs/backlog#1835): the feature ships in no
|
||||||
|
# default build, so this full-suite lane runs only on the weekly schedule
|
||||||
|
# and manual dispatch. Per-PR cfg-seam coverage stays with
|
||||||
|
# test-and-lint-rio-v2. Lifecycle and the promote-or-delete condition:
|
||||||
|
# docs/architecture/minio-file-format-compat.md ("rio-v2 variant lifecycle").
|
||||||
|
if: github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
|
||||||
needs: [ quick-checks ]
|
needs: [ quick-checks ]
|
||||||
runs-on: sm-standard-4
|
runs-on: sm-standard-4
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
@@ -824,6 +829,9 @@ jobs:
|
|||||||
|
|
||||||
e2e-tests-rio-v2:
|
e2e-tests-rio-v2:
|
||||||
name: End-to-End Tests (rio-v2)
|
name: End-to-End Tests (rio-v2)
|
||||||
|
# Inherits the schedule/dispatch-only gate through needs: on every other
|
||||||
|
# event build-rustfs-debug-binary-rio-v2 is skipped, so this job skips
|
||||||
|
# with it (see the dormant-variant comment on that job).
|
||||||
needs: [ build-rustfs-debug-binary-rio-v2 ]
|
needs: [ build-rustfs-debug-binary-rio-v2 ]
|
||||||
runs-on: sm-standard-2
|
runs-on: sm-standard-2
|
||||||
timeout-minutes: 30
|
timeout-minutes: 30
|
||||||
|
|||||||
@@ -94,6 +94,7 @@ jobs:
|
|||||||
short_sha: ${{ steps.check.outputs.short_sha }}
|
short_sha: ${{ steps.check.outputs.short_sha }}
|
||||||
is_prerelease: ${{ steps.check.outputs.is_prerelease }}
|
is_prerelease: ${{ steps.check.outputs.is_prerelease }}
|
||||||
create_latest: ${{ steps.check.outputs.create_latest }}
|
create_latest: ${{ steps.check.outputs.create_latest }}
|
||||||
|
source_ref: ${{ steps.check.outputs.source_ref }}
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout repository
|
- name: Checkout repository
|
||||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
@@ -118,6 +119,7 @@ jobs:
|
|||||||
short_sha=""
|
short_sha=""
|
||||||
is_prerelease=false
|
is_prerelease=false
|
||||||
create_latest=false
|
create_latest=false
|
||||||
|
source_ref="$GITHUB_SHA"
|
||||||
|
|
||||||
if [[ "${{ github.event_name }}" == "workflow_run" ]]; then
|
if [[ "${{ github.event_name }}" == "workflow_run" ]]; then
|
||||||
# Triggered by build workflow completion
|
# Triggered by build workflow completion
|
||||||
@@ -137,6 +139,7 @@ jobs:
|
|||||||
# Extract version info from commit message or use commit SHA
|
# Extract version info from commit message or use commit SHA
|
||||||
# Use Git to generate consistent short SHA (ensures uniqueness like build.yml)
|
# Use Git to generate consistent short SHA (ensures uniqueness like build.yml)
|
||||||
short_sha=$(git rev-parse --short "$HEAD_SHA")
|
short_sha=$(git rev-parse --short "$HEAD_SHA")
|
||||||
|
source_ref="$HEAD_SHA"
|
||||||
|
|
||||||
# Determine build type based on triggering workflow event and ref
|
# Determine build type based on triggering workflow event and ref
|
||||||
triggering_event="$TRIGGERING_EVENT"
|
triggering_event="$TRIGGERING_EVENT"
|
||||||
@@ -261,6 +264,23 @@ jobs:
|
|||||||
echo "⚠️ Only release versions (latest, v1.0.0, 1.0.0) and prereleases (v1.0.0-alpha1, 1.0.0-beta2) are supported"
|
echo "⚠️ Only release versions (latest, v1.0.0, 1.0.0) and prereleases (v1.0.0-alpha1, 1.0.0-beta2) are supported"
|
||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
|
if [[ "$should_build" == true && "$input_version" != "latest" ]]; then
|
||||||
|
tag_ref="refs/tags/$input_version"
|
||||||
|
if ! git ls-remote --exit-code origin "$tag_ref" >/dev/null 2>&1; then
|
||||||
|
if [[ "$input_version" == v* ]]; then
|
||||||
|
tag_ref="refs/tags/${input_version#v}"
|
||||||
|
else
|
||||||
|
tag_ref="refs/tags/v$input_version"
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ! git ls-remote --exit-code origin "$tag_ref" >/dev/null 2>&1; then
|
||||||
|
echo "❌ Release tag not found for Docker build: $input_version"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
source_ref="$tag_ref"
|
||||||
|
fi
|
||||||
fi
|
fi
|
||||||
|
|
||||||
{
|
{
|
||||||
@@ -271,6 +291,7 @@ jobs:
|
|||||||
echo "short_sha=$short_sha"
|
echo "short_sha=$short_sha"
|
||||||
echo "is_prerelease=$is_prerelease"
|
echo "is_prerelease=$is_prerelease"
|
||||||
echo "create_latest=$create_latest"
|
echo "create_latest=$create_latest"
|
||||||
|
echo "source_ref=$source_ref"
|
||||||
} >> "$GITHUB_OUTPUT"
|
} >> "$GITHUB_OUTPUT"
|
||||||
|
|
||||||
echo "🐳 Docker Build Summary:"
|
echo "🐳 Docker Build Summary:"
|
||||||
@@ -281,6 +302,7 @@ jobs:
|
|||||||
echo " - Short SHA: $short_sha"
|
echo " - Short SHA: $short_sha"
|
||||||
echo " - Is prerelease: $is_prerelease"
|
echo " - Is prerelease: $is_prerelease"
|
||||||
echo " - Create latest: $create_latest"
|
echo " - Create latest: $create_latest"
|
||||||
|
echo " - Source ref: $source_ref"
|
||||||
|
|
||||||
# Build multi-arch Docker images
|
# Build multi-arch Docker images
|
||||||
# Strategy: Build images using pre-built binaries from dl.rustfs.com
|
# Strategy: Build images using pre-built binaries from dl.rustfs.com
|
||||||
@@ -308,6 +330,7 @@ jobs:
|
|||||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
ref: ${{ needs.build-check.outputs.source_ref }}
|
||||||
|
|
||||||
- name: Login to Docker Hub
|
- name: Login to Docker Hub
|
||||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3
|
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3
|
||||||
@@ -397,7 +420,8 @@ jobs:
|
|||||||
LABELS="org.opencontainers.image.title=RustFS"
|
LABELS="org.opencontainers.image.title=RustFS"
|
||||||
LABELS="$LABELS,org.opencontainers.image.description=RustFS distributed object storage system"
|
LABELS="$LABELS,org.opencontainers.image.description=RustFS distributed object storage system"
|
||||||
LABELS="$LABELS,org.opencontainers.image.version=$VERSION"
|
LABELS="$LABELS,org.opencontainers.image.version=$VERSION"
|
||||||
LABELS="$LABELS,org.opencontainers.image.revision=${{ github.sha }}"
|
SOURCE_REVISION="$(git rev-parse HEAD)"
|
||||||
|
LABELS="$LABELS,org.opencontainers.image.revision=$SOURCE_REVISION"
|
||||||
LABELS="$LABELS,org.opencontainers.image.source=${{ github.server_url }}/${{ github.repository }}"
|
LABELS="$LABELS,org.opencontainers.image.source=${{ github.server_url }}/${{ github.repository }}"
|
||||||
LABELS="$LABELS,org.opencontainers.image.created=$(date -u +'%Y-%m-%dT%H:%M:%SZ')"
|
LABELS="$LABELS,org.opencontainers.image.created=$(date -u +'%Y-%m-%dT%H:%M:%SZ')"
|
||||||
LABELS="$LABELS,org.opencontainers.image.build-type=$BUILD_TYPE"
|
LABELS="$LABELS,org.opencontainers.image.build-type=$BUILD_TYPE"
|
||||||
|
|||||||
@@ -55,3 +55,142 @@ jobs:
|
|||||||
|
|
||||||
- name: Build RustFS
|
- name: Build RustFS
|
||||||
run: cargo build --release --locked --target x86_64-unknown-linux-gnu -p rustfs --bins
|
run: cargo build --release --locked --target x86_64-unknown-linux-gnu -p rustfs --bins
|
||||||
|
|
||||||
|
# Live-Vault lane for the rustfs-kms suite (rustfs/backlog#1774).
|
||||||
|
#
|
||||||
|
# RUSTFS_KMS_VAULT_TOKEN is the single switch that adds the Vault KV2 and
|
||||||
|
# Vault Transit backends to every for_each_backend spec in
|
||||||
|
# crates/kms/tests/behavior_*.rs (see crates/kms/AGENTS.md). rotate and
|
||||||
|
# versioning are advertised only by the Vault backends, so without this lane
|
||||||
|
# no CI run ever asserts the working half of behavior_rotation.rs — a
|
||||||
|
# rotation that silently dropped historical key versions would stay green.
|
||||||
|
# The same lane runs the dev-Vault #[ignore] tests and the two self-hosting
|
||||||
|
# live scripts (AppRole login, three-node Raft leader failover).
|
||||||
|
#
|
||||||
|
# GitHub-hosted ubuntu-latest, deliberately not the self-hosted sm-standard
|
||||||
|
# fleet: the HA failover script needs a working Docker daemon, and the
|
||||||
|
# self-hosted fleet is heterogeneous — a docker-dependent workflow has been
|
||||||
|
# burned by it before (see the banner in e2e-s3tests.yml, rustfs/backlog#1149).
|
||||||
|
kms-vault-lane:
|
||||||
|
name: KMS live Vault lane
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
timeout-minutes: 90
|
||||||
|
env:
|
||||||
|
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
||||||
|
# Root token of the ephemeral loopback dev server. Not a secret: the
|
||||||
|
# server lives only for this job, listens on 127.0.0.1, and holds only
|
||||||
|
# keys the tests create. The literal value matters — the dev-Vault
|
||||||
|
# #[ignore] fixtures in crates/kms/src/backends/vault.rs hardcode it.
|
||||||
|
VAULT_LANE_TOKEN: dev-only-token
|
||||||
|
VAULT_LANE_ADDR: http://127.0.0.1:8200
|
||||||
|
# Keeps a runner-level proxy from swallowing the loopback dev-server
|
||||||
|
# traffic (see crates/kms/AGENTS.md). Actions env keys are
|
||||||
|
# case-insensitive, so only the uppercase form is set; reqwest reads
|
||||||
|
# either casing.
|
||||||
|
NO_PROXY: 127.0.0.1,localhost
|
||||||
|
steps:
|
||||||
|
- name: Checkout main branch
|
||||||
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
|
with:
|
||||||
|
persist-credentials: false
|
||||||
|
ref: main
|
||||||
|
|
||||||
|
- name: Setup Rust environment
|
||||||
|
uses: ./.github/actions/setup
|
||||||
|
with:
|
||||||
|
# Dedicated key: rust-cache cannot tell runner images apart, so
|
||||||
|
# sharing a key with an sm-standard lane would let two different
|
||||||
|
# system images overwrite each other's artifacts (same reasoning as
|
||||||
|
# ci.yml's ci-uring lane). Saved from this nightly job itself so the
|
||||||
|
# next night starts warm.
|
||||||
|
cache-shared-key: kms-vault-lane
|
||||||
|
cache-save-if: 'true'
|
||||||
|
install-build-packaging-tools: 'false'
|
||||||
|
install-test-tools: 'false'
|
||||||
|
|
||||||
|
- name: Install Vault CLI
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
wget -qO- https://apt.releases.hashicorp.com/gpg | sudo gpg --dearmor -o /usr/share/keyrings/hashicorp-archive-keyring.gpg
|
||||||
|
echo "deb [signed-by=/usr/share/keyrings/hashicorp-archive-keyring.gpg] https://apt.releases.hashicorp.com $(lsb_release -cs) main" | sudo tee /etc/apt/sources.list.d/hashicorp.list >/dev/null
|
||||||
|
sudo apt-get update -qq
|
||||||
|
sudo apt-get install -y -qq vault
|
||||||
|
vault version
|
||||||
|
|
||||||
|
- name: Start Vault dev server with KV2 and Transit engines
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
nohup vault server -dev \
|
||||||
|
-dev-root-token-id="${VAULT_LANE_TOKEN}" \
|
||||||
|
-dev-listen-address=127.0.0.1:8200 >/tmp/vault-dev.log 2>&1 &
|
||||||
|
for _ in $(seq 1 60); do
|
||||||
|
if curl -fsS "${VAULT_LANE_ADDR}/v1/sys/health" >/dev/null 2>&1; then
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
sleep 1
|
||||||
|
done
|
||||||
|
curl -fsS "${VAULT_LANE_ADDR}/v1/sys/health"
|
||||||
|
export VAULT_ADDR="${VAULT_LANE_ADDR}" VAULT_TOKEN="${VAULT_LANE_TOKEN}"
|
||||||
|
# Dev mode mounts KV v2 at secret/ by default; Transit is explicit.
|
||||||
|
# Prove both engines actually work rather than assuming the defaults.
|
||||||
|
vault secrets enable transit
|
||||||
|
vault kv put secret/rustfs-ci-lane-probe value=ok >/dev/null
|
||||||
|
vault kv get secret/rustfs-ci-lane-probe >/dev/null
|
||||||
|
vault write -f transit/keys/rustfs-ci-lane-probe >/dev/null
|
||||||
|
|
||||||
|
- name: Run rustfs-kms suite with the Vault lane on
|
||||||
|
env:
|
||||||
|
RUSTFS_KMS_VAULT_TOKEN: ${{ env.VAULT_LANE_TOKEN }}
|
||||||
|
RUSTFS_KMS_VAULT_ADDR: ${{ env.VAULT_LANE_ADDR }}
|
||||||
|
run: cargo test -p rustfs-kms --locked
|
||||||
|
|
||||||
|
- name: Run dev-Vault ignored tests
|
||||||
|
env:
|
||||||
|
RUSTFS_KMS_VAULT_TOKEN: ${{ env.VAULT_LANE_TOKEN }}
|
||||||
|
RUSTFS_KMS_VAULT_ADDR: ${{ env.VAULT_LANE_ADDR }}
|
||||||
|
# Filters select the dev-Vault-only #[ignore] tests. The AWS #[ignore]
|
||||||
|
# tests (backends::aws, service_manager) stay excluded — they need real
|
||||||
|
# AWS credentials and create billable keys. The AppRole and HA #[ignore]
|
||||||
|
# tests are excluded here because their own scripts below provision the
|
||||||
|
# Vault topology they need.
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
cargo test -p rustfs-kms --locked --lib backends::contract_tests -- --ignored
|
||||||
|
cargo test -p rustfs-kms --locked --lib backends::vault -- --ignored
|
||||||
|
cargo test -p rustfs-kms --locked --test vault_fault_injection -- --ignored
|
||||||
|
|
||||||
|
- name: Run AppRole live checks (self-hosting ephemeral Vault)
|
||||||
|
run: bash scripts/test/vault_approle_kms_live.sh
|
||||||
|
|
||||||
|
- name: Show Vault dev server log on failure
|
||||||
|
if: failure()
|
||||||
|
run: tail -n 200 /tmp/vault-dev.log || true
|
||||||
|
|
||||||
|
# Three-node Raft leader failover (crates/kms/tests/vault_ha_failover_live.rs,
|
||||||
|
# first validated by rustfs/rustfs#5653). Its own job so an election-timing
|
||||||
|
# flake cannot mask the main lane's verdict, and vice versa. The script
|
||||||
|
# provisions and tears down its own Docker cluster.
|
||||||
|
kms-vault-ha-failover:
|
||||||
|
name: KMS Vault HA failover lane
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
timeout-minutes: 60
|
||||||
|
env:
|
||||||
|
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
||||||
|
NO_PROXY: 127.0.0.1,localhost
|
||||||
|
steps:
|
||||||
|
- name: Checkout main branch
|
||||||
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
|
with:
|
||||||
|
persist-credentials: false
|
||||||
|
ref: main
|
||||||
|
|
||||||
|
- name: Setup Rust environment
|
||||||
|
uses: ./.github/actions/setup
|
||||||
|
with:
|
||||||
|
cache-shared-key: kms-vault-lane
|
||||||
|
cache-save-if: 'false'
|
||||||
|
install-build-packaging-tools: 'false'
|
||||||
|
install-test-tools: 'false'
|
||||||
|
|
||||||
|
- name: Run HA leader failover live checks (three-node Raft cluster in Docker)
|
||||||
|
run: bash scripts/test/vault_ha_kms_live.sh
|
||||||
|
|||||||
+68
-21
@@ -1,6 +1,6 @@
|
|||||||
# ARCHITECTURE.md
|
# ARCHITECTURE.md
|
||||||
|
|
||||||
> Last updated: 2026-07-02 · Revision: 2
|
> Last updated: 2026-08-12 · Revision: 3
|
||||||
>
|
>
|
||||||
> This document describes the high-level architecture of RustFS.
|
> This document describes the high-level architecture of RustFS.
|
||||||
> If you want to familiarize yourself with the code base, you are in the right place!
|
> If you want to familiarize yourself with the code base, you are in the right place!
|
||||||
@@ -101,7 +101,10 @@ refactors.
|
|||||||
|
|
||||||
The `rustfs` binary crate composes these libraries into the running server.
|
The `rustfs` binary crate composes these libraries into the running server.
|
||||||
`ecstore` remains the storage engine at the architectural center; its internal
|
`ecstore` remains the storage engine at the architectural center; its internal
|
||||||
module split is tracked under `docs/architecture/`.
|
module split is tracked under `docs/architecture/`. `rio-v2` is the
|
||||||
|
feature-gated MinIO on-disk format compatibility I/O layer; it ships in no
|
||||||
|
default build (lifecycle:
|
||||||
|
[docs/architecture/minio-file-format-compat.md](docs/architecture/minio-file-format-compat.md)).
|
||||||
|
|
||||||
## Architecture Invariants
|
## Architecture Invariants
|
||||||
|
|
||||||
@@ -119,19 +122,44 @@ module split is tracked under `docs/architecture/`.
|
|||||||
|
|
||||||
3. **Each type has exactly one definition.** Types shared across crates must be defined
|
3. **Each type has exactly one definition.** Types shared across crates must be defined
|
||||||
in one crate and re-exported or imported by others.
|
in one crate and re-exported or imported by others.
|
||||||
- ⚠️ VIOLATED: `ReplicationStats` (4 copies), `LastMinuteLatency` (3 copies),
|
- ⚠️ VIOLATED: `ReplicationStats` names three unrelated types
|
||||||
`BackpressureConfig` (3 copies), `DataUsageInfo` (2 copies).
|
(`crates/data-usage/src/data_usage.rs`,
|
||||||
|
`crates/obs/src/metrics/collectors/replication.rs`,
|
||||||
|
`crates/ecstore/src/bucket/replication/replication_state.rs`) — a naming
|
||||||
|
collision, not copies; renaming is tracked in rustfs/backlog#1847.
|
||||||
|
- `LastMinuteLatency` has two deliberately different implementations: the
|
||||||
|
per-second bucketed accumulator in `crates/common/src/last_minute.rs` and
|
||||||
|
the in-memory endpoint-health sample tracker in
|
||||||
|
`crates/ecstore/src/bucket/bucket_target_sys.rs` (its doc comment explains
|
||||||
|
why it stays local).
|
||||||
|
- ✅ RESOLVED: `BackpressureConfig` and `DataUsageInfo` each have exactly one
|
||||||
|
definition (`crates/io-core/src/backpressure.rs`,
|
||||||
|
`crates/data-usage/src/data_usage.rs`). The zero-consumer
|
||||||
|
`BackpressureSettings` copy that lingered in io-metrics was removed
|
||||||
|
(rustfs/backlog#1833).
|
||||||
|
|
||||||
4. **ecstore does not know about HTTP or S3 protocol details.** It operates on
|
4. **ecstore does not know about HTTP or S3 protocol details.** It operates on
|
||||||
storage-level abstractions (objects, buckets, disks, pools).
|
storage-level abstractions (objects, buckets, disks, pools).
|
||||||
|
- ⚠️ VIOLATED: 58 files under `crates/ecstore/src` reference `s3s`
|
||||||
|
(`rg -l 's3s' crates/ecstore/src | wc -l`), `crates/ecstore/src/client/`
|
||||||
|
is a ~9.4K-line embedded S3 HTTP client, and `crates/ecstore/Cargo.toml`
|
||||||
|
depends on `s3s`, `http`, `hyper`/`hyper-util`/`hyper-rustls`, and
|
||||||
|
`reqwest`. Target state: the engine's need to act as an S3 client
|
||||||
|
(tiering, replication targets) is served by an extracted client crate,
|
||||||
|
and ecstore holds no wire or DTO types.
|
||||||
|
|
||||||
5. **The `rustfs` binary crate is the only place that wires everything together.**
|
5. **The `rustfs` binary crate is the only place that wires everything together.**
|
||||||
Individual crates should be testable in isolation.
|
Individual crates should be testable in isolation.
|
||||||
|
|
||||||
6. **Error types use `thiserror` with descriptive names** (e.g., `StorageError`,
|
6. **Error types use `thiserror` with descriptive names** (e.g., `StorageError`,
|
||||||
not bare `Error`).
|
not bare `Error`).
|
||||||
- ⚠️ VIOLATED: 6 crates use `pub enum Error`; 2 crates use `snafu`;
|
- ✅ RESOLVED (strategy): `snafu` is gone from source
|
||||||
`heal` use `anyhow` in library code.
|
(`rg -l snafu crates/ rustfs/` is empty) and library code no longer uses
|
||||||
|
`anyhow` (remaining hits are test code and the `e2e_test` crate; `heal`
|
||||||
|
uses `thiserror`).
|
||||||
|
- ⚠️ VIOLATED (naming): 6 crates still export a bare `pub enum Error`:
|
||||||
|
`crypto`, `filemeta`, `heal`, `iam`, `policy`, and `replication`
|
||||||
|
(`src/resync.rs`) — all `thiserror`-derived.
|
||||||
|
|
||||||
## Known Structural Issues
|
## Known Structural Issues
|
||||||
|
|
||||||
@@ -140,13 +168,25 @@ module split is tracked under `docs/architecture/`.
|
|||||||
|
|
||||||
### Critical
|
### Critical
|
||||||
|
|
||||||
- **common/scanner code duplication (~3K lines).** `scanner` depends on `common`
|
- **scanner/data-usage duplicate `.usage-cache.bin` serialization types.** The
|
||||||
but maintains its own copies of `DataUsageInfo`, `LastMinuteLatency`, and related
|
original finding ("common/scanner code duplication, ~3K lines") is resolved:
|
||||||
types instead of importing them.
|
`scanner` imports the shared data-usage types from `rustfs-data-usage` (see
|
||||||
|
the `pub use rustfs_data_usage::…` re-exports at the top of
|
||||||
|
`crates/scanner/src/data_usage_define.rs`). What remains: `scanner` and
|
||||||
|
`data-usage` each hold their own serialization types for the scanner cache
|
||||||
|
file (`DataUsageCacheInfo`/`DataUsageEntryInfo` in
|
||||||
|
`crates/scanner/src/data_usage_define.rs` vs
|
||||||
|
`DataUsageCacheInfo`/`DataUsageEntry` in
|
||||||
|
`crates/data-usage/src/data_usage.rs`); convergence is tracked in
|
||||||
|
rustfs/backlog#1828.
|
||||||
|
|
||||||
- **ecstore is a monolith (87K lines, 163 files).** It contains disk management,
|
- **ecstore is a monolith (265 files, ~288K lines — roughly half is inline
|
||||||
bucket management, erasure coding, replication, lifecycle, RPC, and configuration
|
`#[cfg(test)]` code).** Measured with
|
||||||
— all in one crate. It should be decomposed along its existing subdirectories.
|
`find crates/ecstore/src -name '*.rs' | xargs wc -l`. It contains disk
|
||||||
|
management, bucket management, erasure coding, replication, lifecycle, RPC,
|
||||||
|
and configuration — all in one crate. It should be decomposed along its
|
||||||
|
existing subdirectories; the split plan lives in
|
||||||
|
[docs/architecture/ecstore-module-split-plan.md](docs/architecture/ecstore-module-split-plan.md).
|
||||||
|
|
||||||
### High
|
### High
|
||||||
|
|
||||||
@@ -154,19 +194,26 @@ module split is tracked under `docs/architecture/`.
|
|||||||
`common → filemeta/madmin` edges must stay removed so leaf/helper crates do
|
`common → filemeta/madmin` edges must stay removed so leaf/helper crates do
|
||||||
not regain upward dependencies.
|
not regain upward dependencies.
|
||||||
|
|
||||||
- **Three-layer BackpressureConfig/DeadlockConfig duplication** across io-core,
|
- **Three-layer backpressure/deadlock policy bridging** across io-core,
|
||||||
concurrency, and `rustfs/src/storage`. Storage policies now expose and consume
|
concurrency, and `rustfs/src/storage`. The config types are no longer
|
||||||
explicit projections into the concurrency/io-core policy shapes, and workload
|
duplicated (`BackpressureConfig` and `DeadlockDetectorConfig` are each
|
||||||
|
defined once, in io-core). Storage policies expose and consume explicit
|
||||||
|
projections into the concurrency/io-core policy shapes, and workload
|
||||||
admission snapshots are composed through provider registries; later work
|
admission snapshots are composed through provider registries; later work
|
||||||
should use those bridges before deleting compatibility wrappers.
|
should use those bridges before deleting compatibility wrappers.
|
||||||
|
|
||||||
### Medium
|
### Medium
|
||||||
|
|
||||||
- **Inconsistent error handling.** Three strategies (thiserror/snafu/anyhow) and
|
- **Bare `Error` naming.** Error-handling strategy has converged on `thiserror`
|
||||||
mixed naming (bare `Error` vs descriptive names).
|
(no `snafu`, no `anyhow` in library code); the remaining inconsistency is the
|
||||||
|
bare `pub enum Error` naming in the 6 crates listed under Invariant 6.
|
||||||
|
|
||||||
- **Ambiguous common vs utils boundary.** Both described as "utilities and data
|
- **`common` is mostly parked domain code, not shared utilities.** Of its
|
||||||
structures." Need clear ownership rules.
|
6,724 lines, ~83% is scanner/heal domain code stranded there to break
|
||||||
|
dependency cycles (`metrics.rs`, ~4,810 lines of scanner-domain metrics;
|
||||||
|
`heal_channel.rs`, ~776 lines of heal-domain channel types). The
|
||||||
|
"common vs utils" naming ambiguity is secondary to moving that code to its
|
||||||
|
domain owners.
|
||||||
|
|
||||||
## Cross-Cutting Concerns
|
## Cross-Cutting Concerns
|
||||||
|
|
||||||
@@ -232,7 +279,7 @@ The binary (`main.rs`) boots in this order:
|
|||||||
|
|
||||||
```
|
```
|
||||||
┌─────────┐
|
┌─────────┐
|
||||||
│ rustfs │ (binary + lib, 75K lines)
|
│ rustfs │ (binary + lib)
|
||||||
│ main │
|
│ main │
|
||||||
└────┬────┘
|
└────┬────┘
|
||||||
│
|
│
|
||||||
@@ -255,7 +302,7 @@ The binary (`main.rs`) boots in this order:
|
|||||||
│ │ │
|
│ │ │
|
||||||
┌─────▼──────┐ ┌──────▼──────┐ ┌──────▼──────┐
|
┌─────▼──────┐ ┌──────▼──────┐ ┌──────▼──────┐
|
||||||
│ ecstore │ │ rio │ │ io-core │
|
│ ecstore │ │ rio │ │ io-core │
|
||||||
│ (87K,core) │ │ (readers) │ │ (zero-copy) │
|
│ (core) │ │ (readers) │ │ (zero-copy) │
|
||||||
└─────┬──────┘ └─────────────┘ └─────────────┘
|
└─────┬──────┘ └─────────────┘ └─────────────┘
|
||||||
│
|
│
|
||||||
┌─────┬──┼──┬─────┬──────┐
|
┌─────┬──┼──┬─────┬──────┐
|
||||||
|
|||||||
Generated
+241
-178
File diff suppressed because it is too large
Load Diff
+63
-62
@@ -41,7 +41,7 @@ members = [
|
|||||||
"crates/protocols", # Protocol implementations (FTPS, SFTP, etc.)
|
"crates/protocols", # Protocol implementations (FTPS, SFTP, etc.)
|
||||||
"crates/protos", # Protocol buffer definitions
|
"crates/protos", # Protocol buffer definitions
|
||||||
"crates/rio", # Rust I/O utilities and abstractions
|
"crates/rio", # Rust I/O utilities and abstractions
|
||||||
"crates/rio-v2", # Next-generation Rust I/O compatibility layer
|
"crates/rio-v2", # MinIO on-disk format compatibility I/O layer (feature-gated, ships in no default build)
|
||||||
"crates/replication", # Replication contracts and wire formats
|
"crates/replication", # Replication contracts and wire formats
|
||||||
"crates/concurrency", # Concurrency management for RustFS - timeout, locking, backpressure, and I/O scheduling
|
"crates/concurrency", # Concurrency management for RustFS - timeout, locking, backpressure, and I/O scheduling
|
||||||
"crates/s3-types", # S3 event type definitions
|
"crates/s3-types", # S3 event type definitions
|
||||||
@@ -69,7 +69,7 @@ edition = "2024"
|
|||||||
license = "Apache-2.0"
|
license = "Apache-2.0"
|
||||||
repository = "https://github.com/rustfs/rustfs"
|
repository = "https://github.com/rustfs/rustfs"
|
||||||
rust-version = "1.97.1"
|
rust-version = "1.97.1"
|
||||||
version = "1.0.0-rc.1"
|
version = "1.0.0-rc.2"
|
||||||
homepage = "https://rustfs.com"
|
homepage = "https://rustfs.com"
|
||||||
description = "RustFS is a high-performance distributed object storage software built using Rust, one of the most popular languages worldwide. "
|
description = "RustFS is a high-performance distributed object storage software built using Rust, one of the most popular languages worldwide. "
|
||||||
keywords = ["RustFS", "Minio", "object-storage", "filesystem", "s3"]
|
keywords = ["RustFS", "Minio", "object-storage", "filesystem", "s3"]
|
||||||
@@ -86,52 +86,52 @@ redundant_clone = "warn"
|
|||||||
|
|
||||||
[workspace.dependencies]
|
[workspace.dependencies]
|
||||||
# RustFS Internal Crates
|
# RustFS Internal Crates
|
||||||
rustfs = { path = "./rustfs", version = "1.0.0-rc.1" }
|
rustfs = { path = "./rustfs", version = "1.0.0-rc.2" }
|
||||||
rustfs-heal = { path = "crates/heal", version = "1.0.0-rc.1" }
|
rustfs-heal = { path = "crates/heal", version = "1.0.0-rc.2" }
|
||||||
rustfs-audit = { path = "crates/audit", version = "1.0.0-rc.1" }
|
rustfs-audit = { path = "crates/audit", version = "1.0.0-rc.2" }
|
||||||
rustfs-checksums = { path = "crates/checksums", version = "1.0.0-rc.1" }
|
rustfs-checksums = { path = "crates/checksums", version = "1.0.0-rc.2" }
|
||||||
rustfs-common = { path = "crates/common", version = "1.0.0-rc.1" }
|
rustfs-common = { path = "crates/common", version = "1.0.0-rc.2" }
|
||||||
rustfs-data-usage = { path = "crates/data-usage", version = "1.0.0-rc.1" }
|
rustfs-data-usage = { path = "crates/data-usage", version = "1.0.0-rc.2" }
|
||||||
rustfs-config = { path = "./crates/config", version = "1.0.0-rc.1" }
|
rustfs-config = { path = "./crates/config", version = "1.0.0-rc.2" }
|
||||||
rustfs-concurrency = { path = "./crates/concurrency", version = "1.0.0-rc.1" }
|
rustfs-concurrency = { path = "./crates/concurrency", version = "1.0.0-rc.2" }
|
||||||
rustfs-credentials = { path = "crates/credentials", version = "1.0.0-rc.1" }
|
rustfs-credentials = { path = "crates/credentials", version = "1.0.0-rc.2" }
|
||||||
rustfs-crypto = { path = "crates/crypto", version = "1.0.0-rc.1" }
|
rustfs-crypto = { path = "crates/crypto", version = "1.0.0-rc.2" }
|
||||||
rustfs-ecstore = { path = "crates/ecstore", version = "1.0.0-rc.1" }
|
rustfs-ecstore = { path = "crates/ecstore", version = "1.0.0-rc.2" }
|
||||||
rustfs-filemeta = { path = "crates/filemeta", version = "1.0.0-rc.1" }
|
rustfs-filemeta = { path = "crates/filemeta", version = "1.0.0-rc.2" }
|
||||||
rustfs-iam = { path = "crates/iam", version = "1.0.0-rc.1" }
|
rustfs-iam = { path = "crates/iam", version = "1.0.0-rc.2" }
|
||||||
rustfs-keystone = { path = "crates/keystone", version = "1.0.0-rc.1" }
|
rustfs-keystone = { path = "crates/keystone", version = "1.0.0-rc.2" }
|
||||||
rustfs-lifecycle = { path = "crates/lifecycle", version = "1.0.0-rc.1" }
|
rustfs-lifecycle = { path = "crates/lifecycle", version = "1.0.0-rc.2" }
|
||||||
rustfs-kms = { path = "crates/kms", version = "1.0.0-rc.1" }
|
rustfs-kms = { path = "crates/kms", version = "1.0.0-rc.2" }
|
||||||
rustfs-lock = { path = "crates/lock", version = "1.0.0-rc.1" }
|
rustfs-lock = { path = "crates/lock", version = "1.0.0-rc.2" }
|
||||||
rustfs-madmin = { path = "crates/madmin", version = "1.0.0-rc.1" }
|
rustfs-madmin = { path = "crates/madmin", version = "1.0.0-rc.2" }
|
||||||
rustfs-notify = { path = "crates/notify", version = "1.0.0-rc.1" }
|
rustfs-notify = { path = "crates/notify", version = "1.0.0-rc.2" }
|
||||||
rustfs-io-metrics = { path = "crates/io-metrics", version = "1.0.0-rc.1" }
|
rustfs-io-metrics = { path = "crates/io-metrics", version = "1.0.0-rc.2" }
|
||||||
rustfs-io-core = { path = "crates/io-core", version = "1.0.0-rc.1" }
|
rustfs-io-core = { path = "crates/io-core", version = "1.0.0-rc.2" }
|
||||||
rustfs-object-capacity = { path = "crates/object-capacity", version = "1.0.0-rc.1" }
|
rustfs-object-capacity = { path = "crates/object-capacity", version = "1.0.0-rc.2" }
|
||||||
rustfs-object-data-cache = { path = "crates/object-data-cache", version = "1.0.0-rc.1", default-features = false }
|
rustfs-object-data-cache = { path = "crates/object-data-cache", version = "1.0.0-rc.2", default-features = false }
|
||||||
rustfs-log-analyzer = { path = "crates/log-analyzer", version = "1.0.0-rc.1" }
|
rustfs-log-analyzer = { path = "crates/log-analyzer", version = "1.0.0-rc.2" }
|
||||||
rustfs-obs = { path = "crates/obs", version = "1.0.0-rc.1" }
|
rustfs-obs = { path = "crates/obs", version = "1.0.0-rc.2" }
|
||||||
rustfs-policy = { path = "crates/policy", version = "1.0.0-rc.1" }
|
rustfs-policy = { path = "crates/policy", version = "1.0.0-rc.2" }
|
||||||
rustfs-protos = { path = "crates/protos", version = "1.0.0-rc.1" }
|
rustfs-protos = { path = "crates/protos", version = "1.0.0-rc.2" }
|
||||||
rustfs-protocols = { path = "crates/protocols", version = "1.0.0-rc.1" }
|
rustfs-protocols = { path = "crates/protocols", version = "1.0.0-rc.2" }
|
||||||
rustfs-replication = { path = "crates/replication", version = "1.0.0-rc.1" }
|
rustfs-replication = { path = "crates/replication", version = "1.0.0-rc.2" }
|
||||||
rustfs-rio = { path = "crates/rio", version = "1.0.0-rc.1" }
|
rustfs-rio = { path = "crates/rio", version = "1.0.0-rc.2" }
|
||||||
rustfs-rio-v2 = { path = "crates/rio-v2", version = "1.0.0-rc.1" }
|
rustfs-rio-v2 = { path = "crates/rio-v2", version = "1.0.0-rc.2" }
|
||||||
rustfs-s3-types = { path = "crates/s3-types", version = "1.0.0-rc.1" }
|
rustfs-s3-types = { path = "crates/s3-types", version = "1.0.0-rc.2" }
|
||||||
rustfs-s3-ops = { path = "crates/s3-ops", version = "1.0.0-rc.1" }
|
rustfs-s3-ops = { path = "crates/s3-ops", version = "1.0.0-rc.2" }
|
||||||
rustfs-s3select-api = { path = "crates/s3select-api", version = "1.0.0-rc.1" }
|
rustfs-s3select-api = { path = "crates/s3select-api", version = "1.0.0-rc.2" }
|
||||||
rustfs-s3select-query = { path = "crates/s3select-query", version = "1.0.0-rc.1" }
|
rustfs-s3select-query = { path = "crates/s3select-query", version = "1.0.0-rc.2" }
|
||||||
rustfs-scanner = { path = "crates/scanner", version = "1.0.0-rc.1" }
|
rustfs-scanner = { path = "crates/scanner", version = "1.0.0-rc.2" }
|
||||||
rustfs-security-governance = { path = "crates/security-governance", version = "1.0.0-rc.1" }
|
rustfs-security-governance = { path = "crates/security-governance", version = "1.0.0-rc.2" }
|
||||||
rustfs-extension-schema = { path = "crates/extension-schema", version = "1.0.0-rc.1" }
|
rustfs-extension-schema = { path = "crates/extension-schema", version = "1.0.0-rc.2" }
|
||||||
rustfs-signer = { path = "crates/signer", version = "1.0.0-rc.1" }
|
rustfs-signer = { path = "crates/signer", version = "1.0.0-rc.2" }
|
||||||
rustfs-storage-api = { path = "crates/storage-api", version = "1.0.0-rc.1" }
|
rustfs-storage-api = { path = "crates/storage-api", version = "1.0.0-rc.2" }
|
||||||
rustfs-trusted-proxies = { path = "crates/trusted-proxies", version = "1.0.0-rc.1" }
|
rustfs-trusted-proxies = { path = "crates/trusted-proxies", version = "1.0.0-rc.2" }
|
||||||
rustfs-targets = { path = "crates/targets", version = "1.0.0-rc.1" }
|
rustfs-targets = { path = "crates/targets", version = "1.0.0-rc.2" }
|
||||||
rustfs-test-utils = { path = "crates/test-utils", version = "1.0.0-rc.1" }
|
rustfs-test-utils = { path = "crates/test-utils", version = "1.0.0-rc.2" }
|
||||||
rustfs-tls-runtime = { path = "crates/tls-runtime", version = "1.0.0-rc.1" }
|
rustfs-tls-runtime = { path = "crates/tls-runtime", version = "1.0.0-rc.2" }
|
||||||
rustfs-utils = { path = "crates/utils", version = "1.0.0-rc.1" }
|
rustfs-utils = { path = "crates/utils", version = "1.0.0-rc.2" }
|
||||||
rustfs-zip = { path = "./crates/zip", version = "1.0.0-rc.1" }
|
rustfs-zip = { path = "./crates/zip", version = "1.0.0-rc.2" }
|
||||||
|
|
||||||
# Async Runtime and Networking
|
# Async Runtime and Networking
|
||||||
async-channel = "2.5.0"
|
async-channel = "2.5.0"
|
||||||
@@ -142,10 +142,10 @@ async-recursion = "1.1.1"
|
|||||||
async-trait = "0.1.92"
|
async-trait = "0.1.92"
|
||||||
async-nats = { version = "0.50.0", default-features = false }
|
async-nats = { version = "0.50.0", default-features = false }
|
||||||
axum = "0.8.9"
|
axum = "0.8.9"
|
||||||
futures = "0.3.33"
|
futures = "0.3.34"
|
||||||
futures-core = "0.3.33"
|
futures-core = "0.3.34"
|
||||||
futures-lite = "2.6.1"
|
futures-lite = "2.6.1"
|
||||||
futures-util = "0.3.33"
|
futures-util = "0.3.34"
|
||||||
pollster = "1.0.1"
|
pollster = "1.0.1"
|
||||||
pulsar = { default-features = false, version = "6.8.0" }
|
pulsar = { default-features = false, version = "6.8.0" }
|
||||||
lapin = { default-features = false, version = "4.10.0" }
|
lapin = { default-features = false, version = "4.10.0" }
|
||||||
@@ -154,7 +154,7 @@ hyper-rustls = { default-features = false, version = "0.27.9" }
|
|||||||
hyper-util = { version = "0.1.20" }
|
hyper-util = { version = "0.1.20" }
|
||||||
http = "1.5.0"
|
http = "1.5.0"
|
||||||
http-body = "1.1.0"
|
http-body = "1.1.0"
|
||||||
http-body-util = "0.1.4"
|
http-body-util = "0.1.5"
|
||||||
minlz = "1.2.3"
|
minlz = "1.2.3"
|
||||||
reqwest = "0.13.4"
|
reqwest = "0.13.4"
|
||||||
rustfs-kafka-async = { version = "1.2.0" }
|
rustfs-kafka-async = { version = "1.2.0" }
|
||||||
@@ -171,7 +171,7 @@ tower = { version = "0.5.3" }
|
|||||||
tower-http = { version = "0.7.0" }
|
tower-http = { version = "0.7.0" }
|
||||||
|
|
||||||
# Serialization and Data Formats
|
# Serialization and Data Formats
|
||||||
apache-avro = "0.21.0"
|
apache-avro = { version = "0.22.0", features = ["snappy", "zstandard"] }
|
||||||
bytes = { version = "1.12.1" }
|
bytes = { version = "1.12.1" }
|
||||||
bytesize = "2.7.0"
|
bytesize = "2.7.0"
|
||||||
byteorder = "1.5.0"
|
byteorder = "1.5.0"
|
||||||
@@ -182,6 +182,7 @@ quick-xml = "0.41.0"
|
|||||||
rmp = { version = "0.8.15" }
|
rmp = { version = "0.8.15" }
|
||||||
rmp-serde = { version = "1.3.1" }
|
rmp-serde = { version = "1.3.1" }
|
||||||
serde = { version = "1.0.229" }
|
serde = { version = "1.0.229" }
|
||||||
|
serde_ignored = { version = "0.1" }
|
||||||
serde_json = { version = "1.0.151" }
|
serde_json = { version = "1.0.151" }
|
||||||
serde_urlencoded = "0.7.1"
|
serde_urlencoded = "0.7.1"
|
||||||
|
|
||||||
@@ -230,9 +231,9 @@ aws-credential-types = { version = "1.3.0" }
|
|||||||
aws-sdk-kms = { default-features = false, version = "1.114.0" }
|
aws-sdk-kms = { default-features = false, version = "1.114.0" }
|
||||||
aws-sdk-s3 = { default-features = false, version = "1.141.0" }
|
aws-sdk-s3 = { default-features = false, version = "1.141.0" }
|
||||||
aws-sdk-sts = { default-features = false, version = "1.110.0" }
|
aws-sdk-sts = { default-features = false, version = "1.110.0" }
|
||||||
aws-smithy-http-client = { default-features = false, version = "1.2.0" }
|
aws-smithy-http-client = { default-features = false, version = "1.3.0" }
|
||||||
aws-smithy-runtime-api = { version = "1.14.0" }
|
aws-smithy-runtime-api = { version = "1.14.0" }
|
||||||
aws-smithy-types = { version = "1.6.1" }
|
aws-smithy-types = { version = "1.6.2" }
|
||||||
base64 = "0.23.1"
|
base64 = "0.23.1"
|
||||||
base64-simd = "0.8.0"
|
base64-simd = "0.8.0"
|
||||||
brotli = "8.0.4"
|
brotli = "8.0.4"
|
||||||
@@ -268,7 +269,7 @@ lz4 = "1.28.1"
|
|||||||
matchit = "0.9.2"
|
matchit = "0.9.2"
|
||||||
md-5 = "0.11.0"
|
md-5 = "0.11.0"
|
||||||
mime_guess = "2.0.5"
|
mime_guess = "2.0.5"
|
||||||
moka = { version = "0.12.15" }
|
moka = { version = "0.12.16" }
|
||||||
netif = "0.1.6"
|
netif = "0.1.6"
|
||||||
num_cpus = { version = "1.17.0" }
|
num_cpus = { version = "1.17.0" }
|
||||||
nvml-wrapper = "0.12.1"
|
nvml-wrapper = "0.12.1"
|
||||||
@@ -294,7 +295,7 @@ serial_test = "4.0.1"
|
|||||||
shadow-rs = { default-features = false, version = "2.0.0" }
|
shadow-rs = { default-features = false, version = "2.0.0" }
|
||||||
siphasher = "1.0.3"
|
siphasher = "1.0.3"
|
||||||
smallvec = { version = "1.15.2" }
|
smallvec = { version = "1.15.2" }
|
||||||
smartstring = "1.0.1"
|
compact_str = "0.10.0"
|
||||||
snap = "1.1.2"
|
snap = "1.1.2"
|
||||||
starshard = { version = "2.2.2" }
|
starshard = { version = "2.2.2" }
|
||||||
strum = { version = "0.28.0" }
|
strum = { version = "0.28.0" }
|
||||||
@@ -339,17 +340,17 @@ pyroscope = { version = "2.1.1" }
|
|||||||
libunftp = { version = "0.23.0" }
|
libunftp = { version = "0.23.0" }
|
||||||
unftp-core = "0.1.0"
|
unftp-core = "0.1.0"
|
||||||
suppaftp = { version = "10.0.1" }
|
suppaftp = { version = "10.0.1" }
|
||||||
rcgen = { version = "0.14.8", default-features = false, features = ["aws_lc_rs", "crypto", "pem"] }
|
rcgen = { version = "0.14.9", default-features = false, features = ["aws_lc_rs", "crypto", "pem"] }
|
||||||
russh = { version = "0.62.5" }
|
russh = { version = "0.62.6" }
|
||||||
russh-sftp = "2.4.0"
|
russh-sftp = "2.4.0"
|
||||||
|
|
||||||
# WebDAV
|
# WebDAV
|
||||||
dav-server = "0.11.0"
|
dav-server = "0.11.0"
|
||||||
|
|
||||||
# Performance Analysis and Memory Profiling
|
# Performance Analysis and Memory Profiling
|
||||||
mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "ce6338661179c8be22e516b00af7483f151485a7" }
|
mimalloc = { version = "0.1.52", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11" }
|
||||||
libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "ce6338661179c8be22e516b00af7483f151485a7", features = ["extended"] }
|
libmimalloc-sys = { version = "0.1.49", git = "https://github.com/xonatius/mimalloc_rust.git", rev = "6d4c41bb10c6d9da1d1b6f07b38c4cc051667f11", features = ["extended"] }
|
||||||
hotpath = { version = "0.23.1", default-features = false }
|
hotpath = { version = "0.23.2", default-features = false }
|
||||||
# Snapshot testing for output format regression detection
|
# Snapshot testing for output format regression detection
|
||||||
insta = { version = "1.48" }
|
insta = { version = "1.48" }
|
||||||
|
|
||||||
|
|||||||
@@ -116,7 +116,7 @@ chown -R 10001:10001 data logs
|
|||||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
||||||
|
|
||||||
# Using specific version
|
# Using specific version
|
||||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.1
|
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.2
|
||||||
```
|
```
|
||||||
|
|
||||||
If you use [podman](https://github.com/containers/podman) instead of docker, you can install the RustFS with the below command
|
If you use [podman](https://github.com/containers/podman) instead of docker, you can install the RustFS with the below command
|
||||||
|
|||||||
+1
-1
@@ -113,7 +113,7 @@ chown -R 10001:10001 data logs
|
|||||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
||||||
|
|
||||||
# 使用指定版本运行
|
# 使用指定版本运行
|
||||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.1
|
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.2
|
||||||
```
|
```
|
||||||
|
|
||||||
如果您通过绑定挂载启用 TLS 证书目录,也请用同样方式准备该目录:
|
如果您通过绑定挂载启用 TLS 证书目录,也请用同样方式准备该目录:
|
||||||
|
|||||||
@@ -21,6 +21,13 @@ use crate::{
|
|||||||
Xxhash3, Xxhash64, Xxhash128,
|
Xxhash3, Xxhash64, Xxhash128,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// DELIBERATE DUPLICATION of the x-amz-checksum-* names that also exist as
|
||||||
|
// AMZ_CHECKSUM_* in rustfs-utils' headers module (crates/utils/src/http/
|
||||||
|
// headers.rs): this crate is a zero-internal-dependency leaf, so it cannot
|
||||||
|
// import them, and it additionally owns the RustFS extension names
|
||||||
|
// (sha512/xxhash*) that utils does not carry. Values are pinned by the S3
|
||||||
|
// wire protocol; do not merge without a maintainer decision on the leaf
|
||||||
|
// boundary (backlog#1833).
|
||||||
pub const CRC_32_HEADER_NAME: &str = "x-amz-checksum-crc32";
|
pub const CRC_32_HEADER_NAME: &str = "x-amz-checksum-crc32";
|
||||||
pub const CRC_32_C_HEADER_NAME: &str = "x-amz-checksum-crc32c";
|
pub const CRC_32_C_HEADER_NAME: &str = "x-amz-checksum-crc32c";
|
||||||
pub const SHA_1_HEADER_NAME: &str = "x-amz-checksum-sha1";
|
pub const SHA_1_HEADER_NAME: &str = "x-amz-checksum-sha1";
|
||||||
|
|||||||
@@ -41,6 +41,14 @@ pub const XXHASH_64_NAME: &str = "xxhash64";
|
|||||||
pub const XXHASH_128_NAME: &str = "xxhash128";
|
pub const XXHASH_128_NAME: &str = "xxhash128";
|
||||||
pub const MD5_NAME: &str = "md5";
|
pub const MD5_NAME: &str = "md5";
|
||||||
|
|
||||||
|
/// One of three deliberately separate checksum registries (backlog#1833):
|
||||||
|
/// this enum owns the **streaming-hash algorithm registry**, including the
|
||||||
|
/// RustFS extensions (sha512, xxhash3/64/128). The on-disk xl.meta bitset
|
||||||
|
/// lives in `rustfs_rio::ChecksumType` (crates/rio/src/checksum.rs, varint
|
||||||
|
/// bits are append-only), and the MinIO-port client keeps its own
|
||||||
|
/// `ChecksumMode` (crates/ecstore/src/client/checksum.rs). When adding an
|
||||||
|
/// algorithm, extend all three (or record why not) — they do not derive from
|
||||||
|
/// each other.
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||||||
#[non_exhaustive]
|
#[non_exhaustive]
|
||||||
pub enum ChecksumAlgorithm {
|
pub enum ChecksumAlgorithm {
|
||||||
|
|||||||
@@ -1,87 +0,0 @@
|
|||||||
// Copyright 2024 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use crate::last_minute::{self};
|
|
||||||
use std::collections::HashMap;
|
|
||||||
|
|
||||||
pub struct ReplicationLatency {
|
|
||||||
// Delays for single and multipart PUT requests
|
|
||||||
upload_histogram: last_minute::LastMinuteHistogram,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl ReplicationLatency {
|
|
||||||
// Merge two ReplicationLatency
|
|
||||||
pub fn merge(&mut self, other: &mut ReplicationLatency) -> &ReplicationLatency {
|
|
||||||
self.upload_histogram.merge(&other.upload_histogram);
|
|
||||||
self
|
|
||||||
}
|
|
||||||
|
|
||||||
// Get upload delay (categorized by object size interval)
|
|
||||||
pub fn get_upload_latency(&mut self) -> HashMap<String, u64> {
|
|
||||||
let mut ret = HashMap::new();
|
|
||||||
let avg = self.upload_histogram.get_avg_data();
|
|
||||||
for (i, v) in avg.iter().enumerate() {
|
|
||||||
let avg_duration = v.avg();
|
|
||||||
ret.insert(self.size_tag_to_string(i), avg_duration.as_millis() as u64);
|
|
||||||
}
|
|
||||||
ret
|
|
||||||
}
|
|
||||||
pub fn update(&mut self, size: i64, during: std::time::Duration) {
|
|
||||||
self.upload_histogram.add(size, during);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Simulate the conversion from size tag to string
|
|
||||||
fn size_tag_to_string(&self, tag: usize) -> String {
|
|
||||||
match tag {
|
|
||||||
0 => String::from("Size < 1 KiB"),
|
|
||||||
1 => String::from("Size < 1 MiB"),
|
|
||||||
2 => String::from("Size < 10 MiB"),
|
|
||||||
3 => String::from("Size < 100 MiB"),
|
|
||||||
4 => String::from("Size < 1 GiB"),
|
|
||||||
_ => String::from("Size > 1 GiB"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// #[derive(Debug, Clone, Default)]
|
|
||||||
// pub struct ReplicationLastMinute {
|
|
||||||
// pub last_minute: LastMinuteLatency,
|
|
||||||
// }
|
|
||||||
|
|
||||||
// impl ReplicationLastMinute {
|
|
||||||
// pub fn merge(&mut self, other: ReplicationLastMinute) -> ReplicationLastMinute {
|
|
||||||
// let mut nl = ReplicationLastMinute::default();
|
|
||||||
// nl.last_minute = self.last_minute.merge(&mut other.last_minute);
|
|
||||||
// nl
|
|
||||||
// }
|
|
||||||
|
|
||||||
// pub fn add_size(&mut self, n: i64) {
|
|
||||||
// let t = SystemTime::now()
|
|
||||||
// .duration_since(UNIX_EPOCH)
|
|
||||||
// .expect("Time went backwards")
|
|
||||||
// .as_secs();
|
|
||||||
// self.last_minute.add_all(t - 1, &AccElem { total: t - 1, size: n as u64, n: 1 });
|
|
||||||
// }
|
|
||||||
|
|
||||||
// pub fn get_total(&self) -> AccElem {
|
|
||||||
// self.last_minute.get_total()
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
|
|
||||||
// impl fmt::Display for ReplicationLastMinute {
|
|
||||||
// fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
|
|
||||||
// let t = self.last_minute.get_total();
|
|
||||||
// write!(f, "ReplicationLastMinute sz= {}, n= {}, dur= {}", t.size, t.n, t.total)
|
|
||||||
// }
|
|
||||||
// }
|
|
||||||
@@ -572,44 +572,3 @@ mod tests {
|
|||||||
assert_eq!(total.n, 6);
|
assert_eq!(total.n, 6);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
const SIZE_LAST_ELEM_MARKER: usize = 10; // Assumed marker size is 10, modify according to actual situation
|
|
||||||
|
|
||||||
#[allow(dead_code)]
|
|
||||||
#[derive(Debug, Default)]
|
|
||||||
pub struct LastMinuteHistogram {
|
|
||||||
histogram: Vec<LastMinuteLatency>,
|
|
||||||
size: u32,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl LastMinuteHistogram {
|
|
||||||
pub fn merge(&mut self, other: &LastMinuteHistogram) {
|
|
||||||
for i in 0..self.histogram.len() {
|
|
||||||
self.histogram[i].merge(&other.histogram[i]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn add(&mut self, size: i64, t: Duration) {
|
|
||||||
let index = size_to_tag(size);
|
|
||||||
self.histogram[index].add(&t);
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn get_avg_data(&mut self) -> [AccElem; SIZE_LAST_ELEM_MARKER] {
|
|
||||||
let mut res = [AccElem::default(); SIZE_LAST_ELEM_MARKER];
|
|
||||||
for (i, elem) in self.histogram.iter_mut().enumerate() {
|
|
||||||
res[i] = elem.get_total();
|
|
||||||
}
|
|
||||||
res
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn size_to_tag(size: i64) -> usize {
|
|
||||||
match size {
|
|
||||||
_ if size < 1024 => 0, // sizeLessThan1KiB
|
|
||||||
_ if size < 1024 * 1024 => 1, // sizeLessThan1MiB
|
|
||||||
_ if size < 10 * 1024 * 1024 => 2, // sizeLessThan10MiB
|
|
||||||
_ if size < 100 * 1024 * 1024 => 3, // sizeLessThan100MiB
|
|
||||||
_ if size < 1024 * 1024 * 1024 => 4, // sizeLessThan1GiB
|
|
||||||
_ => 5, // sizeGreaterThan1GiB
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -12,13 +12,13 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
pub mod bucket_stats;
|
|
||||||
// pub mod error;
|
// pub mod error;
|
||||||
pub mod globals;
|
pub mod globals;
|
||||||
pub mod heal_channel;
|
pub mod heal_channel;
|
||||||
pub mod last_minute;
|
pub mod last_minute;
|
||||||
pub mod metrics;
|
pub mod metrics;
|
||||||
mod readiness;
|
mod readiness;
|
||||||
|
pub mod table_catalog;
|
||||||
|
|
||||||
pub use globals::*;
|
pub use globals::*;
|
||||||
pub use readiness::{GlobalReadiness, SystemStage};
|
pub use readiness::{GlobalReadiness, SystemStage};
|
||||||
|
|||||||
@@ -915,11 +915,13 @@ const SCAN_CYCLE_RESULT_SUCCESS: u8 = 1;
|
|||||||
const SCAN_CYCLE_RESULT_ERROR: u8 = 2;
|
const SCAN_CYCLE_RESULT_ERROR: u8 = 2;
|
||||||
const SCAN_CYCLE_RESULT_PARTIAL: u8 = 3;
|
const SCAN_CYCLE_RESULT_PARTIAL: u8 = 3;
|
||||||
const SCAN_CYCLE_RESULT_SUPERSEDED: u8 = 4;
|
const SCAN_CYCLE_RESULT_SUPERSEDED: u8 = 4;
|
||||||
|
const SCAN_CYCLE_RESULT_DEFERRED: u8 = 5;
|
||||||
const SCAN_CYCLE_RESULT_UNKNOWN_LABEL: &str = "unknown";
|
const SCAN_CYCLE_RESULT_UNKNOWN_LABEL: &str = "unknown";
|
||||||
const SCAN_CYCLE_RESULT_SUCCESS_LABEL: &str = "success";
|
const SCAN_CYCLE_RESULT_SUCCESS_LABEL: &str = "success";
|
||||||
const SCAN_CYCLE_RESULT_ERROR_LABEL: &str = "error";
|
const SCAN_CYCLE_RESULT_ERROR_LABEL: &str = "error";
|
||||||
const SCAN_CYCLE_RESULT_PARTIAL_LABEL: &str = "partial";
|
const SCAN_CYCLE_RESULT_PARTIAL_LABEL: &str = "partial";
|
||||||
const SCAN_CYCLE_RESULT_SUPERSEDED_LABEL: &str = "superseded";
|
const SCAN_CYCLE_RESULT_SUPERSEDED_LABEL: &str = "superseded";
|
||||||
|
const SCAN_CYCLE_RESULT_DEFERRED_LABEL: &str = "deferred";
|
||||||
|
|
||||||
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
|
||||||
pub enum ScanCyclePartialReason {
|
pub enum ScanCyclePartialReason {
|
||||||
@@ -1424,6 +1426,7 @@ fn scan_cycle_result_label(result: u8) -> &'static str {
|
|||||||
SCAN_CYCLE_RESULT_ERROR => SCAN_CYCLE_RESULT_ERROR_LABEL,
|
SCAN_CYCLE_RESULT_ERROR => SCAN_CYCLE_RESULT_ERROR_LABEL,
|
||||||
SCAN_CYCLE_RESULT_PARTIAL => SCAN_CYCLE_RESULT_PARTIAL_LABEL,
|
SCAN_CYCLE_RESULT_PARTIAL => SCAN_CYCLE_RESULT_PARTIAL_LABEL,
|
||||||
SCAN_CYCLE_RESULT_SUPERSEDED => SCAN_CYCLE_RESULT_SUPERSEDED_LABEL,
|
SCAN_CYCLE_RESULT_SUPERSEDED => SCAN_CYCLE_RESULT_SUPERSEDED_LABEL,
|
||||||
|
SCAN_CYCLE_RESULT_DEFERRED => SCAN_CYCLE_RESULT_DEFERRED_LABEL,
|
||||||
_ => SCAN_CYCLE_RESULT_UNKNOWN_LABEL,
|
_ => SCAN_CYCLE_RESULT_UNKNOWN_LABEL,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1752,6 +1755,11 @@ pub fn emit_scan_cycle_superseded(duration: Duration) {
|
|||||||
metrics::counter!(OTEL_SCANNER_CYCLES, "result" => SCAN_CYCLE_RESULT_SUPERSEDED_LABEL).increment(1);
|
metrics::counter!(OTEL_SCANNER_CYCLES, "result" => SCAN_CYCLE_RESULT_SUPERSEDED_LABEL).increment(1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn emit_scan_cycle_deferred(duration: Duration) {
|
||||||
|
global_metrics().record_scan_cycle_deferred(duration);
|
||||||
|
metrics::counter!(OTEL_SCANNER_CYCLES, "result" => SCAN_CYCLE_RESULT_DEFERRED_LABEL).increment(1);
|
||||||
|
}
|
||||||
|
|
||||||
pub fn emit_scan_bucket_drive_complete(success: bool, bucket: &str, disk: &str, duration: Duration) {
|
pub fn emit_scan_bucket_drive_complete(success: bool, bucket: &str, disk: &str, duration: Duration) {
|
||||||
let result = if success { "success" } else { "error" };
|
let result = if success { "success" } else { "error" };
|
||||||
global_metrics().record_scanner_bucket_drive_result(bucket, disk, result);
|
global_metrics().record_scanner_bucket_drive_result(bucket, disk, result);
|
||||||
@@ -2549,6 +2557,17 @@ impl Metrics {
|
|||||||
.store(duration_millis_saturated(duration), Ordering::Relaxed);
|
.store(duration_millis_saturated(duration), Ordering::Relaxed);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn record_scan_cycle_deferred(&self, duration: Duration) {
|
||||||
|
self.record_scanner_cycle_end_time();
|
||||||
|
self.last_scan_cycle_result
|
||||||
|
.store(SCAN_CYCLE_RESULT_DEFERRED, Ordering::Relaxed);
|
||||||
|
self.last_scan_cycle_partial_reason
|
||||||
|
.store(ScanCyclePartialReason::Unknown as u8, Ordering::Relaxed);
|
||||||
|
self.last_scan_cycle_partial_source.store(0, Ordering::Relaxed);
|
||||||
|
self.last_scan_cycle_duration_millis
|
||||||
|
.store(duration_millis_saturated(duration), Ordering::Relaxed);
|
||||||
|
}
|
||||||
|
|
||||||
pub fn record_scan_cycle_partial(&self, duration: Duration, reason: ScanCyclePartialReason) {
|
pub fn record_scan_cycle_partial(&self, duration: Duration, reason: ScanCyclePartialReason) {
|
||||||
self.record_scan_cycle_partial_with_source(duration, reason, None);
|
self.record_scan_cycle_partial_with_source(duration, reason, None);
|
||||||
}
|
}
|
||||||
@@ -4264,6 +4283,21 @@ mod tests {
|
|||||||
assert_eq!(report.partial_cycles, 0);
|
assert_eq!(report.partial_cycles, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn report_tracks_deferred_cycle_without_failed_increment() {
|
||||||
|
let metrics = Metrics::new();
|
||||||
|
metrics.record_scan_cycle_deferred(Duration::from_millis(250));
|
||||||
|
|
||||||
|
let report = metrics.report().await;
|
||||||
|
|
||||||
|
assert_eq!(report.last_cycle_result, SCAN_CYCLE_RESULT_DEFERRED_LABEL);
|
||||||
|
assert_eq!(report.last_cycle_result_code, u64::from(SCAN_CYCLE_RESULT_DEFERRED));
|
||||||
|
assert_eq!(report.last_cycle_duration_seconds, 0.25);
|
||||||
|
assert_eq!(report.failed_cycles, 0);
|
||||||
|
assert_eq!(report.superseded_cycles, 0);
|
||||||
|
assert_eq!(report.partial_cycles, 0);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn report_tracks_successful_scan_cycle_without_failed_increment() {
|
async fn report_tracks_successful_scan_cycle_without_failed_increment() {
|
||||||
let metrics = Metrics::new();
|
let metrics = Metrics::new();
|
||||||
|
|||||||
@@ -1,4 +1,3 @@
|
|||||||
#![allow(clippy::all)]
|
|
||||||
// Copyright 2024 RustFS Team
|
// Copyright 2024 RustFS Team
|
||||||
//
|
//
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
@@ -13,13 +12,6 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
pub struct TargetID {
|
/// Cross-crate lock identity used to fence table-bucket publication against
|
||||||
id: String,
|
/// object mutations that bypass the S3 request authorization layer.
|
||||||
name: String,
|
pub const TABLE_BUCKET_PUBLICATION_LOCK_PATH: &str = ".rustfs-table/warehouses/default/publication.lock";
|
||||||
}
|
|
||||||
|
|
||||||
impl TargetID {
|
|
||||||
fn to_string(&self) -> String {
|
|
||||||
format!("{}:{}", self.id, self.name)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -353,6 +353,11 @@ pub const DEFAULT_OBS_TRACES_EXPORT_ENABLED: bool = true;
|
|||||||
/// Environment variable: RUSTFS_OBS_METRICS_EXPORT_ENABLED
|
/// Environment variable: RUSTFS_OBS_METRICS_EXPORT_ENABLED
|
||||||
pub const DEFAULT_OBS_METRICS_EXPORT_ENABLED: bool = true;
|
pub const DEFAULT_OBS_METRICS_EXPORT_ENABLED: bool = true;
|
||||||
|
|
||||||
|
/// Default detailed PUT stage metrics enabled
|
||||||
|
/// Default value: false
|
||||||
|
/// Environment variable: RUSTFS_OBS_PUT_STAGE_METRICS_ENABLED
|
||||||
|
pub const DEFAULT_OBS_PUT_STAGE_METRICS_ENABLED: bool = false;
|
||||||
|
|
||||||
/// Default logs export enabled
|
/// Default logs export enabled
|
||||||
/// It is used to enable or disable exporting logs
|
/// It is used to enable or disable exporting logs
|
||||||
/// Default value: true
|
/// Default value: true
|
||||||
|
|||||||
@@ -137,6 +137,37 @@ pub const DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED: bool = false;
|
|||||||
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE);
|
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE);
|
||||||
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED);
|
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED);
|
||||||
|
|
||||||
|
/// Request the object-transaction fencing contract used by storage-owned
|
||||||
|
/// cleanup receipts and lock-window optimizations.
|
||||||
|
///
|
||||||
|
/// This is fail-closed: enabling the writer without a live fleet proof rejects
|
||||||
|
/// the commit rather than silently using a legacy-safe path.
|
||||||
|
pub const ENV_OBJECT_TRANSACTION_FENCING_WRITE: &str = "RUSTFS_OBJECT_TRANSACTION_FENCING_WRITE";
|
||||||
|
pub const DEFAULT_OBJECT_TRANSACTION_FENCING_WRITE: bool = false;
|
||||||
|
|
||||||
|
/// Operator-attested confirmation that every serving node understands the
|
||||||
|
/// object transaction fencing contract.
|
||||||
|
pub const ENV_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED: &str = "RUSTFS_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED";
|
||||||
|
pub const DEFAULT_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED: bool = false;
|
||||||
|
|
||||||
|
const _: () = assert!(!DEFAULT_OBJECT_TRANSACTION_FENCING_WRITE);
|
||||||
|
const _: () = assert!(!DEFAULT_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED);
|
||||||
|
|
||||||
|
/// Request preserving legacy per-part checksum metadata during data movement.
|
||||||
|
///
|
||||||
|
/// This remains ineffective until
|
||||||
|
/// [`ENV_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED`] is also enabled.
|
||||||
|
pub const ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE: &str = "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE";
|
||||||
|
pub const DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_WRITE: bool = false;
|
||||||
|
|
||||||
|
/// Operator-attested confirmation that every serving node understands the
|
||||||
|
/// data-movement per-part checksum sidecar.
|
||||||
|
pub const ENV_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED: &str = "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED";
|
||||||
|
pub const DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED: bool = false;
|
||||||
|
|
||||||
|
const _: () = assert!(!DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_WRITE);
|
||||||
|
const _: () = assert!(!DEFAULT_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED);
|
||||||
|
|
||||||
// =============================================================================
|
// =============================================================================
|
||||||
// Concurrent Request Fix - Timeout and Backpressure Configuration
|
// Concurrent Request Fix - Timeout and Backpressure Configuration
|
||||||
// =============================================================================
|
// =============================================================================
|
||||||
@@ -649,4 +680,22 @@ mod remote_version_state_tests {
|
|||||||
"RUSTFS_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED"
|
"RUSTFS_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn data_movement_part_checksum_gate_uses_stable_environment_names() {
|
||||||
|
assert_eq!(super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE, "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE");
|
||||||
|
assert_eq!(
|
||||||
|
super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED,
|
||||||
|
"RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_FLEET_CONFIRMED"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn object_transaction_fencing_gate_uses_stable_environment_names() {
|
||||||
|
assert_eq!(super::ENV_OBJECT_TRANSACTION_FENCING_WRITE, "RUSTFS_OBJECT_TRANSACTION_FENCING_WRITE");
|
||||||
|
assert_eq!(
|
||||||
|
super::ENV_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED,
|
||||||
|
"RUSTFS_OBJECT_TRANSACTION_FENCING_FLEET_CONFIRMED"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -81,6 +81,9 @@ pub const ENV_TEST_IAM_FAIL_INIT_ATTEMPTS: &str = "RUSTFS_TEST_IAM_FAIL_INIT_ATT
|
|||||||
pub const ENV_TEST_IAM_RETRY_INTERVAL_MS: &str = "RUSTFS_TEST_IAM_RETRY_INTERVAL_MS";
|
pub const ENV_TEST_IAM_RETRY_INTERVAL_MS: &str = "RUSTFS_TEST_IAM_RETRY_INTERVAL_MS";
|
||||||
/// Runtime env var controlling the transition worker count.
|
/// Runtime env var controlling the transition worker count.
|
||||||
pub const ENV_TRANSITION_WORKERS: &str = "RUSTFS_MAX_TRANSITION_WORKERS";
|
pub const ENV_TRANSITION_WORKERS: &str = "RUSTFS_MAX_TRANSITION_WORKERS";
|
||||||
|
/// Runtime env var controlling the ILM expiry worker count. A set, parsable,
|
||||||
|
/// non-zero value wins; anything else falls back to `min(cpus, 16)`.
|
||||||
|
pub const ENV_MAX_EXPIRY_WORKERS: &str = "RUSTFS_MAX_EXPIRY_WORKERS";
|
||||||
/// Runtime env var controlling the absolute maximum transition workers.
|
/// Runtime env var controlling the absolute maximum transition workers.
|
||||||
pub const ENV_TRANSITION_WORKERS_ABSOLUTE_MAX: &str = "RUSTFS_ABSOLUTE_MAX_WORKERS";
|
pub const ENV_TRANSITION_WORKERS_ABSOLUTE_MAX: &str = "RUSTFS_ABSOLUTE_MAX_WORKERS";
|
||||||
/// Runtime env var controlling the transition queue capacity.
|
/// Runtime env var controlling the transition queue capacity.
|
||||||
|
|||||||
@@ -44,6 +44,10 @@ pub const ENV_OBS_METRICS_EXPORT_ENABLED: &str = "RUSTFS_OBS_METRICS_EXPORT_ENAB
|
|||||||
pub const ENV_OBS_LOGS_EXPORT_ENABLED: &str = "RUSTFS_OBS_LOGS_EXPORT_ENABLED";
|
pub const ENV_OBS_LOGS_EXPORT_ENABLED: &str = "RUSTFS_OBS_LOGS_EXPORT_ENABLED";
|
||||||
pub const ENV_OBS_PROFILING_EXPORT_ENABLED: &str = "RUSTFS_OBS_PROFILING_EXPORT_ENABLED";
|
pub const ENV_OBS_PROFILING_EXPORT_ENABLED: &str = "RUSTFS_OBS_PROFILING_EXPORT_ENABLED";
|
||||||
|
|
||||||
|
/// Enables detailed per-stage PUT metrics. Disabled by default because each
|
||||||
|
/// PUT records multiple timers and histograms when attribution is active.
|
||||||
|
pub const ENV_OBS_PUT_STAGE_METRICS_ENABLED: &str = "RUSTFS_OBS_PUT_STAGE_METRICS_ENABLED";
|
||||||
|
|
||||||
pub const ENV_OBS_LOGGER_LEVEL: &str = "RUSTFS_OBS_LOGGER_LEVEL";
|
pub const ENV_OBS_LOGGER_LEVEL: &str = "RUSTFS_OBS_LOGGER_LEVEL";
|
||||||
pub const ENV_OBS_LOG_STDOUT_ENABLED: &str = "RUSTFS_OBS_LOG_STDOUT_ENABLED";
|
pub const ENV_OBS_LOG_STDOUT_ENABLED: &str = "RUSTFS_OBS_LOG_STDOUT_ENABLED";
|
||||||
pub const ENV_OBS_LOG_DIRECTORY: &str = "RUSTFS_OBS_LOG_DIRECTORY";
|
pub const ENV_OBS_LOG_DIRECTORY: &str = "RUSTFS_OBS_LOG_DIRECTORY";
|
||||||
@@ -141,6 +145,7 @@ mod tests {
|
|||||||
assert_eq!(ENV_OBS_METRICS_EXPORT_ENABLED, "RUSTFS_OBS_METRICS_EXPORT_ENABLED");
|
assert_eq!(ENV_OBS_METRICS_EXPORT_ENABLED, "RUSTFS_OBS_METRICS_EXPORT_ENABLED");
|
||||||
assert_eq!(ENV_OBS_LOGS_EXPORT_ENABLED, "RUSTFS_OBS_LOGS_EXPORT_ENABLED");
|
assert_eq!(ENV_OBS_LOGS_EXPORT_ENABLED, "RUSTFS_OBS_LOGS_EXPORT_ENABLED");
|
||||||
assert_eq!(ENV_OBS_PROFILING_EXPORT_ENABLED, "RUSTFS_OBS_PROFILING_EXPORT_ENABLED");
|
assert_eq!(ENV_OBS_PROFILING_EXPORT_ENABLED, "RUSTFS_OBS_PROFILING_EXPORT_ENABLED");
|
||||||
|
assert_eq!(ENV_OBS_PUT_STAGE_METRICS_ENABLED, "RUSTFS_OBS_PUT_STAGE_METRICS_ENABLED");
|
||||||
// Test log cleanup related env keys
|
// Test log cleanup related env keys
|
||||||
assert_eq!(ENV_OBS_LOG_MAX_TOTAL_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_TOTAL_SIZE_BYTES");
|
assert_eq!(ENV_OBS_LOG_MAX_TOTAL_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_TOTAL_SIZE_BYTES");
|
||||||
assert_eq!(ENV_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES");
|
assert_eq!(ENV_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES, "RUSTFS_OBS_LOG_MAX_SINGLE_FILE_SIZE_BYTES");
|
||||||
|
|||||||
@@ -37,7 +37,6 @@ hotpath-cpu = ["hotpath", "hotpath/hotpath-cpu", "rustfs-filemeta/hotpath-cpu"]
|
|||||||
hotpath.workspace = true
|
hotpath.workspace = true
|
||||||
serde = { workspace = true, features = ["derive"] }
|
serde = { workspace = true, features = ["derive"] }
|
||||||
rmp-serde = { workspace = true }
|
rmp-serde = { workspace = true }
|
||||||
async-trait = { workspace = true }
|
|
||||||
rustfs-filemeta = { workspace = true }
|
rustfs-filemeta = { workspace = true }
|
||||||
|
|
||||||
[lib]
|
[lib]
|
||||||
|
|||||||
@@ -846,8 +846,15 @@ impl DataUsageEntry {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Data usage cache info
|
/// Read-only projection of the scanner's `.usage-cache.bin` info block.
|
||||||
#[derive(Clone, Debug, Default, Serialize, Deserialize)]
|
///
|
||||||
|
/// The canonical wire format is written by the hand-written map-encoded
|
||||||
|
/// `Serialize` on the scanner-side `DataUsageCacheInfo`
|
||||||
|
/// (`crates/scanner/src/data_usage_define.rs`), which carries 16 fields.
|
||||||
|
/// This type decodes only the shared subset and is deliberately not
|
||||||
|
/// `Serialize`: a derived (array) encoding of this 6-field subset would
|
||||||
|
/// corrupt the cache for scanner readers, so no write path may exist here.
|
||||||
|
#[derive(Clone, Debug, Default, Deserialize)]
|
||||||
pub struct DataUsageCacheInfo {
|
pub struct DataUsageCacheInfo {
|
||||||
pub name: String,
|
pub name: String,
|
||||||
pub next_cycle: u64,
|
pub next_cycle: u64,
|
||||||
@@ -863,8 +870,12 @@ pub struct DataUsageCacheInfo {
|
|||||||
pub snapshot_complete: bool,
|
pub snapshot_complete: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Data usage cache
|
/// Read-only projection of a scanner-written `.usage-cache.bin` file.
|
||||||
#[derive(Clone, Debug, Default, Serialize, Deserialize)]
|
///
|
||||||
|
/// The scanner-side `DataUsageCache` (`crates/scanner/src/data_usage_define.rs`)
|
||||||
|
/// owns the persisted format; this type only decodes it (see
|
||||||
|
/// [`DataUsageCacheInfo`]) and must never grow a serialization path.
|
||||||
|
#[derive(Clone, Debug, Default, Deserialize)]
|
||||||
pub struct DataUsageCache {
|
pub struct DataUsageCache {
|
||||||
pub info: DataUsageCacheInfo,
|
pub info: DataUsageCacheInfo,
|
||||||
pub cache: HashMap<String, DataUsageEntry>,
|
pub cache: HashMap<String, DataUsageEntry>,
|
||||||
@@ -1186,31 +1197,10 @@ impl DataUsageCache {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn marshal_msg(&self) -> Result<Vec<u8>, Box<dyn std::error::Error + Send + Sync>> {
|
|
||||||
let mut buf = Vec::new();
|
|
||||||
self.serialize(&mut rmp_serde::Serializer::new(&mut buf))?;
|
|
||||||
Ok(buf)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn unmarshal(buf: &[u8]) -> Result<Self, Box<dyn std::error::Error + Send + Sync>> {
|
pub fn unmarshal(buf: &[u8]) -> Result<Self, Box<dyn std::error::Error + Send + Sync>> {
|
||||||
let t: Self = rmp_serde::from_slice(buf)?;
|
let t: Self = rmp_serde::from_slice(buf)?;
|
||||||
Ok(t)
|
Ok(t)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Note: load and save methods are storage-specific and should be implemented
|
|
||||||
// in the ecstore crate where storage access is available
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Trait for storage-specific operations on DataUsageCache
|
|
||||||
#[async_trait::async_trait]
|
|
||||||
pub trait DataUsageCacheStorage {
|
|
||||||
/// Load data usage cache from backend storage
|
|
||||||
async fn load(store: &dyn std::any::Any, name: &str) -> Result<Self, Box<dyn std::error::Error + Send + Sync>>
|
|
||||||
where
|
|
||||||
Self: Sized;
|
|
||||||
|
|
||||||
/// Save data usage cache to backend storage
|
|
||||||
async fn save(&self, name: &str) -> Result<(), Box<dyn std::error::Error + Send + Sync>>;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Helper structs and functions for cache operations
|
// Helper structs and functions for cache operations
|
||||||
@@ -1832,6 +1822,82 @@ mod tests {
|
|||||||
assert!(decoded.all_tier_stats.is_none());
|
assert!(decoded.all_tier_stats.is_none());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Scanner-written `.usage-cache.bin` bytes: a 2-element array of the
|
||||||
|
/// canonical 16-field map-encoded info block and one map-encoded entry.
|
||||||
|
/// Captured from the canonical writer's `marshal_msg` — see
|
||||||
|
/// `usage_cache_wire_format_is_pinned` in
|
||||||
|
/// `crates/scanner/src/data_usage_define.rs`, which pins these exact
|
||||||
|
/// bytes and documents regeneration. Hardcoded here because a
|
||||||
|
/// dev-dependency on rustfs-scanner would pull the whole ecstore tree
|
||||||
|
/// into this crate's test build, and a fixture generated at test runtime
|
||||||
|
/// could not detect writer drift anyway.
|
||||||
|
const SCANNER_USAGE_CACHE_WIRE_FIXTURE: &[u8] = &[
|
||||||
|
0x92, 0xde, 0x00, 0x10, 0xa4, 0x6e, 0x61, 0x6d, 0x65, 0xab, 0x77, 0x69, 0x72, 0x65, 0x2d, 0x62, 0x75, 0x63, 0x6b, 0x65,
|
||||||
|
0x74, 0xaa, 0x6e, 0x65, 0x78, 0x74, 0x5f, 0x63, 0x79, 0x63, 0x6c, 0x65, 0x07, 0xac, 0x6c, 0x65, 0x61, 0x64, 0x65, 0x72,
|
||||||
|
0x5f, 0x65, 0x70, 0x6f, 0x63, 0x68, 0x09, 0xab, 0x6c, 0x61, 0x73, 0x74, 0x5f, 0x75, 0x70, 0x64, 0x61, 0x74, 0x65, 0x92,
|
||||||
|
0xce, 0x65, 0x53, 0xf1, 0x00, 0x00, 0xac, 0x73, 0x6b, 0x69, 0x70, 0x5f, 0x68, 0x65, 0x61, 0x6c, 0x69, 0x6e, 0x67, 0xc3,
|
||||||
|
0xa9, 0x6c, 0x69, 0x66, 0x65, 0x63, 0x79, 0x63, 0x6c, 0x65, 0xc0, 0xab, 0x72, 0x65, 0x70, 0x6c, 0x69, 0x63, 0x61, 0x74,
|
||||||
|
0x69, 0x6f, 0x6e, 0xc0, 0xae, 0x66, 0x61, 0x69, 0x6c, 0x65, 0x64, 0x5f, 0x6f, 0x62, 0x6a, 0x65, 0x63, 0x74, 0x73, 0x81,
|
||||||
|
0xb0, 0x77, 0x69, 0x72, 0x65, 0x2d, 0x62, 0x75, 0x63, 0x6b, 0x65, 0x74, 0x2f, 0x6c, 0x6f, 0x73, 0x74, 0x0b, 0xb1, 0x73,
|
||||||
|
0x63, 0x61, 0x6e, 0x5f, 0x72, 0x65, 0x73, 0x75, 0x6d, 0x65, 0x5f, 0x61, 0x66, 0x74, 0x65, 0x72, 0xb2, 0x77, 0x69, 0x72,
|
||||||
|
0x65, 0x2d, 0x62, 0x75, 0x63, 0x6b, 0x65, 0x74, 0x2f, 0x72, 0x65, 0x73, 0x75, 0x6d, 0x65, 0xaf, 0x73, 0x63, 0x61, 0x6e,
|
||||||
|
0x5f, 0x63, 0x68, 0x65, 0x63, 0x6b, 0x70, 0x6f, 0x69, 0x6e, 0x74, 0xc0, 0xad, 0x70, 0x65, 0x6e, 0x64, 0x69, 0x6e, 0x67,
|
||||||
|
0x5f, 0x68, 0x65, 0x61, 0x6c, 0x73, 0x91, 0x9a, 0xa6, 0x6f, 0x62, 0x6a, 0x65, 0x63, 0x74, 0xab, 0x77, 0x69, 0x72, 0x65,
|
||||||
|
0x2d, 0x62, 0x75, 0x63, 0x6b, 0x65, 0x74, 0xa6, 0x62, 0x72, 0x6f, 0x6b, 0x65, 0x6e, 0xc0, 0x01, 0x64, 0xcc, 0xc8, 0x03,
|
||||||
|
0xa8, 0x64, 0x65, 0x66, 0x65, 0x72, 0x72, 0x65, 0x64, 0xa6, 0x62, 0x75, 0x64, 0x67, 0x65, 0x74, 0xab, 0x6f, 0x62, 0x6a,
|
||||||
|
0x65, 0x63, 0x74, 0x5f, 0x6c, 0x6f, 0x63, 0x6b, 0xc0, 0xa6, 0x73, 0x6f, 0x75, 0x72, 0x63, 0x65, 0x92, 0x01, 0x02, 0xb1,
|
||||||
|
0x73, 0x6e, 0x61, 0x70, 0x73, 0x68, 0x6f, 0x74, 0x5f, 0x63, 0x6f, 0x6d, 0x70, 0x6c, 0x65, 0x74, 0x65, 0xc3, 0xb0, 0x73,
|
||||||
|
0x63, 0x61, 0x6e, 0x5f, 0x70, 0x6c, 0x61, 0x6e, 0x5f, 0x64, 0x69, 0x67, 0x65, 0x73, 0x74, 0xdc, 0x00, 0x20, 0x03, 0x03,
|
||||||
|
0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03,
|
||||||
|
0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0x03, 0xb0, 0x63, 0x61, 0x63, 0x68, 0x65, 0x5f, 0x6b, 0x65, 0x79,
|
||||||
|
0x5f, 0x66, 0x6f, 0x72, 0x6d, 0x61, 0x74, 0x01, 0x81, 0xab, 0x77, 0x69, 0x72, 0x65, 0x2d, 0x62, 0x75, 0x63, 0x6b, 0x65,
|
||||||
|
0x74, 0x8b, 0xa8, 0x63, 0x68, 0x69, 0x6c, 0x64, 0x72, 0x65, 0x6e, 0x90, 0xa4, 0x73, 0x69, 0x7a, 0x65, 0xcd, 0x10, 0x00,
|
||||||
|
0xa7, 0x6f, 0x62, 0x6a, 0x65, 0x63, 0x74, 0x73, 0x03, 0xa8, 0x76, 0x65, 0x72, 0x73, 0x69, 0x6f, 0x6e, 0x73, 0x05, 0xae,
|
||||||
|
0x64, 0x65, 0x6c, 0x65, 0x74, 0x65, 0x5f, 0x6d, 0x61, 0x72, 0x6b, 0x65, 0x72, 0x73, 0x01, 0xa9, 0x6f, 0x62, 0x6a, 0x5f,
|
||||||
|
0x73, 0x69, 0x7a, 0x65, 0x73, 0x9b, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xac, 0x6f, 0x62,
|
||||||
|
0x6a, 0x5f, 0x76, 0x65, 0x72, 0x73, 0x69, 0x6f, 0x6e, 0x73, 0x97, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0xb1, 0x72,
|
||||||
|
0x65, 0x70, 0x6c, 0x69, 0x63, 0x61, 0x74, 0x69, 0x6f, 0x6e, 0x5f, 0x73, 0x74, 0x61, 0x74, 0x73, 0xc0, 0xa9, 0x63, 0x6f,
|
||||||
|
0x6d, 0x70, 0x61, 0x63, 0x74, 0x65, 0x64, 0xc3, 0xae, 0x66, 0x61, 0x69, 0x6c, 0x65, 0x64, 0x5f, 0x6f, 0x62, 0x6a, 0x65,
|
||||||
|
0x63, 0x74, 0x73, 0x02, 0xae, 0x61, 0x6c, 0x6c, 0x5f, 0x74, 0x69, 0x65, 0x72, 0x5f, 0x73, 0x74, 0x61, 0x74, 0x73, 0x91,
|
||||||
|
0x81, 0xa4, 0x57, 0x41, 0x52, 0x4d, 0x93, 0xcd, 0x08, 0x00, 0x02, 0x01,
|
||||||
|
];
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn thin_usage_cache_decodes_scanner_wire_fixture() {
|
||||||
|
let decoded =
|
||||||
|
DataUsageCache::unmarshal(SCANNER_USAGE_CACHE_WIRE_FIXTURE).expect("thin projection decodes a scanner-written cache");
|
||||||
|
|
||||||
|
// The six fields shared with the scanner's 16-field info block; the
|
||||||
|
// remaining ten (lifecycle, replication, checkpoint, heals, ...) must
|
||||||
|
// be skipped, not error.
|
||||||
|
assert_eq!(decoded.info.name, "wire-bucket");
|
||||||
|
assert_eq!(decoded.info.next_cycle, 7);
|
||||||
|
assert_eq!(
|
||||||
|
decoded.info.last_update,
|
||||||
|
Some(SystemTime::UNIX_EPOCH + Duration::from_secs(1_700_000_000))
|
||||||
|
);
|
||||||
|
assert!(decoded.info.skip_healing);
|
||||||
|
assert_eq!(decoded.info.failed_objects.get("wire-bucket/lost"), Some(&11));
|
||||||
|
assert!(decoded.info.snapshot_complete);
|
||||||
|
|
||||||
|
// Entries use the shared canonical map-encoded type end to end.
|
||||||
|
let entry = decoded.cache.get("wire-bucket").expect("fixture entry decodes");
|
||||||
|
assert_eq!(entry.size, 4096);
|
||||||
|
assert_eq!(entry.objects, 3);
|
||||||
|
assert_eq!(entry.versions, 5);
|
||||||
|
assert_eq!(entry.delete_markers, 1);
|
||||||
|
assert!(entry.compacted);
|
||||||
|
assert_eq!(entry.failed_objects, 2);
|
||||||
|
assert_eq!(
|
||||||
|
entry.all_tier_stats.as_ref().and_then(|tiers| tiers.tiers.get("WARM")),
|
||||||
|
Some(&TierStats {
|
||||||
|
total_size: 2048,
|
||||||
|
num_versions: 2,
|
||||||
|
num_objects: 1,
|
||||||
|
})
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn hash_path_uses_portable_slash_semantics() {
|
fn hash_path_uses_portable_slash_semantics() {
|
||||||
for (input, expected) in [
|
for (input, expected) in [
|
||||||
|
|||||||
+102
-18
@@ -40,7 +40,8 @@ use http::header::{CONTENT_TYPE, HOST};
|
|||||||
use rustfs_signer::constants::UNSIGNED_PAYLOAD;
|
use rustfs_signer::constants::UNSIGNED_PAYLOAD;
|
||||||
use rustfs_signer::sign_v4;
|
use rustfs_signer::sign_v4;
|
||||||
use s3s::Body;
|
use s3s::Body;
|
||||||
use std::collections::BTreeSet;
|
use sha2::{Digest, Sha256};
|
||||||
|
use std::collections::{BTreeMap, BTreeSet};
|
||||||
use std::error::Error;
|
use std::error::Error;
|
||||||
use std::path::{Path, PathBuf};
|
use std::path::{Path, PathBuf};
|
||||||
use tracing::info;
|
use tracing::info;
|
||||||
@@ -59,13 +60,26 @@ pub(crate) struct VersionShardCensus {
|
|||||||
pub version_id: Option<String>,
|
pub version_id: Option<String>,
|
||||||
pub has_xl_meta: bool,
|
pub has_xl_meta: bool,
|
||||||
pub data_dir: Option<String>,
|
pub data_dir: Option<String>,
|
||||||
|
pub erasure_index: Option<usize>,
|
||||||
pub expected_part_numbers: BTreeSet<usize>,
|
pub expected_part_numbers: BTreeSet<usize>,
|
||||||
pub present_part_numbers: BTreeSet<usize>,
|
pub present_part_fingerprints: BTreeMap<usize, PartShardFingerprint>,
|
||||||
|
pub inline_data_fingerprint: Option<PartShardFingerprint>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Debug, Eq, PartialEq)]
|
||||||
|
pub(crate) struct PartShardFingerprint {
|
||||||
|
pub size: u64,
|
||||||
|
pub sha256: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl VersionShardCensus {
|
impl VersionShardCensus {
|
||||||
pub(crate) fn is_complete(&self) -> bool {
|
pub(crate) fn is_complete(&self) -> bool {
|
||||||
self.has_xl_meta && self.expected_part_numbers == self.present_part_numbers
|
self.has_xl_meta
|
||||||
|
&& self.expected_part_numbers.len() == self.present_part_fingerprints.len()
|
||||||
|
&& self
|
||||||
|
.expected_part_numbers
|
||||||
|
.iter()
|
||||||
|
.all(|part_number| self.present_part_fingerprints.contains_key(part_number))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub(crate) fn matches_manifest(&self, manifest: &Self) -> bool {
|
pub(crate) fn matches_manifest(&self, manifest: &Self) -> bool {
|
||||||
@@ -73,10 +87,25 @@ impl VersionShardCensus {
|
|||||||
&& self.is_complete()
|
&& self.is_complete()
|
||||||
&& manifest.is_complete()
|
&& manifest.is_complete()
|
||||||
&& self.data_dir == manifest.data_dir
|
&& self.data_dir == manifest.data_dir
|
||||||
|
&& self.erasure_index == manifest.erasure_index
|
||||||
&& self.expected_part_numbers == manifest.expected_part_numbers
|
&& self.expected_part_numbers == manifest.expected_part_numbers
|
||||||
|
&& self.present_part_fingerprints == manifest.present_part_fingerprints
|
||||||
|
&& self.inline_data_fingerprint == manifest.inline_data_fingerprint
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn sha256_hex(data: &[u8]) -> String {
|
||||||
|
let digest = Sha256::digest(data);
|
||||||
|
digest.iter().map(|byte| format!("{byte:02x}")).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn shard_fingerprint(data: &[u8]) -> ChaosResult<PartShardFingerprint> {
|
||||||
|
Ok(PartShardFingerprint {
|
||||||
|
size: u64::try_from(data.len())?,
|
||||||
|
sha256: sha256_hex(data),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
/// Single-node RustFS server with `disk_count` local volume directories that
|
/// Single-node RustFS server with `disk_count` local volume directories that
|
||||||
/// can be faulted individually while the server is running.
|
/// can be faulted individually while the server is running.
|
||||||
pub struct DiskFaultHarness {
|
pub struct DiskFaultHarness {
|
||||||
@@ -283,8 +312,10 @@ pub(crate) fn census_object_version_on_disk(
|
|||||||
version_id,
|
version_id,
|
||||||
has_xl_meta: false,
|
has_xl_meta: false,
|
||||||
data_dir: None,
|
data_dir: None,
|
||||||
|
erasure_index: None,
|
||||||
expected_part_numbers: BTreeSet::new(),
|
expected_part_numbers: BTreeSet::new(),
|
||||||
present_part_numbers: BTreeSet::new(),
|
present_part_fingerprints: BTreeMap::new(),
|
||||||
|
inline_data_fingerprint: None,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -296,20 +327,31 @@ pub(crate) fn census_object_version_on_disk(
|
|||||||
file_info.parts.iter().map(|part| part.number).collect()
|
file_info.parts.iter().map(|part| part.number).collect()
|
||||||
};
|
};
|
||||||
let data_dir = file_info.data_dir.map(|id| id.to_string());
|
let data_dir = file_info.data_dir.map(|id| id.to_string());
|
||||||
|
let erasure_index = Some(file_info.erasure.index);
|
||||||
|
let inline_data_fingerprint = file_info.data.as_deref().map(shard_fingerprint).transpose()?;
|
||||||
let part_dir = data_dir.as_ref().map_or_else(|| object_dir.clone(), |id| object_dir.join(id));
|
let part_dir = data_dir.as_ref().map_or_else(|| object_dir.clone(), |id| object_dir.join(id));
|
||||||
let present_part_numbers = match std::fs::read_dir(&part_dir) {
|
let present_part_fingerprints = match std::fs::read_dir(&part_dir) {
|
||||||
Ok(entries) => entries
|
Ok(entries) => {
|
||||||
.filter_map(Result::ok)
|
let mut fingerprints = BTreeMap::new();
|
||||||
.filter_map(|entry| {
|
for entry in entries {
|
||||||
entry
|
let entry = entry?;
|
||||||
.file_type()
|
if !entry.file_type()?.is_file() {
|
||||||
.ok()
|
continue;
|
||||||
.filter(|kind| kind.is_file())
|
}
|
||||||
.and_then(|_| entry.file_name().to_str().map(str::to_owned))
|
let file_name = entry.file_name();
|
||||||
})
|
let Some(part_number) = file_name
|
||||||
.filter_map(|name| name.strip_prefix("part.").and_then(|number| number.parse::<usize>().ok()))
|
.to_str()
|
||||||
.collect(),
|
.and_then(|name| name.strip_prefix("part."))
|
||||||
Err(error) if error.kind() == std::io::ErrorKind::NotFound => BTreeSet::new(),
|
.and_then(|number| number.parse::<usize>().ok())
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let data = std::fs::read(entry.path())?;
|
||||||
|
fingerprints.insert(part_number, shard_fingerprint(&data)?);
|
||||||
|
}
|
||||||
|
fingerprints
|
||||||
|
}
|
||||||
|
Err(error) if error.kind() == std::io::ErrorKind::NotFound => BTreeMap::new(),
|
||||||
Err(error) => return Err(error.into()),
|
Err(error) => return Err(error.into()),
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -317,8 +359,10 @@ pub(crate) fn census_object_version_on_disk(
|
|||||||
version_id,
|
version_id,
|
||||||
has_xl_meta: true,
|
has_xl_meta: true,
|
||||||
data_dir,
|
data_dir,
|
||||||
|
erasure_index,
|
||||||
expected_part_numbers,
|
expected_part_numbers,
|
||||||
present_part_numbers,
|
present_part_fingerprints,
|
||||||
|
inline_data_fingerprint,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -358,3 +402,43 @@ pub async fn signed_admin_post(url: &str, body: Option<&str>, access_key: &str,
|
|||||||
|
|
||||||
Ok(body)
|
Ok(body)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn complete_census() -> VersionShardCensus {
|
||||||
|
VersionShardCensus {
|
||||||
|
version_id: Some("version".to_string()),
|
||||||
|
has_xl_meta: true,
|
||||||
|
data_dir: Some("data-dir".to_string()),
|
||||||
|
erasure_index: Some(3),
|
||||||
|
expected_part_numbers: BTreeSet::from([1]),
|
||||||
|
present_part_fingerprints: BTreeMap::from([(1, shard_fingerprint(b"part").unwrap())]),
|
||||||
|
inline_data_fingerprint: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn shard_fingerprint_uses_physical_length_and_sha256() {
|
||||||
|
assert_eq!(
|
||||||
|
shard_fingerprint(b"abc").unwrap(),
|
||||||
|
PartShardFingerprint {
|
||||||
|
size: 3,
|
||||||
|
sha256: "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad".to_string(),
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn manifest_requires_matching_inline_payload() {
|
||||||
|
let mut expected = complete_census();
|
||||||
|
expected.expected_part_numbers.clear();
|
||||||
|
expected.present_part_fingerprints.clear();
|
||||||
|
expected.inline_data_fingerprint = Some(shard_fingerprint(b"expected").unwrap());
|
||||||
|
let mut changed = expected.clone();
|
||||||
|
changed.inline_data_fingerprint = Some(shard_fingerprint(b"changed").unwrap());
|
||||||
|
assert!(expected.matches_manifest(&expected));
|
||||||
|
assert!(!changed.matches_manifest(&expected));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -67,6 +67,19 @@ fn configured_capture_log_path(temp_dir: &str) -> Option<String> {
|
|||||||
capture_log_path(Path::new(&log_dir), temp_dir).map(|path| path.to_string_lossy().into_owned())
|
capture_log_path(Path::new(&log_dir), temp_dir).map(|path| path.to_string_lossy().into_owned())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) fn capture_command_logs(
|
||||||
|
command: &mut Command,
|
||||||
|
log_path: Option<&str>,
|
||||||
|
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
let Some(log_path) = log_path else {
|
||||||
|
return Ok(());
|
||||||
|
};
|
||||||
|
let file = stdfs::OpenOptions::new().create(true).append(true).open(log_path)?;
|
||||||
|
let stderr_file = file.try_clone()?;
|
||||||
|
command.stdout(Stdio::from(file)).stderr(Stdio::from(stderr_file));
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) fn build_test_s3_config(
|
pub(crate) fn build_test_s3_config(
|
||||||
endpoint_url: &str,
|
endpoint_url: &str,
|
||||||
access_key: &str,
|
access_key: &str,
|
||||||
@@ -557,13 +570,7 @@ impl RustFSTestEnvironment {
|
|||||||
for (key, value) in extra_env {
|
for (key, value) in extra_env {
|
||||||
command.env(key, value);
|
command.env(key, value);
|
||||||
}
|
}
|
||||||
// Optionally capture the child's stdout+stderr to a file so the test can
|
capture_command_logs(&mut command, self.capture_log_path.as_deref())?;
|
||||||
// grep server logs (e.g. to confirm which GET reader path was taken).
|
|
||||||
if let Some(log_path) = &self.capture_log_path {
|
|
||||||
let file = stdfs::OpenOptions::new().create(true).append(true).open(log_path)?;
|
|
||||||
let stderr_file = file.try_clone()?;
|
|
||||||
command.stdout(Stdio::from(file)).stderr(Stdio::from(stderr_file));
|
|
||||||
}
|
|
||||||
let process = command.args(&args).spawn()?;
|
let process = command.args(&args).spawn()?;
|
||||||
|
|
||||||
self.process = Some(process);
|
self.process = Some(process);
|
||||||
@@ -1051,6 +1058,7 @@ pub struct RustFSTestClusterEnvironment {
|
|||||||
pub secret_key: String,
|
pub secret_key: String,
|
||||||
pub extra_env: Vec<(String, String)>,
|
pub extra_env: Vec<(String, String)>,
|
||||||
pub node_extra_env: Vec<Vec<(String, String)>>,
|
pub node_extra_env: Vec<Vec<(String, String)>>,
|
||||||
|
pub node_capture_log_paths: Vec<Option<String>>,
|
||||||
pub topology: ClusterTopology,
|
pub topology: ClusterTopology,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1150,6 +1158,7 @@ impl RustFSTestClusterEnvironment {
|
|||||||
secret_key: "rustfs-cluster-test-secret".to_string(),
|
secret_key: "rustfs-cluster-test-secret".to_string(),
|
||||||
extra_env,
|
extra_env,
|
||||||
node_extra_env: vec![Vec::new(); topology.node_count],
|
node_extra_env: vec![Vec::new(); topology.node_count],
|
||||||
|
node_capture_log_paths: vec![None; topology.node_count],
|
||||||
topology,
|
topology,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -1179,6 +1188,20 @@ impl RustFSTestClusterEnvironment {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Capture stdout+stderr for a single cluster node process.
|
||||||
|
pub fn set_node_capture_log_path<P>(
|
||||||
|
&mut self,
|
||||||
|
node_idx: usize,
|
||||||
|
path: P,
|
||||||
|
) -> Result<(), Box<dyn std::error::Error + Send + Sync>>
|
||||||
|
where
|
||||||
|
P: Into<String>,
|
||||||
|
{
|
||||||
|
self.ensure_node_index(node_idx)?;
|
||||||
|
self.node_capture_log_paths[node_idx] = Some(path.into());
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
fn ensure_node_index(&self, node_idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
fn ensure_node_index(&self, node_idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
if node_idx >= self.nodes.len() {
|
if node_idx >= self.nodes.len() {
|
||||||
return Err(format!("node_idx {node_idx} is invalid").into());
|
return Err(format!("node_idx {node_idx} is invalid").into());
|
||||||
@@ -1268,6 +1291,7 @@ impl RustFSTestClusterEnvironment {
|
|||||||
for (key, value) in &self.node_extra_env[i] {
|
for (key, value) in &self.node_extra_env[i] {
|
||||||
command.env(key, value);
|
command.env(key, value);
|
||||||
}
|
}
|
||||||
|
capture_command_logs(&mut command, self.node_capture_log_paths[i].as_deref())?;
|
||||||
|
|
||||||
let process = command.current_dir(&node.data_dir).spawn()?;
|
let process = command.current_dir(&node.data_dir).spawn()?;
|
||||||
|
|
||||||
@@ -1294,6 +1318,7 @@ impl RustFSTestClusterEnvironment {
|
|||||||
|
|
||||||
let binary_path = rustfs_binary_path();
|
let binary_path = rustfs_binary_path();
|
||||||
let volumes_arg = self.build_volumes_arg();
|
let volumes_arg = self.build_volumes_arg();
|
||||||
|
let log_path = self.node_capture_log_paths[node_idx].clone();
|
||||||
let node = &mut self.nodes[node_idx];
|
let node = &mut self.nodes[node_idx];
|
||||||
info!("Starting cluster node {} on {}", node_idx, node.address);
|
info!("Starting cluster node {} on {}", node_idx, node.address);
|
||||||
|
|
||||||
@@ -1312,6 +1337,7 @@ impl RustFSTestClusterEnvironment {
|
|||||||
for (key, value) in &self.node_extra_env[node_idx] {
|
for (key, value) in &self.node_extra_env[node_idx] {
|
||||||
command.env(key, value);
|
command.env(key, value);
|
||||||
}
|
}
|
||||||
|
capture_command_logs(&mut command, log_path.as_deref())?;
|
||||||
|
|
||||||
let process = command.current_dir(&node.data_dir).spawn()?;
|
let process = command.current_dir(&node.data_dir).spawn()?;
|
||||||
node.process = Some(process);
|
node.process = Some(process);
|
||||||
@@ -1563,6 +1589,7 @@ mod tests {
|
|||||||
secret_key: DEFAULT_SECRET_KEY.to_string(),
|
secret_key: DEFAULT_SECRET_KEY.to_string(),
|
||||||
extra_env: Vec::new(),
|
extra_env: Vec::new(),
|
||||||
node_extra_env: vec![Vec::new(); topology.node_count],
|
node_extra_env: vec![Vec::new(); topology.node_count],
|
||||||
|
node_capture_log_paths: vec![None; topology.node_count],
|
||||||
topology,
|
topology,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1658,6 +1685,16 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cluster_node_log_capture_supports_per_node_paths() {
|
||||||
|
let mut env = fake_cluster(ClusterTopology::single_pool(3));
|
||||||
|
env.set_node_capture_log_path(1, "/tmp/node1.log").unwrap();
|
||||||
|
assert_eq!(env.node_capture_log_paths[0], None);
|
||||||
|
assert_eq!(env.node_capture_log_paths[1], Some("/tmp/node1.log".to_string()));
|
||||||
|
assert_eq!(env.node_capture_log_paths[2], None);
|
||||||
|
assert!(env.set_node_capture_log_path(3, "/tmp/invalid.log").is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn cluster_node_env_rejects_invalid_index() {
|
fn cluster_node_env_rejects_invalid_index() {
|
||||||
let mut env = fake_cluster(ClusterTopology::single_pool(4));
|
let mut env = fake_cluster(ClusterTopology::single_pool(4));
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
|
|
||||||
use crate::common::{RustFSTestEnvironment, init_logging, rustfs_binary_path};
|
use crate::common::{RustFSTestEnvironment, init_logging, rustfs_binary_path};
|
||||||
use aws_sdk_s3::primitives::ByteStream;
|
use aws_sdk_s3::primitives::ByteStream;
|
||||||
|
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart};
|
||||||
use serial_test::serial;
|
use serial_test::serial;
|
||||||
use std::fs;
|
use std::fs;
|
||||||
use std::path::PathBuf;
|
use std::path::PathBuf;
|
||||||
@@ -25,6 +26,15 @@ fn generate_compressible_data(size: usize) -> Vec<u8> {
|
|||||||
data
|
data
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Deterministic 2048-byte-period binary pattern that compresses extremely well: every part
|
||||||
|
/// yields many compressed blocks, which is exactly the shape that reproduced the mid-payload
|
||||||
|
/// Pending truncation (rustfs/rustfs#5957).
|
||||||
|
fn generate_high_ratio_binary_data(size: usize, seed: u8) -> Vec<u8> {
|
||||||
|
(0..size)
|
||||||
|
.map(|i| ((i as u64).wrapping_mul(2_654_435_761).wrapping_add(seed as u64) >> 3) as u8)
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
fn find_part_files(temp_dir: &str, bucket: &str, object_key: &str) -> Vec<PathBuf> {
|
fn find_part_files(temp_dir: &str, bucket: &str, object_key: &str) -> Vec<PathBuf> {
|
||||||
let bucket_path = PathBuf::from(temp_dir).join(bucket);
|
let bucket_path = PathBuf::from(temp_dir).join(bucket);
|
||||||
let mut part_files = Vec::new();
|
let mut part_files = Vec::new();
|
||||||
@@ -55,9 +65,14 @@ async fn start_rustfs_with_compression(env: &mut RustFSTestEnvironment) -> Resul
|
|||||||
env.cleanup_existing_processes().await?;
|
env.cleanup_existing_processes().await?;
|
||||||
|
|
||||||
let binary_path = rustfs_binary_path();
|
let binary_path = rustfs_binary_path();
|
||||||
let process = Command::new(&binary_path)
|
// Route the child's stdout/stderr through the shared RUSTFS_E2E_LOG_DIR
|
||||||
|
// capture (survives the temp-dir cleanup on Drop and is uploaded as a CI
|
||||||
|
// artifact); without the env var the child inherits stdio as before.
|
||||||
|
let mut command = Command::new(&binary_path);
|
||||||
|
command
|
||||||
.env("RUSTFS_CONSOLE_ENABLE", "false")
|
.env("RUSTFS_CONSOLE_ENABLE", "false")
|
||||||
.env("RUSTFS_COMPRESSION_ENABLED", "true")
|
.env("RUSTFS_COMPRESSION_ENABLED", "true")
|
||||||
|
.env("RUSTFS_COMPRESSION_MULTIPART_ENABLED", "true")
|
||||||
.args([
|
.args([
|
||||||
"--address",
|
"--address",
|
||||||
&env.address,
|
&env.address,
|
||||||
@@ -66,8 +81,9 @@ async fn start_rustfs_with_compression(env: &mut RustFSTestEnvironment) -> Resul
|
|||||||
"--secret-key",
|
"--secret-key",
|
||||||
&env.secret_key,
|
&env.secret_key,
|
||||||
&env.temp_dir,
|
&env.temp_dir,
|
||||||
])
|
]);
|
||||||
.spawn()?;
|
crate::common::capture_command_logs(&mut command, env.capture_log_path.as_deref())?;
|
||||||
|
let process = command.spawn()?;
|
||||||
|
|
||||||
env.process = Some(process);
|
env.process = Some(process);
|
||||||
|
|
||||||
@@ -154,3 +170,647 @@ async fn test_compression_roundtrip() -> Result<(), Box<dyn std::error::Error +
|
|||||||
env.stop_server();
|
env.stop_server();
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const MULTIPART_COMPRESSION_BUCKET: &str = "compression-multipart-bucket";
|
||||||
|
const MPU_PART1_SIZE: usize = 5 * 1024 * 1024;
|
||||||
|
const MPU_PART2_SIZE: usize = 1024 * 1024;
|
||||||
|
|
||||||
|
async fn multipart_upload(
|
||||||
|
client: &aws_sdk_s3::Client,
|
||||||
|
bucket: &str,
|
||||||
|
key: &str,
|
||||||
|
parts: &[&[u8]],
|
||||||
|
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
let create = client.create_multipart_upload().bucket(bucket).key(key).send().await?;
|
||||||
|
let upload_id = create.upload_id().ok_or("missing upload id")?.to_string();
|
||||||
|
|
||||||
|
let mut completed_parts = Vec::with_capacity(parts.len());
|
||||||
|
for (i, part) in parts.iter().enumerate() {
|
||||||
|
let part_number = (i + 1) as i32;
|
||||||
|
let upload = client
|
||||||
|
.upload_part()
|
||||||
|
.bucket(bucket)
|
||||||
|
.key(key)
|
||||||
|
.upload_id(&upload_id)
|
||||||
|
.part_number(part_number)
|
||||||
|
.body(ByteStream::from(part.to_vec()))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
completed_parts.push(
|
||||||
|
CompletedPart::builder()
|
||||||
|
.part_number(part_number)
|
||||||
|
.e_tag(upload.e_tag().unwrap_or_default())
|
||||||
|
.build(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
client
|
||||||
|
.complete_multipart_upload()
|
||||||
|
.bucket(bucket)
|
||||||
|
.key(key)
|
||||||
|
.upload_id(&upload_id)
|
||||||
|
.multipart_upload(CompletedMultipartUpload::builder().set_parts(Some(completed_parts)).build())
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn fetch_range(
|
||||||
|
client: &aws_sdk_s3::Client,
|
||||||
|
bucket: &str,
|
||||||
|
key: &str,
|
||||||
|
range: &str,
|
||||||
|
) -> Result<Vec<u8>, Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
let response = client.get_object().bucket(bucket).key(key).range(range).send().await?;
|
||||||
|
Ok(response.body.collect().await?.into_bytes().to_vec())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Multipart disk compression roundtrip: parts are written as independent
|
||||||
|
/// compressed streams and every GET shape must reassemble the original bytes
|
||||||
|
/// (rustfs/rustfs#5957: multipart uploads previously bypassed disk compression
|
||||||
|
/// entirely).
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn test_compression_multipart_roundtrip() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
init_logging();
|
||||||
|
info!("Starting multipart compression roundtrip test");
|
||||||
|
|
||||||
|
let mut env = RustFSTestEnvironment::new().await?;
|
||||||
|
start_rustfs_with_compression(&mut env).await?;
|
||||||
|
|
||||||
|
let client = env.create_s3_client();
|
||||||
|
env.create_test_bucket(MULTIPART_COMPRESSION_BUCKET).await?;
|
||||||
|
|
||||||
|
let object_key = "multipart-compressible.txt";
|
||||||
|
let part1 = generate_compressible_data(MPU_PART1_SIZE);
|
||||||
|
let part2 = generate_compressible_data(MPU_PART2_SIZE);
|
||||||
|
let mut original_data = part1.clone();
|
||||||
|
original_data.extend_from_slice(&part2);
|
||||||
|
let total_size = original_data.len();
|
||||||
|
|
||||||
|
multipart_upload(&client, MULTIPART_COMPRESSION_BUCKET, object_key, &[&part1, &part2]).await?;
|
||||||
|
|
||||||
|
let head_response = client
|
||||||
|
.head_object()
|
||||||
|
.bucket(MULTIPART_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
head_response.content_length().unwrap_or(0) as usize,
|
||||||
|
total_size,
|
||||||
|
"Content-Length should be the logical object size"
|
||||||
|
);
|
||||||
|
|
||||||
|
let part_files = find_part_files(&env.temp_dir, MULTIPART_COMPRESSION_BUCKET, object_key);
|
||||||
|
assert!(!part_files.is_empty(), "expected on-disk part files for the multipart object");
|
||||||
|
let total_physical_size: u64 = part_files.iter().filter_map(|p| fs::metadata(p).ok()).map(|m| m.len()).sum();
|
||||||
|
assert!(
|
||||||
|
total_physical_size < (total_size / 2) as u64,
|
||||||
|
"Physical size {total_physical_size} should be well below original size {total_size} (multipart compression applied)"
|
||||||
|
);
|
||||||
|
info!("Multipart physical storage size: {total_physical_size} bytes (compressed from {total_size} bytes)");
|
||||||
|
|
||||||
|
// Full GET must reassemble both independently compressed parts.
|
||||||
|
let get_response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MULTIPART_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let downloaded = get_response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(downloaded.len(), total_size);
|
||||||
|
assert_eq!(&downloaded[..], &original_data[..], "full GET data mismatch");
|
||||||
|
|
||||||
|
// Range fully inside part 1.
|
||||||
|
let range_inside_part1 = fetch_range(&client, MULTIPART_COMPRESSION_BUCKET, object_key, "bytes=1024-999423").await?;
|
||||||
|
assert_eq!(&range_inside_part1[..], &original_data[1024..999424], "part-1 range mismatch");
|
||||||
|
|
||||||
|
// Range crossing the part boundary.
|
||||||
|
let boundary_start = MPU_PART1_SIZE - 128 * 1024;
|
||||||
|
let boundary_end = MPU_PART1_SIZE + 128 * 1024 - 1;
|
||||||
|
let range_crossing = fetch_range(
|
||||||
|
&client,
|
||||||
|
MULTIPART_COMPRESSION_BUCKET,
|
||||||
|
object_key,
|
||||||
|
&format!("bytes={boundary_start}-{boundary_end}"),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
&range_crossing[..],
|
||||||
|
&original_data[boundary_start..boundary_end + 1],
|
||||||
|
"boundary-crossing range mismatch"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Range fully inside part 2.
|
||||||
|
let part2_start = MPU_PART1_SIZE + 4096;
|
||||||
|
let part2_end = MPU_PART1_SIZE + 256 * 1024 - 1;
|
||||||
|
let range_inside_part2 = fetch_range(
|
||||||
|
&client,
|
||||||
|
MULTIPART_COMPRESSION_BUCKET,
|
||||||
|
object_key,
|
||||||
|
&format!("bytes={part2_start}-{part2_end}"),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
&range_inside_part2[..],
|
||||||
|
&original_data[part2_start..part2_end + 1],
|
||||||
|
"part-2 range mismatch"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Suffix range (last 128 KiB, entirely in part 2).
|
||||||
|
let suffix_len = 128 * 1024;
|
||||||
|
let suffix = fetch_range(&client, MULTIPART_COMPRESSION_BUCKET, object_key, &format!("bytes=-{suffix_len}")).await?;
|
||||||
|
assert_eq!(&suffix[..], &original_data[total_size - suffix_len..], "suffix range mismatch");
|
||||||
|
|
||||||
|
// partNumber GETs must return each original part.
|
||||||
|
for (part_number, expected) in [(1, &part1), (2, &part2)] {
|
||||||
|
let response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MULTIPART_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.part_number(part_number)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let body = response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(&body[..], &expected[..], "partNumber={part_number} GET mismatch");
|
||||||
|
}
|
||||||
|
|
||||||
|
info!("Multipart compression roundtrip test passed");
|
||||||
|
env.delete_test_bucket(MULTIPART_COMPRESSION_BUCKET).await?;
|
||||||
|
env.stop_server();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
const MPU_HIGH_RATIO_BUCKET: &str = "compression-mpu-high-ratio-bucket";
|
||||||
|
|
||||||
|
/// High-ratio binary multipart payload: the object key is on the compression allow-list, so the
|
||||||
|
/// disk-compression path runs and each part is stored as many compressed blocks — the shape that
|
||||||
|
/// reproduced the mid-payload Pending truncation (rustfs/rustfs#5957). Every GET shape must return
|
||||||
|
/// the exact original bytes, and the stored size must show the data really was compressed.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn test_compression_multipart_high_ratio_binary_roundtrip() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
init_logging();
|
||||||
|
info!("Starting multipart high-ratio binary compression roundtrip test");
|
||||||
|
|
||||||
|
let mut env = RustFSTestEnvironment::new().await?;
|
||||||
|
start_rustfs_with_compression(&mut env).await?;
|
||||||
|
|
||||||
|
let client = env.create_s3_client();
|
||||||
|
env.create_test_bucket(MPU_HIGH_RATIO_BUCKET).await?;
|
||||||
|
|
||||||
|
let object_key = "multipart-high-ratio.txt";
|
||||||
|
let part1 = generate_high_ratio_binary_data(MPU_PART1_SIZE, 7);
|
||||||
|
let part2 = generate_high_ratio_binary_data(MPU_PART2_SIZE, 61);
|
||||||
|
let mut original_data = part1.clone();
|
||||||
|
original_data.extend_from_slice(&part2);
|
||||||
|
let total_size = original_data.len();
|
||||||
|
|
||||||
|
multipart_upload(&client, MPU_HIGH_RATIO_BUCKET, object_key, &[&part1, &part2]).await?;
|
||||||
|
|
||||||
|
let head_response = client
|
||||||
|
.head_object()
|
||||||
|
.bucket(MPU_HIGH_RATIO_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
head_response.content_length().unwrap_or(0) as usize,
|
||||||
|
total_size,
|
||||||
|
"Content-Length should be the logical object size"
|
||||||
|
);
|
||||||
|
|
||||||
|
// This pattern compresses to roughly 1/50 of its logical size, so a comfortably loose 2x
|
||||||
|
// margin still proves the parts were stored compressed rather than raw or double-encoded.
|
||||||
|
let part_files = find_part_files(&env.temp_dir, MPU_HIGH_RATIO_BUCKET, object_key);
|
||||||
|
assert!(!part_files.is_empty(), "expected on-disk part files for the multipart object");
|
||||||
|
let total_physical_size: u64 = part_files.iter().filter_map(|p| fs::metadata(p).ok()).map(|m| m.len()).sum();
|
||||||
|
assert!(
|
||||||
|
total_physical_size < (total_size as u64) / 2,
|
||||||
|
"Physical size {total_physical_size} should be far below the logical size {total_size} for high-ratio data"
|
||||||
|
);
|
||||||
|
info!("High-ratio multipart physical storage size: {total_physical_size} bytes (logical {total_size} bytes)");
|
||||||
|
|
||||||
|
info!("step: full GET");
|
||||||
|
let get_response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MPU_HIGH_RATIO_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let downloaded = get_response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(downloaded.len(), total_size);
|
||||||
|
assert_eq!(&downloaded[..], &original_data[..], "full GET data mismatch");
|
||||||
|
|
||||||
|
// Range crossing the part boundary.
|
||||||
|
info!("step: boundary range GET");
|
||||||
|
let boundary_start = MPU_PART1_SIZE - 128 * 1024;
|
||||||
|
let boundary_end = MPU_PART1_SIZE + 128 * 1024 - 1;
|
||||||
|
let range_crossing = fetch_range(
|
||||||
|
&client,
|
||||||
|
MPU_HIGH_RATIO_BUCKET,
|
||||||
|
object_key,
|
||||||
|
&format!("bytes={boundary_start}-{boundary_end}"),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
&range_crossing[..],
|
||||||
|
&original_data[boundary_start..boundary_end + 1],
|
||||||
|
"boundary-crossing range mismatch"
|
||||||
|
);
|
||||||
|
|
||||||
|
// partNumber GET for the trailing part.
|
||||||
|
info!("step: partNumber GET");
|
||||||
|
let part2_response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MPU_HIGH_RATIO_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.part_number(2)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let part2_body = part2_response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(&part2_body[..], &part2[..], "partNumber=2 GET mismatch");
|
||||||
|
|
||||||
|
info!("Multipart high-ratio binary compression roundtrip test passed");
|
||||||
|
env.delete_test_bucket(MPU_HIGH_RATIO_BUCKET).await?;
|
||||||
|
env.stop_server();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
const MPU_COPY_COMPRESSION_BUCKET: &str = "compression-mpu-copy-bucket";
|
||||||
|
const MPU_COPY_SOURCE_SIZE: usize = 6 * 1024 * 1024;
|
||||||
|
const MPU_COPY_RANGE_LEN: usize = 5 * 1024 * 1024;
|
||||||
|
|
||||||
|
/// UploadPartCopy feeds a part from an already stored (and already compressed) object. The copied
|
||||||
|
/// range must be decompressed on read and re-compressed into the destination part, so the final
|
||||||
|
/// object has to match "source prefix + uploaded tail" byte for byte.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn test_compression_multipart_upload_part_copy_roundtrip() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
init_logging();
|
||||||
|
info!("Starting multipart upload-part-copy compression roundtrip test");
|
||||||
|
|
||||||
|
let mut env = RustFSTestEnvironment::new().await?;
|
||||||
|
start_rustfs_with_compression(&mut env).await?;
|
||||||
|
|
||||||
|
let client = env.create_s3_client();
|
||||||
|
env.create_test_bucket(MPU_COPY_COMPRESSION_BUCKET).await?;
|
||||||
|
|
||||||
|
// Source object: a plain PUT that goes through the single-stream compression path.
|
||||||
|
let source_key = "copy-source.txt";
|
||||||
|
let source_data = generate_compressible_data(MPU_COPY_SOURCE_SIZE);
|
||||||
|
client
|
||||||
|
.put_object()
|
||||||
|
.bucket(MPU_COPY_COMPRESSION_BUCKET)
|
||||||
|
.key(source_key)
|
||||||
|
.body(ByteStream::from(source_data.clone()))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
// Destination object: part 1 copied from the source, part 2 uploaded directly.
|
||||||
|
let target_key = "copy-target.txt";
|
||||||
|
let part2 = generate_compressible_data(MPU_PART2_SIZE);
|
||||||
|
let mut expected_data = source_data[..MPU_COPY_RANGE_LEN].to_vec();
|
||||||
|
expected_data.extend_from_slice(&part2);
|
||||||
|
let total_size = expected_data.len();
|
||||||
|
|
||||||
|
let create = client
|
||||||
|
.create_multipart_upload()
|
||||||
|
.bucket(MPU_COPY_COMPRESSION_BUCKET)
|
||||||
|
.key(target_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let upload_id = create.upload_id().ok_or("missing upload id")?.to_string();
|
||||||
|
|
||||||
|
let copy_part = client
|
||||||
|
.upload_part_copy()
|
||||||
|
.bucket(MPU_COPY_COMPRESSION_BUCKET)
|
||||||
|
.key(target_key)
|
||||||
|
.upload_id(&upload_id)
|
||||||
|
.part_number(1)
|
||||||
|
.copy_source(format!("{MPU_COPY_COMPRESSION_BUCKET}/{source_key}"))
|
||||||
|
.copy_source_range(format!("bytes=0-{}", MPU_COPY_RANGE_LEN - 1))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let copy_etag = copy_part
|
||||||
|
.copy_part_result()
|
||||||
|
.and_then(|r| r.e_tag())
|
||||||
|
.ok_or("missing copy part etag")?
|
||||||
|
.to_string();
|
||||||
|
|
||||||
|
let uploaded_part = client
|
||||||
|
.upload_part()
|
||||||
|
.bucket(MPU_COPY_COMPRESSION_BUCKET)
|
||||||
|
.key(target_key)
|
||||||
|
.upload_id(&upload_id)
|
||||||
|
.part_number(2)
|
||||||
|
.body(ByteStream::from(part2.clone()))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
client
|
||||||
|
.complete_multipart_upload()
|
||||||
|
.bucket(MPU_COPY_COMPRESSION_BUCKET)
|
||||||
|
.key(target_key)
|
||||||
|
.upload_id(&upload_id)
|
||||||
|
.multipart_upload(
|
||||||
|
CompletedMultipartUpload::builder()
|
||||||
|
.parts(CompletedPart::builder().part_number(1).e_tag(copy_etag).build())
|
||||||
|
.parts(
|
||||||
|
CompletedPart::builder()
|
||||||
|
.part_number(2)
|
||||||
|
.e_tag(uploaded_part.e_tag().unwrap_or_default())
|
||||||
|
.build(),
|
||||||
|
)
|
||||||
|
.build(),
|
||||||
|
)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let head_response = client
|
||||||
|
.head_object()
|
||||||
|
.bucket(MPU_COPY_COMPRESSION_BUCKET)
|
||||||
|
.key(target_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
head_response.content_length().unwrap_or(0) as usize,
|
||||||
|
total_size,
|
||||||
|
"Content-Length should be the logical object size"
|
||||||
|
);
|
||||||
|
|
||||||
|
let part_files = find_part_files(&env.temp_dir, MPU_COPY_COMPRESSION_BUCKET, target_key);
|
||||||
|
assert!(!part_files.is_empty(), "expected on-disk part files for the copied object");
|
||||||
|
let total_physical_size: u64 = part_files.iter().filter_map(|p| fs::metadata(p).ok()).map(|m| m.len()).sum();
|
||||||
|
assert!(
|
||||||
|
total_physical_size < (total_size / 2) as u64,
|
||||||
|
"Physical size {total_physical_size} should be well below original size {total_size} (copied part compression applied)"
|
||||||
|
);
|
||||||
|
|
||||||
|
let get_response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MPU_COPY_COMPRESSION_BUCKET)
|
||||||
|
.key(target_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let downloaded = get_response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(downloaded.len(), total_size);
|
||||||
|
assert_eq!(&downloaded[..], &expected_data[..], "copied multipart GET data mismatch");
|
||||||
|
|
||||||
|
info!("Multipart upload-part-copy compression roundtrip test passed");
|
||||||
|
env.delete_test_bucket(MPU_COPY_COMPRESSION_BUCKET).await?;
|
||||||
|
env.stop_server();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
const MPU_THREE_PARTS_BUCKET: &str = "compression-mpu-three-parts-bucket";
|
||||||
|
const MPU_THREE_PARTS_TAIL_SIZE: usize = 512 * 1024;
|
||||||
|
|
||||||
|
/// Three-part upload with uneven part sizes: each partNumber GET must map back to exactly one
|
||||||
|
/// compressed part stream, and a suffix range must resolve inside the trailing part.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn test_compression_multipart_three_parts_part_number_gets() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
init_logging();
|
||||||
|
info!("Starting three-part multipart compression partNumber test");
|
||||||
|
|
||||||
|
let mut env = RustFSTestEnvironment::new().await?;
|
||||||
|
start_rustfs_with_compression(&mut env).await?;
|
||||||
|
|
||||||
|
let client = env.create_s3_client();
|
||||||
|
env.create_test_bucket(MPU_THREE_PARTS_BUCKET).await?;
|
||||||
|
|
||||||
|
let object_key = "multipart-three-parts.txt";
|
||||||
|
let part1 = generate_compressible_data(MPU_PART1_SIZE);
|
||||||
|
let part2 = generate_compressible_data(MPU_PART1_SIZE);
|
||||||
|
let part3 = generate_compressible_data(MPU_THREE_PARTS_TAIL_SIZE);
|
||||||
|
let mut original_data = part1.clone();
|
||||||
|
original_data.extend_from_slice(&part2);
|
||||||
|
original_data.extend_from_slice(&part3);
|
||||||
|
let total_size = original_data.len();
|
||||||
|
|
||||||
|
multipart_upload(&client, MPU_THREE_PARTS_BUCKET, object_key, &[&part1, &part2, &part3]).await?;
|
||||||
|
|
||||||
|
let head_response = client
|
||||||
|
.head_object()
|
||||||
|
.bucket(MPU_THREE_PARTS_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
head_response.content_length().unwrap_or(0) as usize,
|
||||||
|
total_size,
|
||||||
|
"Content-Length should be the logical object size"
|
||||||
|
);
|
||||||
|
|
||||||
|
let part_files = find_part_files(&env.temp_dir, MPU_THREE_PARTS_BUCKET, object_key);
|
||||||
|
assert!(!part_files.is_empty(), "expected on-disk part files for the multipart object");
|
||||||
|
let total_physical_size: u64 = part_files.iter().filter_map(|p| fs::metadata(p).ok()).map(|m| m.len()).sum();
|
||||||
|
assert!(
|
||||||
|
total_physical_size < (total_size / 2) as u64,
|
||||||
|
"Physical size {total_physical_size} should be well below original size {total_size} (multipart compression applied)"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Every partNumber GET must return exactly the bytes of the corresponding uploaded part.
|
||||||
|
for (part_number, expected) in [(1, &part1), (2, &part2), (3, &part3)] {
|
||||||
|
let response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MPU_THREE_PARTS_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.part_number(part_number)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let body = response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(&body[..], &expected[..], "partNumber={part_number} GET mismatch");
|
||||||
|
}
|
||||||
|
|
||||||
|
// Suffix range (last 64 KiB) resolves inside the trailing part.
|
||||||
|
let suffix_len = 64 * 1024;
|
||||||
|
let suffix = fetch_range(&client, MPU_THREE_PARTS_BUCKET, object_key, &format!("bytes=-{suffix_len}")).await?;
|
||||||
|
assert_eq!(&suffix[..], &original_data[total_size - suffix_len..], "suffix range mismatch");
|
||||||
|
|
||||||
|
info!("Three-part multipart compression partNumber test passed");
|
||||||
|
env.delete_test_bucket(MPU_THREE_PARTS_BUCKET).await?;
|
||||||
|
env.stop_server();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
const MPU_SSE_COMPRESSION_BUCKET: &str = "compression-mpu-sse-bucket";
|
||||||
|
|
||||||
|
async fn start_rustfs_with_compression_and_sse(
|
||||||
|
env: &mut RustFSTestEnvironment,
|
||||||
|
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
use base64::Engine;
|
||||||
|
env.cleanup_existing_processes().await?;
|
||||||
|
|
||||||
|
let binary_path = rustfs_binary_path();
|
||||||
|
let master_key = base64::engine::general_purpose::STANDARD.encode([0x42u8; 32]);
|
||||||
|
// Server output goes to a file inside the per-test temp dir so a failing
|
||||||
|
// run can be diagnosed from the child's logs.
|
||||||
|
let server_log = std::fs::File::create(format!("{}/server.log", env.temp_dir))?;
|
||||||
|
let server_log_err = server_log.try_clone()?;
|
||||||
|
let process = Command::new(&binary_path)
|
||||||
|
.env("RUSTFS_CONSOLE_ENABLE", "false")
|
||||||
|
.env("RUSTFS_COMPRESSION_ENABLED", "true")
|
||||||
|
.env("RUSTFS_COMPRESSION_MULTIPART_ENABLED", "true")
|
||||||
|
.env("RUSTFS_SSE_S3_MASTER_KEY", master_key)
|
||||||
|
.env("RUST_LOG", "rustfs=info,rustfs_ecstore=info")
|
||||||
|
.stdout(std::process::Stdio::from(server_log))
|
||||||
|
.stderr(std::process::Stdio::from(server_log_err))
|
||||||
|
.args([
|
||||||
|
"--address",
|
||||||
|
&env.address,
|
||||||
|
"--access-key",
|
||||||
|
&env.access_key,
|
||||||
|
"--secret-key",
|
||||||
|
&env.secret_key,
|
||||||
|
&env.temp_dir,
|
||||||
|
])
|
||||||
|
.spawn()?;
|
||||||
|
|
||||||
|
env.process = Some(process);
|
||||||
|
|
||||||
|
info!("Waiting for RustFS server with compression + SSE-S3 enabled on {}", env.address);
|
||||||
|
for i in 0..30 {
|
||||||
|
if TcpStream::connect(&env.address).await.is_ok() {
|
||||||
|
info!("RustFS server is ready after {} attempts", i + 1);
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
if i == 29 {
|
||||||
|
return Err("RustFS server failed to become ready".into());
|
||||||
|
}
|
||||||
|
sleep(Duration::from_secs(1)).await;
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// SSE-S3 + disk compression multipart: each part is compressed and then encrypted, and every GET
|
||||||
|
/// shape must still return the original plaintext bytes. Physical size must shrink because the
|
||||||
|
/// compression runs before encryption.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn test_compression_multipart_sse_s3_roundtrip() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
use aws_sdk_s3::types::ServerSideEncryption;
|
||||||
|
|
||||||
|
init_logging();
|
||||||
|
info!("Starting SSE-S3 multipart compression roundtrip test");
|
||||||
|
|
||||||
|
let mut env = RustFSTestEnvironment::new().await?;
|
||||||
|
start_rustfs_with_compression_and_sse(&mut env).await?;
|
||||||
|
|
||||||
|
let client = env.create_s3_client();
|
||||||
|
env.create_test_bucket(MPU_SSE_COMPRESSION_BUCKET).await?;
|
||||||
|
|
||||||
|
let object_key = "multipart-sse-compressible.txt";
|
||||||
|
let part1 = generate_compressible_data(MPU_PART1_SIZE);
|
||||||
|
let part2 = generate_compressible_data(MPU_PART2_SIZE);
|
||||||
|
let mut original_data = part1.clone();
|
||||||
|
original_data.extend_from_slice(&part2);
|
||||||
|
let total_size = original_data.len();
|
||||||
|
|
||||||
|
let create = client
|
||||||
|
.create_multipart_upload()
|
||||||
|
.bucket(MPU_SSE_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.server_side_encryption(ServerSideEncryption::Aes256)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let upload_id = create.upload_id().ok_or("missing upload id")?.to_string();
|
||||||
|
|
||||||
|
let mut completed_parts = Vec::new();
|
||||||
|
for (i, part) in [&part1, &part2].into_iter().enumerate() {
|
||||||
|
let part_number = (i + 1) as i32;
|
||||||
|
let upload = client
|
||||||
|
.upload_part()
|
||||||
|
.bucket(MPU_SSE_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.upload_id(&upload_id)
|
||||||
|
.part_number(part_number)
|
||||||
|
.body(ByteStream::from(part.clone()))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
completed_parts.push(
|
||||||
|
CompletedPart::builder()
|
||||||
|
.part_number(part_number)
|
||||||
|
.e_tag(upload.e_tag().unwrap_or_default())
|
||||||
|
.build(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
client
|
||||||
|
.complete_multipart_upload()
|
||||||
|
.bucket(MPU_SSE_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.upload_id(&upload_id)
|
||||||
|
.multipart_upload(CompletedMultipartUpload::builder().set_parts(Some(completed_parts)).build())
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let head_response = client
|
||||||
|
.head_object()
|
||||||
|
.bucket(MPU_SSE_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
head_response.content_length().unwrap_or(0) as usize,
|
||||||
|
total_size,
|
||||||
|
"Content-Length should be the logical object size"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
head_response.server_side_encryption(),
|
||||||
|
Some(&ServerSideEncryption::Aes256),
|
||||||
|
"HEAD must report SSE-S3"
|
||||||
|
);
|
||||||
|
|
||||||
|
let part_files = find_part_files(&env.temp_dir, MPU_SSE_COMPRESSION_BUCKET, object_key);
|
||||||
|
assert!(!part_files.is_empty(), "expected on-disk part files for the multipart object");
|
||||||
|
let total_physical_size: u64 = part_files.iter().filter_map(|p| fs::metadata(p).ok()).map(|m| m.len()).sum();
|
||||||
|
assert!(
|
||||||
|
total_physical_size < (total_size / 2) as u64,
|
||||||
|
"Physical size {total_physical_size} should be well below original size {total_size} (compress-then-encrypt applied)"
|
||||||
|
);
|
||||||
|
|
||||||
|
let get_response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MPU_SSE_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let downloaded = get_response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(downloaded.len(), total_size);
|
||||||
|
assert_eq!(&downloaded[..], &original_data[..], "SSE-S3 multipart full GET data mismatch");
|
||||||
|
|
||||||
|
// Range crossing the part boundary must decrypt and decompress across parts.
|
||||||
|
let boundary_start = MPU_PART1_SIZE - 64 * 1024;
|
||||||
|
let boundary_end = MPU_PART1_SIZE + 64 * 1024 - 1;
|
||||||
|
let range_crossing = fetch_range(
|
||||||
|
&client,
|
||||||
|
MPU_SSE_COMPRESSION_BUCKET,
|
||||||
|
object_key,
|
||||||
|
&format!("bytes={boundary_start}-{boundary_end}"),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
&range_crossing[..],
|
||||||
|
&original_data[boundary_start..boundary_end + 1],
|
||||||
|
"SSE-S3 boundary-crossing range mismatch"
|
||||||
|
);
|
||||||
|
|
||||||
|
// partNumber GET for the trailing part.
|
||||||
|
let part2_response = client
|
||||||
|
.get_object()
|
||||||
|
.bucket(MPU_SSE_COMPRESSION_BUCKET)
|
||||||
|
.key(object_key)
|
||||||
|
.part_number(2)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let part2_body = part2_response.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(&part2_body[..], &part2[..], "SSE-S3 partNumber=2 GET mismatch");
|
||||||
|
|
||||||
|
info!("SSE-S3 multipart compression roundtrip test passed");
|
||||||
|
env.delete_test_bucket(MPU_SSE_COMPRESSION_BUCKET).await?;
|
||||||
|
env.stop_server();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|||||||
@@ -76,6 +76,18 @@ const SOURCE_MTIME_HEADERS: [&str; 2] = ["x-rustfs-source-mtime", "x-minio-sourc
|
|||||||
const SOURCE_REPLICATION_REQUEST_HEADERS: [&str; 2] =
|
const SOURCE_REPLICATION_REQUEST_HEADERS: [&str; 2] =
|
||||||
["x-rustfs-source-replication-request", "x-minio-source-replication-request"];
|
["x-rustfs-source-replication-request", "x-minio-source-replication-request"];
|
||||||
const SOURCE_ETAG_HEADERS: [&str; 2] = ["x-rustfs-source-etag", "x-minio-source-etag"];
|
const SOURCE_ETAG_HEADERS: [&str; 2] = ["x-rustfs-source-etag", "x-minio-source-etag"];
|
||||||
|
const SOURCE_TAGGING_TIMESTAMP_HEADERS: [&str; 2] = [
|
||||||
|
"x-rustfs-source-replication-tagging-timestamp",
|
||||||
|
"x-minio-source-replication-tagging-timestamp",
|
||||||
|
];
|
||||||
|
const SOURCE_RETENTION_TIMESTAMP_HEADERS: [&str; 2] = [
|
||||||
|
"x-rustfs-source-replication-retention-timestamp",
|
||||||
|
"x-minio-source-replication-retention-timestamp",
|
||||||
|
];
|
||||||
|
const SOURCE_LEGALHOLD_TIMESTAMP_HEADERS: [&str; 2] = [
|
||||||
|
"x-rustfs-source-replication-legalhold-timestamp",
|
||||||
|
"x-minio-source-replication-legalhold-timestamp",
|
||||||
|
];
|
||||||
const RESERVED_BUCKET_PREFIXES: [&str; 3] = ["xn--", "sthree-", "amzn-s3-demo-"];
|
const RESERVED_BUCKET_PREFIXES: [&str; 3] = ["xn--", "sthree-", "amzn-s3-demo-"];
|
||||||
const RESERVED_BUCKET_SUFFIXES: [&str; 6] = ["-s3alias", "--ol-s3", ".mrap", "--x-s3", "--table-s3", "-an"];
|
const RESERVED_BUCKET_SUFFIXES: [&str; 6] = ["-s3alias", "--ol-s3", ".mrap", "--x-s3", "--table-s3", "-an"];
|
||||||
|
|
||||||
@@ -118,6 +130,25 @@ pub enum FaultAction {
|
|||||||
WrongEtag,
|
WrongEtag,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Replication LWW timestamp headers observed on a request, journaled so
|
||||||
|
/// sender-side tests can assert what a real target would receive.
|
||||||
|
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||||
|
pub struct ReplicationTimestampHeaders {
|
||||||
|
pub tagging: Option<String>,
|
||||||
|
pub retention: Option<String>,
|
||||||
|
pub legalhold: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ReplicationTimestampHeaders {
|
||||||
|
fn from_headers(headers: &HeaderMap) -> Self {
|
||||||
|
Self {
|
||||||
|
tagging: header_value(headers, &SOURCE_TAGGING_TIMESTAMP_HEADERS).map(bounded_journal_value),
|
||||||
|
retention: header_value(headers, &SOURCE_RETENTION_TIMESTAMP_HEADERS).map(bounded_journal_value),
|
||||||
|
legalhold: header_value(headers, &SOURCE_LEGALHOLD_TIMESTAMP_HEADERS).map(bounded_journal_value),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Credential-free request metadata retained for deterministic assertions.
|
/// Credential-free request metadata retained for deterministic assertions.
|
||||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
pub struct RequestRecord {
|
pub struct RequestRecord {
|
||||||
@@ -131,6 +162,7 @@ pub struct RequestRecord {
|
|||||||
pub part_number: Option<i32>,
|
pub part_number: Option<i32>,
|
||||||
pub content_length: Option<u64>,
|
pub content_length: Option<u64>,
|
||||||
pub consumed_bytes: Option<usize>,
|
pub consumed_bytes: Option<usize>,
|
||||||
|
pub replication_timestamps: ReplicationTimestampHeaders,
|
||||||
pub fault: Option<FaultAction>,
|
pub fault: Option<FaultAction>,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -536,7 +568,15 @@ impl S3Access for FaultAccess {
|
|||||||
.get(CONTENT_LENGTH)
|
.get(CONTENT_LENGTH)
|
||||||
.and_then(|value| value.to_str().ok())
|
.and_then(|value| value.to_str().ok())
|
||||||
.and_then(|value| value.parse().ok());
|
.and_then(|value| value.parse().ok());
|
||||||
let fault = record_request(&self.control, operation, context.method().clone(), parsed, content_length);
|
let replication_timestamps = ReplicationTimestampHeaders::from_headers(context.headers());
|
||||||
|
let fault = record_request(
|
||||||
|
&self.control,
|
||||||
|
operation,
|
||||||
|
context.method().clone(),
|
||||||
|
parsed,
|
||||||
|
content_length,
|
||||||
|
replication_timestamps,
|
||||||
|
);
|
||||||
if let Some(RequestFault {
|
if let Some(RequestFault {
|
||||||
action: FaultAction::Status(status),
|
action: FaultAction::Status(status),
|
||||||
..
|
..
|
||||||
@@ -589,6 +629,7 @@ fn record_request(
|
|||||||
method: Method,
|
method: Method,
|
||||||
parsed: ParsedRequest,
|
parsed: ParsedRequest,
|
||||||
content_length: Option<u64>,
|
content_length: Option<u64>,
|
||||||
|
replication_timestamps: ReplicationTimestampHeaders,
|
||||||
) -> Option<RequestFault> {
|
) -> Option<RequestFault> {
|
||||||
let mut state = lock(control);
|
let mut state = lock(control);
|
||||||
let action = parsed
|
let action = parsed
|
||||||
@@ -613,6 +654,7 @@ fn record_request(
|
|||||||
part_number: parsed.part_number,
|
part_number: parsed.part_number,
|
||||||
content_length,
|
content_length,
|
||||||
consumed_bytes: None,
|
consumed_bytes: None,
|
||||||
|
replication_timestamps,
|
||||||
fault: action.clone(),
|
fault: action.clone(),
|
||||||
});
|
});
|
||||||
action.map(|action| RequestFault { sequence, action })
|
action.map(|action| RequestFault { sequence, action })
|
||||||
@@ -1699,6 +1741,52 @@ mod tests {
|
|||||||
.await?)
|
.await?)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn journals_replication_timestamp_headers() -> Result<(), BoxError> {
|
||||||
|
let target = FakeS3Target::start().await?;
|
||||||
|
target.create_bucket("target-bucket");
|
||||||
|
let client = client(&target);
|
||||||
|
|
||||||
|
client
|
||||||
|
.put_object()
|
||||||
|
.bucket("target-bucket")
|
||||||
|
.key("plain")
|
||||||
|
.body(ByteStream::from_static(b"plain"))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
client
|
||||||
|
.put_object()
|
||||||
|
.bucket("target-bucket")
|
||||||
|
.key("stamped")
|
||||||
|
.body(ByteStream::from_static(b"stamped"))
|
||||||
|
.customize()
|
||||||
|
.map_request(move |mut request| {
|
||||||
|
let headers = request.headers_mut();
|
||||||
|
headers.insert("x-rustfs-source-replication-tagging-timestamp", "2026-01-02T03:04:05Z");
|
||||||
|
headers.insert("x-minio-source-replication-retention-timestamp", "2026-01-02T03:04:06Z");
|
||||||
|
headers.insert("x-rustfs-source-replication-legalhold-timestamp", "2026-01-02T03:04:07Z");
|
||||||
|
Ok::<_, std::convert::Infallible>(request)
|
||||||
|
})
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let requests = target.requests();
|
||||||
|
let plain = requests
|
||||||
|
.iter()
|
||||||
|
.find(|record| record.operation == Operation::PutObject && record.key.as_deref() == Some("plain"))
|
||||||
|
.expect("plain PUT must be journaled");
|
||||||
|
assert_eq!(plain.replication_timestamps, ReplicationTimestampHeaders::default());
|
||||||
|
|
||||||
|
let stamped = requests
|
||||||
|
.iter()
|
||||||
|
.find(|record| record.operation == Operation::PutObject && record.key.as_deref() == Some("stamped"))
|
||||||
|
.expect("stamped PUT must be journaled");
|
||||||
|
assert_eq!(stamped.replication_timestamps.tagging.as_deref(), Some("2026-01-02T03:04:05Z"));
|
||||||
|
assert_eq!(stamped.replication_timestamps.retention.as_deref(), Some("2026-01-02T03:04:06Z"));
|
||||||
|
assert_eq!(stamped.replication_timestamps.legalhold.as_deref(), Some("2026-01-02T03:04:07Z"));
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
macro_rules! assert_sdk_error {
|
macro_rules! assert_sdk_error {
|
||||||
($error:expr, $status:expr, $code:expr) => {{
|
($error:expr, $status:expr, $code:expr) => {{
|
||||||
let error = &$error;
|
let error = &$error;
|
||||||
@@ -2985,6 +3073,7 @@ mod tests {
|
|||||||
part_number: None,
|
part_number: None,
|
||||||
},
|
},
|
||||||
Some(0),
|
Some(0),
|
||||||
|
ReplicationTimestampHeaders::default(),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
let records = lock(&control).requests.clone();
|
let records = lock(&control).requests.clone();
|
||||||
@@ -3006,6 +3095,7 @@ mod tests {
|
|||||||
part_number: None,
|
part_number: None,
|
||||||
},
|
},
|
||||||
None,
|
None,
|
||||||
|
ReplicationTimestampHeaders::default(),
|
||||||
);
|
);
|
||||||
{
|
{
|
||||||
let bounded_records = lock(&bounded_control);
|
let bounded_records = lock(&bounded_control);
|
||||||
|
|||||||
@@ -189,8 +189,6 @@ mod tests {
|
|||||||
("RUSTFS_GET_CODEC_STREAMING_ROLLOUT_PCT", "100"),
|
("RUSTFS_GET_CODEC_STREAMING_ROLLOUT_PCT", "100"),
|
||||||
("RUSTFS_GET_CODEC_STREAMING_BODY_COMPAT_CONFIRMED", "true"),
|
("RUSTFS_GET_CODEC_STREAMING_BODY_COMPAT_CONFIRMED", "true"),
|
||||||
("RUSTFS_GET_CODEC_STREAMING_HEADER_COMPAT_CONFIRMED", "true"),
|
("RUSTFS_GET_CODEC_STREAMING_HEADER_COMPAT_CONFIRMED", "true"),
|
||||||
// Lower the min-size floor so every non-inline object below is eligible.
|
|
||||||
("RUSTFS_GET_CODEC_STREAMING_MIN_SIZE", "4096"),
|
|
||||||
// Route multipart objects through per-part codec streaming too.
|
// Route multipart objects through per-part codec streaming too.
|
||||||
("RUSTFS_GET_CODEC_STREAMING_MULTIPART_ENABLE", "true"),
|
("RUSTFS_GET_CODEC_STREAMING_MULTIPART_ENABLE", "true"),
|
||||||
// Lock optimization is on by default, but pin it so the gate's
|
// Lock optimization is on by default, but pin it so the gate's
|
||||||
@@ -315,6 +313,13 @@ mod tests {
|
|||||||
},
|
},
|
||||||
payload(64 * 1024, 2),
|
payload(64 * 1024, 2),
|
||||||
),
|
),
|
||||||
|
(
|
||||||
|
Shape {
|
||||||
|
key: "small-non-inline-256kib-plus",
|
||||||
|
expect_large: true,
|
||||||
|
},
|
||||||
|
payload(256 * 1024 + 1, 6),
|
||||||
|
),
|
||||||
(
|
(
|
||||||
Shape {
|
Shape {
|
||||||
key: "mid-1_5mib",
|
key: "mid-1_5mib",
|
||||||
|
|||||||
@@ -14,7 +14,7 @@
|
|||||||
|
|
||||||
//! E2E tests for group management (fixes #2028).
|
//! E2E tests for group management (fixes #2028).
|
||||||
|
|
||||||
use crate::common::{RustFSTestEnvironment, awscurl_delete, awscurl_get, awscurl_put, init_logging};
|
use crate::common::{RustFSTestEnvironment, admin_request, awscurl_delete, awscurl_get, awscurl_put, init_logging};
|
||||||
use aws_sdk_s3::config::{Credentials, Region};
|
use aws_sdk_s3::config::{Credentials, Region};
|
||||||
use aws_sdk_s3::{Client, Config};
|
use aws_sdk_s3::{Client, Config};
|
||||||
use serial_test::serial;
|
use serial_test::serial;
|
||||||
@@ -32,6 +32,56 @@ fn create_user_s3_client(env: &RustFSTestEnvironment, access_key: &str, secret_k
|
|||||||
Client::from_conf(config)
|
Client::from_conf(config)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test(flavor = "multi_thread")]
|
||||||
|
async fn update_group_members_rejects_invalid_new_group_names() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
init_logging();
|
||||||
|
|
||||||
|
let mut env = RustFSTestEnvironment::new().await?;
|
||||||
|
env.start_rustfs_server(vec![]).await?;
|
||||||
|
|
||||||
|
let invalid_groups = [
|
||||||
|
("test group", "group name contains whitespace"),
|
||||||
|
("test=group", "group name contains reserved characters =,"),
|
||||||
|
("test,group", "group name contains reserved characters =,"),
|
||||||
|
];
|
||||||
|
|
||||||
|
for (group, expected_message) in invalid_groups {
|
||||||
|
let body = serde_json::json!({
|
||||||
|
"group": group,
|
||||||
|
"members": [],
|
||||||
|
"isRemove": false,
|
||||||
|
"groupStatus": "enabled"
|
||||||
|
})
|
||||||
|
.to_string();
|
||||||
|
let (status, response_body) = admin_request(
|
||||||
|
&env.url,
|
||||||
|
http::Method::PUT,
|
||||||
|
"/rustfs/admin/v3/update-group-members",
|
||||||
|
Some(body),
|
||||||
|
&env.access_key,
|
||||||
|
&env.secret_key,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
status,
|
||||||
|
reqwest::StatusCode::BAD_REQUEST,
|
||||||
|
"invalid group {group:?} must return HTTP 400, body: {response_body}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
response_body.contains("<Code>InvalidArgument</Code>"),
|
||||||
|
"invalid group {group:?} must return InvalidArgument, body: {response_body}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
response_body.contains(&format!("<Message>{expected_message}</Message>")),
|
||||||
|
"invalid group {group:?} returned an unexpected message: {response_body}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
env.stop_server();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Test that deleting a group with members fails, and deleting an empty group succeeds.
|
/// Test that deleting a group with members fails, and deleting an empty group succeeds.
|
||||||
#[tokio::test(flavor = "multi_thread")]
|
#[tokio::test(flavor = "multi_thread")]
|
||||||
#[serial]
|
#[serial]
|
||||||
|
|||||||
@@ -1828,33 +1828,36 @@ async fn four_node_compressed_inline_fallback() -> TestResult {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Multipart disk compression is live again, so a compression-enabled cluster classifies multipart objects as compressed and the roundtrip (full GET plus partNumber GET) must still return the original bytes.
|
||||||
|
/// Reverting the multipart compression fix must fail this test.
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial]
|
#[serial]
|
||||||
async fn four_node_multipart_ignores_disk_compression_fallback() -> TestResult {
|
async fn four_node_multipart_disk_compression_roundtrip() -> TestResult {
|
||||||
init_logging();
|
init_logging();
|
||||||
|
|
||||||
let collector = OtlpMetricCollector::start().await?;
|
let collector = OtlpMetricCollector::start().await?;
|
||||||
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
||||||
configure_reader_metric_cluster(&mut cluster, &collector);
|
configure_reader_metric_cluster(&mut cluster, &collector);
|
||||||
cluster.set_env("RUSTFS_COMPRESSION_ENABLED", "true");
|
cluster.set_env("RUSTFS_COMPRESSION_ENABLED", "true");
|
||||||
|
cluster.set_env("RUSTFS_COMPRESSION_MULTIPART_ENABLED", "true");
|
||||||
cluster.start().await?;
|
cluster.start().await?;
|
||||||
|
|
||||||
let bucket = "inline-multipart-compression-fallback";
|
let bucket = "inline-multipart-compression-roundtrip";
|
||||||
cluster.create_test_bucket(bucket).await?;
|
cluster.create_test_bucket(bucket).await?;
|
||||||
let client = cluster.create_s3_client(0)?;
|
let client = cluster.create_s3_client(0)?;
|
||||||
let key = "multipart/compression-disabled.txt";
|
let key = "multipart/compressed.txt";
|
||||||
let (body, second_part, etag) = put_two_part_multipart(&client, bucket, key).await?;
|
let (body, second_part, etag) = put_two_part_multipart(&client, bucket, key).await?;
|
||||||
|
|
||||||
assert_reader_path(
|
assert_reader_path(
|
||||||
&collector,
|
&collector,
|
||||||
&client,
|
&client,
|
||||||
ReaderPathExpectation::for_class(ReaderObject::new(bucket, key, &body, etag.as_deref(), None), LEGACY_DUPLEX, MULTIPART),
|
ReaderPathExpectation::for_class(ReaderObject::new(bucket, key, &body, etag.as_deref(), None), LEGACY_DUPLEX, COMPRESSED),
|
||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
assert_part_number_reader_path(
|
assert_part_number_reader_path(
|
||||||
&collector,
|
&collector,
|
||||||
&client,
|
&client,
|
||||||
PartNumberReaderPathExpectation::new(bucket, key, &second_part, body.len(), MULTIPART, LEGACY_DUPLEX),
|
PartNumberReaderPathExpectation::new(bucket, key, &second_part, body.len(), COMPRESSED, LEGACY_DUPLEX),
|
||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
@@ -1871,6 +1874,7 @@ async fn four_node_mixed_msgpack_compat_mode_preserves_fallback_controls() -> Te
|
|||||||
let sse_master_key = base64::engine::general_purpose::STANDARD.encode([0x42u8; 32]);
|
let sse_master_key = base64::engine::general_purpose::STANDARD.encode([0x42u8; 32]);
|
||||||
cluster.set_env("RUSTFS_SSE_S3_MASTER_KEY", sse_master_key);
|
cluster.set_env("RUSTFS_SSE_S3_MASTER_KEY", sse_master_key);
|
||||||
cluster.set_env("RUSTFS_COMPRESSION_ENABLED", "true");
|
cluster.set_env("RUSTFS_COMPRESSION_ENABLED", "true");
|
||||||
|
cluster.set_env("RUSTFS_COMPRESSION_MULTIPART_ENABLED", "true");
|
||||||
configure_mixed_msgpack_cluster(&mut cluster, &collector)?;
|
configure_mixed_msgpack_cluster(&mut cluster, &collector)?;
|
||||||
cluster.start().await?;
|
cluster.start().await?;
|
||||||
|
|
||||||
@@ -1890,14 +1894,21 @@ async fn four_node_mixed_msgpack_compat_mode_preserves_fallback_controls() -> Te
|
|||||||
ReaderPathExpectation::for_class(
|
ReaderPathExpectation::for_class(
|
||||||
ReaderObject::new(bucket, multipart_key, &multipart_body, multipart_etag.as_deref(), None),
|
ReaderObject::new(bucket, multipart_key, &multipart_body, multipart_etag.as_deref(), None),
|
||||||
LEGACY_DUPLEX,
|
LEGACY_DUPLEX,
|
||||||
MULTIPART,
|
COMPRESSED,
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
assert_part_number_reader_path(
|
assert_part_number_reader_path(
|
||||||
&collector,
|
&collector,
|
||||||
&client,
|
&client,
|
||||||
PartNumberReaderPathExpectation::new(bucket, multipart_key, &second_part, multipart_body.len(), MULTIPART, LEGACY_DUPLEX),
|
PartNumberReaderPathExpectation::new(
|
||||||
|
bucket,
|
||||||
|
multipart_key,
|
||||||
|
&second_part,
|
||||||
|
multipart_body.len(),
|
||||||
|
COMPRESSED,
|
||||||
|
LEGACY_DUPLEX,
|
||||||
|
),
|
||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
assert_msgpack_decode_observed(&collector, &decode_before).await?;
|
assert_msgpack_decode_observed(&collector, &decode_before).await?;
|
||||||
@@ -2353,7 +2364,11 @@ async fn four_node_mixed_msgpack_compat_mode_preserves_fallback_controls_during_
|
|||||||
hot_client.create_bucket().bucket(bucket).send().await?;
|
hot_client.create_bucket().bucket(bucket).send().await?;
|
||||||
put_lifecycle_with_transition_retry(&hot_client, bucket, &tier_name).await?;
|
put_lifecycle_with_transition_retry(&hot_client, bucket, &tier_name).await?;
|
||||||
|
|
||||||
let key = "transition/mixed-multipart.bin";
|
// `.zip` sits on the disk-compression exclusion list: this test pins
|
||||||
|
// msgpack compat controls across ILM transition, and a compressed object
|
||||||
|
// would classify as `compressed` instead of `remote` (and the warm-tier
|
||||||
|
// read path does not decode compression — tracked separately).
|
||||||
|
let key = "transition/mixed-multipart.zip";
|
||||||
let (body, second_part, etag) = put_two_part_multipart(&hot_client, bucket, key).await?;
|
let (body, second_part, etag) = put_two_part_multipart(&hot_client, bucket, key).await?;
|
||||||
wait_for_transition(&hot_client, bucket, key, &tier_name).await?;
|
wait_for_transition(&hot_client, bucket, key, &tier_name).await?;
|
||||||
assert!(
|
assert!(
|
||||||
|
|||||||
@@ -0,0 +1,611 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! ILM on SSE-KMS buckets while per-key SSE authorization is enforced (backlog#1582).
|
||||||
|
//!
|
||||||
|
//! Per-key KMS authorization (`RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY=true`) scopes the
|
||||||
|
//! SSE-KMS data path to the requesting principal's `kms:GenerateDataKey` /
|
||||||
|
//! `kms:Decrypt` grants. Internal callers — the lifecycle scanner's expiry deletes
|
||||||
|
//! and the tier transition worker's reads — carry no request principal, and
|
||||||
|
//! `authorize_sse_kms_key` (rustfs/src/storage/sse.rs) exempts a `None` principal
|
||||||
|
//! so background maintenance keeps working on encrypted buckets.
|
||||||
|
//!
|
||||||
|
//! These tests pin that exemption end to end. If enforcement ever starts applying
|
||||||
|
//! to the scanner's internal operations, expiry stops happening on SSE-KMS buckets
|
||||||
|
//! and [`ilm_expiration_on_sse_kms_bucket_under_enforcement`] times out; if it
|
||||||
|
//! starts applying to the transition worker or the read-through path,
|
||||||
|
//! [`ilm_transition_on_sse_kms_bucket_under_enforcement_reads_back`] fails at the
|
||||||
|
//! transition wait or the plaintext round-trip.
|
||||||
|
//!
|
||||||
|
//! The replication half of the same acceptance item lives in
|
||||||
|
//! `crates/e2e_test/src/replication_extension_test.rs`
|
||||||
|
//! (`test_bucket_replication_sse_kms_failure_contract`); ILM had no coverage
|
||||||
|
//! before this file.
|
||||||
|
//!
|
||||||
|
//! Deployment constraint pinned by the transition test's setup: the RustFS warm
|
||||||
|
//! backend forwards the object's stored `x-amz-server-side-encryption*` metadata
|
||||||
|
//! as raw headers on the tier data PUT (`build_transition_put_options` +
|
||||||
|
//! `api_put_object.rs` header mapping), so a RustFS tier target must itself have
|
||||||
|
//! KMS enabled and hold the named key or it rejects every transition upload with
|
||||||
|
//! 400 InvalidRequest. That rejection is independent of the enforcement switch;
|
||||||
|
//! the cold server here therefore runs its own Local KMS with the same key id.
|
||||||
|
|
||||||
|
use super::common::{LocalKMSTestEnvironment, create_key_with_specific_id};
|
||||||
|
use crate::common::{RustFSTestEnvironment, admin_request, init_logging};
|
||||||
|
use aws_sdk_s3::Client;
|
||||||
|
use aws_sdk_s3::primitives::ByteStream;
|
||||||
|
use aws_sdk_s3::types::{
|
||||||
|
BucketLifecycleConfiguration, ExpirationStatus, LifecycleExpiration, LifecycleRule, LifecycleRuleFilter, RestoreRequest,
|
||||||
|
ServerSideEncryption, ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Transition,
|
||||||
|
TransitionStorageClass,
|
||||||
|
};
|
||||||
|
use serde::Deserialize;
|
||||||
|
use serial_test::serial;
|
||||||
|
use std::time::{Duration as StdDuration, Instant};
|
||||||
|
use tracing::info;
|
||||||
|
|
||||||
|
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
|
||||||
|
|
||||||
|
const SSE_KEY: &str = "kms-ilm-sse-key";
|
||||||
|
const PAYLOAD: &[u8] = b"kms ilm sse payload: survives enforcement, expires and transitions on schedule";
|
||||||
|
|
||||||
|
const EXPIRY_BUCKET: &str = "kms-ilm-expiry";
|
||||||
|
const EXPIRE_KEY: &str = "expire/object.bin";
|
||||||
|
const SURVIVOR_KEY: &str = "keep/object.bin";
|
||||||
|
|
||||||
|
const TIER_NAME: &str = "KMSCOLD";
|
||||||
|
const TIER_BUCKET: &str = "kms-ilm-cold-tier";
|
||||||
|
const TIER_PREFIX: &str = "tiered";
|
||||||
|
const TRANSITION_BUCKET: &str = "kms-ilm-transition";
|
||||||
|
const TRANSITION_KEY: &str = "tier/object.bin";
|
||||||
|
|
||||||
|
/// Generous CI safety net; with a 1s scanner cycle and 2s lifecycle days the
|
||||||
|
/// terminal state normally lands within a few seconds.
|
||||||
|
const ILM_DEADLINE: StdDuration = StdDuration::from_secs(90);
|
||||||
|
|
||||||
|
/// Start a Local-KMS server with per-key SSE authorization enforced and the
|
||||||
|
/// lifecycle clock accelerated.
|
||||||
|
///
|
||||||
|
/// KMS wiring matches `kms_authorization_negative_matrix_test.rs` (local backend,
|
||||||
|
/// `--kms-default-key-id`, insecure dev defaults). The lifecycle env matches
|
||||||
|
/// `reliant/lifecycle.rs::fast_lifecycle_env` plus `RUSTFS_ILM_DEBUG_DAY_SECS=2`,
|
||||||
|
/// so a `Days=1` rule is due about two seconds after the write.
|
||||||
|
async fn start_enforcing_ilm_server(env: &mut LocalKMSTestEnvironment) -> TestResult {
|
||||||
|
create_key_with_specific_id(&env.kms_keys_dir, SSE_KEY).await?;
|
||||||
|
|
||||||
|
let key_dir = env.kms_keys_dir.clone();
|
||||||
|
let args = vec![
|
||||||
|
"--kms-enable",
|
||||||
|
"--kms-backend",
|
||||||
|
"local",
|
||||||
|
"--kms-key-dir",
|
||||||
|
key_dir.as_str(),
|
||||||
|
"--kms-default-key-id",
|
||||||
|
SSE_KEY,
|
||||||
|
];
|
||||||
|
|
||||||
|
let envs = [
|
||||||
|
("RUSTFS_KMS_ALLOW_INSECURE_DEV_DEFAULTS", "true"),
|
||||||
|
("RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY", "true"),
|
||||||
|
("RUSTFS_SCANNER_CYCLE", "1"),
|
||||||
|
("RUSTFS_ILM_PROCESS_TIME", "1"),
|
||||||
|
("RUSTFS_ILM_DEBUG_DAY_SECS", "2"),
|
||||||
|
];
|
||||||
|
|
||||||
|
env.base_env.start_rustfs_server_with_env(args, &envs).await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Set the bucket's default encryption to SSE-KMS under [`SSE_KEY`], so plain
|
||||||
|
/// PUTs (and internal rewrites) are encrypted without per-request SSE headers.
|
||||||
|
async fn set_bucket_default_sse_kms(client: &Client, bucket: &str) -> TestResult {
|
||||||
|
let encryption_config = ServerSideEncryptionConfiguration::builder()
|
||||||
|
.rules(
|
||||||
|
ServerSideEncryptionRule::builder()
|
||||||
|
.apply_server_side_encryption_by_default(
|
||||||
|
ServerSideEncryptionByDefault::builder()
|
||||||
|
.sse_algorithm(ServerSideEncryption::AwsKms)
|
||||||
|
.kms_master_key_id(SSE_KEY)
|
||||||
|
.build()?,
|
||||||
|
)
|
||||||
|
.build(),
|
||||||
|
)
|
||||||
|
.build()?;
|
||||||
|
client
|
||||||
|
.put_bucket_encryption()
|
||||||
|
.bucket(bucket)
|
||||||
|
.server_side_encryption_configuration(encryption_config)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Assert via `HeadObject` that the stored object is SSE-KMS encrypted under
|
||||||
|
/// [`SSE_KEY`]. Without this, a bucket-default misconfiguration would let the
|
||||||
|
/// tests pass on an unencrypted object and prove nothing about KMS.
|
||||||
|
async fn assert_head_sse_kms(client: &Client, bucket: &str, key: &str) -> TestResult {
|
||||||
|
let head = client.head_object().bucket(bucket).key(key).send().await?;
|
||||||
|
assert_eq!(
|
||||||
|
head.server_side_encryption(),
|
||||||
|
Some(&ServerSideEncryption::AwsKms),
|
||||||
|
"{bucket}/{key} must be SSE-KMS encrypted via the bucket default"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
head.ssekms_key_id(),
|
||||||
|
Some(SSE_KEY),
|
||||||
|
"{bucket}/{key} must be wrapped under the configured KMS key"
|
||||||
|
);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `true` once `GET bucket/key` fails with `NoSuchKey`, `false` while it
|
||||||
|
/// still succeeds. Any other error is surfaced. (Copied from
|
||||||
|
/// `reliant/lifecycle.rs`; that helper is private to the reliant module.)
|
||||||
|
async fn object_is_gone(client: &Client, bucket: &str, key: &str) -> Result<bool, Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
match client.get_object().bucket(bucket).key(key).send().await {
|
||||||
|
Ok(output) => {
|
||||||
|
output.body.collect().await?;
|
||||||
|
Ok(false)
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
if let Some(service_error) = e.as_service_error() {
|
||||||
|
if service_error.is_no_such_key() {
|
||||||
|
return Ok(true);
|
||||||
|
}
|
||||||
|
return Err(format!("expected NoSuchKey, got: {e:?}").into());
|
||||||
|
}
|
||||||
|
Err(format!("expected a service error, got: {e:?}").into())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Poll until `GET bucket/key` returns `NoSuchKey`, or fail after `deadline`.
|
||||||
|
async fn wait_for_object_expired(client: &Client, bucket: &str, key: &str, deadline: StdDuration) -> TestResult {
|
||||||
|
let start = Instant::now();
|
||||||
|
loop {
|
||||||
|
if object_is_gone(client, bucket, key).await? {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
if start.elapsed() >= deadline {
|
||||||
|
return Err(format!(
|
||||||
|
"object {bucket}/{key} was not expired by the lifecycle scanner within {}s; \
|
||||||
|
SSE key-policy enforcement may have started blocking the scanner's internal deletes",
|
||||||
|
deadline.as_secs()
|
||||||
|
)
|
||||||
|
.into());
|
||||||
|
}
|
||||||
|
tokio::time::sleep(StdDuration::from_millis(500)).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Install a prefix-scoped `Days`-based expiration rule.
|
||||||
|
async fn put_expiration_rule(client: &Client, bucket: &str, id: &str, prefix: &str, days: i32) -> TestResult {
|
||||||
|
let rule = LifecycleRule::builder()
|
||||||
|
.id(id)
|
||||||
|
.filter(LifecycleRuleFilter::builder().prefix(prefix).build())
|
||||||
|
.expiration(LifecycleExpiration::builder().days(days).build())
|
||||||
|
.status(ExpirationStatus::Enabled)
|
||||||
|
.build()?;
|
||||||
|
let lifecycle = BucketLifecycleConfiguration::builder().rules(rule).build()?;
|
||||||
|
client
|
||||||
|
.put_bucket_lifecycle_configuration()
|
||||||
|
.bucket(bucket)
|
||||||
|
.lifecycle_configuration(lifecycle)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Install a prefix-scoped `Days`-based transition rule targeting [`TIER_NAME`].
|
||||||
|
async fn put_transition_rule(client: &Client, bucket: &str, id: &str, prefix: &str, days: i32) -> TestResult {
|
||||||
|
let rule = LifecycleRule::builder()
|
||||||
|
.id(id)
|
||||||
|
.filter(LifecycleRuleFilter::builder().prefix(prefix).build())
|
||||||
|
.transitions(
|
||||||
|
Transition::builder()
|
||||||
|
.days(days)
|
||||||
|
.storage_class(TransitionStorageClass::from(TIER_NAME))
|
||||||
|
.build(),
|
||||||
|
)
|
||||||
|
.status(ExpirationStatus::Enabled)
|
||||||
|
.build()?;
|
||||||
|
let lifecycle = BucketLifecycleConfiguration::builder().rules(rule).build()?;
|
||||||
|
client
|
||||||
|
.put_bucket_lifecycle_configuration()
|
||||||
|
.bucket(bucket)
|
||||||
|
.lifecycle_configuration(lifecycle)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start a plain Local-KMS server (no enforcement, no lifecycle acceleration)
|
||||||
|
/// holding [`SSE_KEY`], to serve as the cold tier target.
|
||||||
|
///
|
||||||
|
/// The RustFS warm backend forwards the stored SSE-KMS headers on the tier data
|
||||||
|
/// PUT, so the target re-applies managed SSE-KMS under the named key and must
|
||||||
|
/// be able to resolve it; without KMS it answers 400 InvalidRequest and the
|
||||||
|
/// transition can never complete. Enforcement stays off here: the tier writes
|
||||||
|
/// arrive under `cold`'s root credentials, and one enforcing side is enough to
|
||||||
|
/// pin the exemption.
|
||||||
|
async fn start_cold_tier_kms_server(env: &mut LocalKMSTestEnvironment) -> TestResult {
|
||||||
|
create_key_with_specific_id(&env.kms_keys_dir, SSE_KEY).await?;
|
||||||
|
|
||||||
|
let key_dir = env.kms_keys_dir.clone();
|
||||||
|
let args = vec![
|
||||||
|
"--kms-enable",
|
||||||
|
"--kms-backend",
|
||||||
|
"local",
|
||||||
|
"--kms-key-dir",
|
||||||
|
key_dir.as_str(),
|
||||||
|
"--kms-default-key-id",
|
||||||
|
SSE_KEY,
|
||||||
|
];
|
||||||
|
|
||||||
|
env.base_env
|
||||||
|
.start_rustfs_server_with_env(args, &[("RUSTFS_KMS_ALLOW_INSECURE_DEV_DEFAULTS", "true")])
|
||||||
|
.await?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The subset of the manual transition run report these tests assert on.
|
||||||
|
///
|
||||||
|
/// Unknown fields are ignored, so this stays compatible with report growth; the
|
||||||
|
/// full shape is pinned by `reliant/tiering.rs`.
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
struct ManualTransitionRunReport {
|
||||||
|
#[serde(default)]
|
||||||
|
scanned: u64,
|
||||||
|
#[serde(default)]
|
||||||
|
enqueued: u64,
|
||||||
|
#[serde(default)]
|
||||||
|
skipped_already_in_flight: u64,
|
||||||
|
#[serde(default)]
|
||||||
|
skipped_tier: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
struct ManualTransitionRunResponse {
|
||||||
|
state: String,
|
||||||
|
report: ManualTransitionRunReport,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One synchronous (enqueue-only) manual transition run over `bucket/prefix`,
|
||||||
|
/// via the same admin endpoint `reliant/tiering.rs` drives.
|
||||||
|
async fn manual_transition_run(
|
||||||
|
hot: &RustFSTestEnvironment,
|
||||||
|
bucket: &str,
|
||||||
|
prefix: &str,
|
||||||
|
) -> Result<ManualTransitionRunResponse, Box<dyn std::error::Error + Send + Sync>> {
|
||||||
|
let bucket = urlencoding::encode(bucket);
|
||||||
|
let prefix = urlencoding::encode(prefix);
|
||||||
|
let tier = urlencoding::encode(TIER_NAME);
|
||||||
|
let path =
|
||||||
|
format!("/rustfs/admin/v3/ilm/transition/run?bucket={bucket}&prefix={prefix}&tier={tier}&dryRun=false&maxObjects=10");
|
||||||
|
let (status, body) = admin_request(&hot.url, http::Method::POST, &path, None, &hot.access_key, &hot.secret_key).await?;
|
||||||
|
if !status.is_success() {
|
||||||
|
return Err(format!("manual transition run failed: status={status}, body={body}").into());
|
||||||
|
}
|
||||||
|
Ok(serde_json::from_str(&body)?)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drive manual transition runs until one reports the object as processed.
|
||||||
|
///
|
||||||
|
/// The `Days=1` rule becomes due about two seconds after the write
|
||||||
|
/// (`RUSTFS_ILM_DEBUG_DAY_SECS=2`), so early runs may legitimately report the
|
||||||
|
/// object as not yet eligible; the loop keeps running the endpoint until it
|
||||||
|
/// either enqueues the transition, sees it already in flight (the 1s scanner
|
||||||
|
/// backstop got there first), or finds it already on the tier.
|
||||||
|
async fn run_manual_transition_until_processed(
|
||||||
|
hot: &RustFSTestEnvironment,
|
||||||
|
bucket: &str,
|
||||||
|
prefix: &str,
|
||||||
|
deadline: StdDuration,
|
||||||
|
) -> TestResult {
|
||||||
|
let start = Instant::now();
|
||||||
|
loop {
|
||||||
|
let run = manual_transition_run(hot, bucket, prefix).await?;
|
||||||
|
assert_eq!(run.report.scanned, 1, "manual transition run must scan the object: {run:#?}");
|
||||||
|
if run.report.enqueued + run.report.skipped_already_in_flight + run.report.skipped_tier >= 1 {
|
||||||
|
info!(state = %run.state, report = ?run.report, "manual transition run processed the SSE-KMS object");
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
if start.elapsed() >= deadline {
|
||||||
|
return Err(format!(
|
||||||
|
"manual transition runs never processed {bucket}/{prefix} within {}s; last report: {run:#?}",
|
||||||
|
deadline.as_secs()
|
||||||
|
)
|
||||||
|
.into());
|
||||||
|
}
|
||||||
|
tokio::time::sleep(StdDuration::from_millis(500)).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Wire `hot` -> `cold` as a `TierType::RustFS` remote tier via `AddTier`.
|
||||||
|
///
|
||||||
|
/// No `force`, so the server runs the real connectivity probe against `cold`
|
||||||
|
/// (the tier bucket must already exist there). Mirrors
|
||||||
|
/// `reliant/tiering.rs::add_rustfs_tier`, which is private to that module.
|
||||||
|
async fn add_rustfs_tier(hot: &RustFSTestEnvironment, cold: &RustFSTestEnvironment) -> TestResult {
|
||||||
|
let body = serde_json::json!({
|
||||||
|
"type": "rustfs",
|
||||||
|
"rustfs": {
|
||||||
|
"name": TIER_NAME,
|
||||||
|
"endpoint": cold.url.as_str(),
|
||||||
|
"accessKey": cold.access_key.as_str(),
|
||||||
|
"secretKey": cold.secret_key.as_str(),
|
||||||
|
"bucket": TIER_BUCKET,
|
||||||
|
"prefix": TIER_PREFIX,
|
||||||
|
"region": "us-east-1",
|
||||||
|
"storageClass": ""
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.to_string();
|
||||||
|
|
||||||
|
let (status, resp) = admin_request(
|
||||||
|
&hot.url,
|
||||||
|
http::Method::PUT,
|
||||||
|
"/rustfs/admin/v3/tier",
|
||||||
|
Some(body),
|
||||||
|
&hot.access_key,
|
||||||
|
&hot.secret_key,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
if !status.is_success() {
|
||||||
|
return Err(format!("AddTier(RustFS) failed: status={status}, body={resp}").into());
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Poll `HEAD` until the object's storage class is the tier name (transition
|
||||||
|
/// complete), or fail after `deadline`. (From `reliant/tiering.rs`.)
|
||||||
|
async fn wait_for_transition(client: &Client, bucket: &str, key: &str, deadline: StdDuration) -> TestResult {
|
||||||
|
let start = Instant::now();
|
||||||
|
loop {
|
||||||
|
let head = client.head_object().bucket(bucket).key(key).send().await?;
|
||||||
|
if head.storage_class().map(|sc| sc.as_str()) == Some(TIER_NAME) {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
if start.elapsed() >= deadline {
|
||||||
|
return Err(format!(
|
||||||
|
"object {bucket}/{key} was not transitioned to {TIER_NAME} within {}s (storage_class={:?}); \
|
||||||
|
SSE key-policy enforcement may have started blocking the transition worker's internal reads",
|
||||||
|
deadline.as_secs(),
|
||||||
|
head.storage_class()
|
||||||
|
)
|
||||||
|
.into());
|
||||||
|
}
|
||||||
|
tokio::time::sleep(StdDuration::from_millis(500)).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Poll `HEAD` until `x-amz-restore` reports a finished restore
|
||||||
|
/// (`ongoing-request="false"`), or fail after `deadline`.
|
||||||
|
async fn wait_for_restore_complete(client: &Client, bucket: &str, key: &str, deadline: StdDuration) -> TestResult {
|
||||||
|
let start = Instant::now();
|
||||||
|
loop {
|
||||||
|
let head = client.head_object().bucket(bucket).key(key).send().await?;
|
||||||
|
if head.restore().is_some_and(|r| r.contains("ongoing-request=\"false\"")) {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
if start.elapsed() >= deadline {
|
||||||
|
return Err(format!(
|
||||||
|
"object {bucket}/{key} restore did not complete within {}s (restore={:?}); \
|
||||||
|
SSE key-policy enforcement may have started blocking the restore copy-back's internal reads",
|
||||||
|
deadline.as_secs(),
|
||||||
|
head.restore()
|
||||||
|
)
|
||||||
|
.into());
|
||||||
|
}
|
||||||
|
tokio::time::sleep(StdDuration::from_millis(500)).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ILM expiration keeps working on an SSE-KMS bucket while per-key SSE
|
||||||
|
/// authorization is enforced.
|
||||||
|
///
|
||||||
|
/// The lifecycle scanner deletes expired objects with an internal (no-principal)
|
||||||
|
/// identity that holds no `kms` grant. If enforcement ever starts applying to
|
||||||
|
/// those internal deletes (or to the scanner's metadata reads) on encrypted
|
||||||
|
/// buckets, expiry stops happening and this test times out.
|
||||||
|
///
|
||||||
|
/// A survivor object under a non-matching prefix isolates the rule's prefix
|
||||||
|
/// filter as the cause of the deletion and proves the encrypted bucket stays
|
||||||
|
/// readable end to end after the scanner has run.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn ilm_expiration_on_sse_kms_bucket_under_enforcement() -> TestResult {
|
||||||
|
init_logging();
|
||||||
|
|
||||||
|
let mut env = LocalKMSTestEnvironment::new().await?;
|
||||||
|
start_enforcing_ilm_server(&mut env).await?;
|
||||||
|
env.base_env.create_test_bucket(EXPIRY_BUCKET).await?;
|
||||||
|
|
||||||
|
let client = env.base_env.create_s3_client();
|
||||||
|
set_bucket_default_sse_kms(&client, EXPIRY_BUCKET).await?;
|
||||||
|
|
||||||
|
for key in [EXPIRE_KEY, SURVIVOR_KEY] {
|
||||||
|
client
|
||||||
|
.put_object()
|
||||||
|
.bucket(EXPIRY_BUCKET)
|
||||||
|
.key(key)
|
||||||
|
.body(ByteStream::from_static(PAYLOAD))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_head_sse_kms(&client, EXPIRY_BUCKET, key).await?;
|
||||||
|
}
|
||||||
|
info!("both objects stored SSE-KMS encrypted under enforcement");
|
||||||
|
|
||||||
|
put_expiration_rule(&client, EXPIRY_BUCKET, "kms-ilm-expire", "expire/", 1).await?;
|
||||||
|
|
||||||
|
// The regression this pins: the scanner's internal delete must stay exempt
|
||||||
|
// from per-key SSE authorization, so the encrypted object actually expires.
|
||||||
|
wait_for_object_expired(&client, EXPIRY_BUCKET, EXPIRE_KEY, ILM_DEADLINE).await?;
|
||||||
|
info!("SSE-KMS object expired by the lifecycle scanner under enforcement");
|
||||||
|
|
||||||
|
// Negative control: same bucket, same encryption, non-matching prefix. It
|
||||||
|
// must survive the scanner and still decrypt for the requesting principal.
|
||||||
|
assert!(
|
||||||
|
!object_is_gone(&client, EXPIRY_BUCKET, SURVIVOR_KEY).await?,
|
||||||
|
"non-matching-prefix object must not be expired by a prefix-scoped rule"
|
||||||
|
);
|
||||||
|
let survivor = client.get_object().bucket(EXPIRY_BUCKET).key(SURVIVOR_KEY).send().await?;
|
||||||
|
assert_eq!(
|
||||||
|
survivor.body.collect().await?.into_bytes().as_ref(),
|
||||||
|
PAYLOAD,
|
||||||
|
"surviving SSE-KMS object must still decrypt after the scanner has run"
|
||||||
|
);
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ILM transition to a remote tier keeps working on an SSE-KMS bucket while
|
||||||
|
/// per-key SSE authorization is enforced, and the transitioned object reads
|
||||||
|
/// back as plaintext.
|
||||||
|
///
|
||||||
|
/// The transition worker moves the stored (encrypted) bytes to the cold tier
|
||||||
|
/// with an internal (no-principal) identity; the read-through `GET` then
|
||||||
|
/// decrypts the envelope for the requesting principal. If enforcement ever
|
||||||
|
/// starts applying to the worker's internal reads, the transition wait times
|
||||||
|
/// out; if the stored envelope is mishandled across the tier round trip, the
|
||||||
|
/// plaintext comparison fails.
|
||||||
|
///
|
||||||
|
/// The transition is driven through the manual transition-run admin endpoint
|
||||||
|
/// (the mechanism `reliant/tiering.rs` established), so the test does not
|
||||||
|
/// depend on scanner scheduling; the 1s scanner cycle stays on as a backstop.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn ilm_transition_on_sse_kms_bucket_under_enforcement_reads_back() -> TestResult {
|
||||||
|
init_logging();
|
||||||
|
|
||||||
|
// Cold-tier server: independent credentials, its own Local KMS holding the
|
||||||
|
// same key id (see the module docs for why the tier target needs KMS).
|
||||||
|
// Started first; each server's startup cleanup only matches its own unique
|
||||||
|
// address and temp dir, so the two instances coexist.
|
||||||
|
let mut cold = LocalKMSTestEnvironment::new().await?;
|
||||||
|
cold.base_env.access_key = "kmscoldtieradmin".to_string();
|
||||||
|
cold.base_env.secret_key = "kmscoldtiersecret".to_string();
|
||||||
|
start_cold_tier_kms_server(&mut cold).await?;
|
||||||
|
let cold_client = cold.base_env.create_s3_client();
|
||||||
|
cold_client.create_bucket().bucket(TIER_BUCKET).send().await?;
|
||||||
|
|
||||||
|
// Hot server: Local KMS + enforcement + accelerated lifecycle clock.
|
||||||
|
let mut env = LocalKMSTestEnvironment::new().await?;
|
||||||
|
start_enforcing_ilm_server(&mut env).await?;
|
||||||
|
let hot_client = env.base_env.create_s3_client();
|
||||||
|
|
||||||
|
add_rustfs_tier(&env.base_env, &cold.base_env).await?;
|
||||||
|
|
||||||
|
env.base_env.create_test_bucket(TRANSITION_BUCKET).await?;
|
||||||
|
set_bucket_default_sse_kms(&hot_client, TRANSITION_BUCKET).await?;
|
||||||
|
|
||||||
|
hot_client
|
||||||
|
.put_object()
|
||||||
|
.bucket(TRANSITION_BUCKET)
|
||||||
|
.key(TRANSITION_KEY)
|
||||||
|
.body(ByteStream::from_static(PAYLOAD))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_head_sse_kms(&hot_client, TRANSITION_BUCKET, TRANSITION_KEY).await?;
|
||||||
|
info!("object stored SSE-KMS encrypted under enforcement");
|
||||||
|
|
||||||
|
// Days=1 is due ~2s after the write with RUSTFS_ILM_DEBUG_DAY_SECS=2.
|
||||||
|
put_transition_rule(&hot_client, TRANSITION_BUCKET, "kms-ilm-transition", "tier/", 1).await?;
|
||||||
|
|
||||||
|
// Drive the transition deterministically via the manual run endpoint, then
|
||||||
|
// wait for HEAD to report the tier as the object's storage class.
|
||||||
|
run_manual_transition_until_processed(&env.base_env, TRANSITION_BUCKET, "tier/", ILM_DEADLINE).await?;
|
||||||
|
wait_for_transition(&hot_client, TRANSITION_BUCKET, TRANSITION_KEY, ILM_DEADLINE).await?;
|
||||||
|
info!("SSE-KMS object transitioned to the remote tier under enforcement");
|
||||||
|
|
||||||
|
let head = hot_client
|
||||||
|
.head_object()
|
||||||
|
.bucket(TRANSITION_BUCKET)
|
||||||
|
.key(TRANSITION_KEY)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert!(
|
||||||
|
head.restore().is_none(),
|
||||||
|
"a freshly transitioned object must not advertise x-amz-restore, got {:?}",
|
||||||
|
head.restore()
|
||||||
|
);
|
||||||
|
|
||||||
|
// The remote copy exists on the cold tier. The payload the tier holds is the
|
||||||
|
// hot server's stored ciphertext, wrapped once more under the cold server's
|
||||||
|
// own managed SSE-KMS layer (the forwarded headers re-request encryption).
|
||||||
|
let remote = cold_client.list_objects_v2().bucket(TIER_BUCKET).send().await?;
|
||||||
|
assert!(!remote.contents().is_empty(), "cold-tier bucket must hold the transitioned object's data");
|
||||||
|
|
||||||
|
// Read-through GET under enforcement must succeed (not AccessDenied) and
|
||||||
|
// keep advertising SSE-KMS. Its BODY is deliberately not compared here:
|
||||||
|
// the transitioned read path skips managed-SSE decryption — a product gap
|
||||||
|
// unrelated to enforcement — so a direct GET streams the stored ciphertext
|
||||||
|
// (`new_getobjectreader` in crates/ecstore/src/client/object_api_utils.rs
|
||||||
|
// hardcodes `is_encrypted = false` and never applies the
|
||||||
|
// `ReadTransform::Encrypted` wrapping the hot-read path builds in
|
||||||
|
// crates/ecstore/src/object_api/readers.rs). Plaintext recovery is pinned
|
||||||
|
// through restore semantics below; when the read-through gap is fixed, a
|
||||||
|
// byte assertion can be added here too.
|
||||||
|
let read_through = hot_client
|
||||||
|
.get_object()
|
||||||
|
.bucket(TRANSITION_BUCKET)
|
||||||
|
.key(TRANSITION_KEY)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
read_through.server_side_encryption(),
|
||||||
|
Some(&ServerSideEncryption::AwsKms),
|
||||||
|
"transitioned object must still report SSE-KMS on read-through"
|
||||||
|
);
|
||||||
|
let read_through_body = read_through.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(
|
||||||
|
read_through_body.len(),
|
||||||
|
PAYLOAD.len(),
|
||||||
|
"read-through GET must stream the object's full logical size under enforcement"
|
||||||
|
);
|
||||||
|
|
||||||
|
// RestoreObject copies the ciphertext back from the tier under the original
|
||||||
|
// envelope metadata; the restored copy is then served by the normal
|
||||||
|
// decrypting read path. The copy-back runs with an internal (no-principal)
|
||||||
|
// identity, so this also pins the exemption on the restore path. Days=300
|
||||||
|
// because RUSTFS_ILM_DEBUG_DAY_SECS=2 accelerates the restored copy's
|
||||||
|
// expiry as well (300 accelerated days == 600s of validity).
|
||||||
|
hot_client
|
||||||
|
.restore_object()
|
||||||
|
.bucket(TRANSITION_BUCKET)
|
||||||
|
.key(TRANSITION_KEY)
|
||||||
|
.restore_request(RestoreRequest::builder().days(300).build())
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
wait_for_restore_complete(&hot_client, TRANSITION_BUCKET, TRANSITION_KEY, ILM_DEADLINE).await?;
|
||||||
|
info!("SSE-KMS object restored from the remote tier under enforcement");
|
||||||
|
|
||||||
|
// The KMS-relevant half: the restored envelope decrypts back to the exact
|
||||||
|
// plaintext for the requesting principal.
|
||||||
|
let restored = hot_client
|
||||||
|
.get_object()
|
||||||
|
.bucket(TRANSITION_BUCKET)
|
||||||
|
.key(TRANSITION_KEY)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(
|
||||||
|
restored.server_side_encryption(),
|
||||||
|
Some(&ServerSideEncryption::AwsKms),
|
||||||
|
"restored object must still report SSE-KMS"
|
||||||
|
);
|
||||||
|
let body = restored.body.collect().await?.into_bytes();
|
||||||
|
assert_eq!(body.as_ref(), PAYLOAD, "restored SSE-KMS object must round-trip byte-identical plaintext");
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
@@ -59,3 +59,6 @@ mod configured_roundtrip_test;
|
|||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod kms_authorization_negative_matrix_test;
|
mod kms_authorization_negative_matrix_test;
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod kms_ilm_sse_kms_test;
|
||||||
|
|||||||
@@ -39,6 +39,10 @@ pub mod fault_proxy;
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod reliability_disk_fault_test;
|
mod reliability_disk_fault_test;
|
||||||
|
|
||||||
|
// Privileged Linux-only 3x4 replacement rebuild proof for rustfs#5869/#1791.
|
||||||
|
#[cfg(all(test, target_os = "linux"))]
|
||||||
|
mod replacement_privileged_e2e_test;
|
||||||
|
|
||||||
// dist-13 (backlog#1150/#1155): e2e regression net proving a large-object
|
// dist-13 (backlog#1150/#1155): e2e regression net proving a large-object
|
||||||
// degraded EC read never returns a silently truncated body (rustfs#4594/#4560/#4585).
|
// degraded EC read never returns a silently truncated body (rustfs#4594/#4560/#4585).
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -2854,7 +2854,7 @@ pub(crate) mod cmptst_30 {
|
|||||||
result
|
result
|
||||||
}
|
}
|
||||||
|
|
||||||
#[ignore]
|
#[ignore = "timing-sensitive backend-pressure latency probe; run explicitly with --ignored"]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn regression() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
async fn regression() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
crate::common::init_logging();
|
crate::common::init_logging();
|
||||||
|
|||||||
@@ -252,6 +252,7 @@ impl QuotaTestEnv {
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod integration_tests {
|
mod integration_tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial]
|
#[serial]
|
||||||
@@ -963,9 +964,27 @@ mod integration_tests {
|
|||||||
.send()
|
.send()
|
||||||
.await;
|
.await;
|
||||||
|
|
||||||
assert!(complete_result.is_err());
|
let complete_error = complete_result.expect_err("multipart completion above quota must be rejected");
|
||||||
|
assert_eq!(complete_error.as_service_error().and_then(|error| error.code()), Some("InvalidRequest"));
|
||||||
assert!(!env.object_exists("over_quota.txt").await?);
|
assert!(!env.object_exists("over_quota.txt").await?);
|
||||||
|
|
||||||
|
let staged_parts = env
|
||||||
|
.client
|
||||||
|
.list_parts()
|
||||||
|
.bucket(&env.bucket_name)
|
||||||
|
.key("over_quota.txt")
|
||||||
|
.upload_id(upload_id2)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(staged_parts.parts().len(), 2, "quota rejection must preserve the multipart upload");
|
||||||
|
env.client
|
||||||
|
.abort_multipart_upload()
|
||||||
|
.bucket(&env.bucket_name)
|
||||||
|
.key("over_quota.txt")
|
||||||
|
.upload_id(upload_id2)
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
|
||||||
env.cleanup_bucket().await?;
|
env.cleanup_bucket().await?;
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
|
|||||||
@@ -349,11 +349,32 @@ mod tests {
|
|||||||
.send()
|
.send()
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
|
let first_inline = client
|
||||||
|
.put_object()
|
||||||
|
.bucket(bucket)
|
||||||
|
.key("versions/inline.bin")
|
||||||
|
.body(ByteStream::from(payload(8 * 1024, 40)))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let first_inline_version = first_inline
|
||||||
|
.version_id()
|
||||||
|
.ok_or("first inline PUT did not return a version ID")?;
|
||||||
|
let second_inline = client
|
||||||
|
.put_object()
|
||||||
|
.bucket(bucket)
|
||||||
|
.key("versions/inline.bin")
|
||||||
|
.body(ByteStream::from(payload(8 * 1024, 41)))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
let second_inline_version = second_inline
|
||||||
|
.version_id()
|
||||||
|
.ok_or("second inline PUT did not return a version ID")?;
|
||||||
|
|
||||||
let first = client
|
let first = client
|
||||||
.put_object()
|
.put_object()
|
||||||
.bucket(bucket)
|
.bucket(bucket)
|
||||||
.key(key)
|
.key(key)
|
||||||
.body(ByteStream::from(payload(256 * 1024, 41)))
|
.body(ByteStream::from(payload(128 * 1024, 41)))
|
||||||
.send()
|
.send()
|
||||||
.await?;
|
.await?;
|
||||||
let first_version = first.version_id().ok_or("first PUT did not return a version ID")?;
|
let first_version = first.version_id().ok_or("first PUT did not return a version ID")?;
|
||||||
@@ -361,16 +382,36 @@ mod tests {
|
|||||||
.put_object()
|
.put_object()
|
||||||
.bucket(bucket)
|
.bucket(bucket)
|
||||||
.key(key)
|
.key(key)
|
||||||
.body(ByteStream::from(payload(256 * 1024, 42)))
|
.body(ByteStream::from(payload(3 * 1024 * 1024, 42)))
|
||||||
.send()
|
.send()
|
||||||
.await?;
|
.await?;
|
||||||
let second_version = second.version_id().ok_or("second PUT did not return a version ID")?;
|
let second_version = second.version_id().ok_or("second PUT did not return a version ID")?;
|
||||||
let delete = client.delete_object().bucket(bucket).key(key).send().await?;
|
let delete = client.delete_object().bucket(bucket).key(key).send().await?;
|
||||||
let delete_version = delete.version_id().ok_or("delete marker did not return a version ID")?;
|
let delete_version = delete.version_id().ok_or("delete marker did not return a version ID")?;
|
||||||
|
|
||||||
|
let first_inline_census = harness.census_object_version(0, bucket, "versions/inline.bin", Some(first_inline_version))?;
|
||||||
|
let second_inline_census =
|
||||||
|
harness.census_object_version(0, bucket, "versions/inline.bin", Some(second_inline_version))?;
|
||||||
let first_census = harness.census_object_version(0, bucket, key, Some(first_version))?;
|
let first_census = harness.census_object_version(0, bucket, key, Some(first_version))?;
|
||||||
|
let first_other_disk_census = harness.census_object_version(1, bucket, key, Some(first_version))?;
|
||||||
let second_census = harness.census_object_version(0, bucket, key, Some(second_version))?;
|
let second_census = harness.census_object_version(0, bucket, key, Some(second_version))?;
|
||||||
let delete_census = harness.census_object_version(0, bucket, key, Some(delete_version))?;
|
let delete_census = harness.census_object_version(0, bucket, key, Some(delete_version))?;
|
||||||
|
assert!(
|
||||||
|
first_inline_census.is_complete() && second_inline_census.is_complete(),
|
||||||
|
"inline version physical census is incomplete: first={first_inline_census:?} second={second_inline_census:?}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
first_inline_census.present_part_fingerprints.is_empty() && second_inline_census.present_part_fingerprints.is_empty(),
|
||||||
|
"inline versions must not select external shard files: first={first_inline_census:?} second={second_inline_census:?}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
first_inline_census.inline_data_fingerprint.is_some() && second_inline_census.inline_data_fingerprint.is_some(),
|
||||||
|
"inline versions must fingerprint payload bytes stored in xl.meta"
|
||||||
|
);
|
||||||
|
assert_ne!(
|
||||||
|
first_inline_census.inline_data_fingerprint, second_inline_census.inline_data_fingerprint,
|
||||||
|
"same-size inline versions with different payloads must retain distinct xl.meta fingerprints"
|
||||||
|
);
|
||||||
assert!(
|
assert!(
|
||||||
first_census.is_complete(),
|
first_census.is_complete(),
|
||||||
"first version physical census is incomplete: {first_census:?}"
|
"first version physical census is incomplete: {first_census:?}"
|
||||||
@@ -379,6 +420,14 @@ mod tests {
|
|||||||
second_census.is_complete(),
|
second_census.is_complete(),
|
||||||
"second version physical census is incomplete: {second_census:?}"
|
"second version physical census is incomplete: {second_census:?}"
|
||||||
);
|
);
|
||||||
|
assert!(
|
||||||
|
first_other_disk_census.is_complete(),
|
||||||
|
"first version physical census on the second disk is incomplete: {first_other_disk_census:?}"
|
||||||
|
);
|
||||||
|
assert_ne!(
|
||||||
|
first_census.erasure_index, first_other_disk_census.erasure_index,
|
||||||
|
"physical census must preserve each disk's erasure index"
|
||||||
|
);
|
||||||
assert_ne!(
|
assert_ne!(
|
||||||
first_census.data_dir, second_census.data_dir,
|
first_census.data_dir, second_census.data_dir,
|
||||||
"distinct object versions must select distinct physical data directories"
|
"distinct object versions must select distinct physical data directories"
|
||||||
@@ -387,6 +436,24 @@ mod tests {
|
|||||||
first_census.expected_part_numbers, second_census.expected_part_numbers,
|
first_census.expected_part_numbers, second_census.expected_part_numbers,
|
||||||
"same single-part shape should expose the same part numbers"
|
"same single-part shape should expose the same part numbers"
|
||||||
);
|
);
|
||||||
|
let first_part = first_census
|
||||||
|
.present_part_fingerprints
|
||||||
|
.values()
|
||||||
|
.next()
|
||||||
|
.ok_or("first version did not expose a physical part fingerprint")?;
|
||||||
|
let second_part = second_census
|
||||||
|
.present_part_fingerprints
|
||||||
|
.values()
|
||||||
|
.next()
|
||||||
|
.ok_or("second version did not expose a physical part fingerprint")?;
|
||||||
|
assert_ne!(
|
||||||
|
first_part.size, second_part.size,
|
||||||
|
"different shard lengths must retain their physical sizes"
|
||||||
|
);
|
||||||
|
assert_ne!(
|
||||||
|
first_part.sha256, second_part.sha256,
|
||||||
|
"different shard contents must retain their physical hashes"
|
||||||
|
);
|
||||||
assert!(
|
assert!(
|
||||||
delete_census.is_complete(),
|
delete_census.is_complete(),
|
||||||
"delete marker physical census is incomplete: {delete_census:?}"
|
"delete marker physical census is incomplete: {delete_census:?}"
|
||||||
@@ -396,7 +463,7 @@ mod tests {
|
|||||||
"delete marker must not declare object shards: {delete_census:?}"
|
"delete marker must not declare object shards: {delete_census:?}"
|
||||||
);
|
);
|
||||||
assert!(
|
assert!(
|
||||||
delete_census.present_part_numbers.is_empty(),
|
delete_census.present_part_fingerprints.is_empty(),
|
||||||
"delete marker must not select stale object shards: {delete_census:?}"
|
"delete marker must not select stale object shards: {delete_census:?}"
|
||||||
);
|
);
|
||||||
Ok(())
|
Ok(())
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -2401,15 +2401,20 @@ async fn wait_for_site_replication_info<F>(
|
|||||||
where
|
where
|
||||||
F: Fn(&SiteReplicationInfo) -> bool,
|
F: Fn(&SiteReplicationInfo) -> bool,
|
||||||
{
|
{
|
||||||
for _ in 0..40 {
|
// 30s to match wait_for_replication_state: the three-node site tests run
|
||||||
|
// several full rustfs processes on one runner, so peer-state propagation
|
||||||
|
// can take well over 10s under CI load.
|
||||||
|
let deadline = tokio::time::Instant::now() + Duration::from_secs(30);
|
||||||
|
loop {
|
||||||
let info = site_replication_info(env).await?;
|
let info = site_replication_info(env).await?;
|
||||||
if predicate(&info) {
|
if predicate(&info) {
|
||||||
return Ok(info);
|
return Ok(info);
|
||||||
}
|
}
|
||||||
|
if tokio::time::Instant::now() >= deadline {
|
||||||
|
return Err(format!("site replication info did not reach expected state on {}", env.address).into());
|
||||||
|
}
|
||||||
sleep(Duration::from_millis(250)).await;
|
sleep(Duration::from_millis(250)).await;
|
||||||
}
|
}
|
||||||
|
|
||||||
Err(format!("site replication info did not reach expected state on {}", env.address).into())
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn wait_for_site_replication_status<F>(
|
async fn wait_for_site_replication_status<F>(
|
||||||
@@ -2420,15 +2425,19 @@ async fn wait_for_site_replication_status<F>(
|
|||||||
where
|
where
|
||||||
F: Fn(&SRStatusInfo) -> bool,
|
F: Fn(&SRStatusInfo) -> bool,
|
||||||
{
|
{
|
||||||
for _ in 0..40 {
|
// Same 30s ceiling as wait_for_site_replication_info: the status probes
|
||||||
|
// fan out to every peer, so they see the same multi-process CI load.
|
||||||
|
let deadline = tokio::time::Instant::now() + Duration::from_secs(30);
|
||||||
|
loop {
|
||||||
let status = site_replication_status(env, query).await?;
|
let status = site_replication_status(env, query).await?;
|
||||||
if predicate(&status) {
|
if predicate(&status) {
|
||||||
return Ok(status);
|
return Ok(status);
|
||||||
}
|
}
|
||||||
|
if tokio::time::Instant::now() >= deadline {
|
||||||
|
return Err(format!("site replication status did not reach expected state on {}", env.address).into());
|
||||||
|
}
|
||||||
sleep(Duration::from_millis(250)).await;
|
sleep(Duration::from_millis(250)).await;
|
||||||
}
|
}
|
||||||
|
|
||||||
Err(format!("site replication status did not reach expected state on {}", env.address).into())
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn wait_for_replication_reset_target<F>(
|
async fn wait_for_replication_reset_target<F>(
|
||||||
@@ -4235,37 +4244,49 @@ async fn test_bucket_replication_acceptance_matrix_local_dual_targets() -> TestR
|
|||||||
"tag rule with disabled delete-marker replication created a marker: {tagged_state:?}"
|
"tag rule with disabled delete-marker replication created a marker: {tagged_state:?}"
|
||||||
);
|
);
|
||||||
|
|
||||||
set_bucket_versioning(&source_env, source_bucket, BucketVersioningStatus::Suspended).await?;
|
// AWS S3 and MinIO both reject suspending versioning on a bucket that
|
||||||
set_bucket_versioning(&target_env_a, target_bucket_a, BucketVersioningStatus::Suspended).await?;
|
// carries a replication configuration (InvalidBucketState): suspension
|
||||||
let null_put = source_client
|
// would mint null versions that versioned replication can never converge.
|
||||||
|
let suspend_err = source_client
|
||||||
|
.put_bucket_versioning()
|
||||||
|
.bucket(source_bucket)
|
||||||
|
.versioning_configuration(
|
||||||
|
VersioningConfiguration::builder()
|
||||||
|
.status(BucketVersioningStatus::Suspended)
|
||||||
|
.build(),
|
||||||
|
)
|
||||||
|
.send()
|
||||||
|
.await
|
||||||
|
.expect_err("suspending versioning on a replication source must be rejected");
|
||||||
|
assert_eq!(
|
||||||
|
suspend_err.as_service_error().and_then(|error| error.code()),
|
||||||
|
Some("InvalidBucketState"),
|
||||||
|
"suspension on a replication source must fail with InvalidBucketState: {suspend_err:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// The rejected suspension must leave the versioning + replication state
|
||||||
|
// fully intact: a fresh matched PUT still replicates with a real version.
|
||||||
|
let post_reject_put = source_client
|
||||||
.put_object()
|
.put_object()
|
||||||
.bucket(source_bucket)
|
.bucket(source_bucket)
|
||||||
.key("prefix/null.txt")
|
.key("prefix/after-rejected-suspend.txt")
|
||||||
.body(ByteStream::from_static(b"null version"))
|
.body(ByteStream::from_static(b"still replicating"))
|
||||||
.send()
|
.send()
|
||||||
.await?;
|
.await?;
|
||||||
assert!(null_put.version_id().is_none(), "suspended source PUT must create a null version");
|
let post_reject_version_id = post_reject_put
|
||||||
wait_for_replication_state(&target_client_a, target_bucket_a, "null version did not replicate", |state| {
|
.version_id()
|
||||||
|
.ok_or("PUT after rejected suspension omitted version ID")?
|
||||||
|
.to_string();
|
||||||
|
wait_for_replication_state(
|
||||||
|
&target_client_a,
|
||||||
|
target_bucket_a,
|
||||||
|
"replication stopped after rejected versioning suspension",
|
||||||
|
|state| {
|
||||||
state
|
state
|
||||||
.iter()
|
.iter()
|
||||||
.any(|entry| entry.key == "prefix/null.txt" && entry.version_id == "null" && !entry.delete_marker)
|
.any(|entry| entry.key == "prefix/after-rejected-suspend.txt" && entry.version_id == post_reject_version_id)
|
||||||
})
|
},
|
||||||
.await?;
|
)
|
||||||
let null_delete = source_client
|
|
||||||
.delete_object()
|
|
||||||
.bucket(source_bucket)
|
|
||||||
.key("prefix/null.txt")
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
assert!(
|
|
||||||
null_delete.version_id().is_none(),
|
|
||||||
"suspended source DELETE must create a null delete marker"
|
|
||||||
);
|
|
||||||
wait_for_replication_state(&target_client_a, target_bucket_a, "null delete marker did not replicate", |state| {
|
|
||||||
state
|
|
||||||
.iter()
|
|
||||||
.any(|entry| entry.key == "prefix/null.txt" && entry.version_id == "null" && entry.delete_marker)
|
|
||||||
})
|
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
|
|||||||
@@ -32,6 +32,11 @@ workspace = true
|
|||||||
|
|
||||||
[features]
|
[features]
|
||||||
default = []
|
default = []
|
||||||
|
# Compiles the controlled list-objects namespace-journal chaos injector into a
|
||||||
|
# production binary (it is always available to tests). Off by default so the
|
||||||
|
# RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_* env vars cannot rewrite journal
|
||||||
|
# state in a stock build (backlog#1832).
|
||||||
|
list-chaos = []
|
||||||
rio-v2 = ["dep:rustfs-rio-v2"]
|
rio-v2 = ["dep:rustfs-rio-v2"]
|
||||||
hotpath = [
|
hotpath = [
|
||||||
"hotpath/hotpath",
|
"hotpath/hotpath",
|
||||||
|
|||||||
@@ -69,6 +69,7 @@ fn build_non_inline_writers(config: &BenchConfig) -> Vec<Option<BitrotWriterWrap
|
|||||||
fn bench_single_block_non_inline_fast_path(c: &mut Criterion) {
|
fn bench_single_block_non_inline_fast_path(c: &mut Criterion) {
|
||||||
let configs = vec![
|
let configs = vec![
|
||||||
BenchConfig::new(4 * 1024, 4, 2, 128 * 1024),
|
BenchConfig::new(4 * 1024, 4, 2, 128 * 1024),
|
||||||
|
BenchConfig::new(16 * 1024, 4, 2, 128 * 1024),
|
||||||
BenchConfig::new(64 * 1024, 4, 2, 128 * 1024),
|
BenchConfig::new(64 * 1024, 4, 2, 128 * 1024),
|
||||||
BenchConfig::new(128 * 1024, 4, 2, 128 * 1024),
|
BenchConfig::new(128 * 1024, 4, 2, 128 * 1024),
|
||||||
];
|
];
|
||||||
@@ -112,7 +113,12 @@ fn bench_single_block_non_inline_fast_path(c: &mut Criterion) {
|
|||||||
rt.block_on(async {
|
rt.block_on(async {
|
||||||
erasure
|
erasure
|
||||||
.clone()
|
.clone()
|
||||||
.encode_single_block_non_inline(reader, &mut writers, config.data_shards)
|
.encode_single_block_non_inline_with_size_hint(
|
||||||
|
reader,
|
||||||
|
&mut writers,
|
||||||
|
config.data_shards,
|
||||||
|
config.payload_size,
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.expect("single block candidate benchmark");
|
.expect("single block candidate benchmark");
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -61,9 +61,11 @@ pub mod bucket {
|
|||||||
delete_manual_transition_scope_admission_if_current, load_manual_transition_job_record,
|
delete_manual_transition_scope_admission_if_current, load_manual_transition_job_record,
|
||||||
load_manual_transition_job_record_with_etag, load_manual_transition_scope_admission,
|
load_manual_transition_job_record_with_etag, load_manual_transition_scope_admission,
|
||||||
manual_transition_job_lease_expired, manual_transition_scope_admission_lease_expired,
|
manual_transition_job_lease_expired, manual_transition_scope_admission_lease_expired,
|
||||||
manual_transition_scope_key, persist_manual_transition_job_progress, renew_manual_transition_job_lease,
|
manual_transition_scope_key, persist_manual_transition_job_progress,
|
||||||
request_manual_transition_job_cancel, save_manual_transition_job_record,
|
persist_manual_transition_job_progress_if_owned, renew_manual_transition_job_lease,
|
||||||
save_manual_transition_job_record_if_current, save_manual_transition_scope_admission_if_absent,
|
renew_manual_transition_job_lease_if_owned, request_manual_transition_job_cancel,
|
||||||
|
save_manual_transition_job_record, save_manual_transition_job_record_if_current,
|
||||||
|
save_manual_transition_scope_admission_if_absent, update_manual_transition_job_record,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -129,16 +131,19 @@ pub mod bucket {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub mod metadata_sys {
|
pub mod metadata_sys {
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
pub use crate::bucket::metadata_sys::ConfigWriteLockProbe;
|
||||||
pub use crate::bucket::metadata_sys::{
|
pub use crate::bucket::metadata_sys::{
|
||||||
BucketMetadataMutationGuard, BucketMetadataSys, ObjectLockConfigState, acquire_bucket_metadata_transaction_lock,
|
BucketMetadataMutationGuard, BucketMetadataSys, ObjectLockConfigState, acquire_bucket_metadata_transaction_lock,
|
||||||
capture_bucket_metadata_incarnation, delete, delete_if_incarnation, get, get_accelerate_config, get_bucket_policy,
|
acquire_bucket_metadata_transaction_lock_for_incarnation, capture_bucket_metadata_incarnation, delete,
|
||||||
|
delete_if_incarnation, delete_under_transaction_lock, get, get_accelerate_config, get_bucket_policy,
|
||||||
get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk, get_cors_config, get_durability_config,
|
get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk, get_cors_config, get_durability_config,
|
||||||
get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config, get_notification_config,
|
get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config, get_notification_config,
|
||||||
get_object_lock_config, get_object_lock_config_state, get_public_access_block_config, get_quota_config,
|
get_object_lock_config, get_object_lock_config_state, get_public_access_block_config, get_quota_config,
|
||||||
get_replication_config, get_request_payment_config, get_sse_config, get_tagging_config, get_versioning_config,
|
get_replication_config, get_request_payment_config, get_sse_config, get_tagging_config, get_versioning_config,
|
||||||
get_website_config, init_bucket_metadata_sys, list_bucket_targets, reload_bucket_metadata, remove_bucket_metadata,
|
get_website_config, init_bucket_metadata_sys, list_bucket_targets, reload_bucket_metadata, remove_bucket_metadata,
|
||||||
set_bucket_metadata, update, update_bucket_targets_under_transaction_lock, update_config_with, update_if_incarnation,
|
set_bucket_metadata, update, update_bucket_targets_under_transaction_lock, update_config_with, update_if_incarnation,
|
||||||
update_under_transaction_lock,
|
update_quota_if_incarnation, update_under_transaction_lock,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -180,17 +185,18 @@ pub mod bucket {
|
|||||||
mrf_backlog_observability_snapshot,
|
mrf_backlog_observability_snapshot,
|
||||||
};
|
};
|
||||||
pub use crate::bucket::replication::{
|
pub use crate::bucket::replication::{
|
||||||
BucketReplicationResyncStatus, BucketReplicationStats, BucketStats, DeleteReplicationConfigSnapshot,
|
BucketReplicationResyncStatus, BucketReplicationStat, BucketReplicationStats, BucketStats,
|
||||||
DeletedObjectReplicationInfo, DurableMrfBacklog, DynReplicationPool, MrfOpKind, MrfReplicateEntry,
|
DeleteReplicationConfigSnapshot, DeletedObjectReplicationInfo, DurableMrfBacklog, DynReplicationPool, InQueueMetric,
|
||||||
MustReplicateOptions, ObjectOpts, REMOTE_TARGET_CAPABILITY_CONTRACT_VERSION, REMOTE_TARGET_UNSUPPORTED_FIELDS,
|
MrfOpKind, MrfReplicateEntry, MustReplicateOptions, ObjectOpts, REMOTE_TARGET_CAPABILITY_CONTRACT_VERSION,
|
||||||
REMOTE_TARGET_WRITABLE_FIELDS, REPLICATE_INCOMING_DELETE, REPLICATION_CAPABILITY_CONTRACT_VERSION,
|
REMOTE_TARGET_UNSUPPORTED_FIELDS, REMOTE_TARGET_WRITABLE_FIELDS, REPLICATE_INCOMING_DELETE,
|
||||||
REPLICATION_READ_ONLY_HISTORICAL_FIELDS, REPLICATION_WRITABLE_FIELDS, ReplicateDecision, ReplicateObjectInfo,
|
REPLICATION_CAPABILITY_CONTRACT_VERSION, REPLICATION_READ_ONLY_HISTORICAL_FIELDS, REPLICATION_WRITABLE_FIELDS,
|
||||||
ReplicationBatchAdmission, ReplicationConfig, ReplicationConfigStructureError, ReplicationConfigurationExt,
|
ReplicateDecision, ReplicateObjectInfo, ReplicationBatchAdmission, ReplicationConfig,
|
||||||
ReplicationDeleteScheduleInput, ReplicationDeleteStateSource, ReplicationHealQueueResult, ReplicationObjectBridge,
|
ReplicationConfigStructureError, ReplicationConfigurationExt, ReplicationDeleteScheduleInput,
|
||||||
ReplicationObjectIO, ReplicationOperation, ReplicationPoolTrait, ReplicationPriority, ReplicationQueueAdmission,
|
ReplicationDeleteStateSource, ReplicationHealQueueResult, ReplicationObjectBridge, ReplicationObjectIO,
|
||||||
ReplicationScannerBridge, ReplicationState, ReplicationStats, ReplicationStatusType, ReplicationStorage,
|
ReplicationOperation, ReplicationPoolTrait, ReplicationPriority, ReplicationQueueAdmission, ReplicationScannerBridge,
|
||||||
ReplicationTargetValidationError, ReplicationType, ResyncOpts, ResyncStatusType, RuntimeReplicationTargetBacklog,
|
ReplicationState, ReplicationStats, ReplicationStatusType, ReplicationStorage, ReplicationTargetValidationError,
|
||||||
TargetReplicationResyncStatus, VersionPurgeStatusType, commit_force_delete_intent, complete_force_delete_intent,
|
ReplicationType, ResyncOpts, ResyncStatusType, RuntimeReplicationTargetBacklog, TargetReplicationResyncStatus,
|
||||||
|
VersionPurgeStatusType, XferStats, commit_force_delete_intent, complete_force_delete_intent,
|
||||||
delete_replication_state_from_config, delete_replication_version_id, get_global_replication_pool,
|
delete_replication_state_from_config, delete_replication_version_id, get_global_replication_pool,
|
||||||
get_global_replication_stats, init_background_replication, invalid_replication_config_status_field,
|
get_global_replication_stats, init_background_replication, invalid_replication_config_status_field,
|
||||||
persist_force_delete_intent, read_durable_mrf_backlog, replication_state_to_filemeta, replication_status_to_filemeta,
|
persist_force_delete_intent, read_durable_mrf_backlog, replication_state_to_filemeta, replication_status_to_filemeta,
|
||||||
@@ -274,7 +280,9 @@ pub mod cluster {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub mod compression {
|
pub mod compression {
|
||||||
pub use crate::io_support::compress::{MIN_DISK_COMPRESSIBLE_SIZE, is_disk_compressible, is_disk_compression_enabled};
|
pub use crate::io_support::compress::{
|
||||||
|
MIN_DISK_COMPRESSIBLE_SIZE, is_disk_compressible, is_disk_compression_enabled, is_multipart_disk_compression_enabled,
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
pub mod config {
|
pub mod config {
|
||||||
@@ -308,11 +316,13 @@ pub mod config {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub mod data_usage {
|
pub mod data_usage {
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
pub use crate::data_usage::seed_bucket_usage_memory_for_test;
|
||||||
pub use crate::data_usage::{
|
pub use crate::data_usage::{
|
||||||
DATA_USAGE_CACHE_NAME, apply_bucket_usage_memory_overlay, compute_bucket_usage,
|
DATA_USAGE_CACHE_NAME, apply_bucket_usage_memory_overlay, compute_bucket_usage,
|
||||||
init_compression_total_memory_from_backend, invalidate_admin_data_usage_snapshot_cache,
|
init_compression_total_memory_from_backend, invalidate_admin_data_usage_snapshot_cache,
|
||||||
invalidate_data_usage_snapshot_cache, live_bucket_usage_computations, load_admin_data_usage_from_backend_cached,
|
invalidate_data_usage_snapshot_cache, live_bucket_usage_computations, load_admin_data_usage_from_backend_cached,
|
||||||
load_compression_total_from_memory, load_data_usage_from_backend, load_data_usage_from_backend_cached,
|
load_compression_total_from_memory, load_data_usage_from_backend, load_data_usage_from_backend_cached, quota_object_size,
|
||||||
record_bucket_delete_marker_memory, record_bucket_object_delete_memory, record_bucket_object_version_write_memory,
|
record_bucket_delete_marker_memory, record_bucket_object_delete_memory, record_bucket_object_version_write_memory,
|
||||||
record_bucket_object_write_memory, record_bucket_object_write_unknown_previous_memory, record_compression_total_memory,
|
record_bucket_object_write_memory, record_bucket_object_write_unknown_previous_memory, record_compression_total_memory,
|
||||||
refresh_bucket_usage_from_object_layer, refresh_versioned_bucket_usage_from_object_layer,
|
refresh_bucket_usage_from_object_layer, refresh_versioned_bucket_usage_from_object_layer,
|
||||||
@@ -342,7 +352,7 @@ pub mod disk {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub mod error {
|
pub mod error {
|
||||||
pub use crate::disk::error::{BitrotErrorType, DiskError, Error, FileAccessDeniedWithContext, Result};
|
pub use crate::disk::error::{DiskError, Error, FileAccessDeniedWithContext, Result};
|
||||||
}
|
}
|
||||||
|
|
||||||
pub mod error_reduce {
|
pub mod error_reduce {
|
||||||
@@ -399,8 +409,11 @@ pub mod metrics {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub mod notification {
|
pub mod notification {
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
pub use crate::services::notification_sys::rotate_cross_pool_fence_fleet_proof_for_test;
|
||||||
pub use crate::services::notification_sys::{
|
pub use crate::services::notification_sys::{
|
||||||
NotificationPeerErr, NotificationSys, get_global_notification_sys, new_global_notification_sys,
|
CrossPoolFenceFleetProofToken, NotificationPeerErr, NotificationSys, acquire_cross_pool_fence_fleet_proof,
|
||||||
|
cross_pool_fence_fleet_proof_matches, get_global_notification_sys, new_global_notification_sys,
|
||||||
start_remote_version_state_fleet_probe,
|
start_remote_version_state_fleet_probe,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -409,10 +422,10 @@ pub mod object {
|
|||||||
pub use crate::object_api::{
|
pub use crate::object_api::{
|
||||||
BLOCK_SIZE_V2, ERASURE_ALGORITHM, EncryptionResolutionError, EncryptionResolutionErrorKind, GetObjectBodyCacheHook,
|
BLOCK_SIZE_V2, ERASURE_ALGORITHM, EncryptionResolutionError, EncryptionResolutionErrorKind, GetObjectBodyCacheHook,
|
||||||
GetObjectBodyCacheHookLookup, GetObjectBodySource, GetObjectReader, NamespaceLockFence, ObjectEncryptionResolver,
|
GetObjectBodyCacheHookLookup, GetObjectBodySource, GetObjectReader, NamespaceLockFence, ObjectEncryptionResolver,
|
||||||
ObjectInfo, ObjectLockConfigSnapshot, ObjectMutationHook, ObjectOptions, PutObjReader, RangedDecompressReader,
|
ObjectInfo, ObjectLockConfigSnapshot, ObjectMutationHook, ObjectOptions, PutObjReader, QuotaAdmission,
|
||||||
ReadEncryptionMaterial, ReadEncryptionMode, ReadEncryptionRequest, StreamConsumer, get_object_body_cache_plaintext_len,
|
RangedDecompressReader, ReadEncryptionMaterial, ReadEncryptionMode, ReadEncryptionRequest, StreamConsumer,
|
||||||
lookup_get_object_body_cache_hook, register_get_object_body_cache_hook, register_object_mutation_hook,
|
get_object_body_cache_plaintext_len, lookup_get_object_body_cache_hook, register_get_object_body_cache_hook,
|
||||||
unregister_get_object_body_cache_hook, unregister_object_mutation_hook,
|
register_object_mutation_hook, unregister_get_object_body_cache_hook, unregister_object_mutation_hook,
|
||||||
};
|
};
|
||||||
pub use crate::store::{
|
pub use crate::store::{
|
||||||
PrepareSelectObjectSnapshotError, PreparedGetObjectReader, SelectObjectSnapshot, SelectObjectSnapshotReadError,
|
PrepareSelectObjectSnapshotError, PreparedGetObjectReader, SelectObjectSnapshot, SelectObjectSnapshotReadError,
|
||||||
@@ -460,7 +473,8 @@ pub mod set_disk {
|
|||||||
|
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
pub mod test_util {
|
pub mod test_util {
|
||||||
pub use crate::set_disk::{PutObjectCommitBarrier, PutObjectCommitPause};
|
pub use crate::bucket::quota::reservation::fail_next_quota_ledger_save_for_test;
|
||||||
|
pub use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause, PutObjectCommitBarrier, PutObjectCommitPause};
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -58,7 +58,9 @@ use rustfs_utils::http::{
|
|||||||
};
|
};
|
||||||
use rustfs_utils::http::{
|
use rustfs_utils::http::{
|
||||||
SUFFIX_FORCE_DELETE, SUFFIX_SOURCE_DELETEMARKER, SUFFIX_SOURCE_ETAG, SUFFIX_SOURCE_MTIME, SUFFIX_SOURCE_REPLICATION_CHECK,
|
SUFFIX_FORCE_DELETE, SUFFIX_SOURCE_DELETEMARKER, SUFFIX_SOURCE_ETAG, SUFFIX_SOURCE_MTIME, SUFFIX_SOURCE_REPLICATION_CHECK,
|
||||||
SUFFIX_SOURCE_REPLICATION_REQUEST, SUFFIX_SOURCE_VERSION_ID, insert_header,
|
SUFFIX_SOURCE_REPLICATION_LEGALHOLD_TIMESTAMP, SUFFIX_SOURCE_REPLICATION_REQUEST,
|
||||||
|
SUFFIX_SOURCE_REPLICATION_RETENTION_TIMESTAMP, SUFFIX_SOURCE_REPLICATION_TAGGING_TIMESTAMP, SUFFIX_SOURCE_VERSION_ID,
|
||||||
|
insert_header,
|
||||||
};
|
};
|
||||||
use rustls_pki_types::pem::PemObject;
|
use rustls_pki_types::pem::PemObject;
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
@@ -1476,9 +1478,12 @@ impl Default for AdvancedPutOptions {
|
|||||||
replication_status: ReplicationStatusType::Pending,
|
replication_status: ReplicationStatusType::Pending,
|
||||||
source_mtime: OffsetDateTime::now_utc(),
|
source_mtime: OffsetDateTime::now_utc(),
|
||||||
replication_request: false,
|
replication_request: false,
|
||||||
retention_timestamp: OffsetDateTime::now_utc(),
|
// UNIX_EPOCH means "never modified": header() must not emit a
|
||||||
tagging_timestamp: OffsetDateTime::now_utc(),
|
// timestamp header for it, otherwise a receiver would treat an
|
||||||
legalhold_timestamp: OffsetDateTime::now_utc(),
|
// unset category as a modification made right now.
|
||||||
|
retention_timestamp: OffsetDateTime::UNIX_EPOCH,
|
||||||
|
tagging_timestamp: OffsetDateTime::UNIX_EPOCH,
|
||||||
|
legalhold_timestamp: OffsetDateTime::UNIX_EPOCH,
|
||||||
replication_validity_check: false,
|
replication_validity_check: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1675,6 +1680,16 @@ impl PutObjectOptions {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
for (suffix, timestamp) in [
|
||||||
|
(SUFFIX_SOURCE_REPLICATION_TAGGING_TIMESTAMP, self.internal.tagging_timestamp),
|
||||||
|
(SUFFIX_SOURCE_REPLICATION_RETENTION_TIMESTAMP, self.internal.retention_timestamp),
|
||||||
|
(SUFFIX_SOURCE_REPLICATION_LEGALHOLD_TIMESTAMP, self.internal.legalhold_timestamp),
|
||||||
|
] {
|
||||||
|
if timestamp.unix_timestamp() != 0 {
|
||||||
|
insert_header(&mut header, suffix, timestamp.format(&Rfc3339).unwrap_or_default());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if self.internal.replication_request {
|
if self.internal.replication_request {
|
||||||
insert_header(&mut header, SUFFIX_SOURCE_REPLICATION_REQUEST, "true");
|
insert_header(&mut header, SUFFIX_SOURCE_REPLICATION_REQUEST, "true");
|
||||||
}
|
}
|
||||||
@@ -2842,6 +2857,57 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn put_object_headers_carry_replication_timestamp_headers() {
|
||||||
|
// MinIO receivers resolve concurrent tag/retention/legal-hold edits by
|
||||||
|
// last-writer-wins on these headers (object-api-options.go parses them
|
||||||
|
// as RFC3339); a replica without them loses every conflict resolution.
|
||||||
|
let mut opts = PutObjectOptions::default();
|
||||||
|
opts.internal.replication_request = true;
|
||||||
|
let tagging = OffsetDateTime::from_unix_timestamp(1_700_000_001).expect("valid timestamp");
|
||||||
|
let retention = OffsetDateTime::from_unix_timestamp(1_700_000_002).expect("valid timestamp");
|
||||||
|
let legalhold = OffsetDateTime::from_unix_timestamp(1_700_000_003).expect("valid timestamp");
|
||||||
|
opts.internal.tagging_timestamp = tagging;
|
||||||
|
opts.internal.retention_timestamp = retention;
|
||||||
|
opts.internal.legalhold_timestamp = legalhold;
|
||||||
|
|
||||||
|
let header = opts.header();
|
||||||
|
for (suffix, expected) in [
|
||||||
|
("source-replication-tagging-timestamp", tagging),
|
||||||
|
("source-replication-retention-timestamp", retention),
|
||||||
|
("source-replication-legalhold-timestamp", legalhold),
|
||||||
|
] {
|
||||||
|
assert_eq!(
|
||||||
|
rustfs_utils::http::get_header(&header, suffix).as_deref(),
|
||||||
|
Some(expected.format(&Rfc3339).expect("RFC3339 timestamp").as_str()),
|
||||||
|
"replication put requests must carry the {suffix} header"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn put_object_headers_omit_unset_replication_timestamps() {
|
||||||
|
// UNIX_EPOCH means "never modified on the source"; sending it would
|
||||||
|
// make the receiver treat an unset category as a fresh modification.
|
||||||
|
let mut opts = PutObjectOptions::default();
|
||||||
|
opts.internal.replication_request = true;
|
||||||
|
opts.internal.tagging_timestamp = OffsetDateTime::UNIX_EPOCH;
|
||||||
|
opts.internal.retention_timestamp = OffsetDateTime::UNIX_EPOCH;
|
||||||
|
opts.internal.legalhold_timestamp = OffsetDateTime::UNIX_EPOCH;
|
||||||
|
|
||||||
|
let header = opts.header();
|
||||||
|
for suffix in [
|
||||||
|
"source-replication-tagging-timestamp",
|
||||||
|
"source-replication-retention-timestamp",
|
||||||
|
"source-replication-legalhold-timestamp",
|
||||||
|
] {
|
||||||
|
assert!(
|
||||||
|
rustfs_utils::http::get_header(&header, suffix).is_none(),
|
||||||
|
"unset {suffix} must not be sent to replication targets"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn get_remote_target_client_internal_rejects_loopback_endpoint() {
|
async fn get_remote_target_client_internal_rejects_loopback_endpoint() {
|
||||||
let sys = BucketTargetSys::default();
|
let sys = BucketTargetSys::default();
|
||||||
|
|||||||
@@ -64,10 +64,41 @@ impl BucketDurabilityConfig {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Default durability tier seeded into a newly created bucket's metadata
|
||||||
|
/// (rustfs/backlog#1811). `relaxed` aligns new buckets with MinIO's default
|
||||||
|
/// posture: object data is still fdatasynced, while xl.meta and directory-entry
|
||||||
|
/// fsyncs follow the relaxed durability gate.
|
||||||
|
pub const ENV_NEW_BUCKET_DURABILITY_MODE: &str = "RUSTFS_NEW_BUCKET_DURABILITY_MODE";
|
||||||
|
pub const DEFAULT_NEW_BUCKET_DURABILITY_MODE: &str = BUCKET_DURABILITY_MODE_RELAXED;
|
||||||
|
|
||||||
|
/// The `durability.json` bytes to seed into a freshly created bucket's metadata.
|
||||||
|
/// Empty means "no override" (the bucket then follows the global
|
||||||
|
/// `RUSTFS_DURABILITY_MODE`); otherwise the serialized chosen tier. Operators
|
||||||
|
/// can set `inherit` to disable the new-bucket override. Invalid values also
|
||||||
|
/// fail closed to inherit the global mode instead of seeding a surprising tier.
|
||||||
|
pub fn new_bucket_durability_config_json() -> Vec<u8> {
|
||||||
|
let raw = std::env::var(ENV_NEW_BUCKET_DURABILITY_MODE).unwrap_or_else(|_| DEFAULT_NEW_BUCKET_DURABILITY_MODE.to_string());
|
||||||
|
let mode = raw.trim();
|
||||||
|
if mode.eq_ignore_ascii_case("inherit") || mode.is_empty() || !BucketDurabilityConfig::is_valid_mode(mode) {
|
||||||
|
return Vec::new();
|
||||||
|
}
|
||||||
|
serde_json::to_vec(&BucketDurabilityConfig::new(mode)).expect("BucketDurabilityConfig serialization cannot fail")
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
fn new_bucket_seeded_mode() -> Option<String> {
|
||||||
|
let json = new_bucket_durability_config_json();
|
||||||
|
if json.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
serde_json::from_slice::<BucketDurabilityConfig>(&json)
|
||||||
|
.expect("new-bucket durability config must serialize")
|
||||||
|
.normalized_mode()
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn valid_modes_are_recognized() {
|
fn valid_modes_are_recognized() {
|
||||||
assert!(BucketDurabilityConfig::is_valid_mode("strict"));
|
assert!(BucketDurabilityConfig::is_valid_mode("strict"));
|
||||||
@@ -99,4 +130,33 @@ mod tests {
|
|||||||
let empty: BucketDurabilityConfig = serde_json::from_slice(b"{}").expect("deserialize empty");
|
let empty: BucketDurabilityConfig = serde_json::from_slice(b"{}").expect("deserialize empty");
|
||||||
assert_eq!(empty.normalized_mode(), None);
|
assert_eq!(empty.normalized_mode(), None);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_bucket_default_seeds_relaxed_when_unset() {
|
||||||
|
temp_env::with_var_unset(ENV_NEW_BUCKET_DURABILITY_MODE, || {
|
||||||
|
assert_eq!(new_bucket_seeded_mode().as_deref(), Some(BUCKET_DURABILITY_MODE_RELAXED));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_bucket_default_honors_explicit_tiers() {
|
||||||
|
for mode in [
|
||||||
|
BUCKET_DURABILITY_MODE_STRICT,
|
||||||
|
BUCKET_DURABILITY_MODE_RELAXED,
|
||||||
|
BUCKET_DURABILITY_MODE_NONE,
|
||||||
|
] {
|
||||||
|
temp_env::with_var(ENV_NEW_BUCKET_DURABILITY_MODE, Some(mode), || {
|
||||||
|
assert_eq!(new_bucket_seeded_mode().as_deref(), Some(mode));
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_bucket_default_can_inherit_global_mode() {
|
||||||
|
for mode in ["inherit", "", "bogus"] {
|
||||||
|
temp_env::with_var(ENV_NEW_BUCKET_DURABILITY_MODE, Some(mode), || {
|
||||||
|
assert_eq!(new_bucket_seeded_mode(), None);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -86,6 +86,21 @@ where
|
|||||||
com::save_config_with_opts(api, file, data, opts).await
|
com::save_config_with_opts(api, file, data, opts).await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn save_config_with_opts_quiet<S>(api: Arc<S>, file: &str, data: Vec<u8>, opts: &ObjectOptions) -> Result<()>
|
||||||
|
where
|
||||||
|
S: ObjectIO<
|
||||||
|
Error = Error,
|
||||||
|
RangeSpec = HTTPRangeSpec,
|
||||||
|
HeaderMap = HeaderMap,
|
||||||
|
ObjectOptions = ObjectOptions,
|
||||||
|
ObjectInfo = ObjectInfo,
|
||||||
|
GetObjectReader = GetObjectReader,
|
||||||
|
PutObjectReader = PutObjReader,
|
||||||
|
>,
|
||||||
|
{
|
||||||
|
com::save_config_with_opts_quiet(api, file, data, opts).await
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn delete_config<S>(api: Arc<S>, file: &str) -> Result<()>
|
pub(crate) async fn delete_config<S>(api: Arc<S>, file: &str) -> Result<()>
|
||||||
where
|
where
|
||||||
S: ObjectOperations<
|
S: ObjectOperations<
|
||||||
|
|||||||
@@ -45,6 +45,104 @@ const MANUAL_TRANSITION_JOB_LEASE_SECONDS: i128 = 60;
|
|||||||
const MANUAL_TRANSITION_LEGACY_SCOPE_SCAN_LIMIT: i32 = 1000;
|
const MANUAL_TRANSITION_LEGACY_SCOPE_SCAN_LIMIT: i32 = 1000;
|
||||||
const MANUAL_TRANSITION_TASK_SCAN_LIMIT: i32 = 1000;
|
const MANUAL_TRANSITION_TASK_SCAN_LIMIT: i32 = 1000;
|
||||||
const MANUAL_TRANSITION_WORKER_RESULT_SCAN_LIMIT: i32 = 1000;
|
const MANUAL_TRANSITION_WORKER_RESULT_SCAN_LIMIT: i32 = 1000;
|
||||||
|
const MANUAL_TRANSITION_JOB_CAS_RETRIES: usize = 4;
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
struct ManualTransitionJobCasBarrierState {
|
||||||
|
job_id: Uuid,
|
||||||
|
paused: std::sync::atomic::AtomicBool,
|
||||||
|
arrived: tokio::sync::Notify,
|
||||||
|
release: tokio::sync::Semaphore,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
pub(crate) struct ManualTransitionJobCasBarrier {
|
||||||
|
state: Arc<ManualTransitionJobCasBarrierState>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
static MANUAL_TRANSITION_JOB_CAS_BARRIER: std::sync::OnceLock<std::sync::Mutex<Option<Arc<ManualTransitionJobCasBarrierState>>>> =
|
||||||
|
std::sync::OnceLock::new();
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
impl ManualTransitionJobCasBarrier {
|
||||||
|
pub(crate) fn install(job_id: Uuid) -> Self {
|
||||||
|
let state = Arc::new(ManualTransitionJobCasBarrierState {
|
||||||
|
job_id,
|
||||||
|
paused: std::sync::atomic::AtomicBool::new(false),
|
||||||
|
arrived: tokio::sync::Notify::new(),
|
||||||
|
release: tokio::sync::Semaphore::new(0),
|
||||||
|
});
|
||||||
|
let mut slot = MANUAL_TRANSITION_JOB_CAS_BARRIER
|
||||||
|
.get_or_init(|| std::sync::Mutex::new(None))
|
||||||
|
.lock()
|
||||||
|
.expect("manual transition progress CAS barrier mutex should not poison");
|
||||||
|
assert!(
|
||||||
|
slot.is_none(),
|
||||||
|
"manual transition job CAS barrier must be installed by one test at a time"
|
||||||
|
);
|
||||||
|
*slot = Some(Arc::clone(&state));
|
||||||
|
drop(slot);
|
||||||
|
Self { state }
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn wait_until_paused(&self) {
|
||||||
|
tokio::time::timeout(std::time::Duration::from_secs(30), async {
|
||||||
|
loop {
|
||||||
|
let arrived = self.state.arrived.notified();
|
||||||
|
if self.state.paused.load(std::sync::atomic::Ordering::Acquire) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
arrived.await;
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.expect("manual transition job update should reach the deterministic CAS barrier");
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn release(&self) {
|
||||||
|
self.state.release.add_permits(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
impl Drop for ManualTransitionJobCasBarrier {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
self.release();
|
||||||
|
let mut slot = MANUAL_TRANSITION_JOB_CAS_BARRIER
|
||||||
|
.get_or_init(|| std::sync::Mutex::new(None))
|
||||||
|
.lock()
|
||||||
|
.expect("manual transition progress CAS barrier mutex should not poison");
|
||||||
|
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
|
||||||
|
*slot = None;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
async fn pause_manual_transition_job_before_first_cas(job_id: Uuid) {
|
||||||
|
let barrier = MANUAL_TRANSITION_JOB_CAS_BARRIER
|
||||||
|
.get_or_init(|| std::sync::Mutex::new(None))
|
||||||
|
.lock()
|
||||||
|
.expect("manual transition progress CAS barrier mutex should not poison")
|
||||||
|
.as_ref()
|
||||||
|
.filter(|barrier| barrier.job_id == job_id)
|
||||||
|
.cloned();
|
||||||
|
if let Some(barrier) = barrier
|
||||||
|
&& barrier
|
||||||
|
.paused
|
||||||
|
.compare_exchange(false, true, std::sync::atomic::Ordering::AcqRel, std::sync::atomic::Ordering::Acquire)
|
||||||
|
.is_ok()
|
||||||
|
{
|
||||||
|
barrier.arrived.notify_one();
|
||||||
|
barrier
|
||||||
|
.release
|
||||||
|
.acquire()
|
||||||
|
.await
|
||||||
|
.expect("manual transition job CAS barrier should remain open")
|
||||||
|
.forget();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn is_false(value: &bool) -> bool {
|
fn is_false(value: &bool) -> bool {
|
||||||
!*value
|
!*value
|
||||||
@@ -148,7 +246,6 @@ impl ManualTransitionJobRecord {
|
|||||||
|
|
||||||
pub fn fail(&mut self, error: impl Into<String>) {
|
pub fn fail(&mut self, error: impl Into<String>) {
|
||||||
self.state = ManualTransitionJobState::Failed;
|
self.state = ManualTransitionJobState::Failed;
|
||||||
self.report.tier_failure = self.report.tier_failure.saturating_add(1);
|
|
||||||
self.error = Some(error.into());
|
self.error = Some(error.into());
|
||||||
self.mark_updated_terminal();
|
self.mark_updated_terminal();
|
||||||
}
|
}
|
||||||
@@ -1040,7 +1137,7 @@ pub async fn save_manual_transition_job_record_if_current(
|
|||||||
}
|
}
|
||||||
let object = manual_transition_job_record_object_name(job.job_id).map_err(manual_transition_job_store_error)?;
|
let object = manual_transition_job_record_object_name(job.job_id).map_err(manual_transition_job_store_error)?;
|
||||||
let data = job.encode().map_err(manual_transition_job_store_error)?;
|
let data = job.encode().map_err(manual_transition_job_store_error)?;
|
||||||
config_boundary::save_config_with_opts(
|
config_boundary::save_config_with_opts_quiet(
|
||||||
api,
|
api,
|
||||||
&object,
|
&object,
|
||||||
data,
|
data,
|
||||||
@@ -1056,6 +1153,54 @@ pub async fn save_manual_transition_job_record_if_current(
|
|||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Applies a job-record mutation with optimistic concurrency control.
|
||||||
|
///
|
||||||
|
/// The mutation returns whether the record needs to be persisted. When a lease
|
||||||
|
/// is supplied, ownership is checked again after every conflicting write.
|
||||||
|
pub async fn update_manual_transition_job_record<F>(
|
||||||
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Option<Uuid>,
|
||||||
|
update: F,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord>
|
||||||
|
where
|
||||||
|
F: FnMut(&mut ManualTransitionJobRecord) -> bool,
|
||||||
|
{
|
||||||
|
update_manual_transition_job_record_from(api, job_id, expected_lease_id, None, update).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn update_manual_transition_job_record_from<F>(
|
||||||
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Option<Uuid>,
|
||||||
|
mut current: Option<(ManualTransitionJobRecord, String)>,
|
||||||
|
mut update: F,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord>
|
||||||
|
where
|
||||||
|
F: FnMut(&mut ManualTransitionJobRecord) -> bool,
|
||||||
|
{
|
||||||
|
for _ in 0..MANUAL_TRANSITION_JOB_CAS_RETRIES {
|
||||||
|
let (mut record, etag) = match current.take() {
|
||||||
|
Some(current) => current,
|
||||||
|
None => load_manual_transition_job_record_with_etag(api.clone(), job_id).await?,
|
||||||
|
};
|
||||||
|
if expected_lease_id.is_some_and(|lease_id| record.lease_id != lease_id) {
|
||||||
|
return Err(Error::PreconditionFailed);
|
||||||
|
}
|
||||||
|
if !update(&mut record) {
|
||||||
|
return Ok(record);
|
||||||
|
}
|
||||||
|
#[cfg(test)]
|
||||||
|
pause_manual_transition_job_before_first_cas(job_id).await;
|
||||||
|
match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
|
||||||
|
Ok(()) => return Ok(record),
|
||||||
|
Err(Error::PreconditionFailed) => continue,
|
||||||
|
Err(err) => return Err(err),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(Error::PreconditionFailed)
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn save_manual_transition_worker_result_if_absent(
|
pub(crate) async fn save_manual_transition_worker_result_if_absent(
|
||||||
api: Arc<ECStore>,
|
api: Arc<ECStore>,
|
||||||
record: &ManualTransitionWorkerResultRecord,
|
record: &ManualTransitionWorkerResultRecord,
|
||||||
@@ -1314,99 +1459,113 @@ pub async fn reconcile_manual_transition_worker_results(
|
|||||||
api: Arc<ECStore>,
|
api: Arc<ECStore>,
|
||||||
job_id: Uuid,
|
job_id: Uuid,
|
||||||
queue_snapshot: ManualTransitionQueueSnapshot,
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
|
reconcile_manual_transition_worker_results_inner(api, job_id, None, queue_snapshot, false).await
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn reconcile_manual_transition_worker_results_if_owned(
|
||||||
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Uuid,
|
||||||
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
|
reconcile_manual_transition_worker_results_inner(api, job_id, Some(expected_lease_id), queue_snapshot, false).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn reconcile_manual_transition_worker_results_inner(
|
||||||
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Option<Uuid>,
|
||||||
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
|
mark_missing_results_unknown: bool,
|
||||||
) -> EcstoreResult<ManualTransitionJobRecord> {
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
let task_stats = match scan_manual_transition_task_journal(api.clone(), job_id).await? {
|
let task_stats = match scan_manual_transition_task_journal(api.clone(), job_id).await? {
|
||||||
ManualTransitionTaskJournal::Stats(stats) => stats,
|
ManualTransitionTaskJournal::Stats(stats) => stats,
|
||||||
ManualTransitionTaskJournal::Corrupt(error) => {
|
ManualTransitionTaskJournal::Corrupt(error) => {
|
||||||
return mark_manual_transition_job_unknown_for_task_journal_error(api, job_id, error, queue_snapshot).await;
|
return mark_manual_transition_job_unknown_for_task_journal_error(
|
||||||
|
api,
|
||||||
|
job_id,
|
||||||
|
expected_lease_id,
|
||||||
|
error,
|
||||||
|
queue_snapshot,
|
||||||
|
)
|
||||||
|
.await;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
let stats = match scan_manual_transition_worker_result_journal(api.clone(), job_id).await? {
|
let stats = match scan_manual_transition_worker_result_journal(api.clone(), job_id).await? {
|
||||||
ManualTransitionWorkerResultJournal::Stats(stats) => stats,
|
ManualTransitionWorkerResultJournal::Stats(stats) => stats,
|
||||||
ManualTransitionWorkerResultJournal::Corrupt(error) => {
|
ManualTransitionWorkerResultJournal::Corrupt(error) => {
|
||||||
return mark_manual_transition_job_unknown_for_worker_result_journal_error(api, job_id, error, queue_snapshot).await;
|
return mark_manual_transition_job_unknown_for_worker_result_journal_error(
|
||||||
|
api,
|
||||||
|
job_id,
|
||||||
|
expected_lease_id,
|
||||||
|
error,
|
||||||
|
queue_snapshot,
|
||||||
|
)
|
||||||
|
.await;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
for _ in 0..4 {
|
let mut changed = false;
|
||||||
let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
let record = update_manual_transition_job_record(api.clone(), job_id, expected_lease_id, |record| {
|
||||||
let changed = record.apply_worker_result_counts(
|
let counts_changed = record.apply_worker_result_counts(
|
||||||
stats.stats.completed,
|
stats.stats.completed,
|
||||||
stats.stats.failed,
|
stats.stats.failed,
|
||||||
&stats.stats.tier_failure_by_reason,
|
&stats.stats.tier_failure_by_reason,
|
||||||
task_stats.queued,
|
task_stats.queued,
|
||||||
queue_snapshot,
|
queue_snapshot,
|
||||||
);
|
);
|
||||||
|
let became_unknown = mark_missing_results_unknown && record.mark_unknown_if_worker_results_lost(queue_snapshot);
|
||||||
|
changed = counts_changed || became_unknown;
|
||||||
|
changed
|
||||||
|
})
|
||||||
|
.await?;
|
||||||
if !changed {
|
if !changed {
|
||||||
return Ok(record);
|
return Ok(record);
|
||||||
}
|
}
|
||||||
match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
|
|
||||||
Ok(()) => {
|
|
||||||
if record.is_terminal() {
|
if record.is_terminal() {
|
||||||
delete_manual_transition_scope_admission_if_current(
|
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?;
|
||||||
api.clone(),
|
|
||||||
&record.scope_key,
|
|
||||||
record.job_id,
|
|
||||||
record.lease_id,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
} else {
|
} else {
|
||||||
renew_manual_transition_scope_admission_from_job(api, &record).await?;
|
renew_manual_transition_scope_admission_from_job(api, &record).await?;
|
||||||
}
|
}
|
||||||
return Ok(record);
|
Ok(record)
|
||||||
}
|
|
||||||
Err(Error::PreconditionFailed) => continue,
|
|
||||||
Err(err) => return Err(err),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(Error::PreconditionFailed)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn mark_manual_transition_job_unknown_for_task_journal_error(
|
async fn mark_manual_transition_job_unknown_for_task_journal_error(
|
||||||
api: Arc<ECStore>,
|
api: Arc<ECStore>,
|
||||||
job_id: Uuid,
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Option<Uuid>,
|
||||||
error: String,
|
error: String,
|
||||||
queue_snapshot: ManualTransitionQueueSnapshot,
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
) -> EcstoreResult<ManualTransitionJobRecord> {
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
for _ in 0..4 {
|
let mut changed = false;
|
||||||
let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
let record = update_manual_transition_job_record(api.clone(), job_id, expected_lease_id, |record| {
|
||||||
if !record.mark_unknown_for_task_journal_error(error.clone(), queue_snapshot) {
|
changed = record.mark_unknown_for_task_journal_error(error.clone(), queue_snapshot);
|
||||||
return Ok(record);
|
changed
|
||||||
}
|
})
|
||||||
match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
|
|
||||||
Ok(()) => {
|
|
||||||
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id)
|
|
||||||
.await?;
|
.await?;
|
||||||
return Ok(record);
|
if changed && record.is_terminal() {
|
||||||
|
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?;
|
||||||
}
|
}
|
||||||
Err(Error::PreconditionFailed) => continue,
|
Ok(record)
|
||||||
Err(err) => return Err(err),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(Error::PreconditionFailed)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn mark_manual_transition_job_unknown_for_worker_result_journal_error(
|
async fn mark_manual_transition_job_unknown_for_worker_result_journal_error(
|
||||||
api: Arc<ECStore>,
|
api: Arc<ECStore>,
|
||||||
job_id: Uuid,
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Option<Uuid>,
|
||||||
error: String,
|
error: String,
|
||||||
queue_snapshot: ManualTransitionQueueSnapshot,
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
) -> EcstoreResult<ManualTransitionJobRecord> {
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
for _ in 0..4 {
|
let mut changed = false;
|
||||||
let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
let record = update_manual_transition_job_record(api.clone(), job_id, expected_lease_id, |record| {
|
||||||
if !record.mark_unknown_for_worker_result_journal_error(error.clone(), queue_snapshot) {
|
changed = record.mark_unknown_for_worker_result_journal_error(error.clone(), queue_snapshot);
|
||||||
return Ok(record);
|
changed
|
||||||
}
|
})
|
||||||
match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
|
|
||||||
Ok(()) => {
|
|
||||||
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id)
|
|
||||||
.await?;
|
.await?;
|
||||||
return Ok(record);
|
if changed && record.is_terminal() {
|
||||||
|
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?;
|
||||||
}
|
}
|
||||||
Err(Error::PreconditionFailed) => continue,
|
Ok(record)
|
||||||
Err(err) => return Err(err),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(Error::PreconditionFailed)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn save_manual_transition_scope_admission_if_absent(
|
pub async fn save_manual_transition_scope_admission_if_absent(
|
||||||
@@ -1603,19 +1762,14 @@ async fn find_active_legacy_manual_transition_scope_conflict(
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub async fn request_manual_transition_job_cancel(api: Arc<ECStore>, job_id: Uuid) -> EcstoreResult<ManualTransitionJobRecord> {
|
pub async fn request_manual_transition_job_cancel(api: Arc<ECStore>, job_id: Uuid) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
for _ in 0..4 {
|
update_manual_transition_job_record(api, job_id, None, |record| {
|
||||||
let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
|
||||||
if record.is_terminal() || record.cancel_requested {
|
if record.is_terminal() || record.cancel_requested {
|
||||||
return Ok(record);
|
return false;
|
||||||
}
|
}
|
||||||
record.mark_cancel_requested();
|
record.mark_cancel_requested();
|
||||||
match save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await {
|
true
|
||||||
Ok(()) => return Ok(record),
|
})
|
||||||
Err(Error::PreconditionFailed) => continue,
|
.await
|
||||||
Err(err) => return Err(err),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Err(Error::PreconditionFailed)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn persist_manual_transition_job_progress(
|
pub async fn persist_manual_transition_job_progress(
|
||||||
@@ -1624,10 +1778,39 @@ pub async fn persist_manual_transition_job_progress(
|
|||||||
report: &ManualTransitionRunReport,
|
report: &ManualTransitionRunReport,
|
||||||
queue_snapshot: ManualTransitionQueueSnapshot,
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
) -> EcstoreResult<ManualTransitionJobRecord> {
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
let (mut record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
let current = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
||||||
|
persist_manual_transition_job_progress_inner(api, job_id, current.0.lease_id, Some(current), report, queue_snapshot).await
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn persist_manual_transition_job_progress_if_owned(
|
||||||
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Uuid,
|
||||||
|
report: &ManualTransitionRunReport,
|
||||||
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
|
persist_manual_transition_job_progress_inner(api, job_id, expected_lease_id, None, report, queue_snapshot).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn persist_manual_transition_job_progress_inner(
|
||||||
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Uuid,
|
||||||
|
current: Option<(ManualTransitionJobRecord, String)>,
|
||||||
|
report: &ManualTransitionRunReport,
|
||||||
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
|
let record = update_manual_transition_job_record_from(api.clone(), job_id, Some(expected_lease_id), current, |record| {
|
||||||
|
if record.state != ManualTransitionJobState::Running {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
record.update_running_progress(report.clone(), queue_snapshot);
|
record.update_running_progress(report.clone(), queue_snapshot);
|
||||||
save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await?;
|
true
|
||||||
|
})
|
||||||
|
.await?;
|
||||||
|
if record.state == ManualTransitionJobState::Running {
|
||||||
renew_manual_transition_scope_admission_from_job(api, &record).await?;
|
renew_manual_transition_scope_admission_from_job(api, &record).await?;
|
||||||
|
}
|
||||||
Ok(record)
|
Ok(record)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1661,26 +1844,59 @@ pub async fn renew_manual_transition_job_lease(
|
|||||||
job_id: Uuid,
|
job_id: Uuid,
|
||||||
queue_snapshot: ManualTransitionQueueSnapshot,
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
) -> EcstoreResult<ManualTransitionJobRecord> {
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
let (mut record, mut etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
let current = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
||||||
if record.state == ManualTransitionJobState::Running {
|
renew_manual_transition_job_lease_inner(api, job_id, current.0.lease_id, Some(current), queue_snapshot).await
|
||||||
if record.scan_completed && queue_snapshot.queued == 0 && queue_snapshot.active == 0 {
|
}
|
||||||
record = reconcile_manual_transition_worker_results(api.clone(), job_id, queue_snapshot).await?;
|
|
||||||
if record.is_terminal() || !record.report.worker_transition_pending() {
|
pub async fn renew_manual_transition_job_lease_if_owned(
|
||||||
return Ok(record);
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Uuid,
|
||||||
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
|
renew_manual_transition_job_lease_inner(api, job_id, expected_lease_id, None, queue_snapshot).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn renew_manual_transition_job_lease_inner(
|
||||||
|
api: Arc<ECStore>,
|
||||||
|
job_id: Uuid,
|
||||||
|
expected_lease_id: Uuid,
|
||||||
|
current: Option<(ManualTransitionJobRecord, String)>,
|
||||||
|
queue_snapshot: ManualTransitionQueueSnapshot,
|
||||||
|
) -> EcstoreResult<ManualTransitionJobRecord> {
|
||||||
|
let (current, current_etag) = match current {
|
||||||
|
Some(current) => current,
|
||||||
|
None => load_manual_transition_job_record_with_etag(api.clone(), job_id).await?,
|
||||||
|
};
|
||||||
|
if current.lease_id != expected_lease_id {
|
||||||
|
return Err(Error::PreconditionFailed);
|
||||||
}
|
}
|
||||||
(record, etag) = load_manual_transition_job_record_with_etag(api.clone(), job_id).await?;
|
if current.state != ManualTransitionJobState::Running {
|
||||||
|
return Ok(current);
|
||||||
|
}
|
||||||
|
if current.scan_completed && queue_snapshot.queued == 0 && queue_snapshot.active == 0 {
|
||||||
|
return reconcile_manual_transition_worker_results_inner(api, job_id, Some(expected_lease_id), queue_snapshot, true)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
let record = update_manual_transition_job_record_from(
|
||||||
|
api.clone(),
|
||||||
|
job_id,
|
||||||
|
Some(expected_lease_id),
|
||||||
|
Some((current, current_etag)),
|
||||||
|
|record| {
|
||||||
|
if record.state != ManualTransitionJobState::Running {
|
||||||
|
return false;
|
||||||
}
|
}
|
||||||
let became_terminal = record.mark_unknown_if_worker_results_lost(queue_snapshot);
|
|
||||||
if !became_terminal {
|
|
||||||
record.renew_lease(queue_snapshot);
|
record.renew_lease(queue_snapshot);
|
||||||
}
|
true
|
||||||
save_manual_transition_job_record_if_current(api.clone(), &record, &etag).await?;
|
},
|
||||||
if became_terminal {
|
)
|
||||||
|
.await?;
|
||||||
|
if record.is_terminal() {
|
||||||
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?;
|
delete_manual_transition_scope_admission_if_current(api, &record.scope_key, record.job_id, record.lease_id).await?;
|
||||||
} else {
|
} else if record.state == ManualTransitionJobState::Running {
|
||||||
renew_manual_transition_scope_admission_from_job(api, &record).await?;
|
renew_manual_transition_scope_admission_from_job(api, &record).await?;
|
||||||
}
|
}
|
||||||
}
|
|
||||||
Ok(record)
|
Ok(record)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1688,15 +1904,31 @@ async fn renew_manual_transition_scope_admission_from_job(
|
|||||||
api: Arc<ECStore>,
|
api: Arc<ECStore>,
|
||||||
record: &ManualTransitionJobRecord,
|
record: &ManualTransitionJobRecord,
|
||||||
) -> EcstoreResult<()> {
|
) -> EcstoreResult<()> {
|
||||||
if let Ok((admission, admission_etag)) =
|
for _ in 0..MANUAL_TRANSITION_JOB_CAS_RETRIES {
|
||||||
load_manual_transition_scope_admission_with_etag(api.clone(), &record.scope_key).await
|
let (admission, admission_etag) =
|
||||||
&& admission.job_id == record.job_id
|
match load_manual_transition_scope_admission_with_etag(api.clone(), &record.scope_key).await {
|
||||||
&& admission.lease_id == record.lease_id
|
Ok(admission) => admission,
|
||||||
{
|
Err(Error::ConfigNotFound) => return Ok(()),
|
||||||
let renewed_admission = ManualTransitionScopeAdmission::from_job(record);
|
Err(err) => return Err(err),
|
||||||
save_manual_transition_scope_admission_if_current(api, &renewed_admission, &admission_etag).await?;
|
};
|
||||||
|
if admission.job_id != record.job_id || admission.lease_id != record.lease_id {
|
||||||
|
return Err(Error::PreconditionFailed);
|
||||||
}
|
}
|
||||||
Ok(())
|
let mut renewed_admission = ManualTransitionScopeAdmission::from_job(record);
|
||||||
|
renewed_admission.lease_expires_at_unix_nanos = renewed_admission
|
||||||
|
.lease_expires_at_unix_nanos
|
||||||
|
.max(admission.lease_expires_at_unix_nanos);
|
||||||
|
renewed_admission.updated_at_unix_nanos = renewed_admission.updated_at_unix_nanos.max(admission.updated_at_unix_nanos);
|
||||||
|
if renewed_admission == admission {
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
match save_manual_transition_scope_admission_if_current(api.clone(), &renewed_admission, &admission_etag).await {
|
||||||
|
Ok(()) => return Ok(()),
|
||||||
|
Err(Error::PreconditionFailed) => continue,
|
||||||
|
Err(err) => return Err(err),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(Error::PreconditionFailed)
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn delete_manual_transition_scope_admission_if_current(
|
pub async fn delete_manual_transition_scope_admission_if_current(
|
||||||
@@ -2386,14 +2618,14 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn manual_transition_job_record_failure_counts_tier_failure() {
|
fn manual_transition_job_record_control_plane_failure_does_not_count_tier_failure() {
|
||||||
let options = ManualTransitionRunOptions::default();
|
let options = ManualTransitionRunOptions::default();
|
||||||
let mut record = ManualTransitionJobRecord::new(Uuid::new_v4(), "bucket", &options, TEST_OWNER);
|
let mut record = ManualTransitionJobRecord::new(Uuid::new_v4(), "bucket", &options, TEST_OWNER);
|
||||||
|
|
||||||
record.fail("missing tier");
|
record.fail("missing tier");
|
||||||
|
|
||||||
assert_eq!(record.state, ManualTransitionJobState::Failed);
|
assert_eq!(record.state, ManualTransitionJobState::Failed);
|
||||||
assert_eq!(record.report.tier_failure, 1);
|
assert_eq!(record.report.tier_failure, 0);
|
||||||
assert_eq!(record.error.as_deref(), Some("missing tier"));
|
assert_eq!(record.error.as_deref(), Some("missing tier"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ use s3s::dto::{BucketLifecycleConfiguration, ObjectLockConfiguration};
|
|||||||
use time::OffsetDateTime;
|
use time::OffsetDateTime;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
use crate::bucket::metadata::BucketMetadata;
|
||||||
use crate::bucket::metadata_sys::{self, ObjectLockConfigState};
|
use crate::bucket::metadata_sys::{self, ObjectLockConfigState};
|
||||||
use crate::error::{Error, Result};
|
use crate::error::{Error, Result};
|
||||||
|
|
||||||
@@ -26,16 +27,37 @@ pub(crate) struct LifecycleExpiryConfigs {
|
|||||||
pub(crate) lifecycle: Option<Arc<BucketLifecycleConfiguration>>,
|
pub(crate) lifecycle: Option<Arc<BucketLifecycleConfiguration>>,
|
||||||
pub(crate) object_lock: Option<Arc<ObjectLockConfiguration>>,
|
pub(crate) object_lock: Option<Arc<ObjectLockConfiguration>>,
|
||||||
pub(crate) bucket_incarnation_id: Uuid,
|
pub(crate) bucket_incarnation_id: Uuid,
|
||||||
|
pub(crate) table_bucket_enabled: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
pub(crate) async fn get_expiry_configs(api: &crate::store::ECStore, bucket: &str) -> Result<LifecycleExpiryConfigs> {
|
async fn get_authoritative_metadata(
|
||||||
let bucket_incarnation_id = api.bucket_incarnation_id_from_disk(bucket).await?;
|
api: &crate::store::ECStore,
|
||||||
|
bucket: &str,
|
||||||
|
bucket_incarnation_id: Uuid,
|
||||||
|
) -> Result<Arc<BucketMetadata>> {
|
||||||
let sys = metadata_sys::bucket_metadata_sys_of(&api.ctx)?;
|
let sys = metadata_sys::bucket_metadata_sys_of(&api.ctx)?;
|
||||||
let sys = sys.read().await.clone();
|
let sys = sys.read().await.clone();
|
||||||
let metadata = sys.get_authoritative_metadata(bucket).await?;
|
let metadata = sys.get_authoritative_metadata(bucket).await?;
|
||||||
if !metadata.bucket_incarnation_sidecar || metadata.bucket_incarnation_id != bucket_incarnation_id {
|
if !metadata.bucket_incarnation_sidecar || metadata.bucket_incarnation_id != bucket_incarnation_id {
|
||||||
return Err(Error::other(format!("bucket lifecycle metadata is not authoritative: {bucket}")));
|
return Err(Error::other(format!("bucket lifecycle metadata is not authoritative: {bucket}")));
|
||||||
}
|
}
|
||||||
|
Ok(metadata)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn lifecycle_expiry_allowed(
|
||||||
|
api: &crate::store::ECStore,
|
||||||
|
bucket: &str,
|
||||||
|
bucket_incarnation_id: Uuid,
|
||||||
|
) -> Result<bool> {
|
||||||
|
Ok(!get_authoritative_metadata(api, bucket, bucket_incarnation_id)
|
||||||
|
.await?
|
||||||
|
.table_bucket_enabled())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn get_expiry_configs(api: &crate::store::ECStore, bucket: &str) -> Result<LifecycleExpiryConfigs> {
|
||||||
|
let bucket_incarnation_id = api.bucket_incarnation_id_from_disk(bucket).await?;
|
||||||
|
let metadata = get_authoritative_metadata(api, bucket, bucket_incarnation_id).await?;
|
||||||
|
let table_bucket_enabled = metadata.table_bucket_enabled();
|
||||||
|
|
||||||
let lifecycle = if metadata.lifecycle_config.is_none() && !metadata.lifecycle_config_xml.is_empty() {
|
let lifecycle = if metadata.lifecycle_config.is_none() && !metadata.lifecycle_config_xml.is_empty() {
|
||||||
return Err(Error::other("persisted bucket lifecycle configuration is invalid"));
|
return Err(Error::other("persisted bucket lifecycle configuration is invalid"));
|
||||||
@@ -51,6 +73,7 @@ pub(crate) async fn get_expiry_configs(api: &crate::store::ECStore, bucket: &str
|
|||||||
lifecycle: None,
|
lifecycle: None,
|
||||||
object_lock: None,
|
object_lock: None,
|
||||||
bucket_incarnation_id,
|
bucket_incarnation_id,
|
||||||
|
table_bucket_enabled,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
let object_lock = match metadata_sys::object_lock_config_state_from_authoritative_metadata(&metadata)? {
|
let object_lock = match metadata_sys::object_lock_config_state_from_authoritative_metadata(&metadata)? {
|
||||||
@@ -65,6 +88,7 @@ pub(crate) async fn get_expiry_configs(api: &crate::store::ECStore, bucket: &str
|
|||||||
lifecycle,
|
lifecycle,
|
||||||
object_lock,
|
object_lock,
|
||||||
bucket_incarnation_id,
|
bucket_incarnation_id,
|
||||||
|
table_bucket_enabled,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -125,6 +149,7 @@ mod tests {
|
|||||||
let lifecycle = lifecycle_config();
|
let lifecycle = lifecycle_config();
|
||||||
metadata.lifecycle_config_xml = crate::bucket::utils::serialize(&lifecycle).unwrap();
|
metadata.lifecycle_config_xml = crate::bucket::utils::serialize(&lifecycle).unwrap();
|
||||||
metadata.lifecycle_config = Some(lifecycle);
|
metadata.lifecycle_config = Some(lifecycle);
|
||||||
|
metadata.table_bucket_config_json = br#"{"enabled":true}"#.to_vec();
|
||||||
metadata_sys::set_new_bucket_metadata_in(&store_a.ctx, metadata)
|
metadata_sys::set_new_bucket_metadata_in(&store_a.ctx, metadata)
|
||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
@@ -132,7 +157,14 @@ mod tests {
|
|||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
|
|
||||||
assert!(get_expiry_configs(&store_a, bucket).await.unwrap().lifecycle.is_some());
|
let configs = get_expiry_configs(&store_a, bucket).await.unwrap();
|
||||||
|
assert!(configs.lifecycle.is_some());
|
||||||
|
assert!(configs.table_bucket_enabled);
|
||||||
|
assert!(
|
||||||
|
!lifecycle_expiry_allowed(&store_a, bucket, configs.bucket_incarnation_id)
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
);
|
||||||
assert!(get_expiry_configs(&store_b, bucket).await.unwrap().lifecycle.is_none());
|
assert!(get_expiry_configs(&store_b, bucket).await.unwrap().lifecycle.is_none());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -425,6 +425,15 @@ impl BucketMetadata {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Metadata for a physically new user bucket. Existing or fabricated legacy
|
||||||
|
/// metadata must use [`Self::new`] so upgrades do not rewrite their
|
||||||
|
/// durability posture.
|
||||||
|
pub fn new_with_default_durability(name: &str) -> Self {
|
||||||
|
let mut metadata = Self::new(name);
|
||||||
|
metadata.durability_config_json = super::durability::new_bucket_durability_config_json();
|
||||||
|
metadata
|
||||||
|
}
|
||||||
|
|
||||||
pub fn save_file_path(&self) -> String {
|
pub fn save_file_path(&self) -> String {
|
||||||
format!("{}/{}/{}", BUCKET_META_PREFIX, self.name.as_str(), BUCKET_METADATA_FILE)
|
format!("{}/{}/{}", BUCKET_META_PREFIX, self.name.as_str(), BUCKET_METADATA_FILE)
|
||||||
}
|
}
|
||||||
@@ -1302,7 +1311,7 @@ mod test {
|
|||||||
assert!(bm.object_locking(), "object lock active via parsed config");
|
assert!(bm.object_locking(), "object lock active via parsed config");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// backlog#580: KNOWN GAP (weisd 2026-03-06 "inline_data 前缀不同"). RustFS's
|
/// backlog#580: KNOWN GAP (flagged 2026-03-06: "inline_data 前缀不同"). RustFS's
|
||||||
/// inline-data extraction does not yet recover the object body from a
|
/// inline-data extraction does not yet recover the object body from a
|
||||||
/// MinIO-written bucket-metadata object: `into_fileinfo(read_data=true).data`
|
/// MinIO-written bucket-metadata object: `into_fileinfo(read_data=true).data`
|
||||||
/// returns bytes that are not the `.metadata.bin` blob (no `format|version`
|
/// returns bytes that are not the `.metadata.bin` blob (no `format|version`
|
||||||
@@ -1310,7 +1319,7 @@ mod test {
|
|||||||
/// inline-data framing is handled on the read path.
|
/// inline-data framing is handled on the read path.
|
||||||
/// backlog#580: prove RustFS reads a MinIO-written **inlined** bucket-metadata
|
/// backlog#580: prove RustFS reads a MinIO-written **inlined** bucket-metadata
|
||||||
/// object end-to-end. MinIO stores inline data as `[bitrot hash][object body]`
|
/// object end-to-end. MinIO stores inline data as `[bitrot hash][object body]`
|
||||||
/// (the "`inline_data` 前缀不同" that weisd flagged on 2026-03-06 is that
|
/// (the "`inline_data` 前缀不同" gap flagged on 2026-03-06 is that
|
||||||
/// bitrot prefix, not a format incompatibility). Running the raw inline shard
|
/// bitrot prefix, not a format incompatibility). Running the raw inline shard
|
||||||
/// through RustFS's `BitrotReader` with the default `HighwayHash256S` must
|
/// through RustFS's `BitrotReader` with the default `HighwayHash256S` must
|
||||||
/// verify the checksum and yield the exact `.metadata.bin` blob.
|
/// verify the checksum and yield the exact `.metadata.bin` blob.
|
||||||
@@ -1378,6 +1387,43 @@ mod test {
|
|||||||
assert_ne!(old.bucket_incarnation_id, new.bucket_incarnation_id);
|
assert_ne!(old.bucket_incarnation_id, new.bucket_incarnation_id);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn regular_bucket_metadata_constructor_does_not_seed_durability() {
|
||||||
|
temp_env::with_var_unset(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, || {
|
||||||
|
let metadata = BucketMetadata::new("legacy-or-fabricated");
|
||||||
|
assert!(metadata.durability_config_json.is_empty());
|
||||||
|
assert!(metadata.durability_config().is_none());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_bucket_metadata_constructor_seeds_default_durability() {
|
||||||
|
temp_env::with_var_unset(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, || {
|
||||||
|
let metadata = BucketMetadata::new_with_default_durability("new-user-bucket");
|
||||||
|
assert_eq!(
|
||||||
|
metadata.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
|
||||||
|
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
|
||||||
|
);
|
||||||
|
|
||||||
|
let encoded = metadata.marshal_msg().expect("marshal metadata");
|
||||||
|
let decoded = BucketMetadata::unmarshal(&encoded).expect("unmarshal metadata");
|
||||||
|
assert_eq!(decoded.durability_config_json, metadata.durability_config_json);
|
||||||
|
assert_eq!(
|
||||||
|
decoded.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
|
||||||
|
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
|
||||||
|
);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn new_bucket_metadata_constructor_can_inherit_global_durability() {
|
||||||
|
temp_env::with_var(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, Some("inherit"), || {
|
||||||
|
let metadata = BucketMetadata::new_with_default_durability("strict-fleet-new-bucket");
|
||||||
|
assert!(metadata.durability_config_json.is_empty());
|
||||||
|
assert!(metadata.durability_config().is_none());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn site_replication_config_updates_cannot_replace_bucket_incarnation() {
|
fn site_replication_config_updates_cannot_replace_bucket_incarnation() {
|
||||||
let mut metadata = BucketMetadata::new("site-replication-update");
|
let mut metadata = BucketMetadata::new("site-replication-update");
|
||||||
|
|||||||
@@ -50,6 +50,72 @@ use uuid::Uuid;
|
|||||||
|
|
||||||
const BUCKET_METADATA_REFRESH_INTERVAL: Duration = Duration::from_secs(15 * 60);
|
const BUCKET_METADATA_REFRESH_INTERVAL: Duration = Duration::from_secs(15 * 60);
|
||||||
|
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
struct ConfigWriteLockProbeState {
|
||||||
|
bucket: String,
|
||||||
|
arrived: tokio::sync::Notify,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
static CONFIG_WRITE_LOCK_PROBES: std::sync::OnceLock<StdMutex<Vec<Arc<ConfigWriteLockProbeState>>>> = std::sync::OnceLock::new();
|
||||||
|
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
pub struct ConfigWriteLockProbe {
|
||||||
|
state: Arc<ConfigWriteLockProbeState>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
impl ConfigWriteLockProbe {
|
||||||
|
pub fn install(bucket: &str) -> Self {
|
||||||
|
let state = Arc::new(ConfigWriteLockProbeState {
|
||||||
|
bucket: bucket.to_string(),
|
||||||
|
arrived: tokio::sync::Notify::new(),
|
||||||
|
});
|
||||||
|
let mut probes = CONFIG_WRITE_LOCK_PROBES
|
||||||
|
.get_or_init(|| StdMutex::new(Vec::new()))
|
||||||
|
.lock()
|
||||||
|
.expect("config write lock probe mutex should not poison");
|
||||||
|
assert!(
|
||||||
|
!probes.iter().any(|current| current.bucket == state.bucket),
|
||||||
|
"config write lock probe must be unique for a bucket"
|
||||||
|
);
|
||||||
|
probes.push(Arc::clone(&state));
|
||||||
|
drop(probes);
|
||||||
|
Self { state }
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn wait_until_attempted(&self) {
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), self.state.arrived.notified())
|
||||||
|
.await
|
||||||
|
.expect("bucket config update should attempt the transaction lock");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
impl Drop for ConfigWriteLockProbe {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
let mut probes = CONFIG_WRITE_LOCK_PROBES
|
||||||
|
.get_or_init(|| StdMutex::new(Vec::new()))
|
||||||
|
.lock()
|
||||||
|
.expect("config write lock probe mutex should not poison");
|
||||||
|
probes.retain(|state| !Arc::ptr_eq(state, &self.state));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
fn notify_config_write_lock_attempt(bucket: &str) {
|
||||||
|
let probe = CONFIG_WRITE_LOCK_PROBES
|
||||||
|
.get_or_init(|| StdMutex::new(Vec::new()))
|
||||||
|
.lock()
|
||||||
|
.expect("config write lock probe mutex should not poison")
|
||||||
|
.iter()
|
||||||
|
.find(|probe| probe.bucket == bucket)
|
||||||
|
.cloned();
|
||||||
|
if let Some(probe) = probe {
|
||||||
|
probe.arrived.notify_one();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy)]
|
#[derive(Clone, Copy)]
|
||||||
enum MetadataLoadMode {
|
enum MetadataLoadMode {
|
||||||
Initial,
|
Initial,
|
||||||
@@ -288,6 +354,13 @@ pub(crate) fn bucket_metadata_sys_of(ctx: &crate::runtime::instance::InstanceCon
|
|||||||
get_bucket_metadata_sys()
|
get_bucket_metadata_sys()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) fn require_bucket_metadata_sys_in(
|
||||||
|
ctx: &crate::runtime::instance::InstanceContext,
|
||||||
|
) -> Result<Arc<RwLock<BucketMetadataSys>>> {
|
||||||
|
ctx.bucket_metadata_sys()
|
||||||
|
.ok_or_else(|| Error::other("bucket metadata sys not initialized for this instance"))
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn object_store_in(ctx: &crate::runtime::instance::InstanceContext) -> Result<Arc<ECStore>> {
|
pub(crate) async fn object_store_in(ctx: &crate::runtime::instance::InstanceContext) -> Result<Arc<ECStore>> {
|
||||||
let sys = bucket_metadata_sys_of(ctx)?;
|
let sys = bucket_metadata_sys_of(ctx)?;
|
||||||
Ok(sys.read().await.api.clone())
|
Ok(sys.read().await.api.clone())
|
||||||
@@ -376,6 +449,15 @@ pub async fn update(bucket: &str, config_file: &str, data: Vec<u8>) -> Result<Of
|
|||||||
Box::pin(update_with_sys(get_bucket_metadata_sys()?, bucket, config_file, data)).await
|
Box::pin(update_with_sys(get_bucket_metadata_sys()?, bucket, config_file, data)).await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn update_in(
|
||||||
|
ctx: &crate::runtime::instance::InstanceContext,
|
||||||
|
bucket: &str,
|
||||||
|
config_file: &str,
|
||||||
|
data: Vec<u8>,
|
||||||
|
) -> Result<OffsetDateTime> {
|
||||||
|
Box::pin(update_with_sys(require_bucket_metadata_sys_in(ctx)?, bucket, config_file, data)).await
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn delete(bucket: &str, config_file: &str) -> Result<OffsetDateTime> {
|
pub async fn delete(bucket: &str, config_file: &str) -> Result<OffsetDateTime> {
|
||||||
delete_with_sys(get_bucket_metadata_sys()?, bucket, config_file).await
|
delete_with_sys(get_bucket_metadata_sys()?, bucket, config_file).await
|
||||||
}
|
}
|
||||||
@@ -574,6 +656,41 @@ pub async fn update_under_transaction_lock(
|
|||||||
update_under_config_write_guard(get_bucket_metadata_sys()?, guard, config_file, data).await
|
update_under_config_write_guard(get_bucket_metadata_sys()?, guard, config_file, data).await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Clear one config file while the caller holds this bucket's transaction lock.
|
||||||
|
pub async fn delete_under_transaction_lock(
|
||||||
|
guard: &BucketMetadataMutationGuard,
|
||||||
|
bucket: &str,
|
||||||
|
config_file: &str,
|
||||||
|
) -> Result<OffsetDateTime> {
|
||||||
|
guard.ensure_valid(bucket)?;
|
||||||
|
delete_under_config_write_guard(get_bucket_metadata_sys()?, guard, config_file).await
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn update_quota_if_incarnation(
|
||||||
|
bucket: &str,
|
||||||
|
data: Vec<u8>,
|
||||||
|
expected_incarnation_id: Uuid,
|
||||||
|
proof: &crate::services::notification_sys::CrossPoolFenceFleetProofToken,
|
||||||
|
) -> Result<OffsetDateTime> {
|
||||||
|
let sys = get_bucket_metadata_sys()?;
|
||||||
|
let guard = Box::pin(acquire_config_write_guard_for_incarnation(
|
||||||
|
sys.clone(),
|
||||||
|
bucket,
|
||||||
|
Some(expected_incarnation_id),
|
||||||
|
))
|
||||||
|
.await?;
|
||||||
|
if !crate::services::notification_sys::cross_pool_fence_fleet_proof_matches(proof) {
|
||||||
|
return Err(Error::NamespaceLockQuorumUnavailable {
|
||||||
|
mode: "quota_capability",
|
||||||
|
bucket: bucket.to_string(),
|
||||||
|
object: rustfs_config::QUOTA_CONFIG_FILE.to_string(),
|
||||||
|
required: 1,
|
||||||
|
achieved: 0,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
update_under_config_write_guard(sys, &guard, rustfs_config::QUOTA_CONFIG_FILE, data).await
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn update_bucket_targets_under_transaction_lock(
|
pub async fn update_bucket_targets_under_transaction_lock(
|
||||||
guard: &BucketMetadataMutationGuard,
|
guard: &BucketMetadataMutationGuard,
|
||||||
bucket: &str,
|
bucket: &str,
|
||||||
@@ -688,6 +805,14 @@ pub async fn acquire_bucket_metadata_transaction_lock(bucket: &str) -> Result<Bu
|
|||||||
acquire_config_write_guard(get_bucket_metadata_sys()?, bucket).await
|
acquire_config_write_guard(get_bucket_metadata_sys()?, bucket).await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Acquire the bucket transaction lock only if its incarnation still matches.
|
||||||
|
pub async fn acquire_bucket_metadata_transaction_lock_for_incarnation(
|
||||||
|
bucket: &str,
|
||||||
|
expected_incarnation_id: Uuid,
|
||||||
|
) -> Result<BucketMetadataMutationGuard> {
|
||||||
|
acquire_config_write_guard_for_incarnation(get_bucket_metadata_sys()?, bucket, Some(expected_incarnation_id)).await
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn acquire_bucket_metadata_transaction_lock_in(
|
pub(crate) async fn acquire_bucket_metadata_transaction_lock_in(
|
||||||
ctx: &crate::runtime::instance::InstanceContext,
|
ctx: &crate::runtime::instance::InstanceContext,
|
||||||
bucket: &str,
|
bucket: &str,
|
||||||
@@ -718,7 +843,26 @@ async fn acquire_transaction_lock_with_sys(
|
|||||||
let lock = api
|
let lock = api
|
||||||
.new_ns_lock(RUSTFS_META_BUCKET, &bucket_metadata_transaction_lock_key(bucket))
|
.new_ns_lock(RUSTFS_META_BUCKET, &bucket_metadata_transaction_lock_key(bucket))
|
||||||
.await?;
|
.await?;
|
||||||
Ok(lock.get_write_lock(crate::set_disk::get_lock_acquire_timeout()).await?)
|
let acquire = lock.get_write_lock(crate::set_disk::get_lock_acquire_timeout());
|
||||||
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
|
{
|
||||||
|
tokio::pin!(acquire);
|
||||||
|
let mut notified = false;
|
||||||
|
let guard = futures::future::poll_fn(|cx| match std::future::Future::poll(acquire.as_mut(), cx) {
|
||||||
|
std::task::Poll::Pending => {
|
||||||
|
if !notified {
|
||||||
|
notify_config_write_lock_attempt(bucket);
|
||||||
|
notified = true;
|
||||||
|
}
|
||||||
|
std::task::Poll::Pending
|
||||||
|
}
|
||||||
|
std::task::Poll::Ready(result) => std::task::Poll::Ready(result),
|
||||||
|
})
|
||||||
|
.await?;
|
||||||
|
Ok(guard)
|
||||||
|
}
|
||||||
|
#[cfg(not(any(test, feature = "test-util")))]
|
||||||
|
Ok(acquire.await?)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The lock resource name is deliberately still the `bucket-targets` one it
|
/// The lock resource name is deliberately still the `bucket-targets` one it
|
||||||
@@ -873,6 +1017,37 @@ pub(crate) async fn get_object_lock_config_and_incarnation_from_disk_in(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Re-read the quota configuration and bucket incarnation from the same
|
||||||
|
/// authoritative metadata blob while the caller holds the bucket metadata
|
||||||
|
/// transaction read lock.
|
||||||
|
pub(crate) async fn get_quota_config_and_incarnation_from_disk_in(
|
||||||
|
ctx: &crate::runtime::instance::InstanceContext,
|
||||||
|
bucket: &str,
|
||||||
|
) -> Result<(Option<BucketQuota>, Uuid, OffsetDateTime)> {
|
||||||
|
let bucket_meta_sys_lock = bucket_metadata_sys_of(ctx)?;
|
||||||
|
let bucket_meta_sys = bucket_meta_sys_lock.read().await.clone();
|
||||||
|
|
||||||
|
match bucket_meta_sys
|
||||||
|
.read_authoritative_metadata_from_disk_under_transaction_lock(bucket)
|
||||||
|
.await?
|
||||||
|
{
|
||||||
|
BucketMetadataAuthority::Authoritative(metadata)
|
||||||
|
if metadata.bucket_incarnation_sidecar && !metadata.bucket_incarnation_id.is_nil() =>
|
||||||
|
{
|
||||||
|
Ok((
|
||||||
|
metadata.quota_config.clone(),
|
||||||
|
metadata.bucket_incarnation_id,
|
||||||
|
metadata.quota_config_updated_at,
|
||||||
|
))
|
||||||
|
}
|
||||||
|
BucketMetadataAuthority::Authoritative(_) => {
|
||||||
|
Err(Error::other(format!("bucket incarnation metadata is not authoritative: {bucket}")))
|
||||||
|
}
|
||||||
|
BucketMetadataAuthority::MissingBucket => Err(Error::BucketNotFound(bucket.to_string())),
|
||||||
|
BucketMetadataAuthority::Fabricated => Err(Error::other(format!("bucket quota metadata is not authoritative: {bucket}"))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn get_replication_config(bucket: &str) -> Result<(ReplicationConfiguration, OffsetDateTime)> {
|
pub async fn get_replication_config(bucket: &str) -> Result<(ReplicationConfiguration, OffsetDateTime)> {
|
||||||
let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
|
let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
|
||||||
let bucket_meta_sys = bucket_meta_sys_lock.read().await;
|
let bucket_meta_sys = bucket_meta_sys_lock.read().await;
|
||||||
|
|||||||
@@ -14,6 +14,7 @@
|
|||||||
|
|
||||||
use super::metadata_sys::get_bucket_metadata_sys;
|
use super::metadata_sys::get_bucket_metadata_sys;
|
||||||
use crate::error::{Result, StorageError};
|
use crate::error::{Result, StorageError};
|
||||||
|
use crate::store::ECStore;
|
||||||
use rustfs_policy::policy::{BucketPolicy, BucketPolicyArgs};
|
use rustfs_policy::policy::{BucketPolicy, BucketPolicyArgs};
|
||||||
|
|
||||||
pub struct PolicySys {}
|
pub struct PolicySys {}
|
||||||
@@ -27,6 +28,10 @@ impl PolicySys {
|
|||||||
Self::is_allowed_with_policy(args, Self::get(args.bucket).await).await
|
Self::is_allowed_with_policy(args, Self::get(args.bucket).await).await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub async fn try_is_allowed_for_store(store: &ECStore, args: &BucketPolicyArgs<'_>) -> Result<bool> {
|
||||||
|
Self::is_allowed_with_policy(args, store.get_bucket_policy(args.bucket).await.map(|(policy, _)| policy)).await
|
||||||
|
}
|
||||||
|
|
||||||
async fn is_allowed_with_policy(args: &BucketPolicyArgs<'_>, policy: Result<BucketPolicy>) -> Result<bool> {
|
async fn is_allowed_with_policy(args: &BucketPolicyArgs<'_>, policy: Result<BucketPolicy>) -> Result<bool> {
|
||||||
match policy {
|
match policy {
|
||||||
Ok(policy) => Ok(policy.is_allowed(args).await),
|
Ok(policy) => Ok(policy.is_allowed(args).await),
|
||||||
|
|||||||
@@ -52,6 +52,7 @@ impl QuotaChecker {
|
|||||||
) -> Result<QuotaCheckResult, QuotaError> {
|
) -> Result<QuotaCheckResult, QuotaError> {
|
||||||
let start_time = Instant::now();
|
let start_time = Instant::now();
|
||||||
let quota_config = self.get_quota_config(bucket).await?;
|
let quota_config = self.get_quota_config(bucket).await?;
|
||||||
|
let uses_durable_reservations = quota_config.uses_durable_reservations();
|
||||||
|
|
||||||
// If no quota limit is set, allow operation
|
// If no quota limit is set, allow operation
|
||||||
let quota_limit = match quota_config.quota {
|
let quota_limit = match quota_config.quota {
|
||||||
@@ -67,6 +68,7 @@ impl QuotaChecker {
|
|||||||
quota_limit: None,
|
quota_limit: None,
|
||||||
operation_size,
|
operation_size,
|
||||||
remaining: None,
|
remaining: None,
|
||||||
|
uses_durable_reservations,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
Some(q) => q,
|
Some(q) => q,
|
||||||
@@ -74,14 +76,17 @@ impl QuotaChecker {
|
|||||||
|
|
||||||
let current_usage = self.get_real_time_usage(bucket).await?;
|
let current_usage = self.get_real_time_usage(bucket).await?;
|
||||||
|
|
||||||
|
let admission_size = if uses_durable_reservations { 0 } else { operation_size };
|
||||||
let expected_usage = match operation {
|
let expected_usage = match operation {
|
||||||
QuotaOperation::PutObject | QuotaOperation::PostObject | QuotaOperation::CopyObject => current_usage + operation_size,
|
QuotaOperation::PutObject | QuotaOperation::PostObject | QuotaOperation::CopyObject => {
|
||||||
|
current_usage.saturating_add(admission_size)
|
||||||
|
}
|
||||||
QuotaOperation::DeleteObject => current_usage.saturating_sub(operation_size),
|
QuotaOperation::DeleteObject => current_usage.saturating_sub(operation_size),
|
||||||
};
|
};
|
||||||
|
|
||||||
let allowed = match operation {
|
let allowed = match operation {
|
||||||
QuotaOperation::PutObject | QuotaOperation::PostObject | QuotaOperation::CopyObject => {
|
QuotaOperation::PutObject | QuotaOperation::PostObject | QuotaOperation::CopyObject => {
|
||||||
quota_config.check_operation_allowed(current_usage, operation_size)
|
quota_config.check_operation_allowed(current_usage, admission_size)
|
||||||
}
|
}
|
||||||
QuotaOperation::DeleteObject => true,
|
QuotaOperation::DeleteObject => true,
|
||||||
};
|
};
|
||||||
@@ -105,6 +110,7 @@ impl QuotaChecker {
|
|||||||
quota_limit: Some(quota_limit),
|
quota_limit: Some(quota_limit),
|
||||||
operation_size,
|
operation_size,
|
||||||
remaining,
|
remaining,
|
||||||
|
uses_durable_reservations,
|
||||||
};
|
};
|
||||||
|
|
||||||
let duration = start_time.elapsed();
|
let duration = start_time.elapsed();
|
||||||
@@ -158,6 +164,26 @@ impl QuotaChecker {
|
|||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub async fn set_durable_quota_config_if_incarnation(
|
||||||
|
&mut self,
|
||||||
|
bucket: &str,
|
||||||
|
quota: BucketQuota,
|
||||||
|
expected_incarnation_id: uuid::Uuid,
|
||||||
|
proof: &crate::services::notification_sys::CrossPoolFenceFleetProofToken,
|
||||||
|
) -> Result<OffsetDateTime, QuotaError> {
|
||||||
|
let json_data = serde_json::to_vec("a).map_err(|e| QuotaError::InvalidConfig {
|
||||||
|
reason: format!("Failed to serialize quota config: {}", e),
|
||||||
|
})?;
|
||||||
|
let start_time = Instant::now();
|
||||||
|
let updated_at =
|
||||||
|
crate::bucket::metadata_sys::update_quota_if_incarnation(bucket, json_data, expected_incarnation_id, proof)
|
||||||
|
.await
|
||||||
|
.map_err(QuotaError::StorageError)?;
|
||||||
|
|
||||||
|
rustfs_common::metrics::Metrics::inc_time(Metric::QuotaSync, start_time.elapsed());
|
||||||
|
Ok(updated_at)
|
||||||
|
}
|
||||||
|
|
||||||
async fn set_quota_config_for_incarnation(
|
async fn set_quota_config_for_incarnation(
|
||||||
&mut self,
|
&mut self,
|
||||||
bucket: &str,
|
bucket: &str,
|
||||||
@@ -355,6 +381,7 @@ mod tests {
|
|||||||
quota_limit: None,
|
quota_limit: None,
|
||||||
operation_size: 1024,
|
operation_size: 1024,
|
||||||
remaining: None,
|
remaining: None,
|
||||||
|
uses_durable_reservations: false,
|
||||||
};
|
};
|
||||||
|
|
||||||
assert!(result.allowed);
|
assert!(result.allowed);
|
||||||
@@ -378,4 +405,13 @@ mod tests {
|
|||||||
let allowed = quota.check_operation_allowed(512, 1024);
|
let allowed = quota.check_operation_allowed(512, 1024);
|
||||||
assert!(!allowed);
|
assert!(!allowed);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn legacy_quota_rejects_full_operation_while_v1_defers_net_growth() {
|
||||||
|
let legacy: BucketQuota = serde_json::from_str(r#"{"quota":5}"#).expect("legacy quota should parse");
|
||||||
|
let durable = BucketQuota::new(Some(5));
|
||||||
|
|
||||||
|
assert!(!legacy.check_operation_allowed(4, 2));
|
||||||
|
assert!(durable.uses_durable_reservations());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -13,40 +13,100 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
pub mod checker;
|
pub mod checker;
|
||||||
|
pub(crate) mod reservation;
|
||||||
|
|
||||||
use crate::error::Result;
|
use crate::error::Result;
|
||||||
use rustfs_config::{
|
use rustfs_config::{
|
||||||
QUOTA_API_PATH, QUOTA_EXCEEDED_ERROR_CODE, QUOTA_INTERNAL_ERROR_CODE, QUOTA_INVALID_CONFIG_ERROR_CODE,
|
QUOTA_API_PATH, QUOTA_EXCEEDED_ERROR_CODE, QUOTA_INTERNAL_ERROR_CODE, QUOTA_INVALID_CONFIG_ERROR_CODE,
|
||||||
QUOTA_NOT_FOUND_ERROR_CODE,
|
QUOTA_NOT_FOUND_ERROR_CODE,
|
||||||
};
|
};
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Deserializer, Serialize, Serializer, de::Error as _};
|
||||||
use thiserror::Error;
|
use thiserror::Error;
|
||||||
use time::OffsetDateTime;
|
use time::OffsetDateTime;
|
||||||
|
|
||||||
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize, Default)]
|
||||||
pub enum QuotaType {
|
pub enum QuotaType {
|
||||||
/// Hard quota: reject immediately when exceeded
|
/// Hard quota accounting.
|
||||||
#[default]
|
#[default]
|
||||||
#[serde(alias = "HARD", alias = "hard")]
|
#[serde(alias = "HARD", alias = "hard")]
|
||||||
Hard,
|
Hard,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) const QUOTA_RESERVATION_PROTOCOL_V1: u32 = 1;
|
||||||
|
|
||||||
/// Bucket quota configuration. quota_type defaults to Hard when omitted.
|
/// Bucket quota configuration. quota_type defaults to Hard when omitted.
|
||||||
#[derive(Debug, Deserialize, Serialize, Default, Clone, PartialEq)]
|
#[derive(Debug, Default, Clone, PartialEq)]
|
||||||
pub struct BucketQuota {
|
pub struct BucketQuota {
|
||||||
#[serde(default)]
|
|
||||||
pub quota: Option<u64>,
|
pub quota: Option<u64>,
|
||||||
/// Defaults to Hard when missing.
|
/// Defaults to Hard when missing.
|
||||||
#[serde(default)]
|
|
||||||
pub quota_type: QuotaType,
|
pub quota_type: QuotaType,
|
||||||
|
/// Optional durable reservation protocol. The wire format gives older
|
||||||
|
/// nodes a zero hard quota so a mixed-version fleet fails closed.
|
||||||
|
pub reservation_protocol: Option<u32>,
|
||||||
/// Timestamp when this quota configuration was set (for audit purposes)
|
/// Timestamp when this quota configuration was set (for audit purposes)
|
||||||
#[serde(default, with = "time::serde::rfc3339::option")]
|
|
||||||
pub created_at: Option<OffsetDateTime>,
|
pub created_at: Option<OffsetDateTime>,
|
||||||
/// Accept updated_at for compatibility; not used.
|
/// Accept updated_at for compatibility; not used.
|
||||||
#[serde(default, with = "time::serde::rfc3339::option", skip_serializing_if = "Option::is_none")]
|
|
||||||
pub updated_at: Option<OffsetDateTime>,
|
pub updated_at: Option<OffsetDateTime>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Deserialize, Serialize)]
|
||||||
|
struct BucketQuotaWire {
|
||||||
|
#[serde(default)]
|
||||||
|
quota: Option<u64>,
|
||||||
|
#[serde(default)]
|
||||||
|
quota_type: QuotaType,
|
||||||
|
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||||
|
reservation_protocol: Option<u32>,
|
||||||
|
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||||
|
reservation_quota: Option<u64>,
|
||||||
|
#[serde(default, with = "time::serde::rfc3339::option")]
|
||||||
|
created_at: Option<OffsetDateTime>,
|
||||||
|
#[serde(default, with = "time::serde::rfc3339::option", skip_serializing_if = "Option::is_none")]
|
||||||
|
updated_at: Option<OffsetDateTime>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Serialize for BucketQuota {
|
||||||
|
fn serialize<S>(&self, serializer: S) -> std::result::Result<S::Ok, S::Error>
|
||||||
|
where
|
||||||
|
S: Serializer,
|
||||||
|
{
|
||||||
|
let durable = self.uses_durable_reservations();
|
||||||
|
BucketQuotaWire {
|
||||||
|
quota: if durable { Some(0) } else { self.quota },
|
||||||
|
quota_type: self.quota_type.clone(),
|
||||||
|
reservation_protocol: self.reservation_protocol,
|
||||||
|
reservation_quota: if durable { self.quota } else { None },
|
||||||
|
created_at: self.created_at,
|
||||||
|
updated_at: self.updated_at,
|
||||||
|
}
|
||||||
|
.serialize(serializer)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'de> Deserialize<'de> for BucketQuota {
|
||||||
|
fn deserialize<D>(deserializer: D) -> std::result::Result<Self, D::Error>
|
||||||
|
where
|
||||||
|
D: Deserializer<'de>,
|
||||||
|
{
|
||||||
|
let wire = BucketQuotaWire::deserialize(deserializer)?;
|
||||||
|
let quota = if wire.reservation_protocol == Some(QUOTA_RESERVATION_PROTOCOL_V1) {
|
||||||
|
Some(
|
||||||
|
wire.reservation_quota
|
||||||
|
.ok_or_else(|| D::Error::custom("reservation_quota is required for reservation protocol v1"))?,
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
wire.quota
|
||||||
|
};
|
||||||
|
Ok(Self {
|
||||||
|
quota,
|
||||||
|
quota_type: wire.quota_type,
|
||||||
|
reservation_protocol: wire.reservation_protocol,
|
||||||
|
created_at: wire.created_at,
|
||||||
|
updated_at: wire.updated_at,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl BucketQuota {
|
impl BucketQuota {
|
||||||
/// Serialize to JSON bytes. Same format as parse_all_configs.
|
/// Serialize to JSON bytes. Same format as parse_all_configs.
|
||||||
pub fn marshal_msg(&self) -> Result<Vec<u8>> {
|
pub fn marshal_msg(&self) -> Result<Vec<u8>> {
|
||||||
@@ -63,6 +123,7 @@ impl BucketQuota {
|
|||||||
Self {
|
Self {
|
||||||
quota,
|
quota,
|
||||||
quota_type: QuotaType::Hard,
|
quota_type: QuotaType::Hard,
|
||||||
|
reservation_protocol: quota.map(|_| QUOTA_RESERVATION_PROTOCOL_V1),
|
||||||
created_at: Some(now),
|
created_at: Some(now),
|
||||||
updated_at: None,
|
updated_at: None,
|
||||||
}
|
}
|
||||||
@@ -72,7 +133,19 @@ impl BucketQuota {
|
|||||||
self.quota
|
self.quota
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn uses_durable_reservations(&self) -> bool {
|
||||||
|
self.reservation_protocol == Some(QUOTA_RESERVATION_PROTOCOL_V1)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn has_unsupported_reservation_protocol(&self) -> bool {
|
||||||
|
self.reservation_protocol
|
||||||
|
.is_some_and(|version| version != QUOTA_RESERVATION_PROTOCOL_V1)
|
||||||
|
}
|
||||||
|
|
||||||
pub fn check_operation_allowed(&self, current_usage: u64, operation_size: u64) -> bool {
|
pub fn check_operation_allowed(&self, current_usage: u64, operation_size: u64) -> bool {
|
||||||
|
if operation_size == 0 {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
if let Some(quota_limit) = self.quota {
|
if let Some(quota_limit) = self.quota {
|
||||||
current_usage.saturating_add(operation_size) <= quota_limit
|
current_usage.saturating_add(operation_size) <= quota_limit
|
||||||
} else {
|
} else {
|
||||||
@@ -94,6 +167,7 @@ pub struct QuotaCheckResult {
|
|||||||
pub quota_limit: Option<u64>,
|
pub quota_limit: Option<u64>,
|
||||||
pub operation_size: u64,
|
pub operation_size: u64,
|
||||||
pub remaining: Option<u64>,
|
pub remaining: Option<u64>,
|
||||||
|
pub uses_durable_reservations: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
@@ -210,7 +284,59 @@ mod tests {
|
|||||||
let buf = q.marshal_msg().expect("marshal");
|
let buf = q.marshal_msg().expect("marshal");
|
||||||
let restored = BucketQuota::unmarshal(&buf).expect("unmarshal");
|
let restored = BucketQuota::unmarshal(&buf).expect("unmarshal");
|
||||||
assert_eq!(q.quota, restored.quota);
|
assert_eq!(q.quota, restored.quota);
|
||||||
assert_eq!(q.quota_type, restored.quota_type);
|
assert_eq!(restored.quota_type, QuotaType::Hard);
|
||||||
|
assert_eq!(restored.reservation_protocol, Some(QUOTA_RESERVATION_PROTOCOL_V1));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn clearing_quota_keeps_the_legacy_compatible_type() {
|
||||||
|
let quota = BucketQuota::new(None);
|
||||||
|
|
||||||
|
assert_eq!(quota.quota_type, QuotaType::Hard);
|
||||||
|
assert_eq!(quota.reservation_protocol, None);
|
||||||
|
assert!(!quota.uses_durable_reservations());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn durable_quota_makes_legacy_nodes_fail_closed() {
|
||||||
|
let json = serde_json::to_vec(&BucketQuota::new(Some(2048))).expect("durable quota should serialize");
|
||||||
|
let quota: BucketQuota = serde_json::from_slice(&json).expect("current quota version should parse");
|
||||||
|
assert!(quota.uses_durable_reservations());
|
||||||
|
assert_eq!(quota.quota, Some(2048));
|
||||||
|
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
enum LegacyQuotaType {
|
||||||
|
Hard,
|
||||||
|
}
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
struct LegacyBucketQuota {
|
||||||
|
#[allow(dead_code)]
|
||||||
|
quota: Option<u64>,
|
||||||
|
#[allow(dead_code)]
|
||||||
|
quota_type: LegacyQuotaType,
|
||||||
|
}
|
||||||
|
let legacy = serde_json::from_slice::<LegacyBucketQuota>(&json)
|
||||||
|
.expect("legacy readers should ignore the reservation protocol field");
|
||||||
|
assert_eq!(legacy.quota, Some(0));
|
||||||
|
assert!(matches!(legacy.quota_type, LegacyQuotaType::Hard));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unknown_reservation_protocol_does_not_activate_v1() {
|
||||||
|
let quota: BucketQuota =
|
||||||
|
serde_json::from_str(r#"{"quota":0,"quota_type":"Hard","reservation_protocol":2,"reservation_quota":2048}"#)
|
||||||
|
.expect("future protocol should remain parseable");
|
||||||
|
|
||||||
|
assert!(!quota.uses_durable_reservations());
|
||||||
|
assert!(quota.has_unsupported_reservation_protocol());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn reservation_protocol_v1_requires_reservation_quota() {
|
||||||
|
let err = serde_json::from_str::<BucketQuota>(r#"{"quota":0,"quota_type":"Hard","reservation_protocol":1}"#)
|
||||||
|
.expect_err("v1 without its authoritative quota must fail closed");
|
||||||
|
|
||||||
|
assert!(err.to_string().contains("reservation_quota is required"));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// unmarshal accepts format without quota_type
|
/// unmarshal accepts format without quota_type
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -81,6 +81,6 @@ pub use replication_queue_boundary::{
|
|||||||
pub use replication_resync_boundary::{BucketReplicationResyncStatus, ResyncOpts, TargetReplicationResyncStatus};
|
pub use replication_resync_boundary::{BucketReplicationResyncStatus, ResyncOpts, TargetReplicationResyncStatus};
|
||||||
pub use replication_scanner_bridge::ReplicationScannerBridge;
|
pub use replication_scanner_bridge::ReplicationScannerBridge;
|
||||||
pub use replication_state::{ReplicationStats, RuntimeReplicationTargetBacklog};
|
pub use replication_state::{ReplicationStats, RuntimeReplicationTargetBacklog};
|
||||||
pub use replication_stats_boundary::{BucketReplicationStats, BucketStats};
|
pub use replication_stats_boundary::{BucketReplicationStat, BucketReplicationStats, BucketStats, InQueueMetric, XferStats};
|
||||||
pub use replication_storage_boundary::{ReplicationObjectIO, ReplicationStorage};
|
pub use replication_storage_boundary::{ReplicationObjectIO, ReplicationStorage};
|
||||||
pub(crate) use replication_target_config_bridge::ReplicationTargetConfigBridge;
|
pub(crate) use replication_target_config_bridge::ReplicationTargetConfigBridge;
|
||||||
|
|||||||
@@ -704,6 +704,12 @@ impl ReplicationStats {
|
|||||||
} else {
|
} else {
|
||||||
BucketReplicationStats::new()
|
BucketReplicationStats::new()
|
||||||
};
|
};
|
||||||
|
// Stamp the serializable failure windows from the live samples: the
|
||||||
|
// samples themselves do not cross the peer-RPC wire, so this snapshot
|
||||||
|
// is what cluster aggregation and the metrics endpoints see.
|
||||||
|
for stat in replication_stats.stats.values_mut() {
|
||||||
|
stat.fail_stats.refresh_windows();
|
||||||
|
}
|
||||||
let uptime = if cache.contains_key(bucket) {
|
let uptime = if cache.contains_key(bucket) {
|
||||||
SystemTime::now()
|
SystemTime::now()
|
||||||
.duration_since(SystemTime::UNIX_EPOCH)
|
.duration_since(SystemTime::UNIX_EPOCH)
|
||||||
|
|||||||
@@ -15,7 +15,9 @@
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
pub(crate) use rustfs_replication::FailStats;
|
pub(crate) use rustfs_replication::FailStats;
|
||||||
pub(crate) use rustfs_replication::{
|
pub(crate) use rustfs_replication::{
|
||||||
ActiveWorkerStat, BucketReplicationStat, InQueueMetric, ProxyMetric, ProxyStatsCache, QueueCache, ReplicationMetricScope,
|
ActiveWorkerStat, ProxyMetric, ProxyStatsCache, QueueCache, ReplicationMetricScope, SRMetricsSummary,
|
||||||
SRMetricsSummary, XferStats,
|
|
||||||
};
|
};
|
||||||
pub use rustfs_replication::{BucketReplicationStats, BucketStats};
|
// Public so the admin wire DTOs (rustfs/src/admin/replication_metrics_wire.rs)
|
||||||
|
// can project the internal stats onto the minio-go response shapes through
|
||||||
|
// the storage_api facade chain.
|
||||||
|
pub use rustfs_replication::{BucketReplicationStat, BucketReplicationStats, BucketStats, InQueueMetric, XferStats};
|
||||||
|
|||||||
@@ -27,8 +27,10 @@ use rustfs_utils::http::{
|
|||||||
AMZ_OBJECT_TAGGING, AMZ_SERVER_SIDE_ENCRYPTION, AMZ_SERVER_SIDE_ENCRYPTION_KMS_CONTEXT, AMZ_SERVER_SIDE_ENCRYPTION_KMS_ID,
|
AMZ_OBJECT_TAGGING, AMZ_SERVER_SIDE_ENCRYPTION, AMZ_SERVER_SIDE_ENCRYPTION_KMS_CONTEXT, AMZ_SERVER_SIDE_ENCRYPTION_KMS_ID,
|
||||||
AMZ_STORAGE_CLASS, AMZ_TAG_COUNT, CACHE_CONTROL, CONTENT_DISPOSITION, CONTENT_ENCODING, CONTENT_LANGUAGE, CONTENT_TYPE,
|
AMZ_STORAGE_CLASS, AMZ_TAG_COUNT, CACHE_CONTROL, CONTENT_DISPOSITION, CONTENT_ENCODING, CONTENT_LANGUAGE, CONTENT_TYPE,
|
||||||
HeaderExt as _, SUFFIX_OBJECTLOCK_LEGALHOLD_TIMESTAMP, SUFFIX_OBJECTLOCK_RETENTION_TIMESTAMP,
|
HeaderExt as _, SUFFIX_OBJECTLOCK_LEGALHOLD_TIMESTAMP, SUFFIX_OBJECTLOCK_RETENTION_TIMESTAMP,
|
||||||
SUFFIX_REPLICATION_ACTUAL_OBJECT_SIZE, SUFFIX_REPLICATION_SSEC_CRC, SUFFIX_TAGGING_TIMESTAMP, get_str, insert_header_map,
|
SUFFIX_REPLICATION_ACTUAL_OBJECT_SIZE, SUFFIX_REPLICATION_SSEC_CRC, SUFFIX_SOURCE_REPLICATION_LEGALHOLD_TIMESTAMP,
|
||||||
is_internal_key, is_object_encryption_marker, is_replication_stripped_encryption_key, ssec_replication_transport_header,
|
SUFFIX_SOURCE_REPLICATION_RETENTION_TIMESTAMP, SUFFIX_SOURCE_REPLICATION_TAGGING_TIMESTAMP, SUFFIX_TAGGING_TIMESTAMP,
|
||||||
|
get_str, insert_header_map, is_internal_key, is_object_encryption_marker, is_replication_stripped_encryption_key,
|
||||||
|
ssec_replication_transport_header,
|
||||||
};
|
};
|
||||||
use time::OffsetDateTime;
|
use time::OffsetDateTime;
|
||||||
use time::format_description::well_known::Rfc3339;
|
use time::format_description::well_known::Rfc3339;
|
||||||
@@ -119,6 +121,27 @@ fn classify_replication_source_encryption(metadata: &HashMap<String, String>) ->
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn is_legacy_source_replication_timestamp_key(key: &str) -> bool {
|
||||||
|
fn has_prefix_and_suffix(key: &str, prefix: &str, suffix: &str) -> bool {
|
||||||
|
let key = key.as_bytes();
|
||||||
|
key.len() == prefix.len() + suffix.len()
|
||||||
|
&& key[..prefix.len()].eq_ignore_ascii_case(prefix.as_bytes())
|
||||||
|
&& key[prefix.len()..].eq_ignore_ascii_case(suffix.as_bytes())
|
||||||
|
}
|
||||||
|
|
||||||
|
[
|
||||||
|
SUFFIX_SOURCE_REPLICATION_TAGGING_TIMESTAMP,
|
||||||
|
SUFFIX_SOURCE_REPLICATION_RETENTION_TIMESTAMP,
|
||||||
|
SUFFIX_SOURCE_REPLICATION_LEGALHOLD_TIMESTAMP,
|
||||||
|
]
|
||||||
|
.iter()
|
||||||
|
.any(|suffix| {
|
||||||
|
["x-rustfs-", "x-minio-"]
|
||||||
|
.iter()
|
||||||
|
.any(|prefix| has_prefix_and_suffix(key, prefix, suffix))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) fn replication_object_is_ssec_encrypted(user_defined: &HashMap<String, String>) -> bool {
|
pub(crate) fn replication_object_is_ssec_encrypted(user_defined: &HashMap<String, String>) -> bool {
|
||||||
rustfs_replication::is_ssec_encrypted(user_defined)
|
rustfs_replication::is_ssec_encrypted(user_defined)
|
||||||
}
|
}
|
||||||
@@ -176,6 +199,11 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if is_legacy_source_replication_timestamp_key(key) {
|
||||||
|
meta.insert(format!("x-amz-meta-{key}"), value.to_string());
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
if is_internal_key(key) || is_standard_header(key) {
|
if is_internal_key(key) || is_standard_header(key) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -259,15 +287,23 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
|
|||||||
|
|
||||||
if !tags.is_empty() {
|
if !tags.is_empty() {
|
||||||
put_options.user_tags = tags;
|
put_options.user_tags = tags;
|
||||||
put_options.internal.tagging_timestamp =
|
}
|
||||||
if let Some(timestamp) = get_str(&object_info.user_defined, SUFFIX_TAGGING_TIMESTAMP) {
|
}
|
||||||
|
// Load the stored tagging timestamp independently of whether any tags
|
||||||
|
// remain: DeleteObjectTagging leaves the object tagless but stamps this
|
||||||
|
// key, and the deletion's LWW timestamp must still reach the replica.
|
||||||
|
// With no stored key, fall back to mod_time only while tags exist
|
||||||
|
// (MinIO parity); a tagless object without the key was never tagged and
|
||||||
|
// keeps the epoch default (no header).
|
||||||
|
put_options.internal.tagging_timestamp = if let Some(timestamp) = get_str(&object_info.user_defined, SUFFIX_TAGGING_TIMESTAMP)
|
||||||
|
{
|
||||||
OffsetDateTime::parse(×tamp, &Rfc3339)
|
OffsetDateTime::parse(×tamp, &Rfc3339)
|
||||||
.map_err(|err| Error::other(format!("Failed to parse tagging timestamp: {err}")))?
|
.map_err(|err| Error::other(format!("Failed to parse tagging timestamp: {err}")))?
|
||||||
} else {
|
} else if !put_options.user_tags.is_empty() {
|
||||||
object_info.mod_time.unwrap_or(OffsetDateTime::UNIX_EPOCH)
|
object_info.mod_time.unwrap_or(OffsetDateTime::UNIX_EPOCH)
|
||||||
|
} else {
|
||||||
|
OffsetDateTime::UNIX_EPOCH
|
||||||
};
|
};
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
let metadata = &*object_info.user_defined;
|
let metadata = &*object_info.user_defined;
|
||||||
|
|
||||||
@@ -283,13 +319,15 @@ pub(crate) fn replication_put_object_options(sc: &str, object_info: &ObjectInfo)
|
|||||||
put_options.cache_control = cache_control.to_string();
|
put_options.cache_control = cache_control.to_string();
|
||||||
}
|
}
|
||||||
|
|
||||||
if let Some(mode) = metadata.lookup(AMZ_OBJECT_LOCK_MODE) {
|
if let Some(mode) = metadata.lookup(AMZ_OBJECT_LOCK_MODE).filter(|mode| !mode.is_empty()) {
|
||||||
put_options.mode = Some(ObjectLockRetentionMode::from(mode.to_uppercase().as_str()));
|
put_options.mode = Some(ObjectLockRetentionMode::from(mode.to_uppercase().as_str()));
|
||||||
}
|
}
|
||||||
|
|
||||||
if let Some(retain_until_date) = metadata.lookup(AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE) {
|
if let Some(retain_until_date) = metadata.lookup(AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE) {
|
||||||
|
if !retain_until_date.is_empty() {
|
||||||
put_options.retain_until_date = OffsetDateTime::parse(retain_until_date, &Rfc3339)
|
put_options.retain_until_date = OffsetDateTime::parse(retain_until_date, &Rfc3339)
|
||||||
.map_err(|err| Error::other(format!("Failed to parse retain until date: {err}")))?;
|
.map_err(|err| Error::other(format!("Failed to parse retain until date: {err}")))?;
|
||||||
|
}
|
||||||
put_options.internal.retention_timestamp =
|
put_options.internal.retention_timestamp =
|
||||||
if let Some(timestamp) = get_str(&object_info.user_defined, SUFFIX_OBJECTLOCK_RETENTION_TIMESTAMP) {
|
if let Some(timestamp) = get_str(&object_info.user_defined, SUFFIX_OBJECTLOCK_RETENTION_TIMESTAMP) {
|
||||||
OffsetDateTime::parse(×tamp, &Rfc3339).unwrap_or(OffsetDateTime::UNIX_EPOCH)
|
OffsetDateTime::parse(×tamp, &Rfc3339).unwrap_or(OffsetDateTime::UNIX_EPOCH)
|
||||||
@@ -694,6 +732,110 @@ mod tests {
|
|||||||
assert!(options.internal.replication_request);
|
assert!(options.internal.replication_request);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// DeleteObjectTagging leaves the object tagless but stamps the
|
||||||
|
/// tagging-timestamp internal key; the deletion's LWW timestamp must
|
||||||
|
/// still be loaded (and therefore sent) so the replica can order the
|
||||||
|
/// deletion against concurrent tag edits.
|
||||||
|
#[test]
|
||||||
|
fn replication_put_options_carry_tagging_timestamp_after_tag_deletion() {
|
||||||
|
let mut metadata = std::collections::HashMap::new();
|
||||||
|
rustfs_utils::http::insert_str(&mut metadata, SUFFIX_TAGGING_TIMESTAMP, "2026-01-02T03:04:05Z".to_string());
|
||||||
|
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
user_defined: Arc::new(metadata),
|
||||||
|
user_tags: Arc::new(String::new()),
|
||||||
|
mod_time: Some(OffsetDateTime::UNIX_EPOCH),
|
||||||
|
version_id: Some(Uuid::nil()),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let (options, _) = replication_put_object_options("", &object_info).expect("build put options");
|
||||||
|
|
||||||
|
assert!(options.user_tags.is_empty());
|
||||||
|
assert_eq!(
|
||||||
|
options.internal.tagging_timestamp,
|
||||||
|
OffsetDateTime::parse("2026-01-02T03:04:05Z", &Rfc3339).expect("valid timestamp"),
|
||||||
|
"the stored tagging timestamp must load independently of remaining tags"
|
||||||
|
);
|
||||||
|
|
||||||
|
// A tagless object without the stored key was never tagged: the epoch
|
||||||
|
// default keeps the header unsent.
|
||||||
|
let untagged = ObjectInfo {
|
||||||
|
user_tags: Arc::new(String::new()),
|
||||||
|
mod_time: Some(OffsetDateTime::from_unix_timestamp(1_700_000_000).expect("timestamp")),
|
||||||
|
version_id: Some(Uuid::nil()),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let (options, _) = replication_put_object_options("", &untagged).expect("build put options");
|
||||||
|
assert_eq!(options.internal.tagging_timestamp, OffsetDateTime::UNIX_EPOCH);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn replication_put_options_do_not_promote_legacy_user_timestamp_metadata() {
|
||||||
|
let legacy_keys = [
|
||||||
|
"x-rustfs-source-replication-tagging-timestamp",
|
||||||
|
"x-rustfs-source-replication-retention-timestamp",
|
||||||
|
"x-rustfs-source-replication-legalhold-timestamp",
|
||||||
|
"x-minio-source-replication-tagging-timestamp",
|
||||||
|
"x-minio-source-replication-retention-timestamp",
|
||||||
|
"x-minio-source-replication-legalhold-timestamp",
|
||||||
|
];
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
user_defined: Arc::new(
|
||||||
|
legacy_keys
|
||||||
|
.iter()
|
||||||
|
.map(|key| (key.to_string(), "2099-01-02T03:04:05Z".to_string()))
|
||||||
|
.collect(),
|
||||||
|
),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let (options, _) = replication_put_object_options("", &object_info).expect("build put options");
|
||||||
|
|
||||||
|
for legacy_key in legacy_keys {
|
||||||
|
assert!(!options.user_metadata.contains_key(legacy_key));
|
||||||
|
assert_eq!(
|
||||||
|
options
|
||||||
|
.user_metadata
|
||||||
|
.get(&format!("x-amz-meta-{legacy_key}"))
|
||||||
|
.map(String::as_str),
|
||||||
|
Some("2099-01-02T03:04:05Z")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
assert_eq!(options.internal.tagging_timestamp, OffsetDateTime::UNIX_EPOCH);
|
||||||
|
assert_eq!(options.internal.retention_timestamp, OffsetDateTime::UNIX_EPOCH);
|
||||||
|
assert_eq!(options.internal.legalhold_timestamp, OffsetDateTime::UNIX_EPOCH);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn replication_put_options_carry_retention_timestamp_after_clear() {
|
||||||
|
let mut metadata = HashMap::from([
|
||||||
|
(AMZ_OBJECT_LOCK_MODE.to_string(), String::new()),
|
||||||
|
(AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE.to_string(), String::new()),
|
||||||
|
]);
|
||||||
|
rustfs_utils::http::insert_str(&mut metadata, SUFFIX_OBJECTLOCK_RETENTION_TIMESTAMP, "2026-01-02T03:04:05Z".to_string());
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
user_defined: Arc::new(metadata),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let (options, _) = replication_put_object_options("", &object_info).expect("retention clear must replicate");
|
||||||
|
|
||||||
|
assert!(options.mode.is_none());
|
||||||
|
assert_eq!(options.retain_until_date, OffsetDateTime::UNIX_EPOCH);
|
||||||
|
assert_eq!(
|
||||||
|
options.internal.retention_timestamp,
|
||||||
|
OffsetDateTime::parse("2026-01-02T03:04:05Z", &Rfc3339).expect("valid timestamp")
|
||||||
|
);
|
||||||
|
let headers = options.header();
|
||||||
|
assert!(!headers.contains_key(AMZ_OBJECT_LOCK_MODE));
|
||||||
|
assert!(!headers.contains_key(AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE));
|
||||||
|
assert_eq!(
|
||||||
|
rustfs_utils::http::get_header(&headers, SUFFIX_SOURCE_REPLICATION_RETENTION_TIMESTAMP).as_deref(),
|
||||||
|
Some("2026-01-02T03:04:05Z")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn replication_put_options_strip_encryption_metadata_from_plaintext_objects() {
|
fn replication_put_options_strip_encryption_metadata_from_plaintext_objects() {
|
||||||
use rustfs_utils::http::object_encryption_keys::{INTERNAL_ENCRYPTION_ORIGINAL_SIZE_HEADER, SSEC_ORIGINAL_SIZE_HEADER};
|
use rustfs_utils::http::object_encryption_keys::{INTERNAL_ENCRYPTION_ORIGINAL_SIZE_HEADER, SSEC_ORIGINAL_SIZE_HEADER};
|
||||||
|
|||||||
@@ -40,7 +40,14 @@ impl ARN {
|
|||||||
|
|
||||||
impl Display for ARN {
|
impl Display for ARN {
|
||||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
write!(f, "arn:rustfs:{}:{}:{}:{}", self.arn_type, self.region, self.id, self.bucket)
|
// The `minio` partition is deliberate: madmin-go's ParseARN
|
||||||
|
// hard-rejects any other partition, so native mc/madmin tooling can
|
||||||
|
// only decode remote-target ARNs minted in this form (backlog#1675
|
||||||
|
// P1-7). Legacy `arn:rustfs:` ARNs persisted by older releases stay
|
||||||
|
// readable via the FromStr whitelist below; runtime matching between
|
||||||
|
// targets and replication rules is by full-string equality, so mixed
|
||||||
|
// partitions coexist safely.
|
||||||
|
write!(f, "arn:minio:{}:{}:{}:{}", self.arn_type, self.region, self.id, self.bucket)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -48,7 +55,12 @@ impl FromStr for ARN {
|
|||||||
type Err = std::io::Error;
|
type Err = std::io::Error;
|
||||||
|
|
||||||
fn from_str(s: &str) -> Result<Self, Self::Err> {
|
fn from_str(s: &str) -> Result<Self, Self::Err> {
|
||||||
if !s.starts_with("arn:rustfs:") {
|
// Partition whitelist, not just an `arn:` check: `BucketTargetType::
|
||||||
|
// from_str(...).unwrap_or_default()` below never fails, so this is
|
||||||
|
// the only structural gate rejecting foreign ARNs. `arn:rustfs:` is
|
||||||
|
// the legacy partition and must stay accepted forever (persisted
|
||||||
|
// bucket-targets.json / replication configs from older releases).
|
||||||
|
if !s.starts_with("arn:minio:") && !s.starts_with("arn:rustfs:") {
|
||||||
return Err(std::io::Error::new(std::io::ErrorKind::InvalidInput, "Invalid ARN format"));
|
return Err(std::io::Error::new(std::io::ErrorKind::InvalidInput, "Invalid ARN format"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -101,14 +113,50 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// RustFS commonly generates ARNs with an empty region:
|
/// RustFS commonly generates ARNs with an empty region:
|
||||||
/// `arn:rustfs:replication::<deployment_id>:<bucket>`.
|
/// `arn:minio:replication::<deployment_id>:<bucket>`.
|
||||||
#[test]
|
#[test]
|
||||||
fn from_str_handles_empty_region_segment() {
|
fn from_str_handles_empty_region_segment() {
|
||||||
let parsed = ARN::from_str("arn:rustfs:replication::depl-123:bucket-a").expect("valid ARN must parse");
|
let parsed = ARN::from_str("arn:minio:replication::depl-123:bucket-a").expect("valid ARN must parse");
|
||||||
|
|
||||||
assert_eq!(parsed.arn_type, BucketTargetType::ReplicationService);
|
assert_eq!(parsed.arn_type, BucketTargetType::ReplicationService);
|
||||||
assert_eq!(parsed.region, "", "region segment is empty in this form");
|
assert_eq!(parsed.region, "", "region segment is empty in this form");
|
||||||
assert_eq!(parsed.id, "depl-123");
|
assert_eq!(parsed.id, "depl-123");
|
||||||
assert_eq!(parsed.bucket, "bucket-a");
|
assert_eq!(parsed.bucket, "bucket-a");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// madmin-go's `ParseARN` hard-rejects anything that does not start with
|
||||||
|
/// `arn:minio:`, so generated ARNs must use the `minio` partition or the
|
||||||
|
/// native mc/madmin tooling cannot decode remote-target listings.
|
||||||
|
#[test]
|
||||||
|
fn display_emits_minio_partition() {
|
||||||
|
let arn = ARN::new(
|
||||||
|
BucketTargetType::ReplicationService,
|
||||||
|
"depl-123".to_string(),
|
||||||
|
String::new(),
|
||||||
|
"bucket-a".to_string(),
|
||||||
|
);
|
||||||
|
|
||||||
|
assert_eq!(arn.to_string(), "arn:minio:replication::depl-123:bucket-a");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Persisted bucket-targets.json files from older RustFS releases carry
|
||||||
|
/// `arn:rustfs:` ARNs; the legacy partition must stay parseable forever.
|
||||||
|
#[test]
|
||||||
|
fn from_str_accepts_legacy_rustfs_partition() {
|
||||||
|
let parsed = ARN::from_str("arn:rustfs:replication:us-east-1:depl-123:bucket-a").expect("legacy ARN must parse");
|
||||||
|
|
||||||
|
assert_eq!(parsed.arn_type, BucketTargetType::ReplicationService);
|
||||||
|
assert_eq!(parsed.region, "us-east-1");
|
||||||
|
assert_eq!(parsed.id, "depl-123");
|
||||||
|
assert_eq!(parsed.bucket, "bucket-a");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The partition whitelist is the only structural gate: `BucketTargetType::
|
||||||
|
/// from_str(...).unwrap_or_default()` never fails, so any 6-segment string
|
||||||
|
/// would otherwise parse as `type=None`.
|
||||||
|
#[test]
|
||||||
|
fn from_str_rejects_unknown_partition() {
|
||||||
|
assert!(ARN::from_str("arn:aws:replication::depl-123:bucket-a").is_err());
|
||||||
|
assert!(ARN::from_str("not-an-arn").is_err());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -15,6 +15,7 @@
|
|||||||
use crate::disk::disk_store::{get_drive_walkdir_peek_timeout, get_drive_walkdir_stall_timeout};
|
use crate::disk::disk_store::{get_drive_walkdir_peek_timeout, get_drive_walkdir_stall_timeout};
|
||||||
use crate::disk::error::DiskError;
|
use crate::disk::error::DiskError;
|
||||||
use crate::disk::{self, DiskAPI, DiskStore, WalkDirOptions};
|
use crate::disk::{self, DiskAPI, DiskStore, WalkDirOptions};
|
||||||
|
use futures::future::join_all;
|
||||||
use metrics::counter;
|
use metrics::counter;
|
||||||
use rustfs_filemeta::{MetaCacheEntries, MetaCacheEntry, MetacacheReader, is_io_eof};
|
use rustfs_filemeta::{MetaCacheEntries, MetaCacheEntry, MetacacheReader, is_io_eof};
|
||||||
use std::{
|
use std::{
|
||||||
@@ -655,6 +656,7 @@ async fn list_path_raw_inner(
|
|||||||
errs.push(None);
|
errs.push(None);
|
||||||
}
|
}
|
||||||
let mut pending_entries: Vec<Option<MetaCacheEntry>> = vec![None; readers.len()];
|
let mut pending_entries: Vec<Option<MetaCacheEntry>> = vec![None; readers.len()];
|
||||||
|
let mut peek_outcomes: Vec<Option<PeekOutcome>> = std::iter::repeat_with(|| None).take(readers.len()).collect();
|
||||||
|
|
||||||
loop {
|
loop {
|
||||||
let mut current = MetaCacheEntry::default();
|
let mut current = MetaCacheEntry::default();
|
||||||
@@ -676,6 +678,21 @@ async fn list_path_raw_inner(
|
|||||||
let mut has_err = 0;
|
let mut has_err = 0;
|
||||||
let mut agree = 0;
|
let mut agree = 0;
|
||||||
|
|
||||||
|
// Start every missing head read in the same round so one stalled
|
||||||
|
// disk cannot multiply the wait budget by the erasure-set width.
|
||||||
|
// Outcomes are still consumed below in stable disk-index order.
|
||||||
|
let concurrent_peeks = readers.iter_mut().enumerate().filter_map(|(i, reader)| {
|
||||||
|
if errs[i].is_some() || pending_entries[i].is_some() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
|
||||||
|
let cancel = &revjob_rx;
|
||||||
|
Some(async move { (i, peek_with_timeout(cancel, reader, peek_timeout).await) })
|
||||||
|
});
|
||||||
|
for (i, outcome) in join_all(concurrent_peeks).await {
|
||||||
|
peek_outcomes[i] = Some(outcome);
|
||||||
|
}
|
||||||
|
|
||||||
for (i, r) in readers.iter_mut().enumerate() {
|
for (i, r) in readers.iter_mut().enumerate() {
|
||||||
if errs[i].is_some() {
|
if errs[i].is_some() {
|
||||||
has_err += 1;
|
has_err += 1;
|
||||||
@@ -685,7 +702,10 @@ async fn list_path_raw_inner(
|
|||||||
let entry = if let Some(entry) = pending_entries[i].take() {
|
let entry = if let Some(entry) = pending_entries[i].take() {
|
||||||
entry
|
entry
|
||||||
} else {
|
} else {
|
||||||
match peek_with_timeout(&revjob_rx, r, peek_timeout).await {
|
let Some(outcome) = peek_outcomes[i].take() else {
|
||||||
|
return Err(DiskError::Unexpected);
|
||||||
|
};
|
||||||
|
match outcome {
|
||||||
PeekOutcome::Ready(res) => {
|
PeekOutcome::Ready(res) => {
|
||||||
if let Some(entry) = res {
|
if let Some(entry) = res {
|
||||||
// info!("read entry disk: {}, name: {}", i, entry.name);
|
// info!("read entry disk: {}, name: {}", i, entry.name);
|
||||||
@@ -1295,6 +1315,36 @@ mod tests {
|
|||||||
assert_eq!(err, DiskError::Timeout);
|
assert_eq!(err, DiskError::Timeout);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
async fn list_path_raw_bounds_multiple_stalled_readers_by_one_peek_deadline() {
|
||||||
|
let peek_timeout = Duration::from_millis(20);
|
||||||
|
let started = tokio::time::Instant::now();
|
||||||
|
let err = list_path_raw(
|
||||||
|
CancellationToken::new(),
|
||||||
|
ListPathRawOptions {
|
||||||
|
disks: vec![None, None, None, None],
|
||||||
|
min_disks: 1,
|
||||||
|
test_reader_behaviors: vec![
|
||||||
|
TestReaderBehavior::Stall,
|
||||||
|
TestReaderBehavior::Stall,
|
||||||
|
TestReaderBehavior::Stall,
|
||||||
|
TestReaderBehavior::Stall,
|
||||||
|
],
|
||||||
|
peek_timeout: Some(peek_timeout),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("all stalled readers should fail the listing");
|
||||||
|
|
||||||
|
assert_eq!(err, DiskError::Timeout);
|
||||||
|
assert_eq!(
|
||||||
|
started.elapsed(),
|
||||||
|
peek_timeout,
|
||||||
|
"reader deadlines must overlap instead of accumulating once per disk"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn list_path_raw_waits_past_producer_stall_for_slow_progressing_reader() {
|
async fn list_path_raw_waits_past_producer_stall_for_slow_progressing_reader() {
|
||||||
let entry = MetaCacheEntry {
|
let entry = MetaCacheEntry {
|
||||||
|
|||||||
@@ -1,171 +0,0 @@
|
|||||||
// Copyright 2024 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
#![allow(unused_imports)]
|
|
||||||
#![allow(unused_variables)]
|
|
||||||
#![allow(unused_mut)]
|
|
||||||
#![allow(unused_assignments)]
|
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use http::{HeaderMap, StatusCode};
|
|
||||||
use http_body_util::BodyExt;
|
|
||||||
use hyper::body::Body;
|
|
||||||
use hyper::body::Bytes;
|
|
||||||
use std::collections::HashMap;
|
|
||||||
|
|
||||||
use crate::client::{
|
|
||||||
api_error_response::http_resp_to_error_response,
|
|
||||||
transition_api::{ReaderImpl, RequestMetadata, TransitionClient},
|
|
||||||
};
|
|
||||||
use rustfs_utils::hash::EMPTY_STRING_SHA256_HASH;
|
|
||||||
|
|
||||||
impl TransitionClient {
|
|
||||||
pub async fn set_bucket_policy(&self, bucket_name: &str, policy: &str) -> Result<(), std::io::Error> {
|
|
||||||
if policy == "" {
|
|
||||||
return self.remove_bucket_policy(bucket_name).await;
|
|
||||||
}
|
|
||||||
|
|
||||||
self.put_bucket_policy(bucket_name, policy).await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn put_bucket_policy(&self, bucket_name: &str, policy: &str) -> Result<(), std::io::Error> {
|
|
||||||
let mut url_values = HashMap::new();
|
|
||||||
url_values.insert("policy".to_string(), "".to_string());
|
|
||||||
|
|
||||||
let mut req_metadata = RequestMetadata {
|
|
||||||
bucket_name: bucket_name.to_string(),
|
|
||||||
query_values: url_values,
|
|
||||||
content_body: ReaderImpl::Body(Bytes::from(policy.as_bytes().to_vec())),
|
|
||||||
content_length: policy.len() as i64,
|
|
||||||
object_name: "".to_string(),
|
|
||||||
custom_header: HeaderMap::new(),
|
|
||||||
content_md5_base64: "".to_string(),
|
|
||||||
content_sha256_hex: "".to_string(),
|
|
||||||
stream_sha256: false,
|
|
||||||
trailer: HeaderMap::new(),
|
|
||||||
pre_sign_url: Default::default(),
|
|
||||||
add_crc: Default::default(),
|
|
||||||
extra_pre_sign_header: Default::default(),
|
|
||||||
bucket_location: Default::default(),
|
|
||||||
expires: Default::default(),
|
|
||||||
};
|
|
||||||
|
|
||||||
let resp = self.execute_method(http::Method::PUT, &mut req_metadata).await?;
|
|
||||||
//defer closeResponse(resp)
|
|
||||||
|
|
||||||
let resp_status = resp.status();
|
|
||||||
let h = resp.headers().clone();
|
|
||||||
|
|
||||||
//if resp != nil {
|
|
||||||
if resp_status != StatusCode::NO_CONTENT && resp.status() != StatusCode::OK {
|
|
||||||
return Err(std::io::Error::other(http_resp_to_error_response(
|
|
||||||
resp_status,
|
|
||||||
&h,
|
|
||||||
vec![],
|
|
||||||
bucket_name,
|
|
||||||
"",
|
|
||||||
)));
|
|
||||||
}
|
|
||||||
//}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn remove_bucket_policy(&self, bucket_name: &str) -> Result<(), std::io::Error> {
|
|
||||||
let mut url_values = HashMap::new();
|
|
||||||
url_values.insert("policy".to_string(), "".to_string());
|
|
||||||
|
|
||||||
let resp = self
|
|
||||||
.execute_method(
|
|
||||||
http::Method::DELETE,
|
|
||||||
&mut RequestMetadata {
|
|
||||||
bucket_name: bucket_name.to_string(),
|
|
||||||
query_values: url_values,
|
|
||||||
content_sha256_hex: EMPTY_STRING_SHA256_HASH.to_string(),
|
|
||||||
object_name: "".to_string(),
|
|
||||||
custom_header: HeaderMap::new(),
|
|
||||||
content_body: ReaderImpl::Body(Bytes::new()),
|
|
||||||
content_length: 0,
|
|
||||||
content_md5_base64: "".to_string(),
|
|
||||||
stream_sha256: false,
|
|
||||||
trailer: HeaderMap::new(),
|
|
||||||
pre_sign_url: Default::default(),
|
|
||||||
add_crc: Default::default(),
|
|
||||||
extra_pre_sign_header: Default::default(),
|
|
||||||
bucket_location: Default::default(),
|
|
||||||
expires: Default::default(),
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
//defer closeResponse(resp)
|
|
||||||
|
|
||||||
let resp_status = resp.status();
|
|
||||||
let h = resp.headers().clone();
|
|
||||||
|
|
||||||
if resp_status != StatusCode::NO_CONTENT {
|
|
||||||
return Err(std::io::Error::other(http_resp_to_error_response(
|
|
||||||
resp_status,
|
|
||||||
&h,
|
|
||||||
vec![],
|
|
||||||
bucket_name,
|
|
||||||
"",
|
|
||||||
)));
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_bucket_policy(&self, bucket_name: &str) -> Result<String, std::io::Error> {
|
|
||||||
let bucket_policy = self.get_bucket_policy_inner(bucket_name).await?;
|
|
||||||
Ok(bucket_policy)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_bucket_policy_inner(&self, bucket_name: &str) -> Result<String, std::io::Error> {
|
|
||||||
let mut url_values = HashMap::new();
|
|
||||||
url_values.insert("policy".to_string(), "".to_string());
|
|
||||||
|
|
||||||
let resp = self
|
|
||||||
.execute_method(
|
|
||||||
http::Method::GET,
|
|
||||||
&mut RequestMetadata {
|
|
||||||
bucket_name: bucket_name.to_string(),
|
|
||||||
query_values: url_values,
|
|
||||||
content_sha256_hex: EMPTY_STRING_SHA256_HASH.to_string(),
|
|
||||||
object_name: "".to_string(),
|
|
||||||
custom_header: HeaderMap::new(),
|
|
||||||
content_body: ReaderImpl::Body(Bytes::new()),
|
|
||||||
content_length: 0,
|
|
||||||
content_md5_base64: "".to_string(),
|
|
||||||
stream_sha256: false,
|
|
||||||
trailer: HeaderMap::new(),
|
|
||||||
pre_sign_url: Default::default(),
|
|
||||||
add_crc: Default::default(),
|
|
||||||
extra_pre_sign_header: Default::default(),
|
|
||||||
bucket_location: Default::default(),
|
|
||||||
expires: Default::default(),
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let mut body_vec = Vec::new();
|
|
||||||
let mut body = resp.into_body();
|
|
||||||
while let Some(frame) = body.frame().await {
|
|
||||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
|
||||||
if let Some(data) = frame.data_ref() {
|
|
||||||
body_vec.extend_from_slice(data);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
let policy = String::from_utf8_lossy(&body_vec).to_string();
|
|
||||||
Ok(policy)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -229,17 +229,6 @@ pub fn http_resp_to_error_response(
|
|||||||
err_resp
|
err_resp
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn err_transfer_acceleration_bucket(bucket_name: &str) -> ErrorResponse {
|
|
||||||
ErrorResponse {
|
|
||||||
status_code: StatusCode::BAD_REQUEST,
|
|
||||||
code: S3ErrorCode::InvalidArgument,
|
|
||||||
message: "The name of the bucket used for Transfer Acceleration must be DNS-compliant and must not contain periods ‘.’."
|
|
||||||
.to_string(),
|
|
||||||
bucket_name: bucket_name.to_string(),
|
|
||||||
..Default::default()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn err_entity_too_large(total_size: i64, max_object_size: i64, bucket_name: &str, object_name: &str) -> ErrorResponse {
|
pub fn err_entity_too_large(total_size: i64, max_object_size: i64, bucket_name: &str, object_name: &str) -> ErrorResponse {
|
||||||
let msg = format!(
|
let msg = format!(
|
||||||
"Your proposed upload size ‘{}’ exceeds the maximum allowed object size ‘{}’ for single PUT operation.",
|
"Your proposed upload size ‘{}’ exceeds the maximum allowed object size ‘{}’ for single PUT operation.",
|
||||||
@@ -295,16 +284,6 @@ pub fn err_invalid_argument(message: &str) -> ErrorResponse {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn err_api_not_supported(message: &str) -> ErrorResponse {
|
|
||||||
ErrorResponse {
|
|
||||||
status_code: StatusCode::NOT_IMPLEMENTED,
|
|
||||||
code: S3ErrorCode::Custom("APINotSupported".into()),
|
|
||||||
message: message.to_string(),
|
|
||||||
request_id: "rustfs".to_string(),
|
|
||||||
..Default::default()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|||||||
@@ -135,6 +135,10 @@ impl Object {
|
|||||||
Self { ..Default::default() }
|
Self { ..Default::default() }
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity reader surface with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn do_get_request(&self, request: &GetRequest) -> Result<GetResponse, std::io::Error> {
|
fn do_get_request(&self, request: &GetRequest) -> Result<GetResponse, std::io::Error> {
|
||||||
let _ = request.did_offset_change;
|
let _ = request.did_offset_change;
|
||||||
let _ = request.offset;
|
let _ = request.offset;
|
||||||
@@ -150,12 +154,20 @@ impl Object {
|
|||||||
))
|
))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity Object reader method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn set_offset(&mut self, bytes_read: i64) -> Result<(), std::io::Error> {
|
fn set_offset(&mut self, bytes_read: i64) -> Result<(), std::io::Error> {
|
||||||
self.curr_offset += bytes_read;
|
self.curr_offset += bytes_read;
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity Object reader method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn read(&mut self, b: &[u8]) -> Result<i64, std::io::Error> {
|
fn read(&mut self, b: &[u8]) -> Result<i64, std::io::Error> {
|
||||||
let mut read_req = GetRequest {
|
let mut read_req = GetRequest {
|
||||||
is_read_op: true,
|
is_read_op: true,
|
||||||
@@ -180,6 +192,10 @@ impl Object {
|
|||||||
Ok(response.size)
|
Ok(response.size)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity Object reader method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn stat(&self) -> Result<ObjectInfo, std::io::Error> {
|
fn stat(&self) -> Result<ObjectInfo, std::io::Error> {
|
||||||
if !self.is_started || !self.object_info_set {
|
if !self.is_started || !self.object_info_set {
|
||||||
let _ = self.do_get_request(&GetRequest {
|
let _ = self.do_get_request(&GetRequest {
|
||||||
@@ -192,6 +208,10 @@ impl Object {
|
|||||||
Ok(self.object_info.clone())
|
Ok(self.object_info.clone())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity Object reader method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn read_at(&mut self, b: &[u8], offset: i64) -> Result<i64, std::io::Error> {
|
fn read_at(&mut self, b: &[u8], offset: i64) -> Result<i64, std::io::Error> {
|
||||||
self.curr_offset = offset;
|
self.curr_offset = offset;
|
||||||
|
|
||||||
@@ -219,6 +239,10 @@ impl Object {
|
|||||||
Ok(response.size)
|
Ok(response.size)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity Object reader method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn seek(&mut self, offset: i64, whence: i64) -> Result<i64, std::io::Error> {
|
fn seek(&mut self, offset: i64, whence: i64) -> Result<i64, std::io::Error> {
|
||||||
if !self.is_started || !self.object_info_set {
|
if !self.is_started || !self.object_info_set {
|
||||||
let seek_req = GetRequest {
|
let seek_req = GetRequest {
|
||||||
@@ -253,6 +277,10 @@ impl Object {
|
|||||||
Ok(self.curr_offset)
|
Ok(self.curr_offset)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity Object reader method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn close(&mut self) -> Result<(), std::io::Error> {
|
fn close(&mut self) -> Result<(), std::io::Error> {
|
||||||
self.is_closed = true;
|
self.is_closed = true;
|
||||||
Ok(())
|
Ok(())
|
||||||
|
|||||||
@@ -1,199 +0,0 @@
|
|||||||
// Copyright 2024 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
#![allow(unused_imports)]
|
|
||||||
#![allow(unused_variables)]
|
|
||||||
#![allow(unused_mut)]
|
|
||||||
#![allow(unused_assignments)]
|
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use crate::client::{
|
|
||||||
api_error_response::http_resp_to_error_response,
|
|
||||||
api_get_options::GetObjectOptions,
|
|
||||||
transition_api::{ObjectInfo, ReaderImpl, RequestMetadata, TransitionClient},
|
|
||||||
};
|
|
||||||
use bytes::Bytes;
|
|
||||||
use http::{HeaderMap, HeaderValue};
|
|
||||||
use http_body_util::BodyExt;
|
|
||||||
use rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE;
|
|
||||||
use rustfs_utils::EMPTY_STRING_SHA256_HASH;
|
|
||||||
use s3s::dto::Owner;
|
|
||||||
use std::collections::HashMap;
|
|
||||||
|
|
||||||
#[derive(Clone, Debug, Default, serde::Serialize, serde::Deserialize)]
|
|
||||||
pub struct Grantee {
|
|
||||||
pub id: String,
|
|
||||||
pub display_name: String,
|
|
||||||
pub uri: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Clone, Debug, Default, serde::Serialize, serde::Deserialize)]
|
|
||||||
pub struct Grant {
|
|
||||||
pub grantee: Grantee,
|
|
||||||
pub permission: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Serialize, serde::Deserialize)]
|
|
||||||
pub struct AccessControlList {
|
|
||||||
pub grant: Vec<Grant>,
|
|
||||||
pub permission: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Deserialize)]
|
|
||||||
pub struct AccessControlPolicy {
|
|
||||||
#[serde(skip)]
|
|
||||||
owner: Owner,
|
|
||||||
pub access_control_list: AccessControlList,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TransitionClient {
|
|
||||||
pub async fn get_object_acl(&self, bucket_name: &str, object_name: &str) -> Result<ObjectInfo, std::io::Error> {
|
|
||||||
let mut url_values = HashMap::new();
|
|
||||||
url_values.insert("acl".to_string(), "".to_string());
|
|
||||||
let mut resp = self
|
|
||||||
.execute_method(
|
|
||||||
http::Method::GET,
|
|
||||||
&mut RequestMetadata {
|
|
||||||
bucket_name: bucket_name.to_string(),
|
|
||||||
object_name: object_name.to_string(),
|
|
||||||
query_values: url_values,
|
|
||||||
custom_header: HeaderMap::new(),
|
|
||||||
content_sha256_hex: EMPTY_STRING_SHA256_HASH.to_string(),
|
|
||||||
content_body: ReaderImpl::Body(Bytes::new()),
|
|
||||||
content_length: 0,
|
|
||||||
content_md5_base64: "".to_string(),
|
|
||||||
stream_sha256: false,
|
|
||||||
trailer: HeaderMap::new(),
|
|
||||||
pre_sign_url: Default::default(),
|
|
||||||
add_crc: Default::default(),
|
|
||||||
extra_pre_sign_header: Default::default(),
|
|
||||||
bucket_location: Default::default(),
|
|
||||||
expires: Default::default(),
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let resp_status = resp.status();
|
|
||||||
let h = resp.headers().clone();
|
|
||||||
|
|
||||||
let mut body_vec = Vec::new();
|
|
||||||
let mut body = resp.into_body();
|
|
||||||
while let Some(frame) = body.frame().await {
|
|
||||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
|
||||||
if let Some(data) = frame.data_ref() {
|
|
||||||
body_vec.extend_from_slice(data);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if resp_status != http::StatusCode::OK {
|
|
||||||
return Err(std::io::Error::other(http_resp_to_error_response(
|
|
||||||
resp_status,
|
|
||||||
&h,
|
|
||||||
body_vec,
|
|
||||||
bucket_name,
|
|
||||||
object_name,
|
|
||||||
)));
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut res = match quick_xml::de::from_str::<AccessControlPolicy>(&String::from_utf8(body_vec).unwrap()) {
|
|
||||||
Ok(result) => result,
|
|
||||||
Err(err) => {
|
|
||||||
return Err(std::io::Error::other(err.to_string()));
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
let mut obj_info = self
|
|
||||||
.stat_object(bucket_name, object_name, &GetObjectOptions::default())
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
obj_info.owner.display_name = res.owner.display_name.clone();
|
|
||||||
obj_info.owner.id = res.owner.id.clone();
|
|
||||||
|
|
||||||
//obj_info.grant.extend(res.access_control_list.grant);
|
|
||||||
|
|
||||||
let canned_acl = get_canned_acl(&res);
|
|
||||||
if canned_acl != "" {
|
|
||||||
obj_info
|
|
||||||
.metadata
|
|
||||||
.insert("X-Amz-Acl", HeaderValue::from_str(&canned_acl).unwrap());
|
|
||||||
return Ok(obj_info);
|
|
||||||
}
|
|
||||||
|
|
||||||
let grant_acl = get_amz_grant_acl(&res);
|
|
||||||
/*for (k, v) in grant_acl {
|
|
||||||
obj_info.metadata.insert(HeaderName::from_bytes(k.as_bytes()).unwrap(), HeaderValue::from_str(&v.to_string()).unwrap());
|
|
||||||
}*/
|
|
||||||
|
|
||||||
Ok(obj_info)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn get_canned_acl(ac_policy: &AccessControlPolicy) -> String {
|
|
||||||
let grants = ac_policy.access_control_list.grant.clone();
|
|
||||||
|
|
||||||
if grants.len() == 1 {
|
|
||||||
if grants[0].grantee.uri == "" && grants[0].permission == "FULL_CONTROL" {
|
|
||||||
return "private".to_string();
|
|
||||||
}
|
|
||||||
} else if grants.len() == 2 {
|
|
||||||
for g in grants {
|
|
||||||
if g.grantee.uri == "http://acs.amazonaws.com/groups/global/AuthenticatedUsers" && &g.permission == "READ" {
|
|
||||||
return "authenticated-read".to_string();
|
|
||||||
}
|
|
||||||
if g.grantee.uri == "http://acs.amazonaws.com/groups/global/AllUsers" && &g.permission == "READ" {
|
|
||||||
return "public-read".to_string();
|
|
||||||
}
|
|
||||||
if g.permission == "READ" && g.grantee.id == ac_policy.owner.id.clone().unwrap() {
|
|
||||||
return "bucket-owner-read".to_string();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else if grants.len() == 3 {
|
|
||||||
for g in grants {
|
|
||||||
if g.grantee.uri == "http://acs.amazonaws.com/groups/global/AllUsers" && g.permission == "WRITE" {
|
|
||||||
return "public-read-write".to_string();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
"".to_string()
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn get_amz_grant_acl(ac_policy: &AccessControlPolicy) -> HashMap<String, Vec<String>> {
|
|
||||||
let grants = ac_policy.access_control_list.grant.clone();
|
|
||||||
let mut res = HashMap::<String, Vec<String>>::new();
|
|
||||||
|
|
||||||
for g in grants {
|
|
||||||
let mut id = "id=".to_string();
|
|
||||||
id.push_str(&g.grantee.id);
|
|
||||||
let permission: &str = &g.permission;
|
|
||||||
match permission {
|
|
||||||
"READ" => {
|
|
||||||
res.entry("X-Amz-Grant-Read".to_string()).or_insert(vec![]).push(id);
|
|
||||||
}
|
|
||||||
"WRITE" => {
|
|
||||||
res.entry("X-Amz-Grant-Write".to_string()).or_insert(vec![]).push(id);
|
|
||||||
}
|
|
||||||
"READ_ACP" => {
|
|
||||||
res.entry("X-Amz-Grant-Read-Acp".to_string()).or_insert(vec![]).push(id);
|
|
||||||
}
|
|
||||||
"WRITE_ACP" => {
|
|
||||||
res.entry("X-Amz-Grant-Write-Acp".to_string()).or_insert(vec![]).push(id);
|
|
||||||
}
|
|
||||||
"FULL_CONTROL" => {
|
|
||||||
res.entry("X-Amz-Grant-Full-Control".to_string()).or_insert(vec![]).push(id);
|
|
||||||
}
|
|
||||||
_ => (),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
res
|
|
||||||
}
|
|
||||||
@@ -1,266 +0,0 @@
|
|||||||
// Copyright 2024 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
#![allow(unused_imports)]
|
|
||||||
#![allow(unused_variables)]
|
|
||||||
#![allow(unused_mut)]
|
|
||||||
#![allow(unused_assignments)]
|
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use http::{HeaderMap, HeaderValue};
|
|
||||||
use std::collections::HashMap;
|
|
||||||
use time::OffsetDateTime;
|
|
||||||
|
|
||||||
use crate::client::constants::{GET_OBJECT_ATTRIBUTES_MAX_PARTS, GET_OBJECT_ATTRIBUTES_TAGS, ISO8601_DATEFORMAT};
|
|
||||||
use crate::client::{
|
|
||||||
api_get_object_acl::AccessControlPolicy,
|
|
||||||
transition_api::{ReaderImpl, RequestMetadata, TransitionClient},
|
|
||||||
};
|
|
||||||
use http_body_util::BodyExt;
|
|
||||||
use hyper::body::Body;
|
|
||||||
use hyper::body::Bytes;
|
|
||||||
use hyper::body::Incoming;
|
|
||||||
use rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE;
|
|
||||||
use rustfs_utils::EMPTY_STRING_SHA256_HASH;
|
|
||||||
use s3s::header::{X_AMZ_MAX_PARTS, X_AMZ_OBJECT_ATTRIBUTES, X_AMZ_PART_NUMBER_MARKER, X_AMZ_VERSION_ID};
|
|
||||||
|
|
||||||
pub struct ObjectAttributesOptions {
|
|
||||||
pub max_parts: i64,
|
|
||||||
pub version_id: String,
|
|
||||||
pub part_number_marker: i64,
|
|
||||||
//server_side_encryption: encrypt::ServerSide,
|
|
||||||
}
|
|
||||||
|
|
||||||
pub struct ObjectAttributes {
|
|
||||||
pub version_id: String,
|
|
||||||
pub last_modified: OffsetDateTime,
|
|
||||||
pub object_attributes_response: ObjectAttributesResponse,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl ObjectAttributes {
|
|
||||||
fn new() -> Self {
|
|
||||||
Self {
|
|
||||||
version_id: "".to_string(),
|
|
||||||
last_modified: OffsetDateTime::now_utc(),
|
|
||||||
object_attributes_response: ObjectAttributesResponse::new(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Deserialize)]
|
|
||||||
pub struct Checksum {
|
|
||||||
checksum_crc32: String,
|
|
||||||
checksum_crc32c: String,
|
|
||||||
checksum_sha1: String,
|
|
||||||
checksum_sha256: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Checksum {
|
|
||||||
fn new() -> Self {
|
|
||||||
Self {
|
|
||||||
checksum_crc32: "".to_string(),
|
|
||||||
checksum_crc32c: "".to_string(),
|
|
||||||
checksum_sha1: "".to_string(),
|
|
||||||
checksum_sha256: "".to_string(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Deserialize)]
|
|
||||||
pub struct ObjectParts {
|
|
||||||
pub parts_count: i64,
|
|
||||||
pub part_number_marker: i64,
|
|
||||||
pub next_part_number_marker: i64,
|
|
||||||
pub max_parts: i64,
|
|
||||||
is_truncated: bool,
|
|
||||||
parts: Vec<ObjectAttributePart>,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl ObjectParts {
|
|
||||||
fn new() -> Self {
|
|
||||||
Self {
|
|
||||||
parts_count: 0,
|
|
||||||
part_number_marker: 0,
|
|
||||||
next_part_number_marker: 0,
|
|
||||||
max_parts: 0,
|
|
||||||
is_truncated: false,
|
|
||||||
parts: Vec::new(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Deserialize)]
|
|
||||||
pub struct ObjectAttributesResponse {
|
|
||||||
pub etag: String,
|
|
||||||
pub storage_class: String,
|
|
||||||
pub object_size: i64,
|
|
||||||
pub checksum: Checksum,
|
|
||||||
pub object_parts: ObjectParts,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl ObjectAttributesResponse {
|
|
||||||
fn new() -> Self {
|
|
||||||
Self {
|
|
||||||
etag: "".to_string(),
|
|
||||||
storage_class: "".to_string(),
|
|
||||||
object_size: 0,
|
|
||||||
checksum: Checksum::new(),
|
|
||||||
object_parts: ObjectParts::new(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Deserialize)]
|
|
||||||
struct ObjectAttributePart {
|
|
||||||
checksum_crc32: String,
|
|
||||||
checksum_crc32c: String,
|
|
||||||
checksum_sha1: String,
|
|
||||||
checksum_sha256: String,
|
|
||||||
part_number: i64,
|
|
||||||
size: i64,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl ObjectAttributes {
|
|
||||||
pub async fn parse_response(&mut self, h: &HeaderMap, body_vec: Vec<u8>) -> Result<(), std::io::Error> {
|
|
||||||
let last_modified = h
|
|
||||||
.get("Last-Modified")
|
|
||||||
.ok_or_else(|| std::io::Error::other("missing Last-Modified header"))?
|
|
||||||
.to_str()
|
|
||||||
.map_err(|e| std::io::Error::other(format!("invalid Last-Modified header: {e}")))?;
|
|
||||||
let mod_time = OffsetDateTime::parse(last_modified, ISO8601_DATEFORMAT)
|
|
||||||
.map_err(|e| std::io::Error::other(format!("invalid Last-Modified date: {e}")))?;
|
|
||||||
self.last_modified = mod_time;
|
|
||||||
|
|
||||||
let version_id = h
|
|
||||||
.get(X_AMZ_VERSION_ID)
|
|
||||||
.ok_or_else(|| std::io::Error::other("missing version ID header"))?
|
|
||||||
.to_str()
|
|
||||||
.map_err(|e| std::io::Error::other(format!("invalid version ID header: {e}")))?;
|
|
||||||
self.version_id = version_id.to_string();
|
|
||||||
|
|
||||||
let body_str = String::from_utf8(body_vec).map_err(|e| std::io::Error::other(format!("invalid UTF-8 body: {e}")))?;
|
|
||||||
let mut response = match quick_xml::de::from_str::<ObjectAttributesResponse>(&body_str) {
|
|
||||||
Ok(result) => result,
|
|
||||||
Err(err) => {
|
|
||||||
return Err(std::io::Error::other(err.to_string()));
|
|
||||||
}
|
|
||||||
};
|
|
||||||
self.object_attributes_response = response;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TransitionClient {
|
|
||||||
pub async fn get_object_attributes(
|
|
||||||
&self,
|
|
||||||
bucket_name: &str,
|
|
||||||
object_name: &str,
|
|
||||||
opts: ObjectAttributesOptions,
|
|
||||||
) -> Result<ObjectAttributes, std::io::Error> {
|
|
||||||
let mut url_values = HashMap::new();
|
|
||||||
url_values.insert("attributes".to_string(), "".to_string());
|
|
||||||
if opts.version_id != "" {
|
|
||||||
url_values.insert("versionId".to_string(), opts.version_id);
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut headers = HeaderMap::new();
|
|
||||||
headers.insert(
|
|
||||||
X_AMZ_OBJECT_ATTRIBUTES,
|
|
||||||
HeaderValue::from_str(GET_OBJECT_ATTRIBUTES_TAGS).expect("valid header value"),
|
|
||||||
);
|
|
||||||
|
|
||||||
if opts.part_number_marker > 0 {
|
|
||||||
headers.insert(
|
|
||||||
X_AMZ_PART_NUMBER_MARKER,
|
|
||||||
HeaderValue::from_str(&opts.part_number_marker.to_string()).expect("valid header value"),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
if opts.max_parts > 0 {
|
|
||||||
headers.insert(
|
|
||||||
X_AMZ_MAX_PARTS,
|
|
||||||
HeaderValue::from_str(&opts.max_parts.to_string()).expect("valid header value"),
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
headers.insert(
|
|
||||||
X_AMZ_MAX_PARTS,
|
|
||||||
HeaderValue::from_str(&GET_OBJECT_ATTRIBUTES_MAX_PARTS.to_string()).expect("valid header value"),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
/*if opts.server_side_encryption.is_some() {
|
|
||||||
opts.server_side_encryption.Marshal(headers);
|
|
||||||
}*/
|
|
||||||
|
|
||||||
let mut resp = self
|
|
||||||
.execute_method(
|
|
||||||
http::Method::HEAD,
|
|
||||||
&mut RequestMetadata {
|
|
||||||
bucket_name: bucket_name.to_string(),
|
|
||||||
object_name: object_name.to_string(),
|
|
||||||
query_values: url_values,
|
|
||||||
custom_header: headers,
|
|
||||||
content_sha256_hex: EMPTY_STRING_SHA256_HASH.to_string(),
|
|
||||||
content_md5_base64: "".to_string(),
|
|
||||||
content_body: ReaderImpl::Body(Bytes::new()),
|
|
||||||
content_length: 0,
|
|
||||||
stream_sha256: false,
|
|
||||||
trailer: HeaderMap::new(),
|
|
||||||
pre_sign_url: Default::default(),
|
|
||||||
add_crc: Default::default(),
|
|
||||||
extra_pre_sign_header: Default::default(),
|
|
||||||
bucket_location: Default::default(),
|
|
||||||
expires: Default::default(),
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let resp_status = resp.status();
|
|
||||||
let h = resp.headers().clone();
|
|
||||||
let has_etag = h.get("ETag").and_then(|v| v.to_str().ok()).unwrap_or("");
|
|
||||||
if !has_etag.is_empty() {
|
|
||||||
return Err(std::io::Error::other(
|
|
||||||
"get_object_attributes is not supported by the current endpoint version",
|
|
||||||
));
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut body_vec = Vec::new();
|
|
||||||
let mut body = resp.into_body();
|
|
||||||
while let Some(frame) = body.frame().await {
|
|
||||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
|
||||||
if let Some(data) = frame.data_ref() {
|
|
||||||
body_vec.extend_from_slice(data);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if resp_status != http::StatusCode::OK {
|
|
||||||
let err_body =
|
|
||||||
String::from_utf8(body_vec).map_err(|e| std::io::Error::other(format!("invalid UTF-8 error body: {e}")))?;
|
|
||||||
let mut er = match quick_xml::de::from_str::<AccessControlPolicy>(&err_body) {
|
|
||||||
Ok(result) => result,
|
|
||||||
Err(err) => {
|
|
||||||
return Err(std::io::Error::other(err.to_string()));
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
return Err(std::io::Error::other(er.access_control_list.permission));
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut oa = ObjectAttributes::new();
|
|
||||||
oa.parse_response(&h, body_vec).await?;
|
|
||||||
|
|
||||||
Ok(oa)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,159 +0,0 @@
|
|||||||
// Copyright 2024 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use std::io;
|
|
||||||
use std::path::{Path, PathBuf};
|
|
||||||
|
|
||||||
#[cfg(not(windows))]
|
|
||||||
use std::os::unix::fs::PermissionsExt;
|
|
||||||
|
|
||||||
use tokio::fs::{self, OpenOptions};
|
|
||||||
use tokio::io::{AsyncSeekExt, AsyncWriteExt, SeekFrom};
|
|
||||||
|
|
||||||
use crate::client::{
|
|
||||||
api_error_response::err_invalid_argument, api_get_options::GetObjectOptions, transition_api::TransitionClient,
|
|
||||||
};
|
|
||||||
|
|
||||||
async fn prepare_download_target(file_path: &Path) -> io::Result<()> {
|
|
||||||
match fs::metadata(file_path).await {
|
|
||||||
Ok(metadata) if metadata.is_dir() => {
|
|
||||||
return Err(io::Error::other(err_invalid_argument("filename is a directory.")));
|
|
||||||
}
|
|
||||||
Ok(_) => {}
|
|
||||||
Err(err) if err.kind() == io::ErrorKind::NotFound => {}
|
|
||||||
Err(err) => return Err(err),
|
|
||||||
}
|
|
||||||
|
|
||||||
if let Some(parent) = file_path.parent()
|
|
||||||
&& !parent.as_os_str().is_empty()
|
|
||||||
{
|
|
||||||
fs::create_dir_all(parent).await?;
|
|
||||||
|
|
||||||
#[cfg(not(windows))]
|
|
||||||
{
|
|
||||||
let mut permissions = fs::metadata(parent).await?.permissions();
|
|
||||||
permissions.set_mode(0o700);
|
|
||||||
fs::set_permissions(parent, permissions).await?;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn build_part_path(file_path: &Path) -> PathBuf {
|
|
||||||
PathBuf::from(format!("{}.part.rustfs", file_path.display()))
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn open_download_part_file(file_part_path: &Path) -> io::Result<tokio::fs::File> {
|
|
||||||
let mut options = OpenOptions::new();
|
|
||||||
options.create(true).truncate(false).read(true).write(true);
|
|
||||||
|
|
||||||
#[cfg(not(windows))]
|
|
||||||
options.mode(0o600);
|
|
||||||
|
|
||||||
options.open(file_part_path).await
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn cleanup_part_file(file_part_path: &Path) {
|
|
||||||
let _ = fs::remove_file(file_part_path).await;
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TransitionClient {
|
|
||||||
pub async fn fget_object(
|
|
||||||
&self,
|
|
||||||
bucket_name: &str,
|
|
||||||
object_name: &str,
|
|
||||||
file_path: &str,
|
|
||||||
mut opts: GetObjectOptions,
|
|
||||||
) -> Result<(), io::Error> {
|
|
||||||
let file_path = Path::new(file_path);
|
|
||||||
prepare_download_target(file_path).await?;
|
|
||||||
|
|
||||||
let file_part_path = build_part_path(file_path);
|
|
||||||
let mut file_part = open_download_part_file(&file_part_path).await?;
|
|
||||||
let existing_len = file_part.metadata().await?.len();
|
|
||||||
if existing_len > 0 {
|
|
||||||
opts.set_range(existing_len as i64, 0)?;
|
|
||||||
file_part.seek(SeekFrom::Start(existing_len)).await?;
|
|
||||||
}
|
|
||||||
|
|
||||||
let (_object_info, _headers, mut object_reader) = self.get_object_inner(bucket_name, object_name, &opts).await?;
|
|
||||||
if let Err(err) = tokio::io::copy(&mut object_reader, &mut file_part).await {
|
|
||||||
cleanup_part_file(&file_part_path).await;
|
|
||||||
return Err(err);
|
|
||||||
}
|
|
||||||
|
|
||||||
if let Err(err) = file_part.flush().await {
|
|
||||||
cleanup_part_file(&file_part_path).await;
|
|
||||||
return Err(err);
|
|
||||||
}
|
|
||||||
drop(file_part);
|
|
||||||
|
|
||||||
if let Err(err) = fs::rename(&file_part_path, file_path).await {
|
|
||||||
cleanup_part_file(&file_part_path).await;
|
|
||||||
return Err(err);
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
|
||||||
mod tests {
|
|
||||||
use super::*;
|
|
||||||
use tempfile::tempdir;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn prepare_download_target_allows_missing_file_and_creates_parent_dirs() {
|
|
||||||
let dir = tempdir().expect("temp dir");
|
|
||||||
let target = dir.path().join("nested").join("object.bin");
|
|
||||||
|
|
||||||
prepare_download_target(&target)
|
|
||||||
.await
|
|
||||||
.expect("missing target should be accepted");
|
|
||||||
|
|
||||||
assert!(target.parent().expect("parent").exists(), "parent directory should be created");
|
|
||||||
assert!(
|
|
||||||
fs::metadata(&target).await.is_err(),
|
|
||||||
"preparing the target should not create the final file eagerly"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn prepare_download_target_rejects_directory_paths() {
|
|
||||||
let dir = tempdir().expect("temp dir");
|
|
||||||
let target_dir = dir.path().join("download-dir");
|
|
||||||
fs::create_dir_all(&target_dir).await.expect("target dir");
|
|
||||||
|
|
||||||
let err = prepare_download_target(&target_dir)
|
|
||||||
.await
|
|
||||||
.expect_err("directory targets must be rejected");
|
|
||||||
|
|
||||||
assert!(err.to_string().contains("directory"), "unexpected error for directory target: {err}");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn open_download_part_file_creates_part_file() {
|
|
||||||
let dir = tempdir().expect("temp dir");
|
|
||||||
let target = dir.path().join("object.bin");
|
|
||||||
let part_path = build_part_path(&target);
|
|
||||||
|
|
||||||
let file = open_download_part_file(&part_path)
|
|
||||||
.await
|
|
||||||
.expect("part file should be created");
|
|
||||||
drop(file);
|
|
||||||
|
|
||||||
assert!(part_path.exists(), "part file should exist after creation");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -37,7 +37,7 @@ use crate::client::{
|
|||||||
api_put_object_common::optimal_part_info,
|
api_put_object_common::optimal_part_info,
|
||||||
api_put_object_multipart::UploadPartParams,
|
api_put_object_multipart::UploadPartParams,
|
||||||
api_s3_datatypes::{CompleteMultipartUpload, CompletePart, ObjectPart},
|
api_s3_datatypes::{CompleteMultipartUpload, CompletePart, ObjectPart},
|
||||||
constants::{ISO8601_DATEFORMAT, MAX_MULTIPART_PUT_OBJECT_SIZE, MIN_PART_SIZE, TOTAL_WORKERS},
|
constants::{ISO8601_DATEFORMAT, MAX_MULTIPART_PUT_OBJECT_SIZE, MIN_PART_SIZE},
|
||||||
credentials::SignatureType,
|
credentials::SignatureType,
|
||||||
transition_api::{ReaderImpl, TransitionClient, UploadInfo},
|
transition_api::{ReaderImpl, TransitionClient, UploadInfo},
|
||||||
utils::{is_amz_header, is_minio_header, is_rustfs_header, is_standard_header, is_storageclass_header},
|
utils::{is_amz_header, is_minio_header, is_rustfs_header, is_standard_header, is_storageclass_header},
|
||||||
|
|||||||
@@ -30,10 +30,6 @@ pub fn is_object(reader: &ReaderImpl) -> bool {
|
|||||||
matches!(reader, ReaderImpl::ObjectBody(_))
|
matches!(reader, ReaderImpl::ObjectBody(_))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn is_read_at(reader: ReaderImpl) -> bool {
|
|
||||||
matches!(reader, ReaderImpl::ObjectBody(_))
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn optimal_part_info(object_size: i64, configured_part_size: u64) -> Result<(i64, i64, i64), std::io::Error> {
|
pub fn optimal_part_info(object_size: i64, configured_part_size: u64) -> Result<(i64, i64, i64), std::io::Error> {
|
||||||
let unknown_size;
|
let unknown_size;
|
||||||
let mut object_size = object_size;
|
let mut object_size = object_size;
|
||||||
|
|||||||
@@ -81,18 +81,6 @@ async fn read_multipart_part(reader: &mut ReaderImpl, want: usize) -> Result<Vec
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub struct UploadedPartRes {
|
|
||||||
pub error: std::io::Error,
|
|
||||||
pub part_num: i64,
|
|
||||||
pub size: i64,
|
|
||||||
pub part: ObjectPart,
|
|
||||||
}
|
|
||||||
|
|
||||||
pub struct UploadPartReq {
|
|
||||||
pub part_num: i64,
|
|
||||||
pub part: ObjectPart,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TransitionClient {
|
impl TransitionClient {
|
||||||
pub async fn put_object_multipart_stream(
|
pub async fn put_object_multipart_stream(
|
||||||
self: Arc<Self>,
|
self: Arc<Self>,
|
||||||
|
|||||||
@@ -1,134 +0,0 @@
|
|||||||
// Copyright 2024 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
#![allow(unused_imports)]
|
|
||||||
#![allow(unused_variables)]
|
|
||||||
#![allow(unused_mut)]
|
|
||||||
#![allow(unused_assignments)]
|
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use crate::client::{
|
|
||||||
api_error_response::{err_invalid_argument, http_resp_to_error_response},
|
|
||||||
api_get_object_acl::AccessControlList,
|
|
||||||
api_get_options::GetObjectOptions,
|
|
||||||
transition_api::{ObjectInfo, ReadCloser, ReaderImpl, RequestMetadata, TransitionClient, to_object_info},
|
|
||||||
};
|
|
||||||
use http::HeaderMap;
|
|
||||||
use http_body_util::BodyExt;
|
|
||||||
use hyper::body::Body;
|
|
||||||
use hyper::body::Bytes;
|
|
||||||
use s3s::dto::RestoreRequest;
|
|
||||||
use std::collections::HashMap;
|
|
||||||
use std::io::Cursor;
|
|
||||||
use tokio::io::BufReader;
|
|
||||||
|
|
||||||
const TIER_STANDARD: &str = "Standard";
|
|
||||||
const TIER_BULK: &str = "Bulk";
|
|
||||||
const TIER_EXPEDITED: &str = "Expedited";
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Serialize, serde::Deserialize)]
|
|
||||||
pub struct Encryption {
|
|
||||||
pub encryption_type: String,
|
|
||||||
pub kms_context: String,
|
|
||||||
pub kms_key_id: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Serialize, serde::Deserialize)]
|
|
||||||
pub struct MetadataEntry {
|
|
||||||
pub name: String,
|
|
||||||
pub value: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Serialize)]
|
|
||||||
pub struct S3 {
|
|
||||||
pub access_control_list: AccessControlList,
|
|
||||||
pub bucket_name: String,
|
|
||||||
pub prefix: String,
|
|
||||||
pub canned_acl: String,
|
|
||||||
pub encryption: Encryption,
|
|
||||||
pub storage_class: String,
|
|
||||||
//tagging: Tags,
|
|
||||||
pub user_metadata: MetadataEntry,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TransitionClient {
|
|
||||||
pub async fn restore_object(
|
|
||||||
&self,
|
|
||||||
bucket_name: &str,
|
|
||||||
object_name: &str,
|
|
||||||
version_id: &str,
|
|
||||||
restore_req: &RestoreRequest,
|
|
||||||
) -> Result<(), std::io::Error> {
|
|
||||||
/*let restore_request = match quick_xml::se::to_string(restore_req) {
|
|
||||||
Ok(buf) => buf,
|
|
||||||
Err(e) => {
|
|
||||||
return Err(std::io::Error::other(e));
|
|
||||||
}
|
|
||||||
};*/
|
|
||||||
let restore_request = "".to_string();
|
|
||||||
let restore_request_bytes = restore_request.as_bytes().to_vec();
|
|
||||||
|
|
||||||
let mut url_values = HashMap::new();
|
|
||||||
url_values.insert("restore".to_string(), "".to_string());
|
|
||||||
if version_id != "" {
|
|
||||||
url_values.insert("versionId".to_string(), version_id.to_string());
|
|
||||||
}
|
|
||||||
|
|
||||||
let restore_request_buffer = Bytes::from(restore_request_bytes.clone());
|
|
||||||
let resp = self
|
|
||||||
.execute_method(
|
|
||||||
http::Method::HEAD,
|
|
||||||
&mut RequestMetadata {
|
|
||||||
bucket_name: bucket_name.to_string(),
|
|
||||||
object_name: object_name.to_string(),
|
|
||||||
query_values: url_values,
|
|
||||||
custom_header: HeaderMap::new(),
|
|
||||||
content_sha256_hex: "".to_string(), //sum_sha256_hex(&restore_request_bytes),
|
|
||||||
content_md5_base64: "".to_string(), //sum_md5_base64(&restore_request_bytes),
|
|
||||||
content_body: ReaderImpl::Body(restore_request_buffer),
|
|
||||||
content_length: restore_request_bytes.len() as i64,
|
|
||||||
stream_sha256: false,
|
|
||||||
trailer: HeaderMap::new(),
|
|
||||||
pre_sign_url: Default::default(),
|
|
||||||
add_crc: Default::default(),
|
|
||||||
extra_pre_sign_header: Default::default(),
|
|
||||||
bucket_location: Default::default(),
|
|
||||||
expires: Default::default(),
|
|
||||||
},
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let resp_status = resp.status();
|
|
||||||
let h = resp.headers().clone();
|
|
||||||
|
|
||||||
let mut body_vec = Vec::new();
|
|
||||||
let mut body = resp.into_body();
|
|
||||||
while let Some(frame) = body.frame().await {
|
|
||||||
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
|
|
||||||
if let Some(data) = frame.data_ref() {
|
|
||||||
body_vec.extend_from_slice(data);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if resp_status != http::StatusCode::ACCEPTED && resp_status != http::StatusCode::OK {
|
|
||||||
return Err(std::io::Error::other(http_resp_to_error_response(
|
|
||||||
resp_status,
|
|
||||||
&h,
|
|
||||||
body_vec,
|
|
||||||
bucket_name,
|
|
||||||
"",
|
|
||||||
)));
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -29,10 +29,6 @@ use crate::client::utils::base64_decode;
|
|||||||
|
|
||||||
use super::transition_api;
|
use super::transition_api;
|
||||||
|
|
||||||
pub struct ListAllMyBucketsResult {
|
|
||||||
pub owner: Owner,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, Serialize, Deserialize)]
|
#[derive(Debug, Default, Serialize, Deserialize)]
|
||||||
pub struct CommonPrefix {
|
pub struct CommonPrefix {
|
||||||
pub prefix: String,
|
pub prefix: String,
|
||||||
@@ -89,6 +85,10 @@ pub struct ListVersionsResult {
|
|||||||
pub next_version_id_marker: String,
|
pub next_version_id_marker: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "fields of a MinIO-parity list result that this port builds but never reads back (backlog#1823)"
|
||||||
|
)]
|
||||||
pub struct ListBucketResult {
|
pub struct ListBucketResult {
|
||||||
common_prefixes: Vec<CommonPrefix>,
|
common_prefixes: Vec<CommonPrefix>,
|
||||||
contents: Vec<transition_api::ObjectInfo>,
|
contents: Vec<transition_api::ObjectInfo>,
|
||||||
@@ -102,6 +102,10 @@ pub struct ListBucketResult {
|
|||||||
prefix: String,
|
prefix: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "fields of a MinIO-parity list result that this port builds but never reads back (backlog#1823)"
|
||||||
|
)]
|
||||||
pub struct ListMultipartUploadsResult {
|
pub struct ListMultipartUploadsResult {
|
||||||
bucket: String,
|
bucket: String,
|
||||||
key_marker: String,
|
key_marker: String,
|
||||||
@@ -117,16 +121,15 @@ pub struct ListMultipartUploadsResult {
|
|||||||
common_prefixes: Vec<CommonPrefix>,
|
common_prefixes: Vec<CommonPrefix>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "fields of a MinIO-parity list result that this port builds but never reads back (backlog#1823)"
|
||||||
|
)]
|
||||||
pub struct Initiator {
|
pub struct Initiator {
|
||||||
id: String,
|
id: String,
|
||||||
display_name: String,
|
display_name: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
pub struct CopyObjectResult {
|
|
||||||
pub etag: String,
|
|
||||||
pub last_modified: OffsetDateTime,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct ObjectPart {
|
pub struct ObjectPart {
|
||||||
pub etag: String,
|
pub etag: String,
|
||||||
@@ -260,6 +263,7 @@ pub struct CompletePart {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl CompletePart {
|
impl CompletePart {
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity accessor with no caller in this port (backlog#1823)")]
|
||||||
fn checksum(&self, t: &ChecksumMode) -> String {
|
fn checksum(&self, t: &ChecksumMode) -> String {
|
||||||
match t {
|
match t {
|
||||||
ChecksumMode::ChecksumCRC32C => {
|
ChecksumMode::ChecksumCRC32C => {
|
||||||
@@ -284,11 +288,6 @@ impl CompletePart {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub struct CopyObjectPartResult {
|
|
||||||
pub etag: String,
|
|
||||||
pub last_modified: OffsetDateTime,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug, Default, serde::Serialize)]
|
#[derive(Debug, Default, serde::Serialize)]
|
||||||
#[serde(rename = "CompleteMultipartUpload")]
|
#[serde(rename = "CompleteMultipartUpload")]
|
||||||
pub struct CompleteMultipartUpload {
|
pub struct CompleteMultipartUpload {
|
||||||
@@ -357,10 +356,10 @@ impl CompleteMultipartUpload {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub struct CreateBucketConfiguration {
|
#[allow(
|
||||||
pub location: String,
|
dead_code,
|
||||||
}
|
reason = "live via quick_xml::de::from_str in bucket_cache.rs; serde deserialization is not a construction (backlog#1823)"
|
||||||
|
)]
|
||||||
#[derive(serde::Serialize)]
|
#[derive(serde::Serialize)]
|
||||||
pub struct DeleteObject {
|
pub struct DeleteObject {
|
||||||
//api has
|
//api has
|
||||||
@@ -368,21 +367,6 @@ pub struct DeleteObject {
|
|||||||
pub version_id: String,
|
pub version_id: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
pub struct DeletedObject {
|
|
||||||
//s3s has
|
|
||||||
pub key: String,
|
|
||||||
pub version_id: String,
|
|
||||||
pub deletemarker: bool,
|
|
||||||
pub deletemarker_version_id: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
pub struct NonDeletedObject {
|
|
||||||
pub key: String,
|
|
||||||
pub code: String,
|
|
||||||
pub message: String,
|
|
||||||
pub version_id: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(serde::Serialize)]
|
#[derive(serde::Serialize)]
|
||||||
pub struct DeleteMultiObjects {
|
pub struct DeleteMultiObjects {
|
||||||
pub quiet: bool,
|
pub quiet: bool,
|
||||||
@@ -402,6 +386,7 @@ impl DeleteMultiObjects {
|
|||||||
Ok(buf)
|
Ok(buf)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity XML helper with no caller in this port (backlog#1823)")]
|
||||||
pub fn unmarshal(buf: &[u8]) -> Result<Self, std::io::Error> {
|
pub fn unmarshal(buf: &[u8]) -> Result<Self, std::io::Error> {
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize)]
|
||||||
struct WireDeleteObject {
|
struct WireDeleteObject {
|
||||||
@@ -436,8 +421,3 @@ impl DeleteMultiObjects {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub struct DeleteMultiObjectsResult {
|
|
||||||
pub deleted_objects: Vec<DeletedObject>,
|
|
||||||
pub undeleted_objects: Vec<NonDeletedObject>,
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -27,12 +27,24 @@ use crate::client::utils::base64_decode;
|
|||||||
use crate::client::utils::base64_encode;
|
use crate::client::utils::base64_encode;
|
||||||
use crate::client::{api_put_object::PutObjectOptions, api_s3_datatypes::ObjectPart};
|
use crate::client::{api_put_object::PutObjectOptions, api_s3_datatypes::ObjectPart};
|
||||||
use crate::{disk::DiskAPI, object_api::GetObjectReader};
|
use crate::{disk::DiskAPI, object_api::GetObjectReader};
|
||||||
|
// s3s::header has no CRC64NVME constant yet; the canonical RustFS copy lives
|
||||||
|
// in rustfs-utils' headers module.
|
||||||
|
use rustfs_utils::http::headers::AMZ_CHECKSUM_CRC64NVME;
|
||||||
use s3s::header::{
|
use s3s::header::{
|
||||||
X_AMZ_CHECKSUM_ALGORITHM, X_AMZ_CHECKSUM_CRC32, X_AMZ_CHECKSUM_CRC32C, X_AMZ_CHECKSUM_SHA1, X_AMZ_CHECKSUM_SHA256,
|
X_AMZ_CHECKSUM_ALGORITHM, X_AMZ_CHECKSUM_CRC32, X_AMZ_CHECKSUM_CRC32C, X_AMZ_CHECKSUM_SHA1, X_AMZ_CHECKSUM_SHA256,
|
||||||
};
|
};
|
||||||
|
|
||||||
use enumset::{EnumSet, EnumSetType, enum_set};
|
use enumset::{EnumSet, EnumSetType, enum_set};
|
||||||
|
|
||||||
|
/// One of three deliberately separate checksum registries (backlog#1833):
|
||||||
|
/// this enum is the MinIO-port client's wire vocabulary and stops at the
|
||||||
|
/// standard S3 set (CRC64NVME is its newest member; the RustFS extensions do
|
||||||
|
/// not exist on this client path). The streaming-hash registry lives in
|
||||||
|
/// `rustfs_checksums::ChecksumAlgorithm` (crates/checksums/src/lib.rs) and
|
||||||
|
/// the on-disk xl.meta bitset in `rustfs_rio::ChecksumType`
|
||||||
|
/// (crates/rio/src/checksum.rs, varint bits are append-only). When adding an
|
||||||
|
/// algorithm, extend all three (or record why not) — they do not derive from
|
||||||
|
/// each other.
|
||||||
#[derive(Debug, EnumSetType, Default)]
|
#[derive(Debug, EnumSetType, Default)]
|
||||||
#[enumset(repr = "u8")]
|
#[enumset(repr = "u8")]
|
||||||
pub enum ChecksumMode {
|
pub enum ChecksumMode {
|
||||||
@@ -57,8 +69,6 @@ lazy_static! {
|
|||||||
static ref C_ChecksumFullObjectCRC32C: EnumSet<ChecksumMode> =
|
static ref C_ChecksumFullObjectCRC32C: EnumSet<ChecksumMode> =
|
||||||
enum_set!(ChecksumMode::ChecksumCRC32C | ChecksumMode::ChecksumFullObject);
|
enum_set!(ChecksumMode::ChecksumCRC32C | ChecksumMode::ChecksumFullObject);
|
||||||
}
|
}
|
||||||
const AMZ_CHECKSUM_CRC64NVME: &str = "x-amz-checksum-crc64nvme";
|
|
||||||
|
|
||||||
impl ChecksumMode {
|
impl ChecksumMode {
|
||||||
//pub const CRC64_NVME_POLYNOMIAL: i64 = 0xad93d23594c93659;
|
//pub const CRC64_NVME_POLYNOMIAL: i64 = 0xad93d23594c93659;
|
||||||
|
|
||||||
@@ -355,6 +365,10 @@ mod tests {
|
|||||||
pub struct Checksum {
|
pub struct Checksum {
|
||||||
checksum_type: ChecksumMode,
|
checksum_type: ChecksumMode,
|
||||||
r: Vec<u8>,
|
r: Vec<u8>,
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "checksum bookkeeping field kept beside the value it guards (backlog#1823)"
|
||||||
|
)]
|
||||||
computed: bool,
|
computed: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -32,11 +32,5 @@ pub const MAX_MULTIPART_PUT_OBJECT_SIZE: i64 = 1024 * 1024 * 1024 * 1024 * 5;
|
|||||||
pub const UNSIGNED_PAYLOAD: &str = "UNSIGNED-PAYLOAD";
|
pub const UNSIGNED_PAYLOAD: &str = "UNSIGNED-PAYLOAD";
|
||||||
pub const UNSIGNED_PAYLOAD_TRAILER: &str = "STREAMING-UNSIGNED-PAYLOAD-TRAILER";
|
pub const UNSIGNED_PAYLOAD_TRAILER: &str = "STREAMING-UNSIGNED-PAYLOAD-TRAILER";
|
||||||
|
|
||||||
pub const TOTAL_WORKERS: i64 = 4;
|
|
||||||
|
|
||||||
pub const SIGN_V4_ALGORITHM: &str = "AWS4-HMAC-SHA256";
|
|
||||||
pub const ISO8601_DATEFORMAT: &[FormatItem<'_>] =
|
pub const ISO8601_DATEFORMAT: &[FormatItem<'_>] =
|
||||||
format_description!("[year]-[month]-[day]T[hour]:[minute]:[second].[subsecond]Z");
|
format_description!("[year]-[month]-[day]T[hour]:[minute]:[second].[subsecond]Z");
|
||||||
|
|
||||||
pub const GET_OBJECT_ATTRIBUTES_TAGS: &str = "ETag,Checksum,StorageClass,ObjectSize,ObjectParts";
|
|
||||||
pub const GET_OBJECT_ATTRIBUTES_MAX_PARTS: i64 = 1000;
|
|
||||||
|
|||||||
@@ -67,6 +67,10 @@ impl<P: Provider + Default> Credentials<P> {
|
|||||||
Ok(self.creds.clone())
|
Ok(self.creds.clone())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity credential surface with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn expire(&mut self) {
|
fn expire(&mut self) {
|
||||||
self.force_refresh = true;
|
self.force_refresh = true;
|
||||||
}
|
}
|
||||||
@@ -133,6 +137,10 @@ impl Provider for Static {
|
|||||||
|
|
||||||
#[derive(Debug, Clone, Default)]
|
#[derive(Debug, Clone, Default)]
|
||||||
pub struct STSError {
|
pub struct STSError {
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity STS error detail that this port never reads back (backlog#1823)"
|
||||||
|
)]
|
||||||
pub r#type: String,
|
pub r#type: String,
|
||||||
pub code: String,
|
pub code: String,
|
||||||
pub message: String,
|
pub message: String,
|
||||||
@@ -141,6 +149,10 @@ pub struct STSError {
|
|||||||
#[derive(Debug, Clone, thiserror::Error)]
|
#[derive(Debug, Clone, thiserror::Error)]
|
||||||
pub struct ErrorResponse {
|
pub struct ErrorResponse {
|
||||||
pub sts_error: STSError,
|
pub sts_error: STSError,
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity STS error detail that this port never reads back (backlog#1823)"
|
||||||
|
)]
|
||||||
pub request_id: String,
|
pub request_id: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -158,22 +170,3 @@ impl ErrorResponse {
|
|||||||
return self.sts_error.message.clone();
|
return self.sts_error.message.clone();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn xml_decoder<T>(body: &[u8]) -> Result<T, Error>
|
|
||||||
where
|
|
||||||
for<'de> T: Deserialize<'de>,
|
|
||||||
{
|
|
||||||
match std::str::from_utf8(body) {
|
|
||||||
Ok(xml_body) => quick_xml::de::from_str::<T>(xml_body).map_err(|err| Error::new(ErrorKind::InvalidData, err.to_string())),
|
|
||||||
Err(err) => Err(Error::new(ErrorKind::InvalidData, err.to_string())),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn xml_decode_and_body<T>(body_reader: &[u8]) -> Result<(Vec<u8>, T), std::io::Error>
|
|
||||||
where
|
|
||||||
for<'de> T: Deserialize<'de>,
|
|
||||||
{
|
|
||||||
let body = body_reader.to_vec();
|
|
||||||
let parsed = xml_decoder(&body)?;
|
|
||||||
Ok((body, parsed))
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -13,15 +13,10 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
// #730: S3 client compatibility models are kept while ECStore callers move to narrower facades.
|
// #730: S3 client compatibility models are kept while ECStore callers move to narrower facades.
|
||||||
#![allow(dead_code)]
|
|
||||||
|
|
||||||
pub mod admin_handler_utils;
|
pub mod admin_handler_utils;
|
||||||
pub mod api_bucket_policy;
|
|
||||||
pub mod api_error_response;
|
pub mod api_error_response;
|
||||||
pub mod api_get_object;
|
pub mod api_get_object;
|
||||||
pub mod api_get_object_acl;
|
|
||||||
pub mod api_get_object_attributes;
|
|
||||||
pub mod api_get_object_file;
|
|
||||||
pub mod api_get_options;
|
pub mod api_get_options;
|
||||||
pub mod api_list;
|
pub mod api_list;
|
||||||
pub mod api_put_object;
|
pub mod api_put_object;
|
||||||
@@ -29,7 +24,6 @@ pub mod api_put_object_common;
|
|||||||
pub mod api_put_object_multipart;
|
pub mod api_put_object_multipart;
|
||||||
pub mod api_put_object_streaming;
|
pub mod api_put_object_streaming;
|
||||||
pub mod api_remove;
|
pub mod api_remove;
|
||||||
pub mod api_restore;
|
|
||||||
pub mod api_s3_datatypes;
|
pub mod api_s3_datatypes;
|
||||||
pub mod api_stat;
|
pub mod api_stat;
|
||||||
pub mod bucket_cache;
|
pub mod bucket_cache;
|
||||||
|
|||||||
@@ -77,39 +77,6 @@ fn part_number_to_rangespec(oi: ObjectInfo, part_number: usize) -> Option<HTTPRa
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn get_compressed_offsets(oi: ObjectInfo, offset: i64) -> (i64, i64, i64, i64, u64) {
|
|
||||||
let mut skip_length: i64 = 0;
|
|
||||||
let mut cumulative_actual_size: i64 = 0;
|
|
||||||
let mut first_part_idx: i64 = 0;
|
|
||||||
let mut compressed_offset: i64 = 0;
|
|
||||||
let mut part_skip: i64 = 0;
|
|
||||||
let mut decrypt_skip: i64 = 0;
|
|
||||||
let mut seq_num: u64 = 0;
|
|
||||||
for (i, part) in oi.parts.iter().enumerate() {
|
|
||||||
cumulative_actual_size += part.actual_size as i64;
|
|
||||||
if cumulative_actual_size <= offset {
|
|
||||||
compressed_offset += part.size as i64;
|
|
||||||
} else {
|
|
||||||
first_part_idx = i as i64;
|
|
||||||
skip_length = cumulative_actual_size - part.actual_size as i64;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
skip_length = offset - skip_length;
|
|
||||||
|
|
||||||
let parts: &[ObjectPartInfo] = &oi.parts;
|
|
||||||
if skip_length > 0
|
|
||||||
&& parts.len() > first_part_idx as usize
|
|
||||||
&& parts[first_part_idx as usize].index.as_ref().is_some_and(|idx| idx.len() > 0)
|
|
||||||
{
|
|
||||||
let _ = part_skip;
|
|
||||||
let _ = decrypt_skip;
|
|
||||||
let _ = seq_num;
|
|
||||||
}
|
|
||||||
|
|
||||||
(compressed_offset, part_skip, first_part_idx, decrypt_skip, seq_num)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn new_getobjectreader<'a>(
|
pub fn new_getobjectreader<'a>(
|
||||||
rs: &Option<HTTPRangeSpec>,
|
rs: &Option<HTTPRangeSpec>,
|
||||||
oi: &'a ObjectInfo,
|
oi: &'a ObjectInfo,
|
||||||
|
|||||||
@@ -23,6 +23,7 @@ const X_OBS_VERSION_ID: &str = "x-obs-version-id";
|
|||||||
const MAX_REMOTE_VERSION_ID_LEN: usize = 1024;
|
const MAX_REMOTE_VERSION_ID_LEN: usize = 1024;
|
||||||
|
|
||||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||||
|
#[allow(dead_code, reason = "bucket versioning states kept as a complete vocabulary (backlog#1823)")]
|
||||||
pub(crate) enum BucketVersioningState {
|
pub(crate) enum BucketVersioningState {
|
||||||
Unknown,
|
Unknown,
|
||||||
Disabled,
|
Disabled,
|
||||||
@@ -47,6 +48,7 @@ impl RemoteVersion {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity accessor with no caller in this port (backlog#1823)")]
|
||||||
pub(crate) fn exact_request_id(&self) -> Result<Option<&str>, Error> {
|
pub(crate) fn exact_request_id(&self) -> Result<Option<&str>, Error> {
|
||||||
match self {
|
match self {
|
||||||
Self::Unknown => Err(Error::new(
|
Self::Unknown => Err(Error::new(
|
||||||
|
|||||||
@@ -101,6 +101,10 @@ where
|
|||||||
|
|
||||||
const C_UNKNOWN: i32 = -1;
|
const C_UNKNOWN: i32 = -1;
|
||||||
const C_OFFLINE: i32 = 0;
|
const C_OFFLINE: i32 = 0;
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "reachable only from the unused transition client methods below (backlog#1823)"
|
||||||
|
)]
|
||||||
const C_ONLINE: i32 = 1;
|
const C_ONLINE: i32 = 1;
|
||||||
|
|
||||||
fn invalid_utf8_header_error(scope: &str, header_name: &str) -> std::io::Error {
|
fn invalid_utf8_header_error(scope: &str, header_name: &str) -> std::io::Error {
|
||||||
@@ -320,6 +324,10 @@ impl TransitionClient {
|
|||||||
Ok(client)
|
Ok(client)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client surface with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn endpoint_url(&self) -> Url {
|
fn endpoint_url(&self) -> Url {
|
||||||
self.endpoint_url.clone()
|
self.endpoint_url.clone()
|
||||||
}
|
}
|
||||||
@@ -348,12 +356,20 @@ impl TransitionClient {
|
|||||||
.to_string())
|
.to_string())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn trace_errors_only_off(&self) {
|
fn trace_errors_only_off(&self) {
|
||||||
if let Ok(mut trace_errors_only) = self.trace_errors_only.lock() {
|
if let Ok(mut trace_errors_only) = self.trace_errors_only.lock() {
|
||||||
*trace_errors_only = false;
|
*trace_errors_only = false;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn trace_off(&self) {
|
fn trace_off(&self) {
|
||||||
if let Ok(mut is_trace_enabled) = self.is_trace_enabled.lock() {
|
if let Ok(mut is_trace_enabled) = self.is_trace_enabled.lock() {
|
||||||
*is_trace_enabled = false;
|
*is_trace_enabled = false;
|
||||||
@@ -363,12 +379,20 @@ impl TransitionClient {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn set_s3_transfer_accelerate(&self, accelerate_endpoint: &str) {
|
fn set_s3_transfer_accelerate(&self, accelerate_endpoint: &str) {
|
||||||
if let Ok(mut endpoint) = self.s3_accelerate_endpoint.lock() {
|
if let Ok(mut endpoint) = self.s3_accelerate_endpoint.lock() {
|
||||||
*endpoint = accelerate_endpoint.to_string();
|
*endpoint = accelerate_endpoint.to_string();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn set_s3_enable_dual_stack(&self, enabled: bool) {
|
fn set_s3_enable_dual_stack(&self, enabled: bool) {
|
||||||
if let Ok(mut dual_stack) = self.s3_dual_stack_enabled.lock() {
|
if let Ok(mut dual_stack) = self.s3_dual_stack_enabled.lock() {
|
||||||
*dual_stack = enabled;
|
*dual_stack = enabled;
|
||||||
@@ -398,10 +422,18 @@ impl TransitionClient {
|
|||||||
(hash_algos, hash_sums)
|
(hash_algos, hash_sums)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn is_online(&self) -> bool {
|
fn is_online(&self) -> bool {
|
||||||
!self.is_offline()
|
!self.is_offline()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn mark_offline(&self) {
|
fn mark_offline(&self) {
|
||||||
self.health_status
|
self.health_status
|
||||||
.compare_exchange(C_ONLINE, C_OFFLINE, Ordering::SeqCst, Ordering::SeqCst);
|
.compare_exchange(C_ONLINE, C_OFFLINE, Ordering::SeqCst, Ordering::SeqCst);
|
||||||
@@ -411,10 +443,18 @@ impl TransitionClient {
|
|||||||
self.health_status.load(Ordering::SeqCst) == C_OFFLINE
|
self.health_status.load(Ordering::SeqCst) == C_OFFLINE
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn health_check(hc_duration: Duration) {
|
fn health_check(hc_duration: Duration) {
|
||||||
let _ = hc_duration;
|
let _ = hc_duration;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "MinIO-parity transition client method with no caller in this port (backlog#1823)"
|
||||||
|
)]
|
||||||
fn dump_http(&self, req: &Request<s3s::Body>, resp: &Response<Incoming>) -> Result<(), std::io::Error> {
|
fn dump_http(&self, req: &Request<s3s::Body>, resp: &Response<Incoming>) -> Result<(), std::io::Error> {
|
||||||
let mut resp_trace: Vec<u8>;
|
let mut resp_trace: Vec<u8>;
|
||||||
|
|
||||||
@@ -1006,16 +1046,6 @@ impl TransitionCore {
|
|||||||
client.abort_multipart_upload(bucket_name, object, upload_id).await
|
client.abort_multipart_upload(bucket_name, object, upload_id).await
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn get_bucket_policy(&self, bucket_name: &str) -> Result<String, std::io::Error> {
|
|
||||||
let client = self.0.clone();
|
|
||||||
client.get_bucket_policy(bucket_name).await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn put_bucket_policy(&self, bucket_name: &str, bucket_policy: &str) -> Result<(), std::io::Error> {
|
|
||||||
let client = self.0.clone();
|
|
||||||
client.put_bucket_policy(bucket_name, bucket_policy).await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn get_object(
|
pub async fn get_object(
|
||||||
&self,
|
&self,
|
||||||
bucket_name: &str,
|
bucket_name: &str,
|
||||||
@@ -1112,6 +1142,7 @@ impl Default for ObjectInfo {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl ObjectInfo {
|
impl ObjectInfo {
|
||||||
|
#[allow(dead_code, reason = "MinIO-parity accessor with no caller in this port (backlog#1823)")]
|
||||||
pub(crate) fn remote_version(
|
pub(crate) fn remote_version(
|
||||||
&self,
|
&self,
|
||||||
capabilities: ProviderVersionCapabilities,
|
capabilities: ProviderVersionCapabilities,
|
||||||
|
|||||||
@@ -48,10 +48,6 @@ lazy_static! {
|
|||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn is_standard_query_value(qs_key: &str) -> bool {
|
|
||||||
SUPPORTED_QUERY_VALUES[qs_key]
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn is_storageclass_header(header_key: &str) -> bool {
|
pub fn is_storageclass_header(header_key: &str) -> bool {
|
||||||
header_key.to_lowercase() == X_AMZ_STORAGE_CLASS.as_str().to_lowercase()
|
header_key.to_lowercase() == X_AMZ_STORAGE_CLASS.as_str().to_lowercase()
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -13,7 +13,6 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
// #730: cluster/RPC migration leaves transport capabilities staged for upcoming owners.
|
// #730: cluster/RPC migration leaves transport capabilities staged for upcoming owners.
|
||||||
#![allow(dead_code)]
|
|
||||||
|
|
||||||
mod control_plane;
|
mod control_plane;
|
||||||
pub(crate) mod rpc;
|
pub(crate) mod rpc;
|
||||||
|
|||||||
@@ -256,6 +256,7 @@ impl<S> ReplayScopeChannel<S> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "replay-state probe asserted by this file's tests (backlog#1823)")]
|
||||||
fn peer_replay_state(audience: &str) -> PeerReplayState {
|
fn peer_replay_state(audience: &str) -> PeerReplayState {
|
||||||
PEER_REPLAY_STATES
|
PEER_REPLAY_STATES
|
||||||
.lock()
|
.lock()
|
||||||
|
|||||||
@@ -31,7 +31,7 @@ use rustfs_config::{
|
|||||||
DEFAULT_INTERNODE_DATA_TRANSPORT, ENV_RUSTFS_INTERNODE_DATA_TRANSPORT, INTERNODE_DATA_TRANSPORT_TCP,
|
DEFAULT_INTERNODE_DATA_TRANSPORT, ENV_RUSTFS_INTERNODE_DATA_TRANSPORT, INTERNODE_DATA_TRANSPORT_TCP,
|
||||||
KNOWN_INTERNODE_DATA_TRANSPORT_BACKENDS,
|
KNOWN_INTERNODE_DATA_TRANSPORT_BACKENDS,
|
||||||
};
|
};
|
||||||
use rustfs_rio::{HttpReader, HttpWriter};
|
use rustfs_rio::{ChunkReaderBox, HttpChunkReader, HttpReader, HttpWriter};
|
||||||
use sha2::{Digest, Sha256};
|
use sha2::{Digest, Sha256};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
use std::future::Future;
|
use std::future::Future;
|
||||||
@@ -43,6 +43,10 @@ use tokio::io::{AsyncReadExt, AsyncWrite};
|
|||||||
use tokio::sync::OnceCell;
|
use tokio::sync::OnceCell;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "live in the cfg(not(test)) half of build_internode_data_transport_from_env (backlog#1823)"
|
||||||
|
)]
|
||||||
static INTERNODE_DATA_TRANSPORT: OnceLock<std::result::Result<Arc<dyn InternodeDataTransport>, String>> = OnceLock::new();
|
static INTERNODE_DATA_TRANSPORT: OnceLock<std::result::Result<Arc<dyn InternodeDataTransport>, String>> = OnceLock::new();
|
||||||
|
|
||||||
const READ_FILE_STREAM_PATH: &str = "/rustfs/rpc/read_file_stream";
|
const READ_FILE_STREAM_PATH: &str = "/rustfs/rpc/read_file_stream";
|
||||||
@@ -134,6 +138,10 @@ fn put_file_capability_status_is_legacy(status: u16) -> bool {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, Copy, Eq, PartialEq)]
|
#[derive(Debug, Clone, Copy, Eq, PartialEq)]
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "capability-negotiation seam; constructed only by transport test doubles (backlog#1823)"
|
||||||
|
)]
|
||||||
pub struct InternodeDataTransportCapabilities {
|
pub struct InternodeDataTransportCapabilities {
|
||||||
/// Backend can open a streaming remote disk reader.
|
/// Backend can open a streaming remote disk reader.
|
||||||
pub streaming_read: bool,
|
pub streaming_read: bool,
|
||||||
@@ -150,6 +158,10 @@ pub struct InternodeDataTransportCapabilities {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl InternodeDataTransportCapabilities {
|
impl InternodeDataTransportCapabilities {
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "capability-negotiation seam; used by transport test doubles (backlog#1823)"
|
||||||
|
)]
|
||||||
pub const fn tcp_http() -> Self {
|
pub const fn tcp_http() -> Self {
|
||||||
Self {
|
Self {
|
||||||
streaming_read: true,
|
streaming_read: true,
|
||||||
@@ -221,6 +233,11 @@ pub struct NsScannerCapabilityRequest {
|
|||||||
#[async_trait]
|
#[async_trait]
|
||||||
pub trait InternodeDataTransport: Send + Sync + std::fmt::Debug {
|
pub trait InternodeDataTransport: Send + Sync + std::fmt::Debug {
|
||||||
async fn open_read(&self, request: ReadStreamRequest) -> Result<FileReader>;
|
async fn open_read(&self, request: ReadStreamRequest) -> Result<FileReader>;
|
||||||
|
/// Opens an owned-chunk stream when this transport can retain receive-buffer
|
||||||
|
/// ownership. `None` preserves the established `open_read` fallback.
|
||||||
|
async fn open_read_chunks(&self, _request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
|
||||||
|
Ok(None)
|
||||||
|
}
|
||||||
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter>;
|
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter>;
|
||||||
async fn open_walk_dir(&self, request: WalkDirStreamRequest) -> Result<FileReader>;
|
async fn open_walk_dir(&self, request: WalkDirStreamRequest) -> Result<FileReader>;
|
||||||
async fn open_ns_scanner(&self, _request: NsScannerStreamRequest) -> Result<FileReader> {
|
async fn open_ns_scanner(&self, _request: NsScannerStreamRequest) -> Result<FileReader> {
|
||||||
@@ -229,7 +246,12 @@ pub trait InternodeDataTransport: Send + Sync + std::fmt::Debug {
|
|||||||
async fn probe_ns_scanner(&self, _request: NsScannerCapabilityRequest) -> Result<Uuid> {
|
async fn probe_ns_scanner(&self, _request: NsScannerCapabilityRequest) -> Result<Uuid> {
|
||||||
Err(Error::MethodNotAllowed)
|
Err(Error::MethodNotAllowed)
|
||||||
}
|
}
|
||||||
|
// Interface facet nobody calls yet: every transport implements both, but no
|
||||||
|
// caller negotiates on them. Kept for the internode transport split
|
||||||
|
// (backlog#1350); deleting them would delete the seam and six impls.
|
||||||
|
#[allow(dead_code, reason = "unused capability-negotiation facet (backlog#1823)")]
|
||||||
fn name(&self) -> &'static str;
|
fn name(&self) -> &'static str;
|
||||||
|
#[allow(dead_code, reason = "unused capability-negotiation facet (backlog#1823)")]
|
||||||
fn capabilities(&self) -> InternodeDataTransportCapabilities;
|
fn capabilities(&self) -> InternodeDataTransportCapabilities;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -247,6 +269,15 @@ impl InternodeDataTransport for TcpHttpInternodeDataTransport {
|
|||||||
))
|
))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn open_read_chunks(&self, request: ReadStreamRequest) -> Result<Option<ChunkReaderBox>> {
|
||||||
|
let url = build_read_file_stream_url(&request);
|
||||||
|
let mut headers = json_headers();
|
||||||
|
build_auth_headers(&url, &Method::GET, &mut headers)?;
|
||||||
|
Ok(Some(Box::new(
|
||||||
|
HttpChunkReader::new_with_stall_timeout(url, Method::GET, headers, None, request.stall_timeout).await?,
|
||||||
|
)))
|
||||||
|
}
|
||||||
|
|
||||||
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter> {
|
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter> {
|
||||||
let server_epoch = self.put_file_auth_capability(&request.endpoint).await?;
|
let server_epoch = self.put_file_auth_capability(&request.endpoint).await?;
|
||||||
let nonce = server_epoch.map(|_| Uuid::new_v4());
|
let nonce = server_epoch.map(|_| Uuid::new_v4());
|
||||||
@@ -656,6 +687,10 @@ fn build_internode_data_transport_result(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "live in the cfg(test) half of build_internode_data_transport_from_env, which bypasses the process static (backlog#1823)"
|
||||||
|
)]
|
||||||
pub fn build_internode_data_transport(configured_transport: Option<&str>) -> Result<Arc<dyn InternodeDataTransport>> {
|
pub fn build_internode_data_transport(configured_transport: Option<&str>) -> Result<Arc<dyn InternodeDataTransport>> {
|
||||||
build_internode_data_transport_result(configured_transport).map_err(Error::other)
|
build_internode_data_transport_result(configured_transport).map_err(Error::other)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -248,6 +248,16 @@ fn decode_remote_version_state_capability(expected_member: &str, result: &[u8])
|
|||||||
Ok(server_epoch)
|
Ok(server_epoch)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn decode_cross_pool_fence_capability(expected_member: &str, result: &[u8]) -> Result<(u32, Uuid)> {
|
||||||
|
let version = result
|
||||||
|
.get(..4)
|
||||||
|
.and_then(|value| value.try_into().ok())
|
||||||
|
.map(u32::from_be_bytes)
|
||||||
|
.ok_or_else(|| Error::other("peer returned an invalid cross-pool fence capability version"))?;
|
||||||
|
let epoch = decode_remote_version_state_capability(expected_member, &result[4..])?;
|
||||||
|
Ok((version, epoch))
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Clone, Debug)]
|
#[derive(Clone, Debug)]
|
||||||
pub struct PeerLiveEventsBatch {
|
pub struct PeerLiveEventsBatch {
|
||||||
pub events: Vec<u8>,
|
pub events: Vec<u8>,
|
||||||
@@ -1288,6 +1298,16 @@ impl PeerRestClient {
|
|||||||
Ok((self.topology_member.clone(), epoch))
|
Ok((self.topology_member.clone(), epoch))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub async fn probe_cross_pool_fence(&self, topology_fingerprint: String) -> Result<(String, u32, Uuid)> {
|
||||||
|
let mut probe = rustfs_protos::CROSS_POOL_FENCE_CAPABILITY_PROBE_PREFIX.to_vec();
|
||||||
|
probe.extend_from_slice(Uuid::new_v4().as_bytes());
|
||||||
|
let result = self
|
||||||
|
.heal_control(rustfs_protos::HEAL_CONTROL_PROTOCOL_VERSION, topology_fingerprint, probe)
|
||||||
|
.await?;
|
||||||
|
let (supported_version, epoch) = decode_cross_pool_fence_capability(&self.topology_member, &result)?;
|
||||||
|
Ok((self.topology_member.clone(), supported_version, epoch))
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn load_bucket_metadata(&self, bucket: &str, scanner_maintenance_change: bool) -> Result<()> {
|
pub async fn load_bucket_metadata(&self, bucket: &str, scanner_maintenance_change: bool) -> Result<()> {
|
||||||
self.finalize_result(
|
self.finalize_result(
|
||||||
async {
|
async {
|
||||||
@@ -2738,6 +2758,24 @@ mod tests {
|
|||||||
assert!(decode_remote_version_state_capability("node-a:9000", &nil).is_err());
|
assert!(decode_remote_version_state_capability("node-a:9000", &nil).is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn cross_pool_fence_capability_decoder_fails_closed() {
|
||||||
|
let epoch = Uuid::new_v4();
|
||||||
|
let result = rustfs_protos::encode_cross_pool_fence_capability(1, "node-a:9000", epoch.as_bytes())
|
||||||
|
.expect("small capability response should encode");
|
||||||
|
assert_eq!(
|
||||||
|
decode_cross_pool_fence_capability("node-a:9000", &result).expect("valid capability should decode"),
|
||||||
|
(1, epoch)
|
||||||
|
);
|
||||||
|
for malformed in [&[][..], &[0, 0, 0][..], &result[..result.len() - 1]] {
|
||||||
|
assert!(decode_cross_pool_fence_capability("node-a:9000", malformed).is_err());
|
||||||
|
}
|
||||||
|
assert!(decode_cross_pool_fence_capability("node-b:9000", &result).is_err());
|
||||||
|
let nil = rustfs_protos::encode_cross_pool_fence_capability(1, "node-a:9000", Uuid::nil().as_bytes())
|
||||||
|
.expect("small capability response should encode");
|
||||||
|
assert!(decode_cross_pool_fence_capability("node-a:9000", &nil).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
struct TierMutationResponseFixture<'a> {
|
struct TierMutationResponseFixture<'a> {
|
||||||
version: u32,
|
version: u32,
|
||||||
phase: TierMutationRpcPhase,
|
phase: TierMutationRpcPhase,
|
||||||
|
|||||||
@@ -854,7 +854,6 @@ impl PeerS3Client for LocalPeerS3Client {
|
|||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
pub struct RemotePeerS3Client {
|
pub struct RemotePeerS3Client {
|
||||||
pub node: Option<Node>,
|
|
||||||
pub pools: Option<Vec<usize>>,
|
pub pools: Option<Vec<usize>>,
|
||||||
addr: String,
|
addr: String,
|
||||||
/// Health tracker for connection monitoring
|
/// Health tracker for connection monitoring
|
||||||
@@ -886,7 +885,6 @@ impl RemotePeerS3Client {
|
|||||||
pub fn new(node: Option<Node>, pools: Option<Vec<usize>>) -> Self {
|
pub fn new(node: Option<Node>, pools: Option<Vec<usize>>) -> Self {
|
||||||
let addr = node.as_ref().map(|v| v.url.to_string()).unwrap_or_default();
|
let addr = node.as_ref().map(|v| v.url.to_string()).unwrap_or_default();
|
||||||
let client = Self {
|
let client = Self {
|
||||||
node,
|
|
||||||
pools,
|
pools,
|
||||||
addr,
|
addr,
|
||||||
health: Arc::new(DiskHealthTracker::new()),
|
health: Arc::new(DiskHealthTracker::new()),
|
||||||
@@ -905,10 +903,6 @@ impl RemotePeerS3Client {
|
|||||||
.map_err(|err| Error::other(format!("can not get client, err: {err}")))
|
.map_err(|err| Error::other(format!("can not get client, err: {err}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn get_addr(&self) -> String {
|
|
||||||
self.addr.clone()
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Start health monitoring for the remote peer
|
/// Start health monitoring for the remote peer
|
||||||
fn start_health_monitoring(&self) {
|
fn start_health_monitoring(&self) {
|
||||||
let health = Arc::clone(&self.health);
|
let health = Arc::clone(&self.health);
|
||||||
@@ -1208,6 +1202,10 @@ impl PeerS3Client for RemotePeerS3Client {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "local bucket-heal path reached only by this file's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
pub async fn heal_bucket_local(bucket: &str, opts: &HealOpts) -> Result<HealResultItem> {
|
pub async fn heal_bucket_local(bucket: &str, opts: &HealOpts) -> Result<HealResultItem> {
|
||||||
let disks = clone_drives().await;
|
let disks = clone_drives().await;
|
||||||
heal_bucket_local_on_disks(bucket, opts, disks).await
|
heal_bucket_local_on_disks(bucket, opts, disks).await
|
||||||
@@ -1404,6 +1402,10 @@ pub(crate) async fn heal_bucket_local_on_disks(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "reached only through heal_bucket_local, which only tests call (backlog#1823)"
|
||||||
|
)]
|
||||||
async fn clone_drives() -> Vec<Option<DiskStore>> {
|
async fn clone_drives() -> Vec<Option<DiskStore>> {
|
||||||
runtime_sources::local_disk_entries().await
|
runtime_sources::local_disk_entries().await
|
||||||
}
|
}
|
||||||
@@ -1585,15 +1587,7 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn test_remote_peer(addr: &str) -> RemotePeerS3Client {
|
fn test_remote_peer(addr: &str) -> RemotePeerS3Client {
|
||||||
let node = Node {
|
|
||||||
url: url::Url::parse(addr).expect("test peer URL should parse"),
|
|
||||||
pools: vec![0],
|
|
||||||
is_local: false,
|
|
||||||
grid_host: addr.to_string(),
|
|
||||||
};
|
|
||||||
|
|
||||||
RemotePeerS3Client {
|
RemotePeerS3Client {
|
||||||
node: Some(node),
|
|
||||||
pools: Some(vec![0]),
|
pools: Some(vec![0]),
|
||||||
addr: addr.to_string(),
|
addr: addr.to_string(),
|
||||||
health: Arc::new(DiskHealthTracker::new()),
|
health: Arc::new(DiskHealthTracker::new()),
|
||||||
|
|||||||
@@ -522,6 +522,33 @@ impl RemoteDisk {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn open_read_chunks_with_retry(&self, request: ReadStreamRequest) -> Result<Option<rustfs_rio::ChunkReaderBox>> {
|
||||||
|
let mut attempt = 1;
|
||||||
|
let mut last_retry_classification = None;
|
||||||
|
loop {
|
||||||
|
match self.data_transport.open_read_chunks(request.clone()).await {
|
||||||
|
Ok(reader) => {
|
||||||
|
if attempt > 1
|
||||||
|
&& let Some(classification) = last_retry_classification
|
||||||
|
{
|
||||||
|
crate::cluster::rpc::runtime_sources::record_remote_disk_open_read_retry_success(classification);
|
||||||
|
}
|
||||||
|
return Ok(reader);
|
||||||
|
}
|
||||||
|
Err(err) if attempt < REMOTE_DISK_OPEN_READ_MAX_ATTEMPTS && Self::is_retryable_open_read_error(&err) => {
|
||||||
|
if let Some(classification) = err.internode_http_error_kind() {
|
||||||
|
let classification = classification.metric_label();
|
||||||
|
crate::cluster::rpc::runtime_sources::record_remote_disk_open_read_retry(classification);
|
||||||
|
last_retry_classification = Some(classification);
|
||||||
|
}
|
||||||
|
tokio::time::sleep(REMOTE_DISK_OPEN_READ_RETRY_BACKOFF).await;
|
||||||
|
attempt += 1;
|
||||||
|
}
|
||||||
|
Err(err) => return Err(err),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
pub fn record_capacity_probe(&self, total: u64, used: u64, free: u64) {
|
pub fn record_capacity_probe(&self, total: u64, used: u64, free: u64) {
|
||||||
self.health.record_capacity_probe(total, used, free);
|
self.health.record_capacity_probe(total, used, free);
|
||||||
}
|
}
|
||||||
@@ -846,31 +873,49 @@ impl RemoteDisk {
|
|||||||
/// default to 1 (see [`internode_idempotent_read_retries`]). MUST NOT be used for write/lock
|
/// default to 1 (see [`internode_idempotent_read_retries`]). MUST NOT be used for write/lock
|
||||||
/// RPCs — those must never auto-retry (quorum/idempotency safety). The `operation` closure is
|
/// RPCs — those must never auto-retry (quorum/idempotency safety). The `operation` closure is
|
||||||
/// re-invoked per attempt, so it must be `Fn` (rebuild the request from borrowed inputs, do not
|
/// re-invoked per attempt, so it must be `Fn` (rebuild the request from borrowed inputs, do not
|
||||||
/// move captured state out).
|
/// move captured state out). Attempts and backoff share one total timeout budget.
|
||||||
async fn execute_read_with_retry<T, F, Fut>(&self, op: &'static str, operation: F, timeout_duration: Duration) -> Result<T>
|
async fn execute_read_with_retry<T, F, Fut>(&self, op: &'static str, operation: F, timeout_duration: Duration) -> Result<T>
|
||||||
where
|
where
|
||||||
F: Fn() -> Fut,
|
F: Fn() -> Fut,
|
||||||
Fut: std::future::Future<Output = Result<T>>,
|
Fut: std::future::Future<Output = Result<T>>,
|
||||||
{
|
{
|
||||||
|
let deadline = (!timeout_duration.is_zero()).then(|| {
|
||||||
|
time::Instant::now()
|
||||||
|
.checked_add(timeout_duration)
|
||||||
|
.unwrap_or_else(|| time::sleep(timeout_duration).deadline())
|
||||||
|
});
|
||||||
let max_retries = internode_idempotent_read_retries();
|
let max_retries = internode_idempotent_read_retries();
|
||||||
let mut attempt = 0usize;
|
let mut attempt = 0usize;
|
||||||
loop {
|
loop {
|
||||||
// Only the final attempt marks the disk faulty / evicts the channel. Earlier retries
|
let attempt_timeout = deadline
|
||||||
// ignore the failure, so a transient error cannot flip the disk into a faulty
|
.map(|deadline| deadline.saturating_duration_since(time::Instant::now()))
|
||||||
// short-circuit (which would defeat the retry) or over-count failures.
|
.unwrap_or(Duration::ZERO);
|
||||||
|
if deadline.is_some() && attempt_timeout.is_zero() {
|
||||||
|
self.record_timeout(op, timeout_duration);
|
||||||
|
return Err(DiskError::Timeout);
|
||||||
|
}
|
||||||
|
|
||||||
let health_action = if attempt >= max_retries {
|
let health_action = if attempt >= max_retries {
|
||||||
FailureHealthAction::MarkFailure
|
FailureHealthAction::MarkFailure
|
||||||
} else {
|
} else {
|
||||||
FailureHealthAction::IgnoreFailure
|
FailureHealthAction::IgnoreFailure
|
||||||
};
|
};
|
||||||
match self
|
match self
|
||||||
.execute_with_timeout_for_op_and_health_action(op, &operation, timeout_duration, health_action)
|
.execute_with_timeout_for_op_and_health_action(op, &operation, attempt_timeout, health_action)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
Err(err) if attempt < max_retries && is_network_like_disk_error(&err) => {
|
Err(err) if attempt < max_retries && is_network_like_disk_error(&err) => {
|
||||||
|
if matches!(err, DiskError::Timeout) && deadline.is_some_and(|deadline| time::Instant::now() >= deadline) {
|
||||||
|
self.mark_faulty("read_operation_deadline");
|
||||||
|
return Err(err);
|
||||||
|
}
|
||||||
attempt += 1;
|
attempt += 1;
|
||||||
let backoff = REMOTE_DISK_READ_RETRY_BASE_BACKOFF
|
let backoff = REMOTE_DISK_READ_RETRY_BASE_BACKOFF
|
||||||
.saturating_mul(1u32 << u32::try_from(attempt - 1).unwrap_or(4).min(4));
|
.saturating_mul(1u32 << u32::try_from(attempt - 1).unwrap_or(4).min(4));
|
||||||
|
if deadline.is_some_and(|deadline| deadline.saturating_duration_since(time::Instant::now()) <= backoff) {
|
||||||
|
attempt = max_retries;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
debug!(
|
debug!(
|
||||||
endpoint = %self.endpoint,
|
endpoint = %self.endpoint,
|
||||||
addr = %self.addr,
|
addr = %self.addr,
|
||||||
@@ -878,7 +923,17 @@ impl RemoteDisk {
|
|||||||
attempt,
|
attempt,
|
||||||
"retrying idempotent read-only RPC after transient network error"
|
"retrying idempotent read-only RPC after transient network error"
|
||||||
);
|
);
|
||||||
tokio::time::sleep(backoff).await;
|
if let Some(deadline) = deadline {
|
||||||
|
if time::timeout_at(deadline, time::sleep(backoff)).await.is_err() {
|
||||||
|
self.record_timeout(op, timeout_duration);
|
||||||
|
return Err(DiskError::Timeout);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
time::sleep(backoff).await;
|
||||||
|
}
|
||||||
|
if self.health.is_faulty() {
|
||||||
|
return Err(DiskError::FaultyDisk);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
other => return other,
|
other => return other,
|
||||||
}
|
}
|
||||||
@@ -957,16 +1012,22 @@ impl RemoteDisk {
|
|||||||
operation_result
|
operation_result
|
||||||
}
|
}
|
||||||
Err(_) => {
|
Err(_) => {
|
||||||
// Timeout occurred, mark disk as potentially faulty
|
self.record_timeout(op, timeout_duration);
|
||||||
|
if failure_health_action == FailureHealthAction::MarkFailure {
|
||||||
|
self.mark_faulty_and_evict("operation_timeout").await;
|
||||||
|
}
|
||||||
|
Err(DiskError::Timeout)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn record_timeout(&self, op: &'static str, timeout_duration: Duration) {
|
||||||
counter!(
|
counter!(
|
||||||
"rustfs_drive_op_timeout_total",
|
"rustfs_drive_op_timeout_total",
|
||||||
"endpoint" => self.endpoint.to_string(),
|
"endpoint" => self.endpoint.to_string(),
|
||||||
"op" => op.to_string()
|
"op" => op.to_string()
|
||||||
)
|
)
|
||||||
.increment(1);
|
.increment(1);
|
||||||
if failure_health_action == FailureHealthAction::MarkFailure {
|
|
||||||
self.mark_faulty_and_evict("operation_timeout").await;
|
|
||||||
}
|
|
||||||
warn!(
|
warn!(
|
||||||
event = EVENT_REMOTE_DISK_RPC,
|
event = EVENT_REMOTE_DISK_RPC,
|
||||||
component = LOG_COMPONENT_ECSTORE,
|
component = LOG_COMPONENT_ECSTORE,
|
||||||
@@ -978,9 +1039,6 @@ impl RemoteDisk {
|
|||||||
state = "timeout",
|
state = "timeout",
|
||||||
"Remote disk operation timed out"
|
"Remote disk operation timed out"
|
||||||
);
|
);
|
||||||
Err(DiskError::Timeout)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn handle_network_like_error<T>(
|
async fn handle_network_like_error<T>(
|
||||||
@@ -1016,7 +1074,7 @@ impl RemoteDisk {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn mark_faulty_and_evict(&self, reason: &'static str) {
|
fn mark_faulty(&self, reason: &'static str) -> bool {
|
||||||
let previous_state = self.runtime_state();
|
let previous_state = self.runtime_state();
|
||||||
let transitioned_to_offline = self.mark_suspect_or_offline(reason);
|
let transitioned_to_offline = self.mark_suspect_or_offline(reason);
|
||||||
let state = self.runtime_state();
|
let state = self.runtime_state();
|
||||||
@@ -1053,6 +1111,12 @@ impl RemoteDisk {
|
|||||||
"Remote disk marked suspect"
|
"Remote disk marked suspect"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
state != previous_state
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn mark_faulty_and_evict(&self, reason: &'static str) {
|
||||||
|
if self.mark_faulty(reason) {
|
||||||
counter!(
|
counter!(
|
||||||
"rustfs_drive_connection_evict_total",
|
"rustfs_drive_connection_evict_total",
|
||||||
"endpoint" => self.endpoint.to_string(),
|
"endpoint" => self.endpoint.to_string(),
|
||||||
@@ -1295,6 +1359,71 @@ fn validate_decoded_file_info(file_info: &FileInfo) -> Result<()> {
|
|||||||
file_info.validate_for_metadata_read().map_err(Into::into)
|
file_info.validate_for_metadata_read().map_err(Into::into)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl RemoteDisk {
|
||||||
|
#[tracing::instrument(level = "trace", skip_all)]
|
||||||
|
pub(crate) async fn rename_data_borrowed(
|
||||||
|
&self,
|
||||||
|
src_volume: &str,
|
||||||
|
src_path: &str,
|
||||||
|
fi: &FileInfo,
|
||||||
|
dst_volume: &str,
|
||||||
|
dst_path: &str,
|
||||||
|
) -> Result<RenameDataResp> {
|
||||||
|
trace!(
|
||||||
|
event = EVENT_REMOTE_DISK_RPC,
|
||||||
|
component = LOG_COMPONENT_ECSTORE,
|
||||||
|
subsystem = LOG_SUBSYSTEM_REMOTE_DISK,
|
||||||
|
endpoint = %self.endpoint,
|
||||||
|
src_volume,
|
||||||
|
src_path,
|
||||||
|
dst_volume,
|
||||||
|
dst_path,
|
||||||
|
op = "rename_data",
|
||||||
|
state = "started",
|
||||||
|
"Remote disk RPC started"
|
||||||
|
);
|
||||||
|
|
||||||
|
self.execute_with_timeout_for_op(
|
||||||
|
"rename_data",
|
||||||
|
|| async {
|
||||||
|
let file_info = compat_json(fi)?;
|
||||||
|
let file_info_bin = encode_file_info_msgpack(fi)?;
|
||||||
|
let mut client = self
|
||||||
|
.get_client()
|
||||||
|
.await
|
||||||
|
.map_err(|err| Error::other(format!("can not get client, err: {err}")))?;
|
||||||
|
let mut request = Request::new(RenameDataRequest {
|
||||||
|
disk: self.endpoint.to_string(),
|
||||||
|
src_volume: src_volume.to_string(),
|
||||||
|
src_path: src_path.to_string(),
|
||||||
|
file_info,
|
||||||
|
dst_volume: dst_volume.to_string(),
|
||||||
|
dst_path: dst_path.to_string(),
|
||||||
|
file_info_bin: file_info_bin.into(),
|
||||||
|
});
|
||||||
|
let canonical_body = rustfs_protos::canonical_rename_data_request_body(request.get_ref());
|
||||||
|
attach_mutation_body_digest(&mut request, canonical_body, "rename_data")?;
|
||||||
|
|
||||||
|
let response = client.rename_data(request).await?.into_inner();
|
||||||
|
|
||||||
|
if !response.success {
|
||||||
|
return Err(response.error.unwrap_or_default().into());
|
||||||
|
}
|
||||||
|
|
||||||
|
let rename_data_resp = decode_msgpack_or_json::<RenameDataResp>(
|
||||||
|
&response.rename_data_resp_bin,
|
||||||
|
&response.rename_data_resp,
|
||||||
|
"RenameDataResp",
|
||||||
|
)?;
|
||||||
|
|
||||||
|
Ok(rename_data_resp)
|
||||||
|
},
|
||||||
|
get_max_timeout_duration(),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[async_trait::async_trait]
|
#[async_trait::async_trait]
|
||||||
impl DiskAPI for RemoteDisk {
|
impl DiskAPI for RemoteDisk {
|
||||||
#[tracing::instrument(level = "trace", skip_all)]
|
#[tracing::instrument(level = "trace", skip_all)]
|
||||||
@@ -2068,7 +2197,7 @@ impl DiskAPI for RemoteDisk {
|
|||||||
|
|
||||||
Ok(file_info)
|
Ok(file_info)
|
||||||
},
|
},
|
||||||
get_max_timeout_duration(),
|
get_drive_metadata_timeout(),
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
@@ -2220,57 +2349,7 @@ impl DiskAPI for RemoteDisk {
|
|||||||
dst_volume: &str,
|
dst_volume: &str,
|
||||||
dst_path: &str,
|
dst_path: &str,
|
||||||
) -> Result<RenameDataResp> {
|
) -> Result<RenameDataResp> {
|
||||||
trace!(
|
self.rename_data_borrowed(src_volume, src_path, &fi, dst_volume, dst_path)
|
||||||
event = EVENT_REMOTE_DISK_RPC,
|
|
||||||
component = LOG_COMPONENT_ECSTORE,
|
|
||||||
subsystem = LOG_SUBSYSTEM_REMOTE_DISK,
|
|
||||||
endpoint = %self.endpoint,
|
|
||||||
src_volume,
|
|
||||||
src_path,
|
|
||||||
dst_volume,
|
|
||||||
dst_path,
|
|
||||||
op = "rename_data",
|
|
||||||
state = "started",
|
|
||||||
"Remote disk RPC started"
|
|
||||||
);
|
|
||||||
|
|
||||||
self.execute_with_timeout_for_op(
|
|
||||||
"rename_data",
|
|
||||||
|| async {
|
|
||||||
let file_info = compat_json(&fi)?;
|
|
||||||
let file_info_bin = encode_file_info_msgpack(&fi)?;
|
|
||||||
let mut client = self
|
|
||||||
.get_client()
|
|
||||||
.await
|
|
||||||
.map_err(|err| Error::other(format!("can not get client, err: {err}")))?;
|
|
||||||
let mut request = Request::new(RenameDataRequest {
|
|
||||||
disk: self.endpoint.to_string(),
|
|
||||||
src_volume: src_volume.to_string(),
|
|
||||||
src_path: src_path.to_string(),
|
|
||||||
file_info,
|
|
||||||
dst_volume: dst_volume.to_string(),
|
|
||||||
dst_path: dst_path.to_string(),
|
|
||||||
file_info_bin: file_info_bin.into(),
|
|
||||||
});
|
|
||||||
let canonical_body = rustfs_protos::canonical_rename_data_request_body(request.get_ref());
|
|
||||||
attach_mutation_body_digest(&mut request, canonical_body, "rename_data")?;
|
|
||||||
|
|
||||||
let response = client.rename_data(request).await?.into_inner();
|
|
||||||
|
|
||||||
if !response.success {
|
|
||||||
return Err(response.error.unwrap_or_default().into());
|
|
||||||
}
|
|
||||||
|
|
||||||
let rename_data_resp = decode_msgpack_or_json::<RenameDataResp>(
|
|
||||||
&response.rename_data_resp_bin,
|
|
||||||
&response.rename_data_resp,
|
|
||||||
"RenameDataResp",
|
|
||||||
)?;
|
|
||||||
|
|
||||||
Ok(rename_data_resp)
|
|
||||||
},
|
|
||||||
get_max_timeout_duration(),
|
|
||||||
)
|
|
||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2417,6 +2496,30 @@ impl DiskAPI for RemoteDisk {
|
|||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn read_file_stream_chunks(
|
||||||
|
&self,
|
||||||
|
volume: &str,
|
||||||
|
path: &str,
|
||||||
|
offset: usize,
|
||||||
|
length: usize,
|
||||||
|
) -> Result<Option<rustfs_rio::ChunkReaderBox>> {
|
||||||
|
if self.health.is_faulty() {
|
||||||
|
return Err(DiskError::FaultyDisk);
|
||||||
|
}
|
||||||
|
let disk = self.disk_ref().await;
|
||||||
|
let stall_timeout = get_object_disk_read_timeout();
|
||||||
|
self.open_read_chunks_with_retry(ReadStreamRequest {
|
||||||
|
endpoint: self.endpoint.grid_host(),
|
||||||
|
disk,
|
||||||
|
volume: volume.to_string(),
|
||||||
|
path: path.to_string(),
|
||||||
|
offset,
|
||||||
|
length,
|
||||||
|
stall_timeout: (!stall_timeout.is_zero()).then_some(stall_timeout),
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
/// Buffered read for remote disks.
|
/// Buffered read for remote disks.
|
||||||
/// The transport stream is collected into owned Bytes for caller sharing.
|
/// The transport stream is collected into owned Bytes for caller sharing.
|
||||||
#[tracing::instrument(level = "trace", skip_all)]
|
#[tracing::instrument(level = "trace", skip_all)]
|
||||||
@@ -5094,6 +5197,452 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_reset_during_backoff_preserves_recovery() {
|
||||||
|
let remote_disk = Arc::new(new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await);
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let first_attempt = Arc::new(tokio::sync::Notify::new());
|
||||||
|
let started = time::Instant::now();
|
||||||
|
|
||||||
|
let task_disk = Arc::clone(&remote_disk);
|
||||||
|
let task_attempts = Arc::clone(&attempts);
|
||||||
|
let task_first_attempt = Arc::clone(&first_attempt);
|
||||||
|
let task = tokio::spawn(async move {
|
||||||
|
task_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
move || {
|
||||||
|
let attempt = task_attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
let first_attempt = Arc::clone(&task_first_attempt);
|
||||||
|
async move {
|
||||||
|
if attempt == 0 {
|
||||||
|
time::sleep(Duration::from_millis(20)).await;
|
||||||
|
first_attempt.notify_one();
|
||||||
|
return Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionRefused,
|
||||||
|
"connection refused",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_millis(100),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
});
|
||||||
|
|
||||||
|
first_attempt.notified().await;
|
||||||
|
tokio::task::yield_now().await;
|
||||||
|
remote_disk.health.reset_for_store_init_retry(&remote_disk.endpoint);
|
||||||
|
let channel = TonicEndpoint::from_shared(remote_disk.addr.clone())
|
||||||
|
.expect("remote disk address should parse")
|
||||||
|
.connect_lazy();
|
||||||
|
runtime_sources::cache_test_node_channel(remote_disk.addr.clone(), channel).await;
|
||||||
|
task.await
|
||||||
|
.expect("retry task should finish")
|
||||||
|
.expect("the retry should succeed after the health reset");
|
||||||
|
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 2);
|
||||||
|
assert_eq!(started.elapsed(), Duration::from_millis(70));
|
||||||
|
assert_eq!(
|
||||||
|
remote_disk.health.waiting_count(),
|
||||||
|
0,
|
||||||
|
"health reset must not underflow the waiting counter"
|
||||||
|
);
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Online);
|
||||||
|
assert!(
|
||||||
|
runtime_sources::test_node_channel_is_cached(&remote_disk.addr).await,
|
||||||
|
"a recovered channel must survive the retry backoff"
|
||||||
|
);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_still_retries_within_shared_deadline() {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let channel = TonicEndpoint::from_shared(remote_disk.addr.clone())
|
||||||
|
.expect("remote disk address should parse")
|
||||||
|
.connect_lazy();
|
||||||
|
runtime_sources::cache_test_node_channel(remote_disk.addr.clone(), channel).await;
|
||||||
|
|
||||||
|
remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
let attempt = attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async move {
|
||||||
|
if attempt == 0 {
|
||||||
|
return Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionReset,
|
||||||
|
"connection reset",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_millis(100),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("a retry that fits the shared deadline should succeed");
|
||||||
|
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 2);
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Online);
|
||||||
|
assert!(runtime_sources::test_node_channel_is_cached(&remote_disk.addr).await);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_uses_remaining_budget_for_final_attempt() {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let started = time::Instant::now();
|
||||||
|
let channel = TonicEndpoint::from_shared(remote_disk.addr.clone())
|
||||||
|
.expect("remote disk address should parse")
|
||||||
|
.connect_lazy();
|
||||||
|
runtime_sources::cache_test_node_channel(remote_disk.addr.clone(), channel).await;
|
||||||
|
|
||||||
|
let err = remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
let attempt = attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async move {
|
||||||
|
if attempt == 0 {
|
||||||
|
time::sleep(Duration::from_millis(20)).await;
|
||||||
|
return Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionRefused,
|
||||||
|
"connection refused",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
std::future::pending::<Result<()>>().await
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_millis(100),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("the final retry should consume only the remaining total budget");
|
||||||
|
|
||||||
|
assert_eq!(err, DiskError::Timeout);
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 2);
|
||||||
|
assert_eq!(started.elapsed(), Duration::from_millis(100));
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Suspect);
|
||||||
|
assert!(!runtime_sources::test_node_channel_is_cached(&remote_disk.addr).await);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_uses_final_attempt_at_exact_backoff_boundary() {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let started = time::Instant::now();
|
||||||
|
|
||||||
|
let err = remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
let attempt = attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async move {
|
||||||
|
if attempt == 0 {
|
||||||
|
time::sleep(Duration::from_millis(50)).await;
|
||||||
|
return Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionRefused,
|
||||||
|
"connection refused",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
std::future::pending::<Result<()>>().await
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_millis(100),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("the exact backoff boundary should be reserved for a final attempt");
|
||||||
|
|
||||||
|
assert_eq!(err, DiskError::Timeout);
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 2);
|
||||||
|
assert_eq!(started.elapsed(), Duration::from_millis(100));
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Suspect);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_uses_final_attempt_below_backoff_budget() {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let started = time::Instant::now();
|
||||||
|
|
||||||
|
let err = remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
let attempt = attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async move {
|
||||||
|
if attempt == 0 {
|
||||||
|
time::sleep(Duration::from_millis(80)).await;
|
||||||
|
return Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionRefused,
|
||||||
|
"connection refused",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
std::future::pending::<Result<()>>().await
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_millis(100),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("remaining budget below backoff should be reserved for a final attempt");
|
||||||
|
|
||||||
|
assert_eq!(err, DiskError::Timeout);
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 2);
|
||||||
|
assert_eq!(started.elapsed(), Duration::from_millis(100));
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Suspect);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_zero_timeout_disables_the_deadline() {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let started = time::Instant::now();
|
||||||
|
|
||||||
|
remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
let attempt = attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async move {
|
||||||
|
if attempt == 0 {
|
||||||
|
return Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionReset,
|
||||||
|
"connection reset",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::ZERO,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("zero timeout should allow a retry without a deadline");
|
||||||
|
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 2);
|
||||||
|
assert_eq!(started.elapsed(), REMOTE_DISK_READ_RETRY_BASE_BACKOFF);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_accepts_max_metadata_timeout() {
|
||||||
|
temp_env::async_with_vars([(rustfs_config::ENV_DRIVE_METADATA_TIMEOUT_SECS, Some(u64::MAX.to_string()))], async {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
|
||||||
|
remote_disk
|
||||||
|
.execute_read_with_retry("read_version", || async { Ok::<(), Error>(()) }, get_drive_metadata_timeout())
|
||||||
|
.await
|
||||||
|
.expect("the maximum configured metadata timeout must not panic");
|
||||||
|
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_zero_retries_runs_once() {
|
||||||
|
temp_env::async_with_vars([(rustfs_config::ENV_INTERNODE_IDEMPOTENT_READ_RETRIES, Some("0"))], async {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let started = time::Instant::now();
|
||||||
|
let channel = TonicEndpoint::from_shared(remote_disk.addr.clone())
|
||||||
|
.expect("remote disk address should parse")
|
||||||
|
.connect_lazy();
|
||||||
|
runtime_sources::cache_test_node_channel(remote_disk.addr.clone(), channel).await;
|
||||||
|
|
||||||
|
let err = remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async {
|
||||||
|
Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionReset,
|
||||||
|
"connection reset",
|
||||||
|
)))
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_secs(1),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("zero retries should return the first network error");
|
||||||
|
|
||||||
|
assert!(matches!(err, DiskError::Io(ref io_err) if io_err.kind() == std_io::ErrorKind::ConnectionReset));
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 1);
|
||||||
|
assert_eq!(started.elapsed(), Duration::ZERO);
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Suspect);
|
||||||
|
assert!(!runtime_sources::test_node_channel_is_cached(&remote_disk.addr).await);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_attempt_timeout_marks_health_without_evicting() {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let recorder = crate::test_metrics::CapturingRecorder::default();
|
||||||
|
let _recorder_guard = metrics::set_default_local_recorder(&recorder);
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let channel = TonicEndpoint::from_shared(remote_disk.addr.clone())
|
||||||
|
.expect("remote disk address should parse")
|
||||||
|
.connect_lazy();
|
||||||
|
runtime_sources::cache_test_node_channel(remote_disk.addr.clone(), channel).await;
|
||||||
|
|
||||||
|
let err = remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
std::future::pending::<Result<()>>()
|
||||||
|
},
|
||||||
|
Duration::from_millis(100),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("an in-flight attempt that consumes the deadline should time out");
|
||||||
|
|
||||||
|
assert_eq!(err, DiskError::Timeout);
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 1);
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Suspect);
|
||||||
|
assert!(runtime_sources::test_node_channel_is_cached(&remote_disk.addr).await);
|
||||||
|
assert_eq!(
|
||||||
|
recorder.counter_value(
|
||||||
|
"rustfs_drive_op_timeout_total",
|
||||||
|
&[
|
||||||
|
("endpoint", remote_disk.endpoint.to_string().as_str()),
|
||||||
|
("op", "read_version")
|
||||||
|
]
|
||||||
|
),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_does_not_retry_business_errors() {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
|
||||||
|
let err = remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async { Err::<(), Error>(DiskError::FileNotFound) }
|
||||||
|
},
|
||||||
|
Duration::from_secs(1),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("business errors should be returned directly");
|
||||||
|
|
||||||
|
assert_eq!(err, DiskError::FileNotFound);
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 1);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_honors_configured_retry_count() {
|
||||||
|
temp_env::async_with_vars([(rustfs_config::ENV_INTERNODE_IDEMPOTENT_READ_RETRIES, Some("2"))], async {
|
||||||
|
let remote_disk = new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await;
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let started = time::Instant::now();
|
||||||
|
let channel = TonicEndpoint::from_shared(remote_disk.addr.clone())
|
||||||
|
.expect("remote disk address should parse")
|
||||||
|
.connect_lazy();
|
||||||
|
runtime_sources::cache_test_node_channel(remote_disk.addr.clone(), channel).await;
|
||||||
|
|
||||||
|
let err = remote_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
|| {
|
||||||
|
attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
async {
|
||||||
|
Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionReset,
|
||||||
|
"connection reset",
|
||||||
|
)))
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_secs(1),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("exhausted retries should return the last network error");
|
||||||
|
|
||||||
|
assert!(matches!(err, DiskError::Io(ref io_err) if io_err.kind() == std_io::ErrorKind::ConnectionReset));
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 3);
|
||||||
|
assert_eq!(started.elapsed(), Duration::from_millis(150));
|
||||||
|
assert_eq!(remote_disk.runtime_state(), RuntimeDriveHealthState::Suspect);
|
||||||
|
assert!(!runtime_sources::test_node_channel_is_cached(&remote_disk.addr).await);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
#[serial(remote_disk_read_retry)]
|
||||||
|
async fn execute_read_with_retry_stops_when_disk_turns_offline_during_backoff() {
|
||||||
|
let remote_disk = Arc::new(new_remote_disk_with_transport(Arc::new(RecordingInternodeDataTransport::default())).await);
|
||||||
|
let attempts = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let first_attempt = Arc::new(tokio::sync::Notify::new());
|
||||||
|
let task_disk = Arc::clone(&remote_disk);
|
||||||
|
let task_attempts = Arc::clone(&attempts);
|
||||||
|
let task_first_attempt = Arc::clone(&first_attempt);
|
||||||
|
|
||||||
|
let task = tokio::spawn(async move {
|
||||||
|
task_disk
|
||||||
|
.execute_read_with_retry(
|
||||||
|
"read_version",
|
||||||
|
move || {
|
||||||
|
let attempt = task_attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
let first_attempt = Arc::clone(&task_first_attempt);
|
||||||
|
async move {
|
||||||
|
if attempt == 0 {
|
||||||
|
first_attempt.notify_one();
|
||||||
|
return Err::<(), Error>(DiskError::Io(std_io::Error::new(
|
||||||
|
std_io::ErrorKind::ConnectionReset,
|
||||||
|
"connection reset",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
},
|
||||||
|
Duration::from_secs(1),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
});
|
||||||
|
|
||||||
|
first_attempt.notified().await;
|
||||||
|
tokio::task::yield_now().await;
|
||||||
|
remote_disk
|
||||||
|
.health
|
||||||
|
.force_runtime_state_for_test(RuntimeDriveHealthState::Offline);
|
||||||
|
time::advance(REMOTE_DISK_READ_RETRY_BASE_BACKOFF).await;
|
||||||
|
let err = task
|
||||||
|
.await
|
||||||
|
.expect("retry task should finish")
|
||||||
|
.expect_err("an offline disk must stop before the next attempt");
|
||||||
|
|
||||||
|
assert_eq!(err, DiskError::FaultyDisk);
|
||||||
|
assert_eq!(attempts.load(Ordering::SeqCst), 1);
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_execute_with_timeout_evicts_cached_connection() {
|
async fn test_execute_with_timeout_evicts_cached_connection() {
|
||||||
let addr = "http://127.0.0.1:59991".to_string();
|
let addr = "http://127.0.0.1:59991".to_string();
|
||||||
@@ -5603,6 +6152,40 @@ mod tests {
|
|||||||
accept_task.abort();
|
accept_task.abort();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn read_version_uses_the_metadata_timeout_on_a_stalled_peer() {
|
||||||
|
runtime_sources::ensure_test_rpc_secret();
|
||||||
|
let Some((base_addr, accept_task)) = spawn_stalled_grpc_peer().await else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let remote_disk = remote_disk_for_addr(&base_addr).await;
|
||||||
|
|
||||||
|
temp_env::async_with_vars(
|
||||||
|
[
|
||||||
|
(rustfs_config::ENV_DRIVE_METADATA_TIMEOUT_SECS, Some("1")),
|
||||||
|
(rustfs_config::ENV_DRIVE_MAX_TIMEOUT_DURATION, Some("10")),
|
||||||
|
],
|
||||||
|
async {
|
||||||
|
let started = time::Instant::now();
|
||||||
|
let err = tokio::time::timeout(
|
||||||
|
Duration::from_secs(5),
|
||||||
|
remote_disk.read_version("bucket", "bucket", "object", "", &ReadOptions::default()),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("read_version must use the shorter metadata deadline")
|
||||||
|
.expect_err("a stalled peer must fail read_version");
|
||||||
|
|
||||||
|
assert!(matches!(err, DiskError::Timeout), "expected the metadata deadline to fire, got {err:?}");
|
||||||
|
assert!(started.elapsed() >= Duration::from_millis(900));
|
||||||
|
assert!(started.elapsed() < Duration::from_secs(2));
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
remote_disk.cancel_token.cancel();
|
||||||
|
accept_task.abort();
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn delete_volume_bounds_the_wait_on_a_stalled_peer() {
|
async fn delete_volume_bounds_the_wait_on_a_stalled_peer() {
|
||||||
runtime_sources::ensure_test_rpc_secret();
|
runtime_sources::ensure_test_rpc_secret();
|
||||||
|
|||||||
@@ -48,10 +48,6 @@ impl RemoteClient {
|
|||||||
Self { addr: endpoint }
|
Self { addr: endpoint }
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn from_url(url: url::Url) -> Self {
|
|
||||||
Self { addr: url.to_string() }
|
|
||||||
}
|
|
||||||
|
|
||||||
fn build_ping_request() -> PingRequest {
|
fn build_ping_request() -> PingRequest {
|
||||||
let mut fbb = flatbuffers::FlatBufferBuilder::new();
|
let mut fbb = flatbuffers::FlatBufferBuilder::new();
|
||||||
let payload = fbb.create_vector(b"health-check");
|
let payload = fbb.create_vector(b"health-check");
|
||||||
|
|||||||
@@ -46,7 +46,6 @@ use rustfs_config::{
|
|||||||
SCANNER_SUB_SYS,
|
SCANNER_SUB_SYS,
|
||||||
};
|
};
|
||||||
use rustfs_filemeta::FileInfo;
|
use rustfs_filemeta::FileInfo;
|
||||||
use rustfs_utils::path::SLASH_SEPARATOR;
|
|
||||||
use serde_json::{Map, Value};
|
use serde_json::{Map, Value};
|
||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::sync::LazyLock;
|
use std::sync::LazyLock;
|
||||||
@@ -200,8 +199,6 @@ pub const STORAGE_CLASS_SUB_SYS: &str = "storage_class";
|
|||||||
|
|
||||||
pub const COMMA_SEPARATED_LISTS: &[&str] = &[rustfs_config::oidc::OIDC_SCOPES, rustfs_config::oidc::OIDC_OTHER_AUDIENCES];
|
pub const COMMA_SEPARATED_LISTS: &[&str] = &[rustfs_config::oidc::OIDC_SCOPES, rustfs_config::oidc::OIDC_OTHER_AUDIENCES];
|
||||||
|
|
||||||
static CONFIG_BUCKET: LazyLock<String> = LazyLock::new(|| format!("{RUSTFS_META_BUCKET}{SLASH_SEPARATOR}{CONFIG_PREFIX}"));
|
|
||||||
|
|
||||||
type ServerConfigDecryptFn = crate::bucket::migration::LegacyBlobDecryptFn;
|
type ServerConfigDecryptFn = crate::bucket::migration::LegacyBlobDecryptFn;
|
||||||
|
|
||||||
static SERVER_CONFIG_DECRYPT_FN: LazyLock<RwLock<Option<ServerConfigDecryptFn>>> = LazyLock::new(|| RwLock::new(None));
|
static SERVER_CONFIG_DECRYPT_FN: LazyLock<RwLock<Option<ServerConfigDecryptFn>>> = LazyLock::new(|| RwLock::new(None));
|
||||||
|
|||||||
@@ -13,7 +13,6 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
// #730: configuration migration keeps legacy subsystem definitions available behind this module.
|
// #730: configuration migration keeps legacy subsystem definitions available behind this module.
|
||||||
#![allow(dead_code)]
|
|
||||||
|
|
||||||
mod audit;
|
mod audit;
|
||||||
pub mod com;
|
pub mod com;
|
||||||
|
|||||||
@@ -101,6 +101,7 @@ const DEFAULT_RRS_STORAGE_CLASS: &str = "EC:1";
|
|||||||
const ZERO_SET_DRIVE_COUNT_ERROR: &str = "set drive count must be greater than zero";
|
const ZERO_SET_DRIVE_COUNT_ERROR: &str = "set drive count must be greater than zero";
|
||||||
|
|
||||||
pub static DEFAULT_INLINE_BLOCK: usize = 128 * 1024;
|
pub static DEFAULT_INLINE_BLOCK: usize = 128 * 1024;
|
||||||
|
const DEFAULT_INLINE_OBJECT_BUDGET: usize = 2 * DEFAULT_INLINE_BLOCK;
|
||||||
|
|
||||||
pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
pub static DEFAULT_KVS: LazyLock<KVS> = LazyLock::new(|| {
|
||||||
let kvs = vec![
|
let kvs = vec![
|
||||||
@@ -150,6 +151,8 @@ pub struct Config {
|
|||||||
optimize: Option<String>,
|
optimize: Option<String>,
|
||||||
inline_block: usize,
|
inline_block: usize,
|
||||||
initialized: bool,
|
initialized: bool,
|
||||||
|
#[serde(default, skip_serializing_if = "std::ops::Not::not")]
|
||||||
|
inline_block_explicit: bool,
|
||||||
#[serde(skip)]
|
#[serde(skip)]
|
||||||
standard_parities: Vec<PoolParity>,
|
standard_parities: Vec<PoolParity>,
|
||||||
#[serde(skip)]
|
#[serde(skip)]
|
||||||
@@ -186,6 +189,10 @@ impl Config {
|
|||||||
/// A topology-bound lookup fails closed for unknown drive counts and for
|
/// A topology-bound lookup fails closed for unknown drive counts and for
|
||||||
/// deserialized legacy configurations that have no pool topology. Legacy
|
/// deserialized legacy configurations that have no pool topology. Legacy
|
||||||
/// callers retain scalar compatibility through [`Self::get_parity_for_sc`].
|
/// callers retain scalar compatibility through [`Self::get_parity_for_sc`].
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "per-set parity resolution asserted by this file's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) fn parity_for_sc(&self, sc: &str, drives_per_set: usize) -> Option<usize> {
|
pub(crate) fn parity_for_sc(&self, sc: &str, drives_per_set: usize) -> Option<usize> {
|
||||||
if !self.initialized {
|
if !self.initialized {
|
||||||
return None;
|
return None;
|
||||||
@@ -233,17 +240,19 @@ impl Config {
|
|||||||
.map(|(pool_index, pool)| (pool_index, pool.drives_per_set))
|
.map(|(pool_index, pool)| (pool_index, pool.drives_per_set))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn should_inline(&self, shard_size: i64, versioned: bool) -> bool {
|
pub fn should_inline(&self, shard_size: i64, data_shards: usize, versioned: bool) -> bool {
|
||||||
if shard_size < 0 {
|
if shard_size < 0 || data_shards == 0 {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
let shard_size = shard_size as usize;
|
let shard_size = shard_size as usize;
|
||||||
|
// Keep the historical two-data-shard object budget while preventing
|
||||||
let mut inline_block = DEFAULT_INLINE_BLOCK;
|
// wider EC layouts from multiplying the maximum inline object size.
|
||||||
if self.initialized {
|
let inline_block = if self.initialized && self.inline_block_explicit {
|
||||||
inline_block = self.inline_block;
|
self.inline_block
|
||||||
}
|
} else {
|
||||||
|
(DEFAULT_INLINE_OBJECT_BUDGET / data_shards).min(DEFAULT_INLINE_BLOCK)
|
||||||
|
};
|
||||||
|
|
||||||
if versioned {
|
if versioned {
|
||||||
shard_size <= inline_block / 8
|
shard_size <= inline_block / 8
|
||||||
@@ -392,6 +401,7 @@ fn lookup_config_for_pools_with_env(
|
|||||||
}
|
}
|
||||||
|
|
||||||
let optimize = overrides.optimize;
|
let optimize = overrides.optimize;
|
||||||
|
let inline_block_explicit = overrides.inline_block.is_some();
|
||||||
let inline_block = if let Some(value) = overrides.inline_block {
|
let inline_block = if let Some(value) = overrides.inline_block {
|
||||||
let block = value
|
let block = value
|
||||||
.parse::<bytesize::ByteSize>()
|
.parse::<bytesize::ByteSize>()
|
||||||
@@ -424,6 +434,7 @@ fn lookup_config_for_pools_with_env(
|
|||||||
optimize,
|
optimize,
|
||||||
inline_block,
|
inline_block,
|
||||||
initialized: true,
|
initialized: true,
|
||||||
|
inline_block_explicit,
|
||||||
standard_parities,
|
standard_parities,
|
||||||
rrs_parities,
|
rrs_parities,
|
||||||
})
|
})
|
||||||
@@ -541,22 +552,26 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn should_inline_preserves_exact_default_shard_boundaries() {
|
fn should_inline_scales_default_threshold_by_data_shards() {
|
||||||
let config = Config::default();
|
let config = lookup_config_for_pools_with_env(&KVS::new(), &[3, 12], no_env_overrides())
|
||||||
|
.expect("default inline policy should resolve for EC2+1 and EC8+4");
|
||||||
|
|
||||||
for (case, shard_size, versioned, expected) in [
|
for (case, shard_size, data_shards, versioned, expected) in [
|
||||||
("unversioned below", 128 * 1024 - 1, false, true),
|
("EC2+1 unversioned exact", 128 * 1024, 2, false, true),
|
||||||
("unversioned exact", 128 * 1024, false, true),
|
("EC2+1 unversioned above", 128 * 1024 + 1, 2, false, false),
|
||||||
("unversioned above", 128 * 1024 + 1, false, false),
|
("EC2+1 versioned exact", 16 * 1024, 2, true, true),
|
||||||
("versioned below", 16 * 1024 - 1, true, true),
|
("EC2+1 versioned above", 16 * 1024 + 1, 2, true, false),
|
||||||
("versioned exact", 16 * 1024, true, true),
|
("EC8+4 unversioned exact", 32 * 1024, 8, false, true),
|
||||||
("versioned above", 16 * 1024 + 1, true, false),
|
("EC8+4 unversioned above", 32 * 1024 + 1, 8, false, false),
|
||||||
("negative", -1, false, false),
|
("EC8+4 versioned exact", 4 * 1024, 8, true, true),
|
||||||
|
("EC8+4 versioned above", 4 * 1024 + 1, 8, true, false),
|
||||||
|
("negative", -1, 2, false, false),
|
||||||
|
("zero data shards", 0, 0, false, false),
|
||||||
] {
|
] {
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
config.should_inline(shard_size, versioned),
|
config.should_inline(shard_size, data_shards, versioned),
|
||||||
expected,
|
expected,
|
||||||
"{case}: shard_size={shard_size}, versioned={versioned}"
|
"{case}: shard_size={shard_size}, data_shards={data_shards}, versioned={versioned}"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -577,13 +592,28 @@ mod tests {
|
|||||||
let shard_size = erasure.shard_file_size(object_size);
|
let shard_size = erasure.shard_file_size(object_size);
|
||||||
assert_eq!(shard_size, expected_shard_size, "{case}: object_size={object_size}");
|
assert_eq!(shard_size, expected_shard_size, "{case}: object_size={object_size}");
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
config.should_inline(shard_size, versioned),
|
config.should_inline(shard_size, erasure.data_shards, versioned),
|
||||||
expected,
|
expected,
|
||||||
"{case}: object_size={object_size}, shard_size={shard_size}, versioned={versioned}"
|
"{case}: object_size={object_size}, shard_size={shard_size}, versioned={versioned}"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn explicit_inline_block_preserves_fixed_per_shard_rollback() {
|
||||||
|
let overrides = StorageClassEnvOverrides {
|
||||||
|
inline_block: Some("128KiB".to_string()),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let config = lookup_config_for_pools_with_env(&KVS::new(), &[12], overrides)
|
||||||
|
.expect("explicit inline block should resolve for EC8+4");
|
||||||
|
|
||||||
|
assert!(config.should_inline(128 * 1024, 8, false));
|
||||||
|
assert!(!config.should_inline(128 * 1024 + 1, 8, false));
|
||||||
|
assert!(config.should_inline(16 * 1024, 8, true));
|
||||||
|
assert!(!config.should_inline(16 * 1024 + 1, 8, true));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn write_capability_contract_only_accepts_implemented_layouts() {
|
fn write_capability_contract_only_accepts_implemented_layouts() {
|
||||||
assert_eq!(SUPPORTED_WRITE_CLASSES, [STANDARD, RRS]);
|
assert_eq!(SUPPORTED_WRITE_CLASSES, [STANDARD, RRS]);
|
||||||
@@ -777,6 +807,7 @@ mod tests {
|
|||||||
let encoded = serde_json::to_string(&cfg).expect("config should serialize");
|
let encoded = serde_json::to_string(&cfg).expect("config should serialize");
|
||||||
assert!(!encoded.contains("standard_parities"));
|
assert!(!encoded.contains("standard_parities"));
|
||||||
assert!(!encoded.contains("rrs_parities"));
|
assert!(!encoded.contains("rrs_parities"));
|
||||||
|
assert!(!encoded.contains("inline_block_explicit"));
|
||||||
|
|
||||||
let decoded: Config = serde_json::from_str(&encoded).expect("legacy scalar config should deserialize");
|
let decoded: Config = serde_json::from_str(&encoded).expect("legacy scalar config should deserialize");
|
||||||
assert_eq!(decoded.get_parity_for_sc(STANDARD), Some(2));
|
assert_eq!(decoded.get_parity_for_sc(STANDARD), Some(2));
|
||||||
@@ -786,6 +817,25 @@ mod tests {
|
|||||||
assert!(validate_parity(0, 0).is_err());
|
assert!(validate_parity(0, 0).is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn explicit_inline_block_survives_config_round_trip() {
|
||||||
|
let cfg = lookup_config_for_pools_with_env(
|
||||||
|
&KVS::new(),
|
||||||
|
&[12],
|
||||||
|
StorageClassEnvOverrides {
|
||||||
|
inline_block: Some("128KiB".to_string()),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.expect("explicit inline block should resolve");
|
||||||
|
assert!(cfg.should_inline(100 * 1024, 8, false));
|
||||||
|
|
||||||
|
let encoded = serde_json::to_string(&cfg).expect("config should serialize");
|
||||||
|
assert!(encoded.contains("\"inline_block_explicit\":true"));
|
||||||
|
let decoded: Config = serde_json::from_str(&encoded).expect("explicit inline config should deserialize");
|
||||||
|
assert!(decoded.should_inline(100 * 1024, 8, false));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn lookup_config_reads_rrs_from_class_rrs_key() {
|
fn lookup_config_reads_rrs_from_class_rrs_key() {
|
||||||
// Regression: kvs.get(RRS) used RRS="REDUCED_REDUNDANCY" instead of
|
// Regression: kvs.get(RRS) used RRS="REDUCED_REDUNDANCY" instead of
|
||||||
|
|||||||
@@ -13,7 +13,6 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
// #730: pool coordination helpers are being migrated behind runtime owners.
|
// #730: pool coordination helpers are being migrated behind runtime owners.
|
||||||
#![allow(dead_code)]
|
|
||||||
|
|
||||||
pub(crate) mod pools;
|
pub(crate) mod pools;
|
||||||
pub(crate) mod sets;
|
pub(crate) mod sets;
|
||||||
|
|||||||
@@ -226,6 +226,7 @@ fn ensure_decommission_start_rebalance_meta_allowed(meta: Option<&RebalanceMeta>
|
|||||||
ensure_decommission_not_rebalancing(meta.is_some_and(is_rebalance_conflicting_with_decommission))
|
ensure_decommission_not_rebalancing(meta.is_some_and(is_rebalance_conflicting_with_decommission))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "leader precondition asserted by this file's tests (backlog#1823)")]
|
||||||
fn ensure_local_decommission_pool_leaders(endpoints: &EndpointServerPools, indices: &[usize]) -> Result<()> {
|
fn ensure_local_decommission_pool_leaders(endpoints: &EndpointServerPools, indices: &[usize]) -> Result<()> {
|
||||||
for idx in indices {
|
for idx in indices {
|
||||||
ensure_local_decommission_pool_leader(endpoints, *idx)?;
|
ensure_local_decommission_pool_leader(endpoints, *idx)?;
|
||||||
@@ -1058,11 +1059,19 @@ fn should_cleanup_decommission_source_entry(decommissioned: usize, total_version
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "terminal-state classification asserted by this file's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
enum DecommissionTerminalState {
|
enum DecommissionTerminalState {
|
||||||
Completed,
|
Completed,
|
||||||
Failed,
|
Failed,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "terminal-state classification asserted by this file's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
fn classify_decommission_terminal_state(failed_items_present: bool) -> DecommissionTerminalState {
|
fn classify_decommission_terminal_state(failed_items_present: bool) -> DecommissionTerminalState {
|
||||||
if failed_items_present {
|
if failed_items_present {
|
||||||
DecommissionTerminalState::Failed
|
DecommissionTerminalState::Failed
|
||||||
@@ -2266,15 +2275,19 @@ fn decommission_delete_marker_opts(
|
|||||||
version: &rustfs_filemeta::FileInfo,
|
version: &rustfs_filemeta::FileInfo,
|
||||||
version_id: Option<String>,
|
version_id: Option<String>,
|
||||||
src_pool_idx: usize,
|
src_pool_idx: usize,
|
||||||
|
expected_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||||
) -> ObjectOptions {
|
) -> ObjectOptions {
|
||||||
|
let version_suspended = version.version_id.is_none() && version_id.is_none();
|
||||||
ObjectOptions {
|
ObjectOptions {
|
||||||
versioned: true,
|
versioned: !version_suspended,
|
||||||
version_id,
|
version_suspended,
|
||||||
|
version_id: version_id.or_else(|| version_suspended.then(|| uuid::Uuid::nil().to_string())),
|
||||||
mod_time: version.mod_time,
|
mod_time: version.mod_time,
|
||||||
src_pool_idx,
|
src_pool_idx,
|
||||||
data_movement: true,
|
data_movement: true,
|
||||||
delete_marker: true,
|
delete_marker: true,
|
||||||
skip_decommissioned: true,
|
skip_decommissioned: true,
|
||||||
|
expected_bucket_incarnation_id,
|
||||||
delete_replication: version
|
delete_replication: version
|
||||||
.replication_state_internal
|
.replication_state_internal
|
||||||
.as_ref()
|
.as_ref()
|
||||||
@@ -2299,6 +2312,7 @@ fn decommission_remote_tiered_opts(
|
|||||||
version: &rustfs_filemeta::FileInfo,
|
version: &rustfs_filemeta::FileInfo,
|
||||||
version_id: Option<String>,
|
version_id: Option<String>,
|
||||||
src_pool_idx: usize,
|
src_pool_idx: usize,
|
||||||
|
expected_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||||
) -> ObjectOptions {
|
) -> ObjectOptions {
|
||||||
ObjectOptions {
|
ObjectOptions {
|
||||||
versioned: version_id.is_some(),
|
versioned: version_id.is_some(),
|
||||||
@@ -2307,6 +2321,9 @@ fn decommission_remote_tiered_opts(
|
|||||||
user_defined: version.metadata.clone(),
|
user_defined: version.metadata.clone(),
|
||||||
src_pool_idx,
|
src_pool_idx,
|
||||||
data_movement: true,
|
data_movement: true,
|
||||||
|
include_part_checksums: true,
|
||||||
|
http_preconditions: Some(crate::data_movement::data_movement_target_precondition()),
|
||||||
|
expected_bucket_incarnation_id,
|
||||||
..Default::default()
|
..Default::default()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -2805,6 +2822,7 @@ impl ECStore {
|
|||||||
lifecycle_config: Option<BucketLifecycleConfiguration>,
|
lifecycle_config: Option<BucketLifecycleConfiguration>,
|
||||||
object_lock_config: Option<ObjectLockConfiguration>,
|
object_lock_config: Option<ObjectLockConfiguration>,
|
||||||
replication_config: Option<(ReplicationConfiguration, OffsetDateTime)>,
|
replication_config: Option<(ReplicationConfiguration, OffsetDateTime)>,
|
||||||
|
expected_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||||
) -> Result<()> {
|
) -> Result<()> {
|
||||||
debug!(
|
debug!(
|
||||||
event = EVENT_DECOMMISSION_ENTRY,
|
event = EVENT_DECOMMISSION_ENTRY,
|
||||||
@@ -2834,6 +2852,11 @@ impl ECStore {
|
|||||||
}
|
}
|
||||||
decommission_cancel_signal_result(rx.is_cancelled())?;
|
decommission_cancel_signal_result(rx.is_cancelled())?;
|
||||||
|
|
||||||
|
let bucket_incarnation_fence = match expected_bucket_incarnation_id {
|
||||||
|
Some(expected) => Some(self.acquire_bucket_incarnation_fence(&bucket, expected).await?),
|
||||||
|
None => None,
|
||||||
|
};
|
||||||
|
|
||||||
let mut fivs = load_decommission_entry_exact_versions(&set, &entry, &bucket, "file_info_versions").await?;
|
let mut fivs = load_decommission_entry_exact_versions(&set, &entry, &bucket, "file_info_versions").await?;
|
||||||
|
|
||||||
fivs.versions
|
fivs.versions
|
||||||
@@ -2894,7 +2917,7 @@ impl ECStore {
|
|||||||
.delete_object(
|
.delete_object(
|
||||||
bucket.as_str(),
|
bucket.as_str(),
|
||||||
&version.name,
|
&version.name,
|
||||||
decommission_delete_marker_opts(version, version_id.clone(), idx),
|
decommission_delete_marker_opts(version, version_id.clone(), idx, expected_bucket_incarnation_id),
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
@@ -2984,7 +3007,7 @@ impl ECStore {
|
|||||||
bucket.as_str(),
|
bucket.as_str(),
|
||||||
&version.name,
|
&version.name,
|
||||||
version,
|
version,
|
||||||
&decommission_remote_tiered_opts(version, version_id.clone(), idx),
|
&decommission_remote_tiered_opts(version, version_id.clone(), idx, expected_bucket_incarnation_id),
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
@@ -3056,7 +3079,11 @@ impl ECStore {
|
|||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
if let Err(err) = self.clone().decommission_object(idx, bucket, rd).await {
|
if let Err(err) = self
|
||||||
|
.clone()
|
||||||
|
.decommission_object(idx, bucket, rd, expected_bucket_incarnation_id)
|
||||||
|
.await
|
||||||
|
{
|
||||||
if is_decommission_copy_cleanup_safe_error(&err) {
|
if is_decommission_copy_cleanup_safe_error(&err) {
|
||||||
ignore = true;
|
ignore = true;
|
||||||
cleanup_ignored = true;
|
cleanup_ignored = true;
|
||||||
@@ -3133,6 +3160,9 @@ impl ECStore {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if should_cleanup_decommission_source_entry(decommissioned, fivs.versions.len(), expired) {
|
if should_cleanup_decommission_source_entry(decommissioned, fivs.versions.len(), expired) {
|
||||||
|
if bucket_incarnation_fence.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
|
||||||
|
return Err(Error::other("decommission bucket incarnation fence was lost before source cleanup"));
|
||||||
|
}
|
||||||
decommission_cancel_signal_result(rx.is_cancelled())?;
|
decommission_cancel_signal_result(rx.is_cancelled())?;
|
||||||
|
|
||||||
self.save_decommission_entry_progress_stage(
|
self.save_decommission_entry_progress_stage(
|
||||||
@@ -3157,6 +3187,12 @@ impl ECStore {
|
|||||||
entry.name.as_str(),
|
entry.name.as_str(),
|
||||||
&fivs,
|
&fivs,
|
||||||
&cleanup_preflight_allowed_missing,
|
&cleanup_preflight_allowed_missing,
|
||||||
|
data_movement::SourceCleanupBucketFence {
|
||||||
|
expected_incarnation_id: expected_bucket_incarnation_id,
|
||||||
|
lifecycle_guard: bucket_incarnation_fence
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|guard| guard.namespace_lock_guard()),
|
||||||
|
},
|
||||||
"decommission",
|
"decommission",
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
@@ -3268,6 +3304,11 @@ impl ECStore {
|
|||||||
let mut lifecycle_config = None;
|
let mut lifecycle_config = None;
|
||||||
let mut object_lock_config = None;
|
let mut object_lock_config = None;
|
||||||
let mut replication_config = None;
|
let mut replication_config = None;
|
||||||
|
let expected_bucket_incarnation_id = if bi.name == RUSTFS_META_BUCKET {
|
||||||
|
None
|
||||||
|
} else {
|
||||||
|
Some(self.bucket_incarnation_id_from_disk(&bi.name).await?)
|
||||||
|
};
|
||||||
|
|
||||||
if bi.name != RUSTFS_META_BUCKET {
|
if bi.name != RUSTFS_META_BUCKET {
|
||||||
let _ = resolve_decommission_optional_bucket_config_result(
|
let _ = resolve_decommission_optional_bucket_config_result(
|
||||||
@@ -3321,6 +3362,7 @@ impl ECStore {
|
|||||||
let lifecycle_config = lifecycle_config.clone();
|
let lifecycle_config = lifecycle_config.clone();
|
||||||
let object_lock_config = object_lock_config.clone();
|
let object_lock_config = object_lock_config.clone();
|
||||||
let replication_config = replication_config.clone();
|
let replication_config = replication_config.clone();
|
||||||
|
let expected_bucket_incarnation_id = expected_bucket_incarnation_id;
|
||||||
let entry_error = entry_error.clone();
|
let entry_error = entry_error.clone();
|
||||||
let callback_rx = callback_rx.clone();
|
let callback_rx = callback_rx.clone();
|
||||||
|
|
||||||
@@ -3383,6 +3425,7 @@ impl ECStore {
|
|||||||
lifecycle_config,
|
lifecycle_config,
|
||||||
object_lock_config,
|
object_lock_config,
|
||||||
replication_config,
|
replication_config,
|
||||||
|
expected_bucket_incarnation_id,
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
@@ -4168,10 +4211,24 @@ impl ECStore {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[tracing::instrument(skip(self, rd))]
|
#[tracing::instrument(skip(self, rd))]
|
||||||
async fn decommission_object(self: Arc<Self>, pool_idx: usize, bucket: String, rd: GetObjectReader) -> Result<()> {
|
async fn decommission_object(
|
||||||
|
self: Arc<Self>,
|
||||||
|
pool_idx: usize,
|
||||||
|
bucket: String,
|
||||||
|
rd: GetObjectReader,
|
||||||
|
expected_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||||
|
) -> Result<()> {
|
||||||
warn!("decommission_object: start {} {}", &bucket, &rd.object_info.name);
|
warn!("decommission_object: start {} {}", &bucket, &rd.object_info.name);
|
||||||
let object_name = rd.object_info.name.clone();
|
let object_name = rd.object_info.name.clone();
|
||||||
let result = data_movement::migrate_object(self, pool_idx, bucket.clone(), rd, "decommission_object").await;
|
let result = data_movement::migrate_object(
|
||||||
|
self,
|
||||||
|
pool_idx,
|
||||||
|
bucket.clone(),
|
||||||
|
rd,
|
||||||
|
expected_bucket_incarnation_id,
|
||||||
|
"decommission_object",
|
||||||
|
)
|
||||||
|
.await;
|
||||||
if result.is_ok() {
|
if result.is_ok() {
|
||||||
warn!("decommission_object: migrated {} {}", &bucket, &object_name);
|
warn!("decommission_object: migrated {} {}", &bucket, &object_name);
|
||||||
}
|
}
|
||||||
@@ -4347,7 +4404,8 @@ mod tests {
|
|||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|
||||||
let opts = decommission_delete_marker_opts(&version, Some("version-id".to_string()), 7);
|
let incarnation = uuid::Uuid::new_v4();
|
||||||
|
let opts = decommission_delete_marker_opts(&version, Some("version-id".to_string()), 7, Some(incarnation));
|
||||||
let replication = opts.delete_replication.expect("replication state should be preserved");
|
let replication = opts.delete_replication.expect("replication state should be preserved");
|
||||||
|
|
||||||
assert!(opts.versioned);
|
assert!(opts.versioned);
|
||||||
@@ -4357,11 +4415,25 @@ mod tests {
|
|||||||
assert_eq!(opts.src_pool_idx, 7);
|
assert_eq!(opts.src_pool_idx, 7);
|
||||||
assert_eq!(opts.version_id.as_deref(), Some("version-id"));
|
assert_eq!(opts.version_id.as_deref(), Some("version-id"));
|
||||||
assert_eq!(opts.mod_time, Some(mod_time));
|
assert_eq!(opts.mod_time, Some(mod_time));
|
||||||
|
assert_eq!(opts.expected_bucket_incarnation_id, Some(incarnation));
|
||||||
assert_eq!(replication.replica_status, ReplicationStatusType::Replica);
|
assert_eq!(replication.replica_status, ReplicationStatusType::Replica);
|
||||||
assert!(replication.delete_marker);
|
assert!(replication.delete_marker);
|
||||||
assert_eq!(replication.replicate_decision_str, "existing");
|
assert_eq!(replication.replicate_decision_str, "existing");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decommission_delete_marker_opts_preserves_suspended_null_version() {
|
||||||
|
let version = rustfs_filemeta::FileInfo {
|
||||||
|
deleted: true,
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let opts = decommission_delete_marker_opts(&version, None, 7, None);
|
||||||
|
|
||||||
|
assert!(!opts.versioned);
|
||||||
|
assert!(opts.version_suspended);
|
||||||
|
assert_eq!(opts.version_id.as_deref(), Some(uuid::Uuid::nil().to_string().as_str()));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_decommission_object_migration_read_opts_are_raw_data_movement() {
|
fn test_decommission_object_migration_read_opts_are_raw_data_movement() {
|
||||||
let opts = decommission_object_migration_read_opts(Some("vid-1".to_string()));
|
let opts = decommission_object_migration_read_opts(Some("vid-1".to_string()));
|
||||||
@@ -4383,7 +4455,8 @@ mod tests {
|
|||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|
||||||
let opts = decommission_remote_tiered_opts(&version, Some("version-id".to_string()), 9);
|
let incarnation = uuid::Uuid::new_v4();
|
||||||
|
let opts = decommission_remote_tiered_opts(&version, Some("version-id".to_string()), 9, Some(incarnation));
|
||||||
|
|
||||||
assert!(opts.versioned);
|
assert!(opts.versioned);
|
||||||
assert!(opts.data_movement);
|
assert!(opts.data_movement);
|
||||||
@@ -4391,6 +4464,9 @@ mod tests {
|
|||||||
assert_eq!(opts.version_id.as_deref(), Some("version-id"));
|
assert_eq!(opts.version_id.as_deref(), Some("version-id"));
|
||||||
assert_eq!(opts.mod_time, Some(mod_time));
|
assert_eq!(opts.mod_time, Some(mod_time));
|
||||||
assert_eq!(opts.user_defined.get("x-amz-meta-key").map(String::as_str), Some("value"));
|
assert_eq!(opts.user_defined.get("x-amz-meta-key").map(String::as_str), Some("value"));
|
||||||
|
assert!(opts.include_part_checksums);
|
||||||
|
assert!(opts.http_preconditions.is_some());
|
||||||
|
assert_eq!(opts.expected_bucket_incarnation_id, Some(incarnation));
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
+1555
-244
File diff suppressed because it is too large
Load Diff
@@ -12,6 +12,16 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! Per-disk usage snapshots persisted under the metadata bucket.
|
||||||
|
//!
|
||||||
|
//! **Nothing calls into this module.** It landed complete with tests in #5307
|
||||||
|
//! (2026-07-27) and its aggregation entry point,
|
||||||
|
//! [`crate::data_usage::aggregate_local_snapshots`], has never had a caller in
|
||||||
|
//! the tree's history. The live data-usage path is
|
||||||
|
//! `load_data_usage_from_backend` / `store_data_usage_in_backend`. The items
|
||||||
|
//! below therefore carry individual `dead_code` allows rather than a module
|
||||||
|
//! blanket, so the gap stays greppable until it is either wired up or removed.
|
||||||
|
|
||||||
use crate::data_usage::BucketUsageInfo;
|
use crate::data_usage::BucketUsageInfo;
|
||||||
use crate::disk::RUSTFS_META_BUCKET;
|
use crate::disk::RUSTFS_META_BUCKET;
|
||||||
use crate::error::{Error, Result};
|
use crate::error::{Error, Result};
|
||||||
@@ -26,10 +36,12 @@ pub const DATA_USAGE_DIR: &str = "datausage";
|
|||||||
/// Directory used to store incremental scan state files under the metadata bucket.
|
/// Directory used to store incremental scan state files under the metadata bucket.
|
||||||
pub const DATA_USAGE_STATE_DIR: &str = "datausage/state";
|
pub const DATA_USAGE_STATE_DIR: &str = "datausage/state";
|
||||||
/// Snapshot file format version, allows forward compatibility if the structure evolves.
|
/// Snapshot file format version, allows forward compatibility if the structure evolves.
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub const LOCAL_USAGE_SNAPSHOT_VERSION: u32 = 1;
|
pub const LOCAL_USAGE_SNAPSHOT_VERSION: u32 = 1;
|
||||||
|
|
||||||
/// Additional metadata describing which disk produced the snapshot.
|
/// Additional metadata describing which disk produced the snapshot.
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub struct LocalUsageSnapshotMeta {
|
pub struct LocalUsageSnapshotMeta {
|
||||||
/// Disk UUID stored as a string for simpler serialization.
|
/// Disk UUID stored as a string for simpler serialization.
|
||||||
pub disk_id: String,
|
pub disk_id: String,
|
||||||
@@ -43,6 +55,7 @@ pub struct LocalUsageSnapshotMeta {
|
|||||||
|
|
||||||
/// Usage snapshot produced by a single disk.
|
/// Usage snapshot produced by a single disk.
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
#[derive(Debug, Clone, Serialize, Deserialize, Default)]
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub struct LocalUsageSnapshot {
|
pub struct LocalUsageSnapshot {
|
||||||
/// Format version recorded in the snapshot.
|
/// Format version recorded in the snapshot.
|
||||||
pub format_version: u32,
|
pub format_version: u32,
|
||||||
@@ -64,6 +77,7 @@ pub struct LocalUsageSnapshot {
|
|||||||
pub objects_total_size: u64,
|
pub objects_total_size: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
impl LocalUsageSnapshot {
|
impl LocalUsageSnapshot {
|
||||||
/// Create an empty snapshot with the default format version filled in.
|
/// Create an empty snapshot with the default format version filled in.
|
||||||
pub fn new(meta: LocalUsageSnapshotMeta) -> Self {
|
pub fn new(meta: LocalUsageSnapshotMeta) -> Self {
|
||||||
@@ -99,11 +113,13 @@ impl LocalUsageSnapshot {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Build the snapshot file name `<disk-id>.json`.
|
/// Build the snapshot file name `<disk-id>.json`.
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub fn snapshot_file_name(disk_id: &str) -> String {
|
pub fn snapshot_file_name(disk_id: &str) -> String {
|
||||||
format!("{disk_id}.json")
|
format!("{disk_id}.json")
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the object path relative to `RUSTFS_META_BUCKET`, e.g. `datausage/<disk-id>.json`.
|
/// Build the object path relative to `RUSTFS_META_BUCKET`, e.g. `datausage/<disk-id>.json`.
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub fn snapshot_object_path(disk_id: &str) -> String {
|
pub fn snapshot_object_path(disk_id: &str) -> String {
|
||||||
format!("{}/{}", DATA_USAGE_DIR, snapshot_file_name(disk_id))
|
format!("{}/{}", DATA_USAGE_DIR, snapshot_file_name(disk_id))
|
||||||
}
|
}
|
||||||
@@ -119,11 +135,13 @@ pub fn data_usage_state_dir(root: &Path) -> PathBuf {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Build the absolute path to the snapshot file for the provided disk ID.
|
/// Build the absolute path to the snapshot file for the provided disk ID.
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub fn snapshot_path(root: &Path, disk_id: &str) -> PathBuf {
|
pub fn snapshot_path(root: &Path, disk_id: &str) -> PathBuf {
|
||||||
data_usage_dir(root).join(snapshot_file_name(disk_id))
|
data_usage_dir(root).join(snapshot_file_name(disk_id))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read a snapshot from disk if it exists.
|
/// Read a snapshot from disk if it exists.
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub async fn read_snapshot(root: &Path, disk_id: &str) -> Result<Option<LocalUsageSnapshot>> {
|
pub async fn read_snapshot(root: &Path, disk_id: &str) -> Result<Option<LocalUsageSnapshot>> {
|
||||||
let path = snapshot_path(root, disk_id);
|
let path = snapshot_path(root, disk_id);
|
||||||
match fs::read(&path).await {
|
match fs::read(&path).await {
|
||||||
@@ -138,6 +156,7 @@ pub async fn read_snapshot(root: &Path, disk_id: &str) -> Result<Option<LocalUsa
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Persist a snapshot to disk, creating directories as needed and overwriting any existing file.
|
/// Persist a snapshot to disk, creating directories as needed and overwriting any existing file.
|
||||||
|
#[allow(dead_code, reason = "unwired local usage-snapshot feature; see module docs (backlog#1823)")]
|
||||||
pub async fn write_snapshot(root: &Path, disk_id: &str, snapshot: &LocalUsageSnapshot) -> Result<()> {
|
pub async fn write_snapshot(root: &Path, disk_id: &str, snapshot: &LocalUsageSnapshot) -> Result<()> {
|
||||||
let dir = data_usage_dir(root);
|
let dir = data_usage_dir(root);
|
||||||
fs::create_dir_all(&dir).await.map_err(Error::other)?;
|
fs::create_dir_all(&dir).await.map_err(Error::other)?;
|
||||||
|
|||||||
@@ -13,7 +13,6 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
// #730: scanner/data-usage state is partially migrated and still owns staged cache helpers.
|
// #730: scanner/data-usage state is partially migrated and still owns staged cache helpers.
|
||||||
#![allow(dead_code)]
|
|
||||||
|
|
||||||
pub mod local_snapshot;
|
pub mod local_snapshot;
|
||||||
|
|
||||||
@@ -34,8 +33,8 @@ use crate::{
|
|||||||
pub use local_snapshot::{LocalUsageSnapshot, read_snapshot as read_local_snapshot, snapshot_path};
|
pub use local_snapshot::{LocalUsageSnapshot, read_snapshot as read_local_snapshot, snapshot_path};
|
||||||
use rustfs_data_usage::{
|
use rustfs_data_usage::{
|
||||||
BucketTargetUsageInfo, BucketUsageInfo, CompressionTotalInfo, DATA_USAGE_OBJECT_NAME, DATA_USAGE_OBSERVED_OBJECT_NAME,
|
BucketTargetUsageInfo, BucketUsageInfo, CompressionTotalInfo, DATA_USAGE_OBJECT_NAME, DATA_USAGE_OBSERVED_OBJECT_NAME,
|
||||||
DataUsageCache, DataUsageEntry, DataUsageInfo, DiskUsageStatus, LEGACY_DATA_USAGE_OBJECT_NAME, SizeHistogram, SizeSummary,
|
DataUsageCache, DataUsageInfo, DiskUsageStatus, LEGACY_DATA_USAGE_OBJECT_NAME, SizeHistogram, VersionsHistogram,
|
||||||
VersionsHistogram, observed_data_usage_is_newer,
|
observed_data_usage_is_newer,
|
||||||
};
|
};
|
||||||
use rustfs_io_metrics::record_system_path_failure;
|
use rustfs_io_metrics::record_system_path_failure;
|
||||||
use rustfs_utils::path::SLASH_SEPARATOR;
|
use rustfs_utils::path::SLASH_SEPARATOR;
|
||||||
@@ -55,7 +54,6 @@ use tracing::{debug, error, info, instrument};
|
|||||||
// Data usage storage constants
|
// Data usage storage constants
|
||||||
pub const DATA_USAGE_ROOT: &str = SLASH_SEPARATOR;
|
pub const DATA_USAGE_ROOT: &str = SLASH_SEPARATOR;
|
||||||
const DATA_COMPRESSION_TOTAL_NAME: &str = ".compression.json";
|
const DATA_COMPRESSION_TOTAL_NAME: &str = ".compression.json";
|
||||||
const DATA_USAGE_BLOOM_NAME: &str = ".bloomcycle.bin";
|
|
||||||
pub const DATA_USAGE_CACHE_NAME: &str = ".usage-cache.bin";
|
pub const DATA_USAGE_CACHE_NAME: &str = ".usage-cache.bin";
|
||||||
const DATA_USAGE_CACHE_TTL_SECS: u64 = 30;
|
const DATA_USAGE_CACHE_TTL_SECS: u64 = 30;
|
||||||
const LIVE_BUCKET_USAGE_MAX_ENTRIES: u64 = 1024;
|
const LIVE_BUCKET_USAGE_MAX_ENTRIES: u64 = 1024;
|
||||||
@@ -313,11 +311,6 @@ lazy_static::lazy_static! {
|
|||||||
LEGACY_DATA_USAGE_OBJECT_NAME
|
LEGACY_DATA_USAGE_OBJECT_NAME
|
||||||
);
|
);
|
||||||
static ref LEGACY_DATA_USAGE_OBJ_BACKUP_PATH: String = format!("{}.bkp", LEGACY_DATA_USAGE_OBJ_NAME_PATH.as_str());
|
static ref LEGACY_DATA_USAGE_OBJ_BACKUP_PATH: String = format!("{}.bkp", LEGACY_DATA_USAGE_OBJ_NAME_PATH.as_str());
|
||||||
pub static ref DATA_USAGE_BLOOM_NAME_PATH: String = format!("{}{}{}",
|
|
||||||
crate::disk::BUCKET_META_PREFIX,
|
|
||||||
SLASH_SEPARATOR,
|
|
||||||
DATA_USAGE_BLOOM_NAME
|
|
||||||
);
|
|
||||||
pub static ref DATA_COMPRESSION_TOTAL_NAME_PATH: String = format!("{}{}{}",
|
pub static ref DATA_COMPRESSION_TOTAL_NAME_PATH: String = format!("{}{}{}",
|
||||||
crate::disk::BUCKET_META_PREFIX,
|
crate::disk::BUCKET_META_PREFIX,
|
||||||
SLASH_SEPARATOR,
|
SLASH_SEPARATOR,
|
||||||
@@ -858,6 +851,10 @@ async fn resolve_loaded_snapshot_pair_with_source(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "primary/backup snapshot fallback asserted by this file's tests (backlog#1823)"
|
||||||
|
)]
|
||||||
async fn resolve_loaded_snapshot(
|
async fn resolve_loaded_snapshot(
|
||||||
primary: Result<Vec<u8>, Error>,
|
primary: Result<Vec<u8>, Error>,
|
||||||
backup: impl Future<Output = Result<Vec<u8>, Error>>,
|
backup: impl Future<Output = Result<Vec<u8>, Error>>,
|
||||||
@@ -1187,6 +1184,10 @@ pub async fn invalidate_admin_data_usage_snapshot_cache() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Aggregate usage information from local disk snapshots.
|
/// Aggregate usage information from local disk snapshots.
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "reached only through aggregate_local_snapshots, which has no caller (backlog#1823)"
|
||||||
|
)]
|
||||||
fn merge_snapshot(aggregated: &mut DataUsageInfo, mut snapshot: LocalUsageSnapshot, latest_update: &mut Option<SystemTime>) {
|
fn merge_snapshot(aggregated: &mut DataUsageInfo, mut snapshot: LocalUsageSnapshot, latest_update: &mut Option<SystemTime>) {
|
||||||
if let Some(update) = snapshot.last_update
|
if let Some(update) = snapshot.last_update
|
||||||
&& latest_update.is_none_or(|current| update > current)
|
&& latest_update.is_none_or(|current| update > current)
|
||||||
@@ -1220,6 +1221,10 @@ fn merge_snapshot(aggregated: &mut DataUsageInfo, mut snapshot: LocalUsageSnapsh
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "entry point of the local usage-snapshot feature, which has had no caller since it landed in #5307 (backlog#1823)"
|
||||||
|
)]
|
||||||
pub async fn aggregate_local_snapshots(store: Arc<ECStore>) -> Result<(Vec<DiskUsageStatus>, DataUsageInfo), Error> {
|
pub async fn aggregate_local_snapshots(store: Arc<ECStore>) -> Result<(Vec<DiskUsageStatus>, DataUsageInfo), Error> {
|
||||||
let mut aggregated = DataUsageInfo::default();
|
let mut aggregated = DataUsageInfo::default();
|
||||||
let mut latest_update: Option<SystemTime> = None;
|
let mut latest_update: Option<SystemTime> = None;
|
||||||
@@ -1355,7 +1360,7 @@ impl BucketUsageAccumulator {
|
|||||||
return Ok(());
|
return Ok(());
|
||||||
}
|
}
|
||||||
|
|
||||||
let object_size = object.size.max(0) as u64;
|
let object_size = quota_object_size(object)?;
|
||||||
self.current_live_versions = self.current_live_versions.saturating_add(1);
|
self.current_live_versions = self.current_live_versions.saturating_add(1);
|
||||||
self.size_histogram.add(object_size);
|
self.size_histogram.add(object_size);
|
||||||
self.total_size = self.total_size.saturating_add(object_size);
|
self.total_size = self.total_size.saturating_add(object_size);
|
||||||
@@ -1385,6 +1390,31 @@ impl BucketUsageAccumulator {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn quota_object_size(object: &ObjectInfo) -> Result<u64, Error> {
|
||||||
|
let logical_size = u64::try_from(object.get_actual_size().map_err(Error::other)?).map_err(|_| Error::PartMissingOrCorrupt)?;
|
||||||
|
let persisted_part_size = if object.parts.is_empty() {
|
||||||
|
u64::try_from(object.size).map_err(|_| Error::PartMissingOrCorrupt)?
|
||||||
|
} else {
|
||||||
|
object.parts.iter().try_fold(0_u64, |total, part| {
|
||||||
|
// Compressed streaming objects persist -1 when the transformed
|
||||||
|
// part size is unknown. The physical part size remains a valid
|
||||||
|
// quota floor; reject only non-negative values that overflow.
|
||||||
|
let actual_size = if part.actual_size < 0 {
|
||||||
|
if object.is_compressed() {
|
||||||
|
0
|
||||||
|
} else {
|
||||||
|
return Err(Error::PartMissingOrCorrupt);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
u64::try_from(part.actual_size).map_err(|_| Error::PartMissingOrCorrupt)?
|
||||||
|
};
|
||||||
|
let part_size = actual_size.max(u64::try_from(part.size).map_err(|_| Error::PartMissingOrCorrupt)?);
|
||||||
|
total.checked_add(part_size).ok_or(Error::PartMissingOrCorrupt)
|
||||||
|
})?
|
||||||
|
};
|
||||||
|
Ok(logical_size.max(persisted_part_size))
|
||||||
|
}
|
||||||
|
|
||||||
type UsageVersionPage = StorageListObjectVersionsInfo<ObjectInfo>;
|
type UsageVersionPage = StorageListObjectVersionsInfo<ObjectInfo>;
|
||||||
|
|
||||||
pub async fn compute_bucket_usage(store: Arc<ECStore>, bucket_name: &str) -> Result<BucketUsageInfo, Error> {
|
pub async fn compute_bucket_usage(store: Arc<ECStore>, bucket_name: &str) -> Result<BucketUsageInfo, Error> {
|
||||||
@@ -1638,7 +1668,7 @@ fn preserve_unknown_dirty_usage(
|
|||||||
Some(preserved)
|
Some(preserved)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
async fn replace_bucket_usage_memory_from_authoritative(bucket: &str, usage: BucketUsageInfo, refresh_started_at: SystemTime) {
|
async fn replace_bucket_usage_memory_from_authoritative(bucket: &str, usage: BucketUsageInfo, refresh_started_at: SystemTime) {
|
||||||
let mut cache = memory_cache().write().await;
|
let mut cache = memory_cache().write().await;
|
||||||
if let Some(existing) = cache.get(bucket)
|
if let Some(existing) = cache.get(bucket)
|
||||||
@@ -1650,6 +1680,19 @@ async fn replace_bucket_usage_memory_from_authoritative(bucket: &str, usage: Buc
|
|||||||
cache.insert(bucket.to_string(), cached_bucket_usage_from_backend(usage, refresh_started_at, true));
|
cache.insert(bucket.to_string(), cached_bucket_usage_from_backend(usage, refresh_started_at, true));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
pub async fn seed_bucket_usage_memory_for_test(bucket: &str, size: u64) {
|
||||||
|
replace_bucket_usage_memory_from_authoritative(
|
||||||
|
bucket,
|
||||||
|
BucketUsageInfo {
|
||||||
|
size,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
SystemTime::now(),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
/// Fast in-memory update for immediate quota and admin usage consistency.
|
/// Fast in-memory update for immediate quota and admin usage consistency.
|
||||||
pub async fn record_bucket_object_write_memory(bucket: &str, previous_current_size: Option<u64>, new_size: u64) {
|
pub async fn record_bucket_object_write_memory(bucket: &str, previous_current_size: Option<u64>, new_size: u64) {
|
||||||
record_bucket_object_write_memory_inner(bucket, previous_current_size, new_size, false).await;
|
record_bucket_object_write_memory_inner(bucket, previous_current_size, new_size, false).await;
|
||||||
@@ -1729,11 +1772,6 @@ pub async fn record_bucket_object_write_unknown_previous_memory(bucket: &str, ne
|
|||||||
entry.pending_scanner_position = None;
|
entry.pending_scanner_position = None;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Fast in-memory increment for immediate quota consistency.
|
|
||||||
pub async fn increment_bucket_usage_memory(bucket: &str, size_increment: u64) {
|
|
||||||
record_bucket_object_write_memory(bucket, None, size_increment).await;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Fast in-memory update for successful object deletes.
|
/// Fast in-memory update for successful object deletes.
|
||||||
pub async fn record_bucket_object_delete_memory(bucket: &str, deleted_size: u64, removed_current_object: bool) {
|
pub async fn record_bucket_object_delete_memory(bucket: &str, deleted_size: u64, removed_current_object: bool) {
|
||||||
ensure_bucket_usage_cached(bucket).await;
|
ensure_bucket_usage_cached(bucket).await;
|
||||||
@@ -1776,11 +1814,6 @@ pub async fn record_bucket_delete_marker_memory(bucket: &str) {
|
|||||||
entry.pending_scanner_position = None;
|
entry.pending_scanner_position = None;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Fast in-memory decrement for immediate quota consistency
|
|
||||||
pub async fn decrement_bucket_usage_memory(bucket: &str, size_decrement: u64) {
|
|
||||||
record_bucket_object_delete_memory(bucket, size_decrement, size_decrement > 0).await;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Get bucket usage from the authoritative cache for this topology.
|
/// Get bucket usage from the authoritative cache for this topology.
|
||||||
async fn get_persisted_bucket_usage(bucket: &str) -> Option<u64> {
|
async fn get_persisted_bucket_usage(bucket: &str) -> Option<u64> {
|
||||||
let store = runtime_sources::object_store_handle()?;
|
let store = runtime_sources::object_store_handle()?;
|
||||||
@@ -1975,91 +2008,6 @@ pub async fn apply_bucket_usage_memory_overlay(data_usage_info: &mut DataUsageIn
|
|||||||
apply_bucket_usage_memory_overlay_if_authoritative(data_usage_info, authoritative).await;
|
apply_bucket_usage_memory_overlay_if_authoritative(data_usage_info, authoritative).await;
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Sync memory cache with backend data (called by scanner)
|
|
||||||
pub async fn sync_memory_cache_with_backend() -> Result<(), Error> {
|
|
||||||
if let Some(store) = runtime_sources::object_store_handle() {
|
|
||||||
match load_data_usage_from_backend(store.clone()).await {
|
|
||||||
Ok(data_usage_info) => {
|
|
||||||
replace_bucket_usage_memory_from_info(&data_usage_info).await;
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
debug!("Failed to sync memory cache with backend: {}", e);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Create a data usage cache entry from size summary
|
|
||||||
pub fn create_cache_entry_from_summary(summary: &SizeSummary) -> DataUsageEntry {
|
|
||||||
let mut entry = DataUsageEntry::default();
|
|
||||||
entry.add_sizes(summary);
|
|
||||||
entry
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Convert data usage cache to DataUsageInfo
|
|
||||||
pub fn cache_to_data_usage_info(
|
|
||||||
cache: &DataUsageCache,
|
|
||||||
path: &str,
|
|
||||||
buckets: &[crate::storage_api_contracts::bucket::BucketInfo],
|
|
||||||
) -> DataUsageInfo {
|
|
||||||
let e = match cache.find(path) {
|
|
||||||
Some(e) => e,
|
|
||||||
None => return DataUsageInfo::default(),
|
|
||||||
};
|
|
||||||
let flat = cache.flatten(&e);
|
|
||||||
|
|
||||||
let mut buckets_usage = HashMap::new();
|
|
||||||
for bucket in buckets.iter() {
|
|
||||||
let e = match cache.find(&bucket.name) {
|
|
||||||
Some(e) => e,
|
|
||||||
None => continue,
|
|
||||||
};
|
|
||||||
let flat = cache.flatten(&e);
|
|
||||||
let mut bui = BucketUsageInfo {
|
|
||||||
size: flat.size as u64,
|
|
||||||
versions_count: flat.versions as u64,
|
|
||||||
objects_count: flat.objects as u64,
|
|
||||||
delete_markers_count: flat.delete_markers as u64,
|
|
||||||
object_size_histogram: flat.obj_sizes.to_map(),
|
|
||||||
object_versions_histogram: flat.obj_versions.to_map(),
|
|
||||||
..Default::default()
|
|
||||||
};
|
|
||||||
|
|
||||||
if let Some(rs) = &flat.replication_stats {
|
|
||||||
bui.replica_size = rs.replica_size;
|
|
||||||
bui.replica_count = rs.replica_count;
|
|
||||||
|
|
||||||
for (arn, stat) in rs.targets.iter() {
|
|
||||||
bui.replication_info.insert(
|
|
||||||
arn.clone(),
|
|
||||||
BucketTargetUsageInfo {
|
|
||||||
replication_pending_size: stat.pending_size,
|
|
||||||
replicated_size: stat.replicated_size,
|
|
||||||
replication_failed_size: stat.failed_size,
|
|
||||||
replication_pending_count: stat.pending_count,
|
|
||||||
replication_failed_count: stat.failed_count,
|
|
||||||
replicated_count: stat.replicated_count,
|
|
||||||
..Default::default()
|
|
||||||
},
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
buckets_usage.insert(bucket.name.clone(), bui);
|
|
||||||
}
|
|
||||||
|
|
||||||
DataUsageInfo {
|
|
||||||
last_update: cache.info.last_update,
|
|
||||||
objects_total_count: flat.objects as u64,
|
|
||||||
versions_total_count: flat.versions as u64,
|
|
||||||
delete_markers_total_count: flat.delete_markers as u64,
|
|
||||||
objects_total_size: flat.size as u64,
|
|
||||||
buckets_count: e.children.len() as u64,
|
|
||||||
buckets_usage,
|
|
||||||
..Default::default()
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Helper functions for DataUsageCache operations
|
// Helper functions for DataUsageCache operations
|
||||||
pub async fn load_data_usage_cache(store: &crate::set_disk::SetDisks, name: &str) -> crate::error::Result<DataUsageCache> {
|
pub async fn load_data_usage_cache(store: &crate::set_disk::SetDisks, name: &str) -> crate::error::Result<DataUsageCache> {
|
||||||
use crate::disk::{BUCKET_META_PREFIX, RUSTFS_META_BUCKET};
|
use crate::disk::{BUCKET_META_PREFIX, RUSTFS_META_BUCKET};
|
||||||
@@ -2137,30 +2085,6 @@ pub async fn load_data_usage_cache(store: &crate::set_disk::SetDisks, name: &str
|
|||||||
Ok(d)
|
Ok(d)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[instrument(skip(cache))]
|
|
||||||
pub async fn save_data_usage_cache(cache: &DataUsageCache, name: &str) -> crate::error::Result<()> {
|
|
||||||
use crate::config::com::save_config;
|
|
||||||
use crate::disk::BUCKET_META_PREFIX;
|
|
||||||
use std::path::Path;
|
|
||||||
|
|
||||||
let Some(store) = runtime_sources::object_store_handle() else {
|
|
||||||
return Err(Error::other("errServerNotInitialized"));
|
|
||||||
};
|
|
||||||
let buf = cache.marshal_msg().map_err(Error::other)?;
|
|
||||||
let buf_clone = buf.clone();
|
|
||||||
|
|
||||||
let store_clone = store.clone();
|
|
||||||
|
|
||||||
let name = Path::new(BUCKET_META_PREFIX).join(name).to_string_lossy().to_string();
|
|
||||||
|
|
||||||
let name_clone = name.clone();
|
|
||||||
tokio::spawn(async move {
|
|
||||||
let _ = save_config(store_clone, &format!("{}{}", name_clone, ".bkp"), buf_clone).await;
|
|
||||||
});
|
|
||||||
save_config(store, &name, buf).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Persist the current in-memory compression total to the backend.
|
/// Persist the current in-memory compression total to the backend.
|
||||||
/// Resets the debounce counter so the next auto-persist won't fire
|
/// Resets the debounce counter so the next auto-persist won't fire
|
||||||
/// immediately after this manual flush (intended for shutdown paths).
|
/// immediately after this manual flush (intended for shutdown paths).
|
||||||
@@ -3135,6 +3059,102 @@ mod tests {
|
|||||||
assert_eq!(usage.object_versions_histogram.get("BETWEEN_1000_AND_10000"), Some(&1));
|
assert_eq!(usage.object_versions_histogram.get("BETWEEN_1000_AND_10000"), Some(&1));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bucket_usage_uses_the_larger_of_logical_and_physical_size() {
|
||||||
|
let mut metadata = HashMap::new();
|
||||||
|
rustfs_utils::http::insert_str(
|
||||||
|
&mut metadata,
|
||||||
|
rustfs_utils::http::SUFFIX_COMPRESSION,
|
||||||
|
"klauspost/compress/s2".to_string(),
|
||||||
|
);
|
||||||
|
rustfs_utils::http::insert_str(&mut metadata, rustfs_utils::http::SUFFIX_ACTUAL_SIZE, "4096".to_string());
|
||||||
|
let object = ObjectInfo {
|
||||||
|
name: "compressed".to_string(),
|
||||||
|
size: 128,
|
||||||
|
user_defined: Arc::new(metadata),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let mut usage = BucketUsageAccumulator::default();
|
||||||
|
usage
|
||||||
|
.record("bucket", &object)
|
||||||
|
.expect("valid compressed metadata should be counted");
|
||||||
|
assert_eq!(usage.finish().size, 4096);
|
||||||
|
|
||||||
|
let mut framed_metadata = HashMap::new();
|
||||||
|
rustfs_utils::http::insert_str(
|
||||||
|
&mut framed_metadata,
|
||||||
|
rustfs_utils::http::SUFFIX_COMPRESSION,
|
||||||
|
"klauspost/compress/s2".to_string(),
|
||||||
|
);
|
||||||
|
rustfs_utils::http::insert_str(&mut framed_metadata, rustfs_utils::http::SUFFIX_ACTUAL_SIZE, "1".to_string());
|
||||||
|
let framed = ObjectInfo {
|
||||||
|
name: "framed".to_string(),
|
||||||
|
size: 17,
|
||||||
|
user_defined: Arc::new(framed_metadata),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
assert_eq!(quota_object_size(&framed).expect("physical framing must remain quota-accounted"), 17);
|
||||||
|
|
||||||
|
let legacy_compressed_part = ObjectInfo {
|
||||||
|
name: "legacy-compressed-part".to_string(),
|
||||||
|
size: 1,
|
||||||
|
user_defined: Arc::new((*framed.user_defined).clone()),
|
||||||
|
parts: Arc::new(vec![rustfs_filemeta::ObjectPartInfo {
|
||||||
|
size: 1,
|
||||||
|
actual_size: -1,
|
||||||
|
..Default::default()
|
||||||
|
}]),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
quota_object_size(&legacy_compressed_part).expect("unknown compressed part size is a valid sentinel"),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
|
||||||
|
let uncompressed_negative_part = ObjectInfo {
|
||||||
|
name: "uncompressed-negative-part".to_string(),
|
||||||
|
size: 1,
|
||||||
|
parts: Arc::new(vec![rustfs_filemeta::ObjectPartInfo {
|
||||||
|
size: 1,
|
||||||
|
actual_size: -1,
|
||||||
|
..Default::default()
|
||||||
|
}]),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
assert!(matches!(quota_object_size(&uncompressed_negative_part), Err(Error::PartMissingOrCorrupt)));
|
||||||
|
|
||||||
|
let mut corrupt_metadata = (*object.user_defined).clone();
|
||||||
|
rustfs_utils::http::insert_str(&mut corrupt_metadata, rustfs_utils::http::SUFFIX_ACTUAL_SIZE, "-1".to_string());
|
||||||
|
let corrupt = ObjectInfo {
|
||||||
|
user_defined: Arc::new(corrupt_metadata),
|
||||||
|
..object
|
||||||
|
};
|
||||||
|
assert!(matches!(quota_object_size(&corrupt), Err(Error::PartMissingOrCorrupt)));
|
||||||
|
|
||||||
|
let mut poisoned_metadata = HashMap::new();
|
||||||
|
rustfs_utils::http::insert_str(
|
||||||
|
&mut poisoned_metadata,
|
||||||
|
rustfs_utils::http::SUFFIX_COMPRESSION,
|
||||||
|
"klauspost/compress/s2".to_string(),
|
||||||
|
);
|
||||||
|
rustfs_utils::http::insert_str(&mut poisoned_metadata, rustfs_utils::http::SUFFIX_ACTUAL_SIZE, "1".to_string());
|
||||||
|
let poisoned = ObjectInfo {
|
||||||
|
name: "legacy-swift-metadata".to_string(),
|
||||||
|
size: 4096,
|
||||||
|
user_defined: Arc::new(poisoned_metadata),
|
||||||
|
parts: Arc::new(vec![rustfs_filemeta::ObjectPartInfo {
|
||||||
|
size: 4096,
|
||||||
|
actual_size: 4096,
|
||||||
|
..Default::default()
|
||||||
|
}]),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
quota_object_size(&poisoned).expect("persisted part accounting must bound legacy user metadata"),
|
||||||
|
4096
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial]
|
#[serial]
|
||||||
async fn live_bucket_usage_refreshes_are_coalesced_only_while_in_flight() {
|
async fn live_bucket_usage_refreshes_are_coalesced_only_while_in_flight() {
|
||||||
|
|||||||
@@ -12,7 +12,9 @@
|
|||||||
// See the License for the specific language governing permissions and
|
// See the License for the specific language governing permissions and
|
||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
use crate::cluster::rpc::{TonicInterceptor, gen_tonic_signature_interceptor, node_service_time_out_client};
|
use crate::cluster::rpc::{
|
||||||
|
ScannerBucketListing, TonicInterceptor, gen_tonic_signature_interceptor, node_service_time_out_client,
|
||||||
|
};
|
||||||
use crate::data_usage::{DATA_USAGE_CACHE_NAME, DATA_USAGE_ROOT, load_data_usage_from_backend_cached};
|
use crate::data_usage::{DATA_USAGE_CACHE_NAME, DATA_USAGE_ROOT, load_data_usage_from_backend_cached};
|
||||||
use crate::error::{Error, Result};
|
use crate::error::{Error, Result};
|
||||||
use crate::{
|
use crate::{
|
||||||
@@ -23,6 +25,7 @@ use crate::{
|
|||||||
|
|
||||||
use crate::data_usage::load_data_usage_cache;
|
use crate::data_usage::load_data_usage_cache;
|
||||||
use crate::storage_api_contracts::admin::StorageAdminApi;
|
use crate::storage_api_contracts::admin::StorageAdminApi;
|
||||||
|
use crate::storage_api_contracts::bucket::BucketOptions;
|
||||||
use rustfs_common::heal_channel::DriveState;
|
use rustfs_common::heal_channel::DriveState;
|
||||||
use rustfs_madmin::{
|
use rustfs_madmin::{
|
||||||
BackendDisks, Disk, ErasureSetInfo, ITEM_INITIALIZING, ITEM_OFFLINE, ITEM_ONLINE, ITEM_UNKNOWN, InfoMessage, MemStats,
|
BackendDisks, Disk, ErasureSetInfo, ITEM_INITIALIZING, ITEM_OFFLINE, ITEM_ONLINE, ITEM_UNKNOWN, InfoMessage, MemStats,
|
||||||
@@ -74,6 +77,19 @@ fn apply_data_usage_result(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn apply_bucket_namespace_count(result: Result<ScannerBucketListing>, buckets: &mut rustfs_madmin::Buckets) {
|
||||||
|
if let Ok(listing) = result
|
||||||
|
&& listing.topology_complete
|
||||||
|
{
|
||||||
|
let count = listing.buckets.iter().filter(|bucket| !bucket.name.starts_with('.')).count();
|
||||||
|
let Ok(count) = u64::try_from(count) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
buckets.count = count;
|
||||||
|
buckets.error = None;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// pub const ITEM_OFFLINE: &str = "offline";
|
// pub const ITEM_OFFLINE: &str = "offline";
|
||||||
// pub const ITEM_INITIALIZING: &str = "initializing";
|
// pub const ITEM_INITIALIZING: &str = "initializing";
|
||||||
// pub const ITEM_ONLINE: &str = "online";
|
// pub const ITEM_ONLINE: &str = "online";
|
||||||
@@ -285,6 +301,18 @@ pub async fn get_server_info(get_pools: bool) -> InfoMessage {
|
|||||||
&mut delete_markers,
|
&mut delete_markers,
|
||||||
&mut usage,
|
&mut usage,
|
||||||
);
|
);
|
||||||
|
if buckets.error.is_some() {
|
||||||
|
apply_bucket_namespace_count(
|
||||||
|
store
|
||||||
|
.list_bucket_for_scanner(&BucketOptions {
|
||||||
|
cached: true,
|
||||||
|
no_metadata: true,
|
||||||
|
..Default::default()
|
||||||
|
})
|
||||||
|
.await,
|
||||||
|
&mut buckets,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
let after3 = OffsetDateTime::now_utc();
|
let after3 = OffsetDateTime::now_utc();
|
||||||
|
|
||||||
@@ -625,6 +653,7 @@ fn reconcile_servers_with_endpoint_topology(
|
|||||||
(added, report)
|
(added, report)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[allow(dead_code, reason = "exercised by this file's topology tests (backlog#1823)")]
|
||||||
fn server_topology_completeness_report(
|
fn server_topology_completeness_report(
|
||||||
servers: &[ServerProperties],
|
servers: &[ServerProperties],
|
||||||
endpoints: &EndpointServerPools,
|
endpoints: &EndpointServerPools,
|
||||||
@@ -705,12 +734,13 @@ mod tests {
|
|||||||
endpoints::{EndpointServerPools, Endpoints, PoolEndpoints},
|
endpoints::{EndpointServerPools, Endpoints, PoolEndpoints},
|
||||||
};
|
};
|
||||||
use crate::runtime::sources as runtime_sources;
|
use crate::runtime::sources as runtime_sources;
|
||||||
|
use crate::storage_api_contracts::bucket::BucketInfo;
|
||||||
use rustfs_madmin::{Disk, ITEM_OFFLINE, ITEM_ONLINE, ITEM_UNKNOWN, ServerProperties};
|
use rustfs_madmin::{Disk, ITEM_OFFLINE, ITEM_ONLINE, ITEM_UNKNOWN, ServerProperties};
|
||||||
|
|
||||||
use super::{
|
use super::{
|
||||||
DATA_USAGE_ROOT, DATA_USAGE_UNAVAILABLE_ERROR, apply_data_usage_result, apply_erasure_set_usage,
|
DATA_USAGE_ROOT, DATA_USAGE_UNAVAILABLE_ERROR, apply_bucket_namespace_count, apply_data_usage_result,
|
||||||
get_local_server_property, get_online_offline_disks_stats, get_server_info, reconcile_servers_with_endpoint_topology,
|
apply_erasure_set_usage, get_local_server_property, get_online_offline_disks_stats, get_server_info,
|
||||||
server_topology_completeness_report,
|
reconcile_servers_with_endpoint_topology, server_topology_completeness_report,
|
||||||
};
|
};
|
||||||
|
|
||||||
fn disk_with_state(endpoint: &str, state: &str) -> Disk {
|
fn disk_with_state(endpoint: &str, state: &str) -> Disk {
|
||||||
@@ -960,6 +990,75 @@ mod tests {
|
|||||||
assert_eq!(usage.error.as_deref(), Some(DATA_USAGE_UNAVAILABLE_ERROR));
|
assert_eq!(usage.error.as_deref(), Some(DATA_USAGE_UNAVAILABLE_ERROR));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn live_bucket_namespace_count_survives_unavailable_data_usage() {
|
||||||
|
let mut buckets = rustfs_madmin::Buckets {
|
||||||
|
count: 0,
|
||||||
|
error: Some(DATA_USAGE_UNAVAILABLE_ERROR.to_string()),
|
||||||
|
};
|
||||||
|
|
||||||
|
apply_bucket_namespace_count(
|
||||||
|
Ok(crate::cluster::rpc::ScannerBucketListing {
|
||||||
|
buckets: vec![
|
||||||
|
BucketInfo {
|
||||||
|
name: "bucket-a".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
BucketInfo {
|
||||||
|
name: ".rustfs.sys".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
BucketInfo {
|
||||||
|
name: "bucket-b".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
],
|
||||||
|
set_buckets: Vec::new(),
|
||||||
|
topology_complete: true,
|
||||||
|
}),
|
||||||
|
&mut buckets,
|
||||||
|
);
|
||||||
|
|
||||||
|
assert_eq!(buckets.count, 2);
|
||||||
|
assert_eq!(buckets.error, None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn incomplete_bucket_namespace_lookup_preserves_usage_state() {
|
||||||
|
let mut buckets = rustfs_madmin::Buckets {
|
||||||
|
count: 7,
|
||||||
|
error: Some(DATA_USAGE_UNAVAILABLE_ERROR.to_string()),
|
||||||
|
};
|
||||||
|
|
||||||
|
apply_bucket_namespace_count(
|
||||||
|
Ok(crate::cluster::rpc::ScannerBucketListing {
|
||||||
|
buckets: vec![BucketInfo {
|
||||||
|
name: "bucket-a".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
}],
|
||||||
|
set_buckets: Vec::new(),
|
||||||
|
topology_complete: false,
|
||||||
|
}),
|
||||||
|
&mut buckets,
|
||||||
|
);
|
||||||
|
|
||||||
|
assert_eq!(buckets.count, 7);
|
||||||
|
assert_eq!(buckets.error.as_deref(), Some(DATA_USAGE_UNAVAILABLE_ERROR));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn failed_bucket_namespace_lookup_preserves_usage_state() {
|
||||||
|
let mut buckets = rustfs_madmin::Buckets {
|
||||||
|
count: 7,
|
||||||
|
error: Some(DATA_USAGE_UNAVAILABLE_ERROR.to_string()),
|
||||||
|
};
|
||||||
|
|
||||||
|
apply_bucket_namespace_count(Err(crate::error::Error::DiskNotFound), &mut buckets);
|
||||||
|
|
||||||
|
assert_eq!(buckets.count, 7);
|
||||||
|
assert_eq!(buckets.error.as_deref(), Some(DATA_USAGE_UNAVAILABLE_ERROR));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn incomplete_erasure_set_cache_is_not_reported_as_zero() {
|
fn incomplete_erasure_set_cache_is_not_reported_as_zero() {
|
||||||
let mut cache = rustfs_data_usage::DataUsageCache::default();
|
let mut cache = rustfs_data_usage::DataUsageCache::default();
|
||||||
|
|||||||
@@ -46,21 +46,49 @@ pub(crate) const GET_CODEC_STREAMING_OBJECT_CLASS_MULTIPART: &str = "multipart";
|
|||||||
pub(crate) const GET_STAGE_DECODE: &str = "decode";
|
pub(crate) const GET_STAGE_DECODE: &str = "decode";
|
||||||
pub(crate) const GET_STAGE_EMIT: &str = "emit";
|
pub(crate) const GET_STAGE_EMIT: &str = "emit";
|
||||||
pub(crate) const GET_STAGE_FILL: &str = "fill";
|
pub(crate) const GET_STAGE_FILL: &str = "fill";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_FIRST_BYTE: &str = "first_byte";
|
pub(crate) const GET_STAGE_FIRST_BYTE: &str = "first_byte";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_FIRST_METADATA_RESPONSE: &str = "first_metadata_response";
|
pub(crate) const GET_STAGE_FIRST_METADATA_RESPONSE: &str = "first_metadata_response";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_FIRST_VALID_METADATA_RESPONSE: &str = "first_valid_metadata_response";
|
pub(crate) const GET_STAGE_FIRST_VALID_METADATA_RESPONSE: &str = "first_valid_metadata_response";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_FIRST_SHARD_READ: &str = "first_shard_read";
|
pub(crate) const GET_STAGE_FIRST_SHARD_READ: &str = "first_shard_read";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_FULL_BODY: &str = "full_body";
|
pub(crate) const GET_STAGE_FULL_BODY: &str = "full_body";
|
||||||
pub(crate) const GET_STAGE_INLINE_PREPARE: &str = "inline_prepare";
|
pub(crate) const GET_STAGE_INLINE_PREPARE: &str = "inline_prepare";
|
||||||
pub(crate) const GET_STAGE_LOCK_ACQUIRE: &str = "lock_acquire";
|
pub(crate) const GET_STAGE_LOCK_ACQUIRE: &str = "lock_acquire";
|
||||||
pub(crate) const GET_STAGE_METADATA: &str = "metadata";
|
pub(crate) const GET_STAGE_METADATA: &str = "metadata";
|
||||||
pub(crate) const GET_STAGE_METADATA_CACHE_LOOKUP: &str = "metadata_cache_lookup";
|
pub(crate) const GET_STAGE_METADATA_CACHE_LOOKUP: &str = "metadata_cache_lookup";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_METADATA_FANOUT: &str = "metadata_fanout";
|
pub(crate) const GET_STAGE_METADATA_FANOUT: &str = "metadata_fanout";
|
||||||
pub(crate) const GET_STAGE_METADATA_RESOLVE: &str = "metadata_resolve";
|
pub(crate) const GET_STAGE_METADATA_RESOLVE: &str = "metadata_resolve";
|
||||||
pub(crate) const GET_STAGE_OBJECT_INFO: &str = "object_info";
|
pub(crate) const GET_STAGE_OBJECT_INFO: &str = "object_info";
|
||||||
pub(crate) const GET_STAGE_OUTPUT_LOCK_WAIT: &str = "output_lock_wait";
|
pub(crate) const GET_STAGE_OUTPUT_LOCK_WAIT: &str = "output_lock_wait";
|
||||||
pub(crate) const GET_STAGE_OUTPUT_POLL: &str = "output_poll";
|
pub(crate) const GET_STAGE_OUTPUT_POLL: &str = "output_poll";
|
||||||
pub(crate) const GET_STAGE_PATH_DECISION: &str = "path_decision";
|
pub(crate) const GET_STAGE_PATH_DECISION: &str = "path_decision";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_QUORUM_REACHED: &str = "quorum_reached";
|
pub(crate) const GET_STAGE_QUORUM_REACHED: &str = "quorum_reached";
|
||||||
pub(crate) const GET_STAGE_RANGE: &str = "range";
|
pub(crate) const GET_STAGE_RANGE: &str = "range";
|
||||||
pub(crate) const GET_STAGE_READER_SETUP: &str = "reader_setup";
|
pub(crate) const GET_STAGE_READER_SETUP: &str = "reader_setup";
|
||||||
@@ -84,12 +112,28 @@ pub(crate) const GET_STAGE_READER_STREAM_FIRST_READ: &str = "reader_stream_first
|
|||||||
pub(crate) const GET_STAGE_READER_TASK_BITROT_READER_INIT: &str = "reader_task_bitrot_reader_init";
|
pub(crate) const GET_STAGE_READER_TASK_BITROT_READER_INIT: &str = "reader_task_bitrot_reader_init";
|
||||||
pub(crate) const GET_STAGE_READER_TASK_FILE_OPEN: &str = "reader_task_file_open";
|
pub(crate) const GET_STAGE_READER_TASK_FILE_OPEN: &str = "reader_task_file_open";
|
||||||
pub(crate) const GET_STAGE_READER_TASK_READER_CONSTRUCTION: &str = "reader_task_reader_construction";
|
pub(crate) const GET_STAGE_READER_TASK_READER_CONSTRUCTION: &str = "reader_task_reader_construction";
|
||||||
|
pub(crate) const GET_STAGE_READ_VERSION_DECODE: &str = "read_version_decode";
|
||||||
|
pub(crate) const GET_STAGE_READ_VERSION_PATH_CHECK: &str = "read_version_path_check";
|
||||||
|
pub(crate) const GET_STAGE_READ_VERSION_PATH_RESOLVE: &str = "read_version_path_resolve";
|
||||||
|
pub(crate) const GET_STAGE_READ_VERSION_XLMETA_READ: &str = "read_version_xlmeta_read";
|
||||||
pub(crate) const GET_STAGE_RECONSTRUCT: &str = "reconstruct";
|
pub(crate) const GET_STAGE_RECONSTRUCT: &str = "reconstruct";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_RESPONSE_HANDOFF: &str = "response_handoff";
|
pub(crate) const GET_STAGE_RESPONSE_HANDOFF: &str = "response_handoff";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_SLOWEST_METADATA_RESPONSE: &str = "slowest_metadata_response";
|
pub(crate) const GET_STAGE_SLOWEST_METADATA_RESPONSE: &str = "slowest_metadata_response";
|
||||||
pub(crate) const GET_STAGE_STRIPE_READ: &str = "stripe_read";
|
pub(crate) const GET_STAGE_STRIPE_READ: &str = "stripe_read";
|
||||||
pub(crate) const GET_STAGE_STRIPE_READ_FIRST_SHARD: &str = "stripe_read_first_shard";
|
pub(crate) const GET_STAGE_STRIPE_READ_FIRST_SHARD: &str = "stripe_read_first_shard";
|
||||||
pub(crate) const GET_STAGE_STRIPE_READ_QUORUM: &str = "stripe_read_quorum";
|
pub(crate) const GET_STAGE_STRIPE_READ_QUORUM: &str = "stripe_read_quorum";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const GET_STAGE_BITROT_VERIFY: &str = "bitrot_verify";
|
pub(crate) const GET_STAGE_BITROT_VERIFY: &str = "bitrot_verify";
|
||||||
|
|
||||||
pub(crate) const GET_READER_BUFFER_OUTPUT: &str = "output";
|
pub(crate) const GET_READER_BUFFER_OUTPUT: &str = "output";
|
||||||
@@ -137,6 +181,7 @@ pub(crate) const GET_METADATA_CACHE_REASON_NO_LOCK: &str = "no_lock";
|
|||||||
pub(crate) const GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED: &str = "not_found_or_expired";
|
pub(crate) const GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED: &str = "not_found_or_expired";
|
||||||
pub(crate) const GET_METADATA_CACHE_REASON_NOT_READ_DATA: &str = "not_read_data";
|
pub(crate) const GET_METADATA_CACHE_REASON_NOT_READ_DATA: &str = "not_read_data";
|
||||||
pub(crate) const GET_METADATA_CACHE_REASON_PART_NUMBER: &str = "part_number";
|
pub(crate) const GET_METADATA_CACHE_REASON_PART_NUMBER: &str = "part_number";
|
||||||
|
pub(crate) const GET_METADATA_CACHE_REASON_PART_CHECKSUMS: &str = "part_checksums";
|
||||||
pub(crate) const GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ: &str = "raw_data_movement_read";
|
pub(crate) const GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ: &str = "raw_data_movement_read";
|
||||||
pub(crate) const GET_METADATA_CACHE_REASON_STALE_PUBLICATION: &str = "stale_publication";
|
pub(crate) const GET_METADATA_CACHE_REASON_STALE_PUBLICATION: &str = "stale_publication";
|
||||||
pub(crate) const GET_METADATA_CACHE_REASON_USABLE: &str = "usable";
|
pub(crate) const GET_METADATA_CACHE_REASON_USABLE: &str = "usable";
|
||||||
@@ -145,6 +190,17 @@ pub(crate) const GET_METADATA_CACHE_REASON_VERSION_SUSPENDED: &str = "version_su
|
|||||||
pub(crate) const GET_METADATA_CACHE_REASON_VERSIONED: &str = "versioned";
|
pub(crate) const GET_METADATA_CACHE_REASON_VERSIONED: &str = "versioned";
|
||||||
pub(crate) const GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA: &str = "conflicting_metadata";
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA: &str = "conflicting_metadata";
|
||||||
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER: &str = "delete_marker";
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER: &str = "delete_marker";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_BODY_VERIFY: &str = "data_read_inline_body_verify";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_DELETED: &str = "data_read_inline_deleted";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_GEOMETRY: &str = "data_read_inline_geometry";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_IDENTITY_MISMATCH: &str = "data_read_inline_identity_mismatch";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_MISSING_PAYLOAD: &str = "data_read_inline_missing_payload";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_MISSING_SHARD: &str = "data_read_inline_missing_shard";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_NOT_INLINE: &str = "data_read_inline_not_inline";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_PART_SHAPE: &str = "data_read_inline_part_shape";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_REMOTE: &str = "data_read_inline_remote";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_SIZE: &str = "data_read_inline_size";
|
||||||
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_TRANSFORMED: &str = "data_read_inline_transformed";
|
||||||
pub(crate) const GET_METADATA_EARLY_STOP_REASON_ERROR: &str = "error";
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_ERROR: &str = "error";
|
||||||
pub(crate) const GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM: &str = "insufficient_quorum";
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM: &str = "insufficient_quorum";
|
||||||
pub(crate) const GET_METADATA_EARLY_STOP_REASON_NOT_FOUND: &str = "not_found";
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_NOT_FOUND: &str = "not_found";
|
||||||
@@ -154,8 +210,20 @@ pub(crate) const GET_METADATA_EARLY_STOP_REASON_VERSION_NOT_FOUND: &str = "versi
|
|||||||
pub(crate) const GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM: &str = "version_match_quorum";
|
pub(crate) const GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM: &str = "version_match_quorum";
|
||||||
|
|
||||||
/// Early-stop active state labels
|
/// Early-stop active state labels
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const EARLY_STOP_ACTIVE_HIT: &str = "hit";
|
pub(crate) const EARLY_STOP_ACTIVE_HIT: &str = "hit";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const EARLY_STOP_ACTIVE_MISS: &str = "miss";
|
pub(crate) const EARLY_STOP_ACTIVE_MISS: &str = "miss";
|
||||||
|
#[allow(
|
||||||
|
dead_code,
|
||||||
|
reason = "GET stage vocabulary; value pinned by this file's tests, no writer yet (backlog#1823)"
|
||||||
|
)]
|
||||||
pub(crate) const EARLY_STOP_ACTIVE_DISABLED: &str = "disabled";
|
pub(crate) const EARLY_STOP_ACTIVE_DISABLED: &str = "disabled";
|
||||||
|
|
||||||
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||||
@@ -441,6 +509,10 @@ mod tests {
|
|||||||
assert_eq!(GET_STAGE_QUORUM_REACHED, "quorum_reached");
|
assert_eq!(GET_STAGE_QUORUM_REACHED, "quorum_reached");
|
||||||
assert_eq!(GET_STAGE_RANGE, "range");
|
assert_eq!(GET_STAGE_RANGE, "range");
|
||||||
assert_eq!(GET_STAGE_READER_SETUP, "reader_setup");
|
assert_eq!(GET_STAGE_READER_SETUP, "reader_setup");
|
||||||
|
assert_eq!(GET_STAGE_READ_VERSION_DECODE, "read_version_decode");
|
||||||
|
assert_eq!(GET_STAGE_READ_VERSION_PATH_CHECK, "read_version_path_check");
|
||||||
|
assert_eq!(GET_STAGE_READ_VERSION_PATH_RESOLVE, "read_version_path_resolve");
|
||||||
|
assert_eq!(GET_STAGE_READ_VERSION_XLMETA_READ, "read_version_xlmeta_read");
|
||||||
assert_eq!(GET_STAGE_RECONSTRUCT, "reconstruct");
|
assert_eq!(GET_STAGE_RECONSTRUCT, "reconstruct");
|
||||||
assert_eq!(GET_STAGE_RESPONSE_HANDOFF, "response_handoff");
|
assert_eq!(GET_STAGE_RESPONSE_HANDOFF, "response_handoff");
|
||||||
assert_eq!(GET_STAGE_SLOWEST_METADATA_RESPONSE, "slowest_metadata_response");
|
assert_eq!(GET_STAGE_SLOWEST_METADATA_RESPONSE, "slowest_metadata_response");
|
||||||
@@ -480,6 +552,7 @@ mod tests {
|
|||||||
assert_eq!(GET_METADATA_CACHE_REASON_NO_LOCK, "no_lock");
|
assert_eq!(GET_METADATA_CACHE_REASON_NO_LOCK, "no_lock");
|
||||||
assert_eq!(GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED, "not_found_or_expired");
|
assert_eq!(GET_METADATA_CACHE_REASON_NOT_FOUND_OR_EXPIRED, "not_found_or_expired");
|
||||||
assert_eq!(GET_METADATA_CACHE_REASON_NOT_READ_DATA, "not_read_data");
|
assert_eq!(GET_METADATA_CACHE_REASON_NOT_READ_DATA, "not_read_data");
|
||||||
|
assert_eq!(GET_METADATA_CACHE_REASON_PART_CHECKSUMS, "part_checksums");
|
||||||
assert_eq!(GET_METADATA_CACHE_REASON_PART_NUMBER, "part_number");
|
assert_eq!(GET_METADATA_CACHE_REASON_PART_NUMBER, "part_number");
|
||||||
assert_eq!(GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ, "raw_data_movement_read");
|
assert_eq!(GET_METADATA_CACHE_REASON_RAW_DATA_MOVEMENT_READ, "raw_data_movement_read");
|
||||||
assert_eq!(GET_METADATA_CACHE_REASON_STALE_PUBLICATION, "stale_publication");
|
assert_eq!(GET_METADATA_CACHE_REASON_STALE_PUBLICATION, "stale_publication");
|
||||||
@@ -489,6 +562,32 @@ mod tests {
|
|||||||
assert_eq!(GET_METADATA_CACHE_REASON_VERSIONED, "versioned");
|
assert_eq!(GET_METADATA_CACHE_REASON_VERSIONED, "versioned");
|
||||||
assert_eq!(GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA, "conflicting_metadata");
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA, "conflicting_metadata");
|
||||||
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER, "delete_marker");
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER, "delete_marker");
|
||||||
|
assert_eq!(
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_BODY_VERIFY,
|
||||||
|
"data_read_inline_body_verify"
|
||||||
|
);
|
||||||
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_DELETED, "data_read_inline_deleted");
|
||||||
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_GEOMETRY, "data_read_inline_geometry");
|
||||||
|
assert_eq!(
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_IDENTITY_MISMATCH,
|
||||||
|
"data_read_inline_identity_mismatch"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_MISSING_PAYLOAD,
|
||||||
|
"data_read_inline_missing_payload"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_MISSING_SHARD,
|
||||||
|
"data_read_inline_missing_shard"
|
||||||
|
);
|
||||||
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_NOT_INLINE, "data_read_inline_not_inline");
|
||||||
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_PART_SHAPE, "data_read_inline_part_shape");
|
||||||
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_REMOTE, "data_read_inline_remote");
|
||||||
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_SIZE, "data_read_inline_size");
|
||||||
|
assert_eq!(
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_DATA_READ_INLINE_TRANSFORMED,
|
||||||
|
"data_read_inline_transformed"
|
||||||
|
);
|
||||||
assert_eq!(GET_METADATA_EARLY_STOP_REASON_ERROR, "error");
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_ERROR, "error");
|
||||||
assert_eq!(GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM, "insufficient_quorum");
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM, "insufficient_quorum");
|
||||||
assert_eq!(GET_METADATA_EARLY_STOP_REASON_NOT_FOUND, "not_found");
|
assert_eq!(GET_METADATA_EARLY_STOP_REASON_NOT_FOUND, "not_found");
|
||||||
|
|||||||
@@ -13,8 +13,6 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
// #730: diagnostics constants are staged for request-path telemetry migration.
|
// #730: diagnostics constants are staged for request-path telemetry migration.
|
||||||
#![allow(dead_code)]
|
|
||||||
|
|
||||||
pub(crate) mod admin_server_info;
|
pub(crate) mod admin_server_info;
|
||||||
pub(crate) mod get;
|
pub(crate) mod get;
|
||||||
pub(crate) mod pool;
|
|
||||||
|
|||||||
@@ -1,30 +0,0 @@
|
|||||||
// Copyright 2024 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
//! BytesPool metric label constants.
|
|
||||||
//!
|
|
||||||
//! These constants are used when recording pool acquisition and return
|
|
||||||
//! metrics to avoid string allocations and ensure label consistency.
|
|
||||||
|
|
||||||
/// BytesPool tier labels
|
|
||||||
pub const POOL_TIER_SMALL: &str = "small";
|
|
||||||
pub const POOL_TIER_MEDIUM: &str = "medium";
|
|
||||||
pub const POOL_TIER_LARGE: &str = "large";
|
|
||||||
pub const POOL_TIER_XLARGE: &str = "xlarge";
|
|
||||||
|
|
||||||
/// BytesPool outcome labels
|
|
||||||
pub const POOL_OUTCOME_HIT: &str = "hit";
|
|
||||||
pub const POOL_OUTCOME_MISS: &str = "miss";
|
|
||||||
pub const POOL_OUTCOME_RECYCLED: &str = "recycled";
|
|
||||||
pub const POOL_OUTCOME_DROPPED: &str = "dropped";
|
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user