mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-06 03:59:14 +00:00
Compare commits
45 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a8cecf6462 | |||
| 980f3abbd3 | |||
| e36650827b | |||
| 09c8e10d5e | |||
| f053862aad | |||
| 0ff03c596c | |||
| e8a7f4bc4a | |||
| a74919db8e | |||
| e77c6f0ca5 | |||
| 3149411943 | |||
| 7ba5cd6888 | |||
| acfeef55ab | |||
| 0d1b312673 | |||
| 42c32381b6 | |||
| e6bf2a4646 | |||
| cf9688898d | |||
| 2d159635ed | |||
| 2f02d1d2d8 | |||
| 8ae8fb7eea | |||
| bbd7b9ef17 | |||
| 15e9bc5ed0 | |||
| 882d9ca8a4 | |||
| 19a29a7027 | |||
| 2e4ab045b6 | |||
| cbfd5b92f4 | |||
| 971f9acdf4 | |||
| a6589c19e3 | |||
| 0885c721fe | |||
| eaf5159d0f | |||
| 2477e31059 | |||
| d8c3b1bb26 | |||
| a3b8183be9 | |||
| 4dbc58887a | |||
| 123967e729 | |||
| 4b0d597d4d | |||
| 13e6424e99 | |||
| b33693fc19 | |||
| 9ed1d46090 | |||
| 6eb60f8e72 | |||
| 8dd3cabd41 | |||
| e648f683bf | |||
| 193b1b7d3f | |||
| 3b920c7999 | |||
| 5f8b097172 | |||
| 445114577f |
@@ -1,2 +0,0 @@
|
|||||||
sha256-linux=4988bad7f5929152e0744f07393bc5d24aba2726e5daafa0eeca9a8c6a1f5683
|
|
||||||
sha256-darwin=4988bad7f5929152e0744f07393bc5d24aba2726e5daafa0eeca9a8c6a1f5683
|
|
||||||
@@ -1,2 +1,2 @@
|
|||||||
sha256-darwin=a881fd7d3f5cb94654221ca85b8b30cce1b95e608824a55a15339cbc294e6d34
|
sha256-darwin=85ee9b9e66e916dbf51857298637b6bcd1c23c2b066aaa4c171ce961c2d002b5
|
||||||
sha256-linux=e9a8d64e73f627c4d26c236dbbba690c9ee03a9e26d42a4244515b4439365535
|
sha256-linux=a2933d83dfe74ffa03410a0959333a1c48288b8469ca9f17273d449d7510c24b
|
||||||
|
|||||||
@@ -0,0 +1,72 @@
|
|||||||
|
{
|
||||||
|
"lane": "ci/test-and-lint",
|
||||||
|
"tests": [
|
||||||
|
{
|
||||||
|
"invariant": "write-quorum",
|
||||||
|
"suite": "rustfs-ecstore",
|
||||||
|
"name": "set_disk::ops::object::inline_put_commit_path_tests::inline_put_direct_commit_accepts_exact_quorum_and_rejects_quorum_minus_one"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"invariant": "metadata-rollback",
|
||||||
|
"suite": "rustfs-ecstore",
|
||||||
|
"name": "set_disk::core::io_primitives::tests::write_unique_file_info_reverts_metadata_when_write_quorum_fails"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"invariant": "stale-writer",
|
||||||
|
"suite": "rustfs-ecstore",
|
||||||
|
"name": "set_disk::ops::object::put_object_tmp_cleanup_tests::put_object_no_lock_aborts_after_outer_namespace_lock_loss"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"invariant": "range-body",
|
||||||
|
"suite": "rustfs-ecstore",
|
||||||
|
"name": "set_disk::ops::object::transition_upload_integrity_tests::transitioned_compressed_object_range_get_returns_plaintext_slice"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"invariant": "multipart-cancellation",
|
||||||
|
"suite": "rustfs-ecstore",
|
||||||
|
"name": "set_disk::ops::multipart::tests::cancelled_complete_keeps_upload_lock_through_tail_cleanup"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"invariant": "list-uncommitted-version",
|
||||||
|
"suite": "rustfs-filemeta",
|
||||||
|
"name": "metacache::tests::resolve_with_write_quorum_slack_keeps_partial_latest_hidden_during_merge"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"invariant": "minio-object-fixture",
|
||||||
|
"suite": "rustfs-filemeta",
|
||||||
|
"name": "filemeta::test::parses_real_minio_object_xlmeta"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"invariant": "corrupt-part-arrays",
|
||||||
|
"suite": "rustfs-filemeta",
|
||||||
|
"name": "filemeta::test::crc_valid_but_part_arrays_corrupt_into_fileinfo_errors_not_panics"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"fixtures": [
|
||||||
|
{
|
||||||
|
"path": "crates/filemeta/tests/fixtures/minio/object_large_bin.xlmeta.hex",
|
||||||
|
"sha256": "e8093767806d701e639b48d023190e858fbc4cde69bcfd83c22af8cba8452ce5",
|
||||||
|
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "crates/filemeta/tests/fixtures/minio/object_small_txt.xlmeta.hex",
|
||||||
|
"sha256": "2a415ad3a3be5a9440035d4026ff880e0e8c1ec1701be9f4e077734e8dce03da",
|
||||||
|
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "crates/filemeta/tests/fixtures/minio/object_versioned_txt.xlmeta.hex",
|
||||||
|
"sha256": "7f21f50c326dd8b0228deb6dbdb7052b3d0a3f8ee6c85d43486f0e6bb7a97261",
|
||||||
|
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "crates/ecstore/tests/fixtures/minio/bucket_metadata.blob.hex",
|
||||||
|
"sha256": "f2b6e260aff106adf6039feb1c645686e84e75404ff725491fb18668be5db203",
|
||||||
|
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"path": "crates/ecstore/tests/fixtures/minio/bucket_metadata_full.xlmeta.hex",
|
||||||
|
"sha256": "3b6de589519c08a1614c8bd409bb8199c17d42043861b07bce513075e6fbfc12",
|
||||||
|
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -38,8 +38,10 @@ script-tests: ## Run shell script tests
|
|||||||
./scripts/test_python_bin.sh
|
./scripts/test_python_bin.sh
|
||||||
./scripts/check_embedded_secrets.sh --self-test
|
./scripts/check_embedded_secrets.sh --self-test
|
||||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test
|
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test
|
||||||
|
$(RUSTFS_PYTHON_BIN) ./scripts/test_e2e_binary.py
|
||||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test
|
$(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test
|
||||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test
|
$(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test
|
||||||
|
$(RUSTFS_PYTHON_BIN) ./scripts/test_security_workflow.py
|
||||||
$(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py
|
$(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py
|
||||||
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
|
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
|
||||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test
|
$(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test
|
||||||
|
|||||||
@@ -183,13 +183,6 @@ test-group = 'e2e-reliability'
|
|||||||
filter = 'package(e2e_test) & test(/^inline_fast_path_cluster_test::/)'
|
filter = 'package(e2e_test) & test(/^inline_fast_path_cluster_test::/)'
|
||||||
test-group = 'e2e-inline-boundaries'
|
test-group = 'e2e-inline-boundaries'
|
||||||
|
|
||||||
# 4-node 4-drive distributed Actions suite: each case starts four rustfs
|
|
||||||
# processes and up to sixteen data directories. Serialize across nextest's
|
|
||||||
# process boundary so several 4x4 clusters never overlap.
|
|
||||||
[[profile.default.overrides]]
|
|
||||||
filter = 'package(e2e_test) & test(/^distributed::/)'
|
|
||||||
test-group = 'e2e-cluster-nightly'
|
|
||||||
|
|
||||||
# Vault KMS tests share the fixed dev-server port 8200. serial_test's #[serial]
|
# Vault KMS tests share the fixed dev-server port 8200. serial_test's #[serial]
|
||||||
# does not cross nextest process boundaries, so keep every Vault-backed test in
|
# does not cross nextest process boundaries, so keep every Vault-backed test in
|
||||||
# one group.
|
# one group.
|
||||||
@@ -533,26 +526,6 @@ path = "junit.xml"
|
|||||||
filter = 'package(e2e_test)'
|
filter = 'package(e2e_test)'
|
||||||
test-group = 'e2e-cluster-nightly'
|
test-group = 'e2e-cluster-nightly'
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# e2e-distributed profile — 4-node 4-disk Actions suite
|
|
||||||
# ---------------------------------------------------------------------------
|
|
||||||
# Nightly / dispatch lane owned by .github/workflows/e2e-distributed.yml.
|
|
||||||
# Each case starts four rustfs processes (and for site replication, two
|
|
||||||
# clusters). Upgrade cases also require RUSTFS_UPGRADE_SOURCE_BINARY.
|
|
||||||
# Serialized via e2e-cluster-nightly. Not a PR merge gate.
|
|
||||||
[profile.e2e-distributed]
|
|
||||||
default-filter = 'package(e2e_test) & test(/^distributed::/)'
|
|
||||||
fail-fast = false
|
|
||||||
# Decommission / rebalance cases poll for up to 180s with little stdout.
|
|
||||||
slow-timeout = { period = "120s", terminate-after = 6 }
|
|
||||||
|
|
||||||
[profile.e2e-distributed.junit]
|
|
||||||
path = "junit.xml"
|
|
||||||
|
|
||||||
[[profile.e2e-distributed.overrides]]
|
|
||||||
filter = 'package(e2e_test)'
|
|
||||||
test-group = 'e2e-cluster-nightly'
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# e2e-odm-interop profile — on-demand migration provider interop lane (ODM-20)
|
# e2e-odm-interop profile — on-demand migration provider interop lane (ODM-20)
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -613,10 +586,6 @@ path = "junit.xml"
|
|||||||
# cluster-fault lane. heal_erasure_disk_rebuild is intentionally not
|
# cluster-fault lane. heal_erasure_disk_rebuild is intentionally not
|
||||||
# excluded here because backlog#2213 promotes core heal rebuild coverage to
|
# excluded here because backlog#2213 promotes core heal rebuild coverage to
|
||||||
# this merge/main lane while retaining nightly coverage.
|
# this merge/main lane while retaining nightly coverage.
|
||||||
# * distributed:: — 4-node 4-disk Actions suite (S3, lock, versioning,
|
|
||||||
# replication, quota, observability, expand/decommission/rebalance, site
|
|
||||||
# replication, chaos, upgrade history/IAM). Owns [profile.e2e-distributed] and
|
|
||||||
# .github/workflows/e2e-distributed.yml.
|
|
||||||
# * on_demand_migration::interop_test — the ODM-20 provider interoperability
|
# * on_demand_migration::interop_test — the ODM-20 provider interoperability
|
||||||
# cases, which are meaningless without a source: they run in the dedicated
|
# cases, which are meaningless without a source: they run in the dedicated
|
||||||
# [profile.e2e-odm-interop] lane below, where the workflow points them at a
|
# [profile.e2e-odm-interop] lane below, where the workflow points them at a
|
||||||
@@ -638,7 +607,6 @@ default-filter = """
|
|||||||
package(e2e_test)
|
package(e2e_test)
|
||||||
& !test(/^protocols::/)
|
& !test(/^protocols::/)
|
||||||
& !test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|degraded_listing_availability_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
& !test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|degraded_listing_availability_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||||
& !test(/^distributed::/)
|
|
||||||
& !test(/^replication_extension_test::/)
|
& !test(/^replication_extension_test::/)
|
||||||
& !test(/^replication_target_matrix_test::/)
|
& !test(/^replication_target_matrix_test::/)
|
||||||
& !test(/^on_demand_migration::(concurrency_test|fault_test|interop_test|real_source_test)::/)
|
& !test(/^on_demand_migration::(concurrency_test|fault_test|interop_test|real_source_test)::/)
|
||||||
|
|||||||
@@ -0,0 +1,116 @@
|
|||||||
|
# Copyright 2024 RustFS Team
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
|
||||||
|
name: Quick Checks
|
||||||
|
description: Run the shared compile-free RustFS quality checks.
|
||||||
|
|
||||||
|
runs:
|
||||||
|
using: composite
|
||||||
|
steps:
|
||||||
|
- name: Install quality tools
|
||||||
|
uses: taiki-e/install-action@bffeee26d4db9be238a4ea78d8826604ebcb594d # v2
|
||||||
|
with:
|
||||||
|
tool: |
|
||||||
|
ripgrep@15.2.0
|
||||||
|
shellcheck@0.11.0
|
||||||
|
|
||||||
|
- name: Install actionlint
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
actionlint_dir="$(mktemp -d "${RUNNER_TEMP}/actionlint.XXXXXX")"
|
||||||
|
curl --fail --location --silent --show-error \
|
||||||
|
--output "$actionlint_dir/actionlint.tar.gz" \
|
||||||
|
https://github.com/rhysd/actionlint/releases/download/v1.7.12/actionlint_1.7.12_linux_amd64.tar.gz
|
||||||
|
echo "8aca8db96f1b94770f1b0d72b6dddcb1ebb8123cb3712530b08cc387b349a3d8 $actionlint_dir/actionlint.tar.gz" | sha256sum --check --status
|
||||||
|
tar -xzf "$actionlint_dir/actionlint.tar.gz" -C "$actionlint_dir" actionlint
|
||||||
|
rm "$actionlint_dir/actionlint.tar.gz"
|
||||||
|
echo "$actionlint_dir" >> "$GITHUB_PATH"
|
||||||
|
|
||||||
|
- name: Install Rust toolchain
|
||||||
|
uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable
|
||||||
|
with:
|
||||||
|
components: rustfmt
|
||||||
|
|
||||||
|
- name: Check workflow syntax and shell scripts
|
||||||
|
shell: bash
|
||||||
|
run: shellcheck --version && actionlint
|
||||||
|
|
||||||
|
- name: Check code formatting
|
||||||
|
shell: bash
|
||||||
|
run: cargo fmt --all --check
|
||||||
|
|
||||||
|
- name: Check unsafe code allowances
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_unsafe_code_allowances.sh
|
||||||
|
|
||||||
|
- name: Check layered dependencies
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_layer_dependencies.sh
|
||||||
|
|
||||||
|
- name: Check architecture migration rules
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_architecture_migration_rules.sh
|
||||||
|
|
||||||
|
- name: Check logging guardrails
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_logging_guardrails.sh
|
||||||
|
|
||||||
|
- name: Check error other(format!) ratchet
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_error_other_format_ratchet.sh
|
||||||
|
|
||||||
|
- name: Check tokio io-uring feature guard
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_no_tokio_io_uring.sh
|
||||||
|
|
||||||
|
- name: Check extension schema boundaries
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_extension_schema_boundaries.sh
|
||||||
|
|
||||||
|
- name: Check body-cache whitelist guard
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_body_cache_whitelist.sh
|
||||||
|
|
||||||
|
- name: Check s3s footprint ratchet
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_s3s_footprint.sh
|
||||||
|
|
||||||
|
- name: Check cryptographic capability wording
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_fips_wording.sh
|
||||||
|
|
||||||
|
- name: Check no embedded secret material
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_embedded_secrets.sh
|
||||||
|
|
||||||
|
- name: Check test wiring
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
python3 ./scripts/check_test_wiring.py --self-test
|
||||||
|
python3 ./scripts/test_e2e_binary.py
|
||||||
|
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
|
||||||
|
python3 ./scripts/test_security_workflow.py
|
||||||
|
python3 ./scripts/check_test_wiring.py
|
||||||
|
|
||||||
|
- name: Check no planning docs committed
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_no_planning_docs.sh
|
||||||
|
|
||||||
|
- name: Check CI paths stay in sync
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_ci_paths_sync.sh
|
||||||
|
|
||||||
|
- name: Check io_uring lane --lib precondition
|
||||||
|
shell: bash
|
||||||
|
run: ./scripts/check_uring_lane_lib_only.sh
|
||||||
@@ -10,16 +10,16 @@ Use N/A when there is no related issue.
|
|||||||
|
|
||||||
## Summary of Changes
|
## Summary of Changes
|
||||||
<!--
|
<!--
|
||||||
Briefly explain what changed and why reviewers should accept it.
|
Describe the concrete problem and resulting behavior. For a behavior change, name the input or state that triggers it and the expected outcome. Explain any new dependency or abstraction that the change needs.
|
||||||
Focus on behavior, compatibility, and review-relevant context.
|
|
||||||
-->
|
-->
|
||||||
|
|
||||||
## Verification
|
## Verification
|
||||||
<!--
|
<!--
|
||||||
List the commands or checks you ran, for example:
|
Give 1–3 concrete pieces of evidence for the changed behavior: the test or command, its observed result, and the regression it catches. For a bug fix, record a failing-before/passing-after check or explain why it was unavailable.
|
||||||
- `make pre-commit`
|
|
||||||
|
|
||||||
Use N/A only when verification is not applicable.
|
Identify the tested commit and any local changes. When testing a prebuilt binary or external service, include its source/version and artifact identity; a successful run against a different build is not evidence for this change.
|
||||||
|
|
||||||
|
List relevant checks not run and the remaining risk. Use the validation tier in AGENTS.md; do not run broader checks solely to fill this section. For documentation-only changes, list the applicable documentation checks. Use N/A only when verification is not applicable.
|
||||||
-->
|
-->
|
||||||
|
|
||||||
## Impact
|
## Impact
|
||||||
|
|||||||
@@ -4,11 +4,6 @@
|
|||||||
{ "workflow": ".github/workflows/ci.yml", "max_age_hours": 192 },
|
{ "workflow": ".github/workflows/ci.yml", "max_age_hours": 192 },
|
||||||
{ "workflow": ".github/workflows/coverage.yml", "max_age_hours": 192 },
|
{ "workflow": ".github/workflows/coverage.yml", "max_age_hours": 192 },
|
||||||
{ "workflow": ".github/workflows/e2e-replication-nightly.yml", "max_age_hours": 36 },
|
{ "workflow": ".github/workflows/e2e-replication-nightly.yml", "max_age_hours": 36 },
|
||||||
{
|
|
||||||
"workflow": ".github/workflows/e2e-distributed.yml",
|
|
||||||
"max_age_hours": 36,
|
|
||||||
"never_ran_grace_until": "2026-09-18T00:00:00Z"
|
|
||||||
},
|
|
||||||
{ "workflow": ".github/workflows/e2e-s3tests.yml", "max_age_hours": 192 },
|
{ "workflow": ".github/workflows/e2e-s3tests.yml", "max_age_hours": 192 },
|
||||||
{ "workflow": ".github/workflows/fuzz.yml", "max_age_hours": 36 },
|
{ "workflow": ".github/workflows/fuzz.yml", "max_age_hours": 36 },
|
||||||
{ "workflow": ".github/workflows/mint.yml", "max_age_hours": 192 },
|
{ "workflow": ".github/workflows/mint.yml", "max_age_hours": 192 },
|
||||||
|
|||||||
@@ -12,24 +12,10 @@
|
|||||||
# See the License for the specific language governing permissions and
|
# See the License for the specific language governing permissions and
|
||||||
# limitations under the License.
|
# limitations under the License.
|
||||||
|
|
||||||
# Companion to ci.yml for required status checks.
|
# Reports the existing required checks for paths excluded by ci.yml.
|
||||||
#
|
# Mixed PRs can trigger both workflows; their Quick Checks jobs use one shared
|
||||||
# ci.yml skips docs-only pull requests via paths-ignore, but the branch ruleset
|
# action to keep validation coverage aligned. Keep this paths list in sync with
|
||||||
# requires a check named "Test and Lint" — without this workflow a docs-only PR
|
# ci.yml's pull_request.paths-ignore via scripts/check_ci_paths_sync.sh.
|
||||||
# would wait on it forever. This workflow triggers on exactly the paths ci.yml
|
|
||||||
# ignores and reports success under the same job name. Mixed PRs trigger both
|
|
||||||
# workflows and the real check still gates: a required check with any failing
|
|
||||||
# run blocks the merge.
|
|
||||||
# https://docs.github.com/en/repositories/configuring-branches-and-merges-in-your-repository/defining-the-mergeability-of-pull-requests/troubleshooting-required-status-checks#handling-skipped-but-required-checks
|
|
||||||
#
|
|
||||||
# "Quick Checks" is mirrored here ahead of the ruleset change that will make it
|
|
||||||
# required too (rustfs/backlog#1599). Until that change lands this job is
|
|
||||||
# inert; mirroring it first is what lets the ruleset change happen without
|
|
||||||
# stranding docs-only PRs on a check nobody reports.
|
|
||||||
#
|
|
||||||
# Keep the paths list below in sync with the pull_request paths-ignore list
|
|
||||||
# in ci.yml, and keep the quick-checks steps below byte-identical to the
|
|
||||||
# quick-checks job in ci.yml.
|
|
||||||
|
|
||||||
name: Continuous Integration (docs only)
|
name: Continuous Integration (docs only)
|
||||||
|
|
||||||
@@ -59,19 +45,6 @@ permissions:
|
|||||||
contents: read
|
contents: read
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
# Deliberately NOT a bare `echo`. Once "Quick Checks" becomes a required
|
|
||||||
# check, ci.yml gates every expensive job behind it, so a mixed PR reports
|
|
||||||
# two check runs with this name: the real one (45-51s) and this companion.
|
|
||||||
# GitHub has no written contract for how it picks between same-named
|
|
||||||
# required check runs ("latest wins" vs "any failure blocks"), so instead of
|
|
||||||
# relying on ordering we make both runs execute the same commands against
|
|
||||||
# the same merge ref — their conclusions are then necessarily identical and
|
|
||||||
# the choice does not matter. Keep these steps byte-identical to the
|
|
||||||
# quick-checks job in ci.yml (a guard script that asserts this, and the paths
|
|
||||||
# sync below, is tracked in rustfs/backlog#1603).
|
|
||||||
#
|
|
||||||
# For a genuinely docs-only PR this adds no strictness (no code changed, so
|
|
||||||
# fmt and the guards always pass) and costs ~50s of ubuntu-latest.
|
|
||||||
quick-checks:
|
quick-checks:
|
||||||
name: Quick Checks
|
name: Quick Checks
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
@@ -82,63 +55,8 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
|
||||||
- name: Install ripgrep
|
- name: Run shared quick checks
|
||||||
uses: taiki-e/install-action@bffeee26d4db9be238a4ea78d8826604ebcb594d # v2
|
uses: ./.github/actions/quick-checks
|
||||||
with:
|
|
||||||
tool: ripgrep@15.2.0
|
|
||||||
|
|
||||||
- name: Install Rust toolchain
|
|
||||||
uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable
|
|
||||||
with:
|
|
||||||
components: rustfmt
|
|
||||||
|
|
||||||
- name: Check code formatting
|
|
||||||
run: cargo fmt --all --check
|
|
||||||
|
|
||||||
- name: Check unsafe code allowances
|
|
||||||
run: ./scripts/check_unsafe_code_allowances.sh
|
|
||||||
|
|
||||||
- name: Check layered dependencies
|
|
||||||
run: ./scripts/check_layer_dependencies.sh
|
|
||||||
|
|
||||||
- name: Check architecture migration rules
|
|
||||||
run: ./scripts/check_architecture_migration_rules.sh
|
|
||||||
|
|
||||||
- name: Check logging guardrails
|
|
||||||
run: ./scripts/check_logging_guardrails.sh
|
|
||||||
|
|
||||||
- name: Check tokio io-uring feature guard
|
|
||||||
run: ./scripts/check_no_tokio_io_uring.sh
|
|
||||||
|
|
||||||
- name: Check extension schema boundaries
|
|
||||||
run: ./scripts/check_extension_schema_boundaries.sh
|
|
||||||
|
|
||||||
- name: Check body-cache whitelist guard
|
|
||||||
run: ./scripts/check_body_cache_whitelist.sh
|
|
||||||
|
|
||||||
- name: Check s3s footprint ratchet
|
|
||||||
run: ./scripts/check_s3s_footprint.sh
|
|
||||||
|
|
||||||
- name: Check cryptographic capability wording
|
|
||||||
run: ./scripts/check_fips_wording.sh
|
|
||||||
|
|
||||||
- name: Check no embedded secret material
|
|
||||||
run: ./scripts/check_embedded_secrets.sh
|
|
||||||
|
|
||||||
- name: Check test wiring
|
|
||||||
run: |
|
|
||||||
python3 ./scripts/check_test_wiring.py --self-test
|
|
||||||
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
|
|
||||||
python3 ./scripts/check_test_wiring.py
|
|
||||||
|
|
||||||
- name: Check no planning docs committed
|
|
||||||
run: ./scripts/check_no_planning_docs.sh
|
|
||||||
|
|
||||||
- name: Check CI paths stay in sync
|
|
||||||
run: ./scripts/check_ci_paths_sync.sh
|
|
||||||
|
|
||||||
- name: Check io_uring lane --lib precondition
|
|
||||||
run: ./scripts/check_uring_lane_lib_only.sh
|
|
||||||
|
|
||||||
test-and-lint:
|
test-and-lint:
|
||||||
name: Test and Lint
|
name: Test and Lint
|
||||||
|
|||||||
+24
-76
@@ -100,12 +100,7 @@ jobs:
|
|||||||
- name: Typos check with custom config file
|
- name: Typos check with custom config file
|
||||||
uses: crate-ci/typos@37bb98842b0d8c4ffebdb75301a13db0267cef89 # master
|
uses: crate-ci/typos@37bb98842b0d8c4ffebdb75301a13db0267cef89 # master
|
||||||
|
|
||||||
# Fast, compile-free checks that fail early so contributors get feedback in
|
# Fail early with compile-free checks shared with docs-only CI.
|
||||||
# ~1 minute instead of waiting for the full test job.
|
|
||||||
#
|
|
||||||
# These steps are mirrored byte-for-byte in ci-docs-only.yml so that a mixed
|
|
||||||
# PR, which reports two check runs named "Quick Checks", cannot get one red
|
|
||||||
# and one green. Edit both jobs together.
|
|
||||||
quick-checks:
|
quick-checks:
|
||||||
name: Quick Checks
|
name: Quick Checks
|
||||||
if: github.event_name != 'pull_request' || github.event.action != 'closed'
|
if: github.event_name != 'pull_request' || github.event.action != 'closed'
|
||||||
@@ -117,66 +112,8 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
persist-credentials: false
|
persist-credentials: false
|
||||||
|
|
||||||
- name: Install ripgrep
|
- name: Run shared quick checks
|
||||||
uses: taiki-e/install-action@bffeee26d4db9be238a4ea78d8826604ebcb594d # v2
|
uses: ./.github/actions/quick-checks
|
||||||
with:
|
|
||||||
tool: ripgrep@15.2.0
|
|
||||||
|
|
||||||
- name: Install Rust toolchain
|
|
||||||
uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable
|
|
||||||
with:
|
|
||||||
components: rustfmt
|
|
||||||
|
|
||||||
- name: Check code formatting
|
|
||||||
run: cargo fmt --all --check
|
|
||||||
|
|
||||||
- name: Check unsafe code allowances
|
|
||||||
run: ./scripts/check_unsafe_code_allowances.sh
|
|
||||||
|
|
||||||
- name: Check layered dependencies
|
|
||||||
run: ./scripts/check_layer_dependencies.sh
|
|
||||||
|
|
||||||
- name: Check architecture migration rules
|
|
||||||
run: ./scripts/check_architecture_migration_rules.sh
|
|
||||||
|
|
||||||
- name: Check logging guardrails
|
|
||||||
run: ./scripts/check_logging_guardrails.sh
|
|
||||||
|
|
||||||
- name: Check error other(format!) ratchet
|
|
||||||
run: ./scripts/check_error_other_format_ratchet.sh
|
|
||||||
|
|
||||||
- name: Check tokio io-uring feature guard
|
|
||||||
run: ./scripts/check_no_tokio_io_uring.sh
|
|
||||||
|
|
||||||
- name: Check extension schema boundaries
|
|
||||||
run: ./scripts/check_extension_schema_boundaries.sh
|
|
||||||
|
|
||||||
- name: Check body-cache whitelist guard
|
|
||||||
run: ./scripts/check_body_cache_whitelist.sh
|
|
||||||
|
|
||||||
- name: Check s3s footprint ratchet
|
|
||||||
run: ./scripts/check_s3s_footprint.sh
|
|
||||||
|
|
||||||
- name: Check cryptographic capability wording
|
|
||||||
run: ./scripts/check_fips_wording.sh
|
|
||||||
|
|
||||||
- name: Check no embedded secret material
|
|
||||||
run: ./scripts/check_embedded_secrets.sh
|
|
||||||
|
|
||||||
- name: Check test wiring
|
|
||||||
run: |
|
|
||||||
python3 ./scripts/check_test_wiring.py --self-test
|
|
||||||
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
|
|
||||||
python3 ./scripts/check_test_wiring.py
|
|
||||||
|
|
||||||
- name: Check no planning docs committed
|
|
||||||
run: ./scripts/check_no_planning_docs.sh
|
|
||||||
|
|
||||||
- name: Check CI paths stay in sync
|
|
||||||
run: ./scripts/check_ci_paths_sync.sh
|
|
||||||
|
|
||||||
- name: Check io_uring lane --lib precondition
|
|
||||||
run: ./scripts/check_uring_lane_lib_only.sh
|
|
||||||
|
|
||||||
test-and-lint:
|
test-and-lint:
|
||||||
name: Test and Lint
|
name: Test and Lint
|
||||||
@@ -269,6 +206,7 @@ jobs:
|
|||||||
CARGO_BUILD_JOBS: ${{ (github.event_name == 'push' || github.event_name == 'workflow_dispatch') && '3' || '2' }}
|
CARGO_BUILD_JOBS: ${{ (github.event_name == 'push' || github.event_name == 'workflow_dispatch') && '3' || '2' }}
|
||||||
run: |
|
run: |
|
||||||
mkdir -p artifacts/test-and-lint
|
mkdir -p artifacts/test-and-lint
|
||||||
|
rm -f target/nextest/ci/junit.xml
|
||||||
./scripts/ci/resource_sampler.sh start nextest
|
./scripts/ci/resource_sampler.sh start nextest
|
||||||
trap './scripts/ci/resource_sampler.sh stop' EXIT
|
trap './scripts/ci/resource_sampler.sh stop' EXIT
|
||||||
set +e
|
set +e
|
||||||
@@ -277,6 +215,12 @@ jobs:
|
|||||||
--status-level all --final-status-level all \
|
--status-level all --final-status-level all \
|
||||||
2>&1 | tee artifacts/test-and-lint/nextest.log
|
2>&1 | tee artifacts/test-and-lint/nextest.log
|
||||||
status=${PIPESTATUS[0]}
|
status=${PIPESTATUS[0]}
|
||||||
|
if [[ "${status}" -eq 0 ]]; then
|
||||||
|
cargo nextest list --profile ci --all --exclude e2e_test --message-format json \
|
||||||
|
> artifacts/test-and-lint/core-test-listing.json \
|
||||||
|
&& python3 scripts/check_test_wiring.py --check-core artifacts/test-and-lint/core-test-listing.json \
|
||||||
|
&& test -s target/nextest/ci/junit.xml || status=$?
|
||||||
|
fi
|
||||||
{
|
{
|
||||||
echo "command=cargo nextest run --profile ci --all --exclude e2e_test"
|
echo "command=cargo nextest run --profile ci --all --exclude e2e_test"
|
||||||
echo "exit_status=${status}"
|
echo "exit_status=${status}"
|
||||||
@@ -638,13 +582,15 @@ jobs:
|
|||||||
install-build-packaging-tools: 'false'
|
install-build-packaging-tools: 'false'
|
||||||
|
|
||||||
- name: Build debug binary
|
- name: Build debug binary
|
||||||
run: cargo build -p rustfs --bins --features e2e-test-hooks
|
run: python3 scripts/e2e_binary.py build --bins --features e2e-test-hooks
|
||||||
|
|
||||||
- name: Upload debug binary
|
- name: Upload debug binary
|
||||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||||
with:
|
with:
|
||||||
name: rustfs-debug-binary
|
name: rustfs-debug-binary
|
||||||
path: target/debug/rustfs
|
path: |
|
||||||
|
target/debug/rustfs
|
||||||
|
target/debug/rustfs.e2e.json
|
||||||
if-no-files-found: error
|
if-no-files-found: error
|
||||||
retention-days: 1
|
retention-days: 1
|
||||||
|
|
||||||
@@ -676,13 +622,15 @@ jobs:
|
|||||||
install-build-packaging-tools: 'false'
|
install-build-packaging-tools: 'false'
|
||||||
|
|
||||||
- name: Build debug binary with rio-v2
|
- name: Build debug binary with rio-v2
|
||||||
run: cargo build -p rustfs --bins --features rio-v2,e2e-test-hooks
|
run: python3 scripts/e2e_binary.py build --bins --features rio-v2,e2e-test-hooks
|
||||||
|
|
||||||
- name: Upload debug binary
|
- name: Upload debug binary
|
||||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||||
with:
|
with:
|
||||||
name: rustfs-debug-binary-rio-v2
|
name: rustfs-debug-binary-rio-v2
|
||||||
path: target/debug/rustfs
|
path: |
|
||||||
|
target/debug/rustfs
|
||||||
|
target/debug/rustfs.e2e.json
|
||||||
if-no-files-found: error
|
if-no-files-found: error
|
||||||
retention-days: 1
|
retention-days: 1
|
||||||
|
|
||||||
@@ -831,7 +779,7 @@ jobs:
|
|||||||
NEXTEST_ARCHIVE: ${{ runner.temp }}/rustfs-e2e-smoke.tar.zst
|
NEXTEST_ARCHIVE: ${{ runner.temp }}/rustfs-e2e-smoke.tar.zst
|
||||||
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-smoke-logs
|
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-smoke-logs
|
||||||
run: |
|
run: |
|
||||||
cargo nextest run --profile e2e-smoke --archive-file "${NEXTEST_ARCHIVE}" \
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-smoke --archive-file "${NEXTEST_ARCHIVE}" \
|
||||||
--status-level all --final-status-level all --failure-output final
|
--status-level all --final-status-level all --failure-output final
|
||||||
|
|
||||||
- name: Upload e2e smoke diagnostics
|
- name: Upload e2e smoke diagnostics
|
||||||
@@ -867,7 +815,7 @@ jobs:
|
|||||||
RUSTFS_TEST_PORT="$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1]); s.close()')"
|
RUSTFS_TEST_PORT="$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1]); s.close()')"
|
||||||
RUSTFS_TEST_PORT="${RUSTFS_TEST_PORT}" \
|
RUSTFS_TEST_PORT="${RUSTFS_TEST_PORT}" \
|
||||||
RUSTFS_TEST_LOG="${RUN_ROOT}/rustfs.log" \
|
RUSTFS_TEST_LOG="${RUN_ROOT}/rustfs.log" \
|
||||||
./scripts/e2e-run.sh ./target/debug/rustfs "${RUN_ROOT}/data"
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- ./scripts/e2e-run.sh ./target/debug/rustfs "${RUN_ROOT}/data"
|
||||||
|
|
||||||
- name: Upload test logs
|
- name: Upload test logs
|
||||||
if: failure()
|
if: failure()
|
||||||
@@ -969,7 +917,7 @@ jobs:
|
|||||||
# extend that filter, never add ad-hoc e2e jobs here. Reuses the downloaded
|
# extend that filter, never add ad-hoc e2e jobs here. Reuses the downloaded
|
||||||
# debug binary; each test spawns its own rustfs server on a random port.
|
# debug binary; each test spawns its own rustfs server on a random port.
|
||||||
- name: Run e2e full suite
|
- name: Run e2e full suite
|
||||||
run: cargo nextest run --profile e2e-full -p e2e_test
|
run: python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-full -p e2e_test
|
||||||
|
|
||||||
- name: Upload junit
|
- name: Upload junit
|
||||||
if: always()
|
if: always()
|
||||||
@@ -1030,7 +978,7 @@ jobs:
|
|||||||
- name: Run end-to-end tests
|
- name: Run end-to-end tests
|
||||||
run: |
|
run: |
|
||||||
s3s-e2e --version
|
s3s-e2e --version
|
||||||
./scripts/e2e-run.sh ./target/debug/rustfs /tmp/rustfs
|
python3 scripts/e2e_binary.py run --features rio-v2,e2e-test-hooks -- ./scripts/e2e-run.sh ./target/debug/rustfs /tmp/rustfs
|
||||||
|
|
||||||
- name: Upload test logs
|
- name: Upload test logs
|
||||||
if: failure()
|
if: failure()
|
||||||
@@ -1073,7 +1021,7 @@ jobs:
|
|||||||
S3_PORT="${S3_PORT}" \
|
S3_PORT="${S3_PORT}" \
|
||||||
DATA_ROOT="${RUN_ROOT}" \
|
DATA_ROOT="${RUN_ROOT}" \
|
||||||
S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \
|
S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \
|
||||||
./scripts/s3-tests/run.sh
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- ./scripts/s3-tests/run.sh
|
||||||
|
|
||||||
- name: Upload s3 test artifacts
|
- name: Upload s3 test artifacts
|
||||||
if: always()
|
if: always()
|
||||||
@@ -1155,7 +1103,7 @@ jobs:
|
|||||||
S3_PORT="${S3_PORT}" \
|
S3_PORT="${S3_PORT}" \
|
||||||
DATA_ROOT="${RUN_ROOT}" \
|
DATA_ROOT="${RUN_ROOT}" \
|
||||||
S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \
|
S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \
|
||||||
./scripts/s3-tests/run.sh
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- ./scripts/s3-tests/run.sh
|
||||||
|
|
||||||
- name: Upload s3 test artifacts
|
- name: Upload s3 test artifacts
|
||||||
if: always()
|
if: always()
|
||||||
|
|||||||
@@ -1,140 +0,0 @@
|
|||||||
# Copyright 2024 RustFS Team
|
|
||||||
#
|
|
||||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
# you may not use this file except in compliance with the License.
|
|
||||||
# You may obtain a copy of the License at
|
|
||||||
#
|
|
||||||
# http://www.apache.org/LICENSE-2.0
|
|
||||||
#
|
|
||||||
# Unless required by applicable law or agreed to in writing, software
|
|
||||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
# See the License for the specific language governing permissions and
|
|
||||||
# limitations under the License.
|
|
||||||
|
|
||||||
# 4-node 4-disk distributed e2e lane.
|
|
||||||
#
|
|
||||||
# Each selected test starts a real localhost cluster via
|
|
||||||
# `RustFSTestClusterEnvironment` (4 processes; 4 drives per node unless the
|
|
||||||
# case is a two-site 4-node 1-drive pair or a 4-node upgrade). Membership is
|
|
||||||
# `[profile.e2e-distributed]` in `.config/nextest.toml`. This is not a required
|
|
||||||
# merge check: it is the scheduled/dispatch counterpart to the hardware
|
|
||||||
# functional chain that currently clones rustfs/auto-testing onto three VMs.
|
|
||||||
# Upgrade cases download the same pinned previous release as e2e-upgrade.yml.
|
|
||||||
|
|
||||||
name: e2e-distributed
|
|
||||||
|
|
||||||
on:
|
|
||||||
workflow_dispatch:
|
|
||||||
inputs:
|
|
||||||
filter:
|
|
||||||
description: "Optional nextest -E filter (default: the whole e2e-distributed profile)"
|
|
||||||
required: false
|
|
||||||
default: ""
|
|
||||||
schedule:
|
|
||||||
# 05:53 UTC nightly — clear of e2e-nightly (04:29) and ODM interop (05:23).
|
|
||||||
- cron: "53 5 * * *"
|
|
||||||
|
|
||||||
permissions:
|
|
||||||
contents: read
|
|
||||||
|
|
||||||
concurrency:
|
|
||||||
group: ${{ github.workflow }}-${{ github.ref }}
|
|
||||||
cancel-in-progress: ${{ github.event_name != 'schedule' }}
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
distributed:
|
|
||||||
name: Distributed 4-node 4-disk e2e
|
|
||||||
runs-on: sm-standard-4
|
|
||||||
timeout-minutes: 180
|
|
||||||
env:
|
|
||||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
|
||||||
NO_PROXY: 127.0.0.1,localhost
|
|
||||||
HTTP_PROXY: ""
|
|
||||||
HTTPS_PROXY: ""
|
|
||||||
# Pinned previous release used by distributed::upgrade_test (same pin as e2e-upgrade.yml).
|
|
||||||
UPGRADE_SOURCE_VERSION: 1.0.0-rc.2
|
|
||||||
UPGRADE_SOURCE_ASSET: rustfs-linux-x86_64-gnu-v1.0.0-rc.2.zip
|
|
||||||
UPGRADE_SOURCE_SHA256: 7c789386bf85278f865b8e0d359bf4edb84d5aa408cc3fa54a18c25ca74cd6e7
|
|
||||||
steps:
|
|
||||||
- name: Checkout repository
|
|
||||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
||||||
with:
|
|
||||||
persist-credentials: false
|
|
||||||
|
|
||||||
- name: Setup Rust environment
|
|
||||||
uses: ./.github/actions/setup
|
|
||||||
with:
|
|
||||||
rust-version: stable
|
|
||||||
cache-shared-key: ci-e2e-distributed
|
|
||||||
cache-save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
||||||
install-build-packaging-tools: 'false'
|
|
||||||
|
|
||||||
- name: Download pinned previous release
|
|
||||||
env:
|
|
||||||
SOURCE_DIR: ${{ runner.temp }}/rustfs-upgrade-source
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
mkdir -p "$SOURCE_DIR"
|
|
||||||
archive="$SOURCE_DIR/$UPGRADE_SOURCE_ASSET"
|
|
||||||
curl --fail --location --retry 3 --output "$archive" \
|
|
||||||
"https://github.com/${GITHUB_REPOSITORY}/releases/download/${UPGRADE_SOURCE_VERSION}/${UPGRADE_SOURCE_ASSET}"
|
|
||||||
echo "$UPGRADE_SOURCE_SHA256 $archive" | sha256sum --check --strict
|
|
||||||
unzip -q "$archive" -d "$SOURCE_DIR"
|
|
||||||
chmod +x "$SOURCE_DIR/rustfs"
|
|
||||||
test -x "$SOURCE_DIR/rustfs"
|
|
||||||
echo "RUSTFS_UPGRADE_SOURCE_BINARY=$SOURCE_DIR/rustfs" >> "$GITHUB_ENV"
|
|
||||||
|
|
||||||
- name: Build rustfs binary
|
|
||||||
run: |
|
|
||||||
cargo build -p rustfs --bins
|
|
||||||
: > target/debug/rustfs.features
|
|
||||||
|
|
||||||
- name: Verify distributed e2e membership
|
|
||||||
env:
|
|
||||||
NEXTEST_LISTING: ${{ runner.temp }}/rustfs-e2e-distributed-list.json
|
|
||||||
run: |
|
|
||||||
cargo nextest list --profile e2e-distributed -p e2e_test --message-format json > "${NEXTEST_LISTING}"
|
|
||||||
python3 ./scripts/check_test_wiring.py --check-profile e2e-distributed "${NEXTEST_LISTING}"
|
|
||||||
|
|
||||||
- name: Run distributed 4-node e2e suite
|
|
||||||
env:
|
|
||||||
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-distributed-logs
|
|
||||||
NEXTEST_FILTER: ${{ github.event.inputs.filter }}
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
if [ -n "${NEXTEST_FILTER}" ]; then
|
|
||||||
cargo nextest run --profile e2e-distributed -p e2e_test -E "${NEXTEST_FILTER}" --no-tests=fail
|
|
||||||
else
|
|
||||||
cargo nextest run --profile e2e-distributed -p e2e_test --no-tests=fail
|
|
||||||
fi
|
|
||||||
|
|
||||||
- name: Upload distributed e2e diagnostics
|
|
||||||
if: always()
|
|
||||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
|
||||||
with:
|
|
||||||
name: e2e-distributed-${{ github.run_number }}
|
|
||||||
path: |
|
|
||||||
target/nextest/e2e-distributed/junit.xml
|
|
||||||
${{ runner.temp }}/rustfs-e2e-distributed-list.json
|
|
||||||
${{ runner.temp }}/rustfs-e2e-distributed-logs/
|
|
||||||
retention-days: 7
|
|
||||||
if-no-files-found: warn
|
|
||||||
|
|
||||||
alert-on-failure:
|
|
||||||
name: Alert on scheduled failure
|
|
||||||
needs: [distributed]
|
|
||||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
timeout-minutes: 10
|
|
||||||
permissions:
|
|
||||||
contents: read
|
|
||||||
issues: write
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
||||||
with:
|
|
||||||
persist-credentials: false
|
|
||||||
- name: Open or update failure-tracking issue
|
|
||||||
uses: ./.github/actions/schedule-failure-issue
|
|
||||||
with:
|
|
||||||
github-token: ${{ secrets.GITHUB_TOKEN }}
|
|
||||||
@@ -89,14 +89,10 @@ jobs:
|
|||||||
- name: Verify awscurl
|
- name: Verify awscurl
|
||||||
run: test -x "$AWSCURL_PATH"
|
run: test -x "$AWSCURL_PATH"
|
||||||
|
|
||||||
# Build the rustfs binary once up front. The e2e tests spawn it as a
|
# Build once and carry its source/binary identity into the test invocation.
|
||||||
# child process (crates/e2e_test/src/common.rs) and will build it on
|
|
||||||
# demand otherwise, but a single explicit build avoids several parallel
|
|
||||||
# nextest test processes racing to build it at once.
|
|
||||||
- name: Build rustfs binary
|
- name: Build rustfs binary
|
||||||
run: |
|
run: |
|
||||||
cargo build -p rustfs --bins
|
python3 scripts/e2e_binary.py build --bins
|
||||||
: > target/debug/rustfs.features
|
|
||||||
|
|
||||||
- name: Verify replication e2e membership
|
- name: Verify replication e2e membership
|
||||||
env:
|
env:
|
||||||
@@ -108,7 +104,7 @@ jobs:
|
|||||||
- name: Run replication e2e nightly suite
|
- name: Run replication e2e nightly suite
|
||||||
env:
|
env:
|
||||||
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-repl-nightly-logs
|
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-repl-nightly-logs
|
||||||
run: cargo nextest run --profile e2e-repl-nightly -p e2e_test
|
run: python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-repl-nightly -p e2e_test
|
||||||
|
|
||||||
- name: Upload nextest junit report
|
- name: Upload nextest junit report
|
||||||
if: always()
|
if: always()
|
||||||
@@ -144,8 +140,7 @@ jobs:
|
|||||||
|
|
||||||
- name: Build rustfs binary
|
- name: Build rustfs binary
|
||||||
run: |
|
run: |
|
||||||
cargo build -p rustfs --bins --features e2e-test-hooks
|
python3 scripts/e2e_binary.py build --bins --features e2e-test-hooks
|
||||||
: > target/debug/rustfs.features
|
|
||||||
|
|
||||||
- name: Verify cluster fault e2e membership
|
- name: Verify cluster fault e2e membership
|
||||||
env:
|
env:
|
||||||
@@ -157,7 +152,7 @@ jobs:
|
|||||||
- name: Run cluster fault e2e nightly suite
|
- name: Run cluster fault e2e nightly suite
|
||||||
env:
|
env:
|
||||||
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-nightly-logs
|
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-nightly-logs
|
||||||
run: cargo nextest run --profile e2e-nightly -p e2e_test
|
run: python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-nightly -p e2e_test
|
||||||
|
|
||||||
- name: Upload cluster fault diagnostics
|
- name: Upload cluster fault diagnostics
|
||||||
if: always()
|
if: always()
|
||||||
@@ -198,6 +193,9 @@ jobs:
|
|||||||
sudo apt-get install -y -qq iproute2
|
sudo apt-get install -y -qq iproute2
|
||||||
ss -tn state CLOSE-WAIT >/dev/null
|
ss -tn state CLOSE-WAIT >/dev/null
|
||||||
|
|
||||||
|
- name: Build protocol server
|
||||||
|
run: python3 scripts/e2e_binary.py build --features "$RUSTFS_BUILD_FEATURES"
|
||||||
|
|
||||||
# The suite owns fixed protocol ports and serializes its internal cases.
|
# The suite owns fixed protocol ports and serializes its internal cases.
|
||||||
- name: Verify protocol e2e membership
|
- name: Verify protocol e2e membership
|
||||||
env:
|
env:
|
||||||
@@ -210,7 +208,7 @@ jobs:
|
|||||||
env:
|
env:
|
||||||
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-protocol-e2e-logs
|
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-protocol-e2e-logs
|
||||||
run: >-
|
run: >-
|
||||||
cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
|
python3 scripts/e2e_binary.py run --features "$RUSTFS_BUILD_FEATURES" -- cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
|
||||||
|
|
||||||
- name: Upload protocol diagnostics
|
- name: Upload protocol diagnostics
|
||||||
if: always()
|
if: always()
|
||||||
|
|||||||
@@ -98,12 +98,11 @@ jobs:
|
|||||||
|
|
||||||
- name: Build current RustFS binary
|
- name: Build current RustFS binary
|
||||||
run: |
|
run: |
|
||||||
cargo build --locked -p rustfs --bin rustfs
|
python3 scripts/e2e_binary.py build
|
||||||
: > target/debug/rustfs.features
|
|
||||||
|
|
||||||
- name: Run upgrade compatibility test
|
- name: Run upgrade compatibility test
|
||||||
run: |
|
run: |
|
||||||
cargo test --locked -p e2e_test \
|
python3 scripts/e2e_binary.py run -- cargo test --locked -p e2e_test \
|
||||||
"upgrade_compatibility_test::${{ matrix.test }}" \
|
"upgrade_compatibility_test::${{ matrix.test }}" \
|
||||||
-- --ignored --exact --nocapture
|
-- --ignored --exact --nocapture
|
||||||
|
|
||||||
|
|||||||
@@ -132,7 +132,7 @@ jobs:
|
|||||||
s3api create-bucket --bucket "${RUSTFS_ODM_INTEROP_BUCKET}"
|
s3api create-bucket --bucket "${RUSTFS_ODM_INTEROP_BUCKET}"
|
||||||
|
|
||||||
- name: Build the RustFS binary under test
|
- name: Build the RustFS binary under test
|
||||||
run: cargo build --locked -p rustfs --bins
|
run: python3 scripts/e2e_binary.py build --bins
|
||||||
|
|
||||||
# The lane selects tests by module, so a rename would quietly shrink it.
|
# The lane selects tests by module, so a rename would quietly shrink it.
|
||||||
# The committed digest in .config/e2e-odm-interop-selection.txt fails
|
# The committed digest in .config/e2e-odm-interop-selection.txt fails
|
||||||
@@ -143,7 +143,7 @@ jobs:
|
|||||||
python3 ./scripts/check_test_wiring.py --check-profile e2e-odm-interop "${NEXTEST_LISTING}"
|
python3 ./scripts/check_test_wiring.py --check-profile e2e-odm-interop "${NEXTEST_LISTING}"
|
||||||
|
|
||||||
- name: Run the interop cases against MinIO
|
- name: Run the interop cases against MinIO
|
||||||
run: cargo nextest run --profile e2e-odm-interop -p e2e_test --no-tests=fail
|
run: python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-odm-interop -p e2e_test --no-tests=fail
|
||||||
|
|
||||||
- name: Build the MinIO interop report
|
- name: Build the MinIO interop report
|
||||||
if: always()
|
if: always()
|
||||||
@@ -251,7 +251,7 @@ jobs:
|
|||||||
|
|
||||||
- name: Build the RustFS binary under test
|
- name: Build the RustFS binary under test
|
||||||
if: steps.credentials.outputs.present == 'true'
|
if: steps.credentials.outputs.present == 'true'
|
||||||
run: cargo build --locked -p rustfs --bins
|
run: python3 scripts/e2e_binary.py build --bins
|
||||||
|
|
||||||
# A filterset that matches nothing is valid, so the count is asserted
|
# A filterset that matches nothing is valid, so the count is asserted
|
||||||
# rather than inferred from a green run.
|
# rather than inferred from a green run.
|
||||||
@@ -272,7 +272,7 @@ jobs:
|
|||||||
- name: Run the three-case minimum
|
- name: Run the three-case minimum
|
||||||
if: steps.credentials.outputs.present == 'true'
|
if: steps.credentials.outputs.present == 'true'
|
||||||
run: |
|
run: |
|
||||||
cargo nextest run --profile e2e-odm-interop -p e2e_test \
|
python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-odm-interop -p e2e_test \
|
||||||
-E "${CLOUD_CASE_FILTER}" --no-tests=fail
|
-E "${CLOUD_CASE_FILTER}" --no-tests=fail
|
||||||
|
|
||||||
- name: Build the ${{ matrix.provider }} interop report
|
- name: Build the ${{ matrix.provider }} interop report
|
||||||
|
|||||||
@@ -152,6 +152,60 @@ jobs:
|
|||||||
else
|
else
|
||||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||||
fi
|
fi
|
||||||
|
STEPS_TABLE="/tmp/rustfs-heal-steps.md"
|
||||||
|
python3 - "${LOG_FILE}" "${STEPS_TABLE}" <<'PY'
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||||
|
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||||
|
step_re = re.compile(r'^\[HEAL-STEP\]\s+(\d+)\s+(.+?)\s+(PASS|FAIL|SKIP)\s*$')
|
||||||
|
ver_re = re.compile(r'^\[HEAL-VERSION\]\s+(\S+)(?:\s+\(node\s+(\S+)\))?\s*$')
|
||||||
|
result_re = re.compile(r'^\[HEAL-RESULT\]\s+(PASS|FAIL)\s+(.*)$')
|
||||||
|
|
||||||
|
steps = {}
|
||||||
|
order = []
|
||||||
|
version = None
|
||||||
|
version_node = None
|
||||||
|
verdict = None
|
||||||
|
verdict_detail = ''
|
||||||
|
try:
|
||||||
|
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||||
|
for raw in fh:
|
||||||
|
line = ansi.sub('', raw).strip()
|
||||||
|
m = step_re.match(line)
|
||||||
|
if m:
|
||||||
|
n, desc, status = m.group(1), m.group(2), m.group(3)
|
||||||
|
if n not in steps:
|
||||||
|
order.append(n)
|
||||||
|
steps[n] = (desc, status) # later lines win (fail after pass)
|
||||||
|
continue
|
||||||
|
m = ver_re.match(line)
|
||||||
|
if m:
|
||||||
|
version, version_node = m.group(1), m.group(2)
|
||||||
|
continue
|
||||||
|
m = result_re.match(line)
|
||||||
|
if m:
|
||||||
|
verdict, verdict_detail = m.group(1), m.group(2)
|
||||||
|
except FileNotFoundError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
with open(out_file, 'w', encoding='utf-8') as out:
|
||||||
|
out.write('## Step Results\n\n')
|
||||||
|
if version:
|
||||||
|
node_note = f' (captured via `rustfs --version` on {version_node})' if version_node else ''
|
||||||
|
out.write(f'- Version under test: **{version}**{node_note}\n')
|
||||||
|
if verdict:
|
||||||
|
out.write(f'- Overall result: **{verdict}** — {verdict_detail}\n')
|
||||||
|
out.write('\n')
|
||||||
|
out.write('| Step | Description | Result |\n')
|
||||||
|
out.write('| --- | --- | --- |\n')
|
||||||
|
for n in sorted(order, key=int):
|
||||||
|
desc, status = steps[n]
|
||||||
|
out.write(f'| {n} | {desc} | {status} |\n')
|
||||||
|
if not order:
|
||||||
|
out.write('| - | - | NOT RUN (no step result lines found) |\n')
|
||||||
|
PY
|
||||||
{
|
{
|
||||||
echo "# RustFS heal test report"
|
echo "# RustFS heal test report"
|
||||||
echo ""
|
echo ""
|
||||||
@@ -160,6 +214,8 @@ jobs:
|
|||||||
echo "- Package: ${PACKAGE_SOURCE}"
|
echo "- Package: ${PACKAGE_SOURCE}"
|
||||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||||
echo ""
|
echo ""
|
||||||
|
cat "${STEPS_TABLE}" || true
|
||||||
|
echo ""
|
||||||
echo "## Log tail"
|
echo "## Log tail"
|
||||||
echo '```text'
|
echo '```text'
|
||||||
tail -n 200 "${LOG_FILE}" || true
|
tail -n 200 "${LOG_FILE}" || true
|
||||||
|
|||||||
@@ -380,6 +380,60 @@ jobs:
|
|||||||
else
|
else
|
||||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||||
fi
|
fi
|
||||||
|
STEPS_TABLE="${POOL_ARTIFACT_DIR}/pool-steps.md"
|
||||||
|
python3 - "${LOG_FILE}" "${STEPS_TABLE}" <<'PY'
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||||
|
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||||
|
step_re = re.compile(r'^\[POOL-STEP\]\s+(\d+)\s+(.+?)\s+(PASS|FAIL|SKIP)\s*$')
|
||||||
|
ver_re = re.compile(r'^\[POOL-VERSION\]\s+(\S+)(?:\s+\(node\s+(\S+)\))?\s*$')
|
||||||
|
result_re = re.compile(r'^\[POOL-RESULT\]\s+(PASS|FAIL)\s+(.*)$')
|
||||||
|
|
||||||
|
steps = {}
|
||||||
|
order = []
|
||||||
|
version = None
|
||||||
|
version_node = None
|
||||||
|
verdict = None
|
||||||
|
verdict_detail = ''
|
||||||
|
try:
|
||||||
|
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||||
|
for raw in fh:
|
||||||
|
line = ansi.sub('', raw).strip()
|
||||||
|
m = step_re.match(line)
|
||||||
|
if m:
|
||||||
|
n, desc, status = m.group(1), m.group(2), m.group(3)
|
||||||
|
if n not in steps:
|
||||||
|
order.append(n)
|
||||||
|
steps[n] = (desc, status) # later lines win (fail after pass)
|
||||||
|
continue
|
||||||
|
m = ver_re.match(line)
|
||||||
|
if m:
|
||||||
|
version, version_node = m.group(1), m.group(2)
|
||||||
|
continue
|
||||||
|
m = result_re.match(line)
|
||||||
|
if m:
|
||||||
|
verdict, verdict_detail = m.group(1), m.group(2)
|
||||||
|
except FileNotFoundError:
|
||||||
|
pass
|
||||||
|
|
||||||
|
with open(out_file, 'w', encoding='utf-8') as out:
|
||||||
|
out.write('## Step Results\n\n')
|
||||||
|
if version:
|
||||||
|
node_note = f' (captured via `rustfs --version` on {version_node})' if version_node else ''
|
||||||
|
out.write(f'- Version under test: **{version}**{node_note}\n')
|
||||||
|
if verdict:
|
||||||
|
out.write(f'- Overall result: **{verdict}** — {verdict_detail}\n')
|
||||||
|
out.write('\n')
|
||||||
|
out.write('| Step | Description | Result |\n')
|
||||||
|
out.write('| --- | --- | --- |\n')
|
||||||
|
for n in sorted(order, key=int):
|
||||||
|
desc, status = steps[n]
|
||||||
|
out.write(f'| {n} | {desc} | {status} |\n')
|
||||||
|
if not order:
|
||||||
|
out.write('| - | - | NOT RUN (no step result lines found) |\n')
|
||||||
|
PY
|
||||||
{
|
{
|
||||||
echo "# RustFS pool expansion test report"
|
echo "# RustFS pool expansion test report"
|
||||||
echo ""
|
echo ""
|
||||||
@@ -389,6 +443,8 @@ jobs:
|
|||||||
echo "- Warp concurrent: ${{ inputs.warp_concurrent || '32' }}"
|
echo "- Warp concurrent: ${{ inputs.warp_concurrent || '32' }}"
|
||||||
echo "- Test Step Outcome: ${{ steps.pool_test.outcome }}"
|
echo "- Test Step Outcome: ${{ steps.pool_test.outcome }}"
|
||||||
echo ""
|
echo ""
|
||||||
|
cat "${STEPS_TABLE}" || true
|
||||||
|
echo ""
|
||||||
echo "## Log tail"
|
echo "## Log tail"
|
||||||
echo '```text'
|
echo '```text'
|
||||||
tail -n 200 "${LOG_FILE}" || true
|
tail -n 200 "${LOG_FILE}" || true
|
||||||
|
|||||||
@@ -74,10 +74,23 @@ env:
|
|||||||
jobs:
|
jobs:
|
||||||
security-test:
|
security-test:
|
||||||
runs-on: smoke-testing
|
runs-on: smoke-testing
|
||||||
continue-on-error: true
|
|
||||||
timeout-minutes: 360
|
timeout-minutes: 360
|
||||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||||
steps:
|
steps:
|
||||||
|
- name: Checkout repository (for the OIDC live gate script)
|
||||||
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||||
|
with:
|
||||||
|
persist-credentials: false
|
||||||
|
|
||||||
|
- name: Initialize security evidence
|
||||||
|
id: evidence
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
umask 077
|
||||||
|
SECURITY_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-security-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
|
||||||
|
mkdir -- "${SECURITY_ARTIFACTS_DIR}"
|
||||||
|
printf 'SECURITY_ARTIFACTS_DIR=%s\n' "${SECURITY_ARTIFACTS_DIR}" >> "${GITHUB_ENV}"
|
||||||
|
|
||||||
# auto-testing is private: clone it with the dedicated PF token (not
|
# auto-testing is private: clone it with the dedicated PF token (not
|
||||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||||
- name: Checkout auto-testing scripts (with retry)
|
- name: Checkout auto-testing scripts (with retry)
|
||||||
@@ -98,11 +111,6 @@ jobs:
|
|||||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||||
exit 1
|
exit 1
|
||||||
|
|
||||||
- name: Checkout repository (for the OIDC live gate script)
|
|
||||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
||||||
with:
|
|
||||||
persist-credentials: false
|
|
||||||
|
|
||||||
- name: Show environment
|
- name: Show environment
|
||||||
run: |
|
run: |
|
||||||
uname -a
|
uname -a
|
||||||
@@ -135,7 +143,8 @@ jobs:
|
|||||||
id: test
|
id: test
|
||||||
continue-on-error: true
|
continue-on-error: true
|
||||||
env:
|
env:
|
||||||
REPORT_FILE: /tmp/rustfs-security-report.md
|
REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/suite-report.md
|
||||||
|
TMPDIR: ${{ env.SECURITY_ARTIFACTS_DIR }}
|
||||||
RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/scripts/test/oidc_keycloak_live.sh
|
RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/scripts/test/oidc_keycloak_live.sh
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
@@ -159,29 +168,48 @@ jobs:
|
|||||||
else
|
else
|
||||||
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||||
fi
|
fi
|
||||||
./auto-testing/rustfs-security-test.sh "${ARGS[@]}"
|
GITHUB_STEP_SUMMARY=/dev/null ./auto-testing/rustfs-security-test.sh "${ARGS[@]}"
|
||||||
|
|
||||||
- name: Generate report
|
- name: Generate report
|
||||||
if: always()
|
id: report
|
||||||
|
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||||
|
env:
|
||||||
|
TEST_OUTCOME: ${{ steps.test.outcome }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
if [ ! -f /tmp/rustfs-security-report.md ]; then
|
RESULT=failure
|
||||||
{
|
if [ "${TEST_OUTCOME}" = "success" ] && [ -s "${SECURITY_ARTIFACTS_DIR}/suite-report.md" ]; then
|
||||||
echo "# RustFS security test report"
|
RESULT=success
|
||||||
echo ""
|
|
||||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
|
||||||
echo "- Trigger: ${{ github.event_name }}"
|
|
||||||
echo "- Test Step Outcome: failure (suite did not produce a report)"
|
|
||||||
} > /tmp/rustfs-security-report.md
|
|
||||||
fi
|
fi
|
||||||
cat /tmp/rustfs-security-report.md >> "${GITHUB_STEP_SUMMARY}"
|
{
|
||||||
|
echo "# RustFS security test report"
|
||||||
|
echo ""
|
||||||
|
echo "- Run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
|
||||||
|
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
|
||||||
|
echo "- Workflow Commit: ${GITHUB_SHA}"
|
||||||
|
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||||
|
echo "- Test Step Outcome: ${RESULT}"
|
||||||
|
echo "- Suite Step Outcome: ${TEST_OUTCOME}"
|
||||||
|
echo ""
|
||||||
|
# The dashboard prioritizes case rows over the step outcome.
|
||||||
|
# Keep partial case results in the artifact when the suite fails.
|
||||||
|
if [ "${RESULT}" = "success" ]; then
|
||||||
|
cat "${SECURITY_ARTIFACTS_DIR}/suite-report.md"
|
||||||
|
elif [ -s "${SECURITY_ARTIFACTS_DIR}/suite-report.md" ]; then
|
||||||
|
echo "The suite did not complete successfully. See suite-report.md in this run's artifact for diagnostics."
|
||||||
|
else
|
||||||
|
echo "The suite did not produce a non-empty report."
|
||||||
|
fi
|
||||||
|
} > "${SECURITY_ARTIFACTS_DIR}/report.md"
|
||||||
|
cat "${SECURITY_ARTIFACTS_DIR}/report.md" >> "${GITHUB_STEP_SUMMARY}"
|
||||||
|
[ "${RESULT}" = "success" ]
|
||||||
|
|
||||||
- name: Upload functional report to dashboard
|
- name: Upload functional report to dashboard
|
||||||
if: always()
|
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||||
continue-on-error: true
|
continue-on-error: true
|
||||||
env:
|
env:
|
||||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||||
REPORT_FILE: /tmp/rustfs-security-report.md
|
REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/report.md
|
||||||
SUITE: security
|
SUITE: security
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
@@ -210,8 +238,9 @@ jobs:
|
|||||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||||
SUITE: 'security'
|
SUITE: 'security'
|
||||||
SUITE_LABEL: 'Security'
|
SUITE_LABEL: 'Security'
|
||||||
|
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
|
||||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||||
REPORT_FILE: '/tmp/rustfs-security-report.md'
|
REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/report.md
|
||||||
LOG_FILE: ''
|
LOG_FILE: ''
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
@@ -245,7 +274,7 @@ jobs:
|
|||||||
echo ""
|
echo ""
|
||||||
echo "## Report (errors and symptoms)"
|
echo "## Report (errors and symptoms)"
|
||||||
echo ""
|
echo ""
|
||||||
if [ -s "${REPORT_FILE}" ]; then
|
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then
|
||||||
redact < "${REPORT_FILE}"
|
redact < "${REPORT_FILE}"
|
||||||
elif [ -s "${LOG_FILE:-}" ]; then
|
elif [ -s "${LOG_FILE:-}" ]; then
|
||||||
echo "(report file missing; log tail below)"
|
echo "(report file missing; log tail below)"
|
||||||
@@ -263,14 +292,12 @@ jobs:
|
|||||||
echo "filed backlog issue for suite ${SUITE}"
|
echo "filed backlog issue for suite ${SUITE}"
|
||||||
|
|
||||||
- name: Upload report and logs
|
- name: Upload report and logs
|
||||||
if: always()
|
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||||
with:
|
with:
|
||||||
name: rustfs-security-test-${{ github.run_id }}
|
name: rustfs-security-test-${{ github.run_id }}-${{ github.run_attempt }}
|
||||||
path: |
|
path: ${{ env.SECURITY_ARTIFACTS_DIR }}/
|
||||||
/tmp/rustfs-security-report.md
|
if-no-files-found: error
|
||||||
/tmp/rustfs-security.*/*
|
|
||||||
if-no-files-found: ignore
|
|
||||||
retention-days: 3
|
retention-days: 3
|
||||||
|
|
||||||
- name: Cleanup environment (after)
|
- name: Cleanup environment (after)
|
||||||
|
|||||||
@@ -18,15 +18,15 @@ on:
|
|||||||
workflow_dispatch:
|
workflow_dispatch:
|
||||||
inputs:
|
inputs:
|
||||||
from_version:
|
from_version:
|
||||||
description: 'OLD RustFS release tag (e.g. 1.0.0-rc.4-preview.1)'
|
description: 'OLD RustFS release tag, e.g. 1.0.0-rc.3 (its release must ship a .deb asset). Leave empty for the default.'
|
||||||
required: false
|
required: false
|
||||||
default: '1.0.0-rc.4-preview.1'
|
default: '1.0.0-rc.3'
|
||||||
from_url:
|
from_url:
|
||||||
description: 'OLD .deb URL. Overrides from_version.'
|
description: 'OLD .deb URL. Overrides from_version.'
|
||||||
required: false
|
required: false
|
||||||
type: string
|
type: string
|
||||||
to_version:
|
to_version:
|
||||||
description: 'NEW RustFS release tag (leave empty for latest nightly)'
|
description: 'NEW RustFS release tag, e.g. 1.0.0-rc.5 (any version with a .deb asset). Leave empty for latest nightly.'
|
||||||
required: false
|
required: false
|
||||||
to_url:
|
to_url:
|
||||||
description: 'NEW .deb URL. Overrides to_version / nightly default.'
|
description: 'NEW .deb URL. Overrides to_version / nightly default.'
|
||||||
@@ -145,6 +145,7 @@ jobs:
|
|||||||
continue-on-error: true
|
continue-on-error: true
|
||||||
env:
|
env:
|
||||||
LOG_FILE: /tmp/rustfs-upgrade.log
|
LOG_FILE: /tmp/rustfs-upgrade.log
|
||||||
|
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||||
run: |
|
run: |
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
chmod +x auto-testing/rustfs-upgrade-test.sh
|
chmod +x auto-testing/rustfs-upgrade-test.sh
|
||||||
@@ -175,6 +176,29 @@ jobs:
|
|||||||
else
|
else
|
||||||
ARGS+=(--to-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
ARGS+=(--to-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||||
fi
|
fi
|
||||||
|
# Fail fast with a clear message when a requested release tag has
|
||||||
|
# no .deb asset (e.g. 1.0.0-rc.4 ships only zips), instead of
|
||||||
|
# letting the suite die mid-run on a 404.
|
||||||
|
check_release_asset() {
|
||||||
|
local version="$1" tag asset url
|
||||||
|
[ -n "${version}" ] && [ "${version}" != "null" ] || return 0
|
||||||
|
tag="${version#v}"
|
||||||
|
asset="rustfs_${tag//-/.}_amd64.deb"
|
||||||
|
url="https://github.com/rustfs/rustfs/releases/download/${tag}/${asset}"
|
||||||
|
if ! gh api "repos/rustfs/rustfs/releases/tags/${tag}" --jq '.assets[].name' 2>/dev/null | grep -qxF "${asset}"; then
|
||||||
|
echo "ERROR: release ${tag} has no downloadable asset ${asset}:" >&2
|
||||||
|
echo " ${url}" >&2
|
||||||
|
echo "Pick a tag whose release ships a .deb (check its release assets)." >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "resolved ${tag} -> ${url}"
|
||||||
|
}
|
||||||
|
if [ -z "${FROM_URL}" ]; then
|
||||||
|
check_release_asset "${FROM_VERSION}"
|
||||||
|
fi
|
||||||
|
if [ -z "${TO_URL}" ]; then
|
||||||
|
check_release_asset "${TO_VERSION}"
|
||||||
|
fi
|
||||||
./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}"
|
./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}"
|
||||||
|
|
||||||
- name: Generate report
|
- name: Generate report
|
||||||
@@ -203,54 +227,75 @@ jobs:
|
|||||||
TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||||
fi
|
fi
|
||||||
CASE_TABLE="/tmp/rustfs-upgrade-cases.md"
|
CASE_TABLE="/tmp/rustfs-upgrade-cases.md"
|
||||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
MATRIX_TABLE="/tmp/rustfs-upgrade-matrix.md"
|
||||||
|
python3 - "${LOG_FILE}" "${CASE_TABLE}" "${MATRIX_TABLE}" <<'PY'
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
log_file, out_file, matrix_file = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||||
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
|
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
|
||||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
|
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
|
||||||
|
topo_re = re.compile(
|
||||||
|
r'^\[UPG-TOPO\]\s+(\S+)\s+(\S+)\s+(\S+)\s+(\S+)\s+PASS=(\d+)\s+FAIL=(\d+)\s*$')
|
||||||
|
|
||||||
rows = []
|
rows = []
|
||||||
index = {}
|
index = {}
|
||||||
|
topo_rows = []
|
||||||
try:
|
try:
|
||||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||||
for raw in fh:
|
for raw in fh:
|
||||||
line = ansi.sub('', raw).strip()
|
line = ansi.sub('', raw).strip()
|
||||||
m = start_re.match(line)
|
m = topo_re.match(line)
|
||||||
if m:
|
if m:
|
||||||
case_id, name = m.group(1), m.group(2)
|
topo_rows.append(m.groups())
|
||||||
if case_id not in index:
|
continue
|
||||||
index[case_id] = len(rows)
|
m = start_re.match(line)
|
||||||
rows.append([case_id, name, 'RUNNING'])
|
if m:
|
||||||
continue
|
case_id, name = m.group(1), m.group(2)
|
||||||
m = done_re.match(line)
|
if case_id not in index:
|
||||||
if m:
|
index[case_id] = len(rows)
|
||||||
status, case_id = m.group(1), m.group(2)
|
rows.append([case_id, name, 'RUNNING'])
|
||||||
if case_id in index:
|
continue
|
||||||
rows[index[case_id]][2] = status
|
m = done_re.match(line)
|
||||||
else:
|
if m:
|
||||||
rows.append([case_id, case_id, status])
|
status, case_id = m.group(1), m.group(2)
|
||||||
index[case_id] = len(rows) - 1
|
if case_id in index:
|
||||||
|
rows[index[case_id]][2] = status
|
||||||
|
else:
|
||||||
|
rows.append([case_id, case_id, status])
|
||||||
|
index[case_id] = len(rows) - 1
|
||||||
except FileNotFoundError:
|
except FileNotFoundError:
|
||||||
rows = []
|
rows = []
|
||||||
|
|
||||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||||
for _, _, status in rows:
|
for _, _, status in rows:
|
||||||
counts[status] = counts.get(status, 0) + 1
|
counts[status] = counts.get(status, 0) + 1
|
||||||
|
|
||||||
with open(out_file, 'w', encoding='utf-8') as out:
|
with open(out_file, 'w', encoding='utf-8') as out:
|
||||||
out.write('## Case Summary\n\n')
|
out.write('## Case Summary\n\n')
|
||||||
out.write(f"- Total: {len(rows)}\\n")
|
out.write(f"- Total: {len(rows)}\\n")
|
||||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||||
out.write('\\n')
|
out.write('\\n')
|
||||||
out.write('| Case | Name | Status |\\n')
|
out.write('| Case | Name | Status |\\n')
|
||||||
out.write('| --- | --- | --- |\\n')
|
out.write('| --- | --- | --- |\\n')
|
||||||
for case_id, name, status in rows:
|
for case_id, name, status in rows:
|
||||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||||
|
|
||||||
|
# Upgrade matrix: one row per topology/backend with the versions
|
||||||
|
# captured on the nodes (rustfs --version) and the aggregated
|
||||||
|
# result. The dashboard renders this table directly.
|
||||||
|
with open(matrix_file, 'w', encoding='utf-8') as out:
|
||||||
|
out.write('## Upgrade Matrix\n\n')
|
||||||
|
out.write('| Topology | KMS Backend | From Version | To Version | Result |\n')
|
||||||
|
out.write('| --- | --- | --- | --- | --- |\n')
|
||||||
|
for topo, backend, old_v, new_v, npass, nfail in topo_rows:
|
||||||
|
result = 'PASS' if nfail == '0' else 'FAIL'
|
||||||
|
out.write(f'| {topo} | {backend} | {old_v} | {new_v} | {result} (PASS={npass} FAIL={nfail}) |\n')
|
||||||
|
if not topo_rows:
|
||||||
|
out.write('| - | - | - | - | NOT RUN (suite failed before upgrade) |\n')
|
||||||
PY
|
PY
|
||||||
{
|
{
|
||||||
echo "# RustFS upgrade compatibility report"
|
echo "# RustFS upgrade compatibility report"
|
||||||
@@ -261,6 +306,8 @@ jobs:
|
|||||||
echo "- To: ${TO_SOURCE}"
|
echo "- To: ${TO_SOURCE}"
|
||||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||||
echo ""
|
echo ""
|
||||||
|
cat "${MATRIX_TABLE}" || true
|
||||||
|
echo ""
|
||||||
cat "${CASE_TABLE}" || true
|
cat "${CASE_TABLE}" || true
|
||||||
echo ""
|
echo ""
|
||||||
echo "## Log tail"
|
echo "## Log tail"
|
||||||
|
|||||||
@@ -42,6 +42,7 @@ jobs:
|
|||||||
- name: Check latest scheduled runs
|
- name: Check latest scheduled runs
|
||||||
env:
|
env:
|
||||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
RUSTFS_DEFAULT_BRANCH: ${{ github.event.repository.default_branch }}
|
||||||
run: |
|
run: |
|
||||||
set +e
|
set +e
|
||||||
python3 scripts/check_scheduled_validation_freshness.py \
|
python3 scripts/check_scheduled_validation_freshness.py \
|
||||||
|
|||||||
@@ -22,7 +22,6 @@ on:
|
|||||||
- "Continuous Integration"
|
- "Continuous Integration"
|
||||||
- "coverage"
|
- "coverage"
|
||||||
- "e2e-nightly"
|
- "e2e-nightly"
|
||||||
- "e2e-distributed"
|
|
||||||
- "e2e-s3tests"
|
- "e2e-s3tests"
|
||||||
- "Fuzz"
|
- "Fuzz"
|
||||||
- "mint"
|
- "mint"
|
||||||
|
|||||||
@@ -33,6 +33,7 @@ profile.json
|
|||||||
*.zst
|
*.zst
|
||||||
.secrets
|
.secrets
|
||||||
*.go
|
*.go
|
||||||
|
!crates/zip/tests/fixtures/snowball/**/generate/*.go
|
||||||
*.pb
|
*.pb
|
||||||
*.svg
|
*.svg
|
||||||
deploy/logs/*.log.*
|
deploy/logs/*.log.*
|
||||||
|
|||||||
Generated
+324
-112
File diff suppressed because it is too large
Load Diff
+10
-6
@@ -168,7 +168,7 @@ reqwest = "0.13.4"
|
|||||||
rustfs-kafka-async = { version = "1.3.1" }
|
rustfs-kafka-async = { version = "1.3.1" }
|
||||||
socket2 = { version = "0.6.5" }
|
socket2 = { version = "0.6.5" }
|
||||||
tokio = { version = "1.53.1" }
|
tokio = { version = "1.53.1" }
|
||||||
tokio-rustls = { default-features = false, version = "0.26.4" }
|
tokio-rustls = { default-features = false, version = "0.26.5" }
|
||||||
tokio-stream = { version = "0.1.19" }
|
tokio-stream = { version = "0.1.19" }
|
||||||
tokio-test = "0.4.5"
|
tokio-test = "0.4.5"
|
||||||
tokio-util = { version = "0.7.19" }
|
tokio-util = { version = "0.7.19" }
|
||||||
@@ -234,15 +234,19 @@ tokio-postgres-rustls = "0.14.0"
|
|||||||
# Utilities and Tools
|
# Utilities and Tools
|
||||||
anyhow = "1.0.104"
|
anyhow = "1.0.104"
|
||||||
arc-swap = "1.9.2"
|
arc-swap = "1.9.2"
|
||||||
# RUSTFS_COMPAT_TODO(tokio-tar-extension-limits): keep the fork pin until every parser hardening used by Snowball is released upstream. Remove after astral-sh/tokio-tar#118 is merged and a published release includes extension, physical-entry, and sparse limits, cancellation-safe sparse parsing, and error-fused entry streams.
|
# RUSTFS_COMPAT_TODO(tokio-tar-extension-limits): keep the fork pin while Snowball and Swift still depend on it. Remove after Snowball uses a released tar-codec/tar-framing API that exposes precedence-resolved MinIO vendor records, RustFS preserves cancellation-safe ownership of large streamed members, footerless minio-go input is accepted only at an authenticated complete request boundary, the existing resource-limit, cancellation, and error-fuse regressions pass, and Swift no longer needs this fork.
|
||||||
astral-tokio-tar = { git = "https://github.com/cxymds/tokio-tar.git", rev = "603756478b7668436e464519c77ccac22a99ba96" }
|
astral-tokio-tar = { git = "https://github.com/cxymds/tokio-tar.git", rev = "603756478b7668436e464519c77ccac22a99ba96" }
|
||||||
|
# Candidate Snowball parser versions exercised by rustfs-zip compatibility fixtures.
|
||||||
|
tar-codec = "0.0.14"
|
||||||
|
tar-framing = "0.0.14"
|
||||||
atoi = "3.1.0"
|
atoi = "3.1.0"
|
||||||
atomic_enum = "0.3.0"
|
atomic_enum = "0.3.0"
|
||||||
aws-config = { version = "1.11.0" }
|
aws-config = { version = "1.12.0" }
|
||||||
aws-credential-types = { version = "1.3.0" }
|
aws-credential-types = { version = "1.3.0" }
|
||||||
aws-sdk-kms = { default-features = false, version = "1.117.0" }
|
aws-sdk-kms = { default-features = false, version = "1.118.0" }
|
||||||
aws-sdk-s3 = { default-features = false, version = "1.144.0" }
|
aws-sdk-s3 = { default-features = false, version = "1.145.0" }
|
||||||
aws-sdk-sts = { default-features = false, version = "1.113.0" }
|
aws-sdk-sts = { default-features = false, version = "1.114.0" }
|
||||||
|
aws-smithy-async = { version = "1.3.0" }
|
||||||
aws-smithy-http-client = { default-features = false, version = "1.4.0" }
|
aws-smithy-http-client = { default-features = false, version = "1.4.0" }
|
||||||
aws-smithy-runtime-api = { version = "1.16.0" }
|
aws-smithy-runtime-api = { version = "1.16.0" }
|
||||||
aws-smithy-types = { version = "1.6.3" }
|
aws-smithy-types = { version = "1.6.3" }
|
||||||
|
|||||||
+33
-48
@@ -1,7 +1,7 @@
|
|||||||
# e2e_test
|
# e2e_test
|
||||||
|
|
||||||
End-to-end test suite for RustFS. Each test spawns a **real `rustfs` binary**
|
End-to-end test suite for RustFS. Each test spawns a **real `rustfs` binary**
|
||||||
(built on demand from the workspace) and drives it over the network with the
|
(built and identified before the test invocation) and drives it over the network with the
|
||||||
AWS SDK (`aws-sdk-s3`), raw HTTP (`reqwest` / `awscurl`), or a protocol client
|
AWS SDK (`aws-sdk-s3`), raw HTTP (`reqwest` / `awscurl`), or a protocol client
|
||||||
(FTPS / WebDAV / SFTP). This is the black-box integration layer: exhaustive
|
(FTPS / WebDAV / SFTP). This is the black-box integration layer: exhaustive
|
||||||
end-to-end behavior lives here, unit behavior stays in the source crates
|
end-to-end behavior lives here, unit behavior stays in the source crates
|
||||||
@@ -26,38 +26,33 @@ Registered in [`src/lib.rs`](src/lib.rs). Grouped by concern:
|
|||||||
| **protocols** | [`src/protocols/`](src/protocols) | FTPS, WebDAV, SFTP compliance. Fixed ports, own guide: [`src/protocols/README.md`](src/protocols/README.md) |
|
| **protocols** | [`src/protocols/`](src/protocols) | FTPS, WebDAV, SFTP compliance. Fixed ports, own guide: [`src/protocols/README.md`](src/protocols/README.md) |
|
||||||
| **reliant** | [`src/reliant/`](src/reliant) | Tests that reuse an **externally started** server (SQL/select, conditional writes, lifecycle, deleted-object reads, node-interact). Run via [`scripts/run_e2e_tests.sh`](../../scripts/run_e2e_tests.sh); see [`src/reliant/README.md`](src/reliant/README.md) |
|
| **reliant** | [`src/reliant/`](src/reliant) | Tests that reuse an **externally started** server (SQL/select, conditional writes, lifecycle, deleted-object reads, node-interact). Run via [`scripts/run_e2e_tests.sh`](../../scripts/run_e2e_tests.sh); see [`src/reliant/README.md`](src/reliant/README.md) |
|
||||||
| **cluster** | `cluster_concurrency_test`, `stale_multipart_cleanup_cluster_test`, `namespace_lock_quorum_test`, `admin_timeout_regression_test`, `object_lambda_test`, `replication_extension_test`, `tier_stats_cluster_test` | Multi-node scenarios via `RustFSTestClusterEnvironment` |
|
| **cluster** | `cluster_concurrency_test`, `stale_multipart_cleanup_cluster_test`, `namespace_lock_quorum_test`, `admin_timeout_regression_test`, `object_lambda_test`, `replication_extension_test`, `tier_stats_cluster_test` | Multi-node scenarios via `RustFSTestClusterEnvironment` |
|
||||||
| **distributed 4×4** | [`src/distributed/`](src/distributed) | Nightly `e2e-distributed` lane: S3, object lock/WORM, versioning, bucket/site replication, quota, expand/decommission/rebalance, concurrency, chaos, 4-node upgrade of historical data and IAM AK/SK. Map: [`docs/testing/distributed-e2e.md`](../../docs/testing/distributed-e2e.md) |
|
|
||||||
| **chaos / reliability** | [`src/chaos.rs`](src/chaos.rs), `reliability_disk_fault_test`, `heal_erasure_disk_rebuild_test`, `server_startup_failfast_test` | Disk offline/replace/corrupt, EC rebuild, heal, fail-fast startup |
|
| **chaos / reliability** | [`src/chaos.rs`](src/chaos.rs), `reliability_disk_fault_test`, `heal_erasure_disk_rebuild_test`, `server_startup_failfast_test` | Disk offline/replace/corrupt, EC rebuild, heal, fail-fast startup |
|
||||||
| **upgrade compatibility** | `upgrade_compatibility_test` | Pinned previous-release writes followed by current-build reads on the same data directory |
|
| **upgrade compatibility** | `upgrade_compatibility_test` | Pinned previous-release writes followed by current-build reads on the same data directory |
|
||||||
|
|
||||||
## How to run
|
## How to run
|
||||||
|
|
||||||
All commands assume repo root. `cargo test` triggers an on-demand build of the
|
All commands assume repo root and Python 3.9 or newer on Linux or macOS. Build the server once through the provenance entry point, then run the test command through the same script:
|
||||||
`rustfs` binary from [`src/common.rs`](src/common.rs) (`rustfs_binary_path`) on
|
|
||||||
first use — the first invocation is slow, later ones reuse the binary.
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Whole crate (default = ignored tests skipped)
|
python3 scripts/e2e_binary.py build --features e2e-test-hooks
|
||||||
cargo nextest run -p e2e_test
|
|
||||||
|
# Whole crate (ignored tests remain skipped)
|
||||||
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run -p e2e_test
|
||||||
|
|
||||||
# One module
|
# One module
|
||||||
cargo nextest run -p e2e_test -E 'test(list_objects_v2_pagination_test)'
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run -p e2e_test -E 'test(list_objects_v2_pagination_test)'
|
||||||
|
|
||||||
# PR smoke subset (see "CI smoke subset" below)
|
|
||||||
cargo nextest run --profile e2e-smoke -p e2e_test
|
|
||||||
|
|
||||||
# ILM serial lane — ignored lifecycle tests, single-threaded (mirrors CI)
|
|
||||||
cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \
|
|
||||||
-E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))'
|
|
||||||
|
|
||||||
|
# PR smoke subset
|
||||||
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-smoke -p e2e_test
|
||||||
```
|
```
|
||||||
|
|
||||||
The protocols suite has its own contract (fixed bind ports 9022–9301,
|
`build` records the source contents, HEAD, resolved Cargo features, profile, toolchain, and binary SHA-256 beside the executable in `rustfs.e2e.json`. `run` validates that identity before and after the command, preserves command failures, and removes its temporary run receipt on completion. The Rust harness checks that receipt before starting each server; it never compiles a server inside a test process. Source or binary changes during a run invalidate the result, even when the test command succeeds. Use an isolated worktree and keep it unchanged until the command finishes.
|
||||||
single-worker execution, feature-gated scheduling) documented in
|
|
||||||
[`src/protocols/README.md`](src/protocols/README.md). `RUSTFS_BUILD_FEATURES`
|
The additional `--features` arguments must match between `build` and `run`; Cargo defaults remain enabled. The wrapper supplies `RUSTFS_BUILD_FEATURES` from Cargo's resolved feature list, including features enabled by `full`. Protocol helpers require a subset of that list. `CARGO_TARGET_DIR` and `--profile release` are supported. An in-workspace target directory must be Git-ignored; tracked files are always included in the source identity. `build --bins` preserves CI lanes that compile all RustFS binary targets. For a downloaded artifact, copy both the executable and its sidecar, then use `run`; do not generate a new identity for an arbitrary prebuilt binary. `CARGO_BIN_EXE_rustfs` cannot override the verified executable.
|
||||||
selects which features the spawned binary is built with; leave it unset to run
|
|
||||||
every protocol entry. Use the exact profile command under
|
Each build/run holds an exclusive `rustfs.e2e.lock` marker beside the binary; concurrent wrappers fail immediately. Use a private target directory and do not run ordinary Cargo builds against it while tests are active: Cargo does not honor this marker. Interrupted runs fail and terminate their command group. After an uncatchable kill, inspect the PID recorded in a leftover marker and remove it only after confirming its owner has stopped. Embedded file symlinks are hashed through their target; embedded directory symlinks are rejected because their contents cannot be enumerated safely by this entry point.
|
||||||
[Troubleshooting](#troubleshooting) for CI-equivalent execution.
|
|
||||||
|
The protocols suite has its own fixed-port and single-worker contract in [`src/protocols/README.md`](src/protocols/README.md). Use its command under [Troubleshooting](#troubleshooting).
|
||||||
|
|
||||||
### `#[ignore]` semantics
|
### `#[ignore]` semantics
|
||||||
|
|
||||||
@@ -123,7 +118,7 @@ via `create_s3_client(idx)` / `create_all_clients()`. See
|
|||||||
| `wait_for_server_ready` | Poll readiness before issuing requests |
|
| `wait_for_server_ready` | Poll readiness before issuing requests |
|
||||||
| `create_s3_client` / `create_test_bucket` / `delete_test_bucket` | aws-sdk-s3 client + bucket lifecycle |
|
| `create_s3_client` / `create_test_bucket` / `delete_test_bucket` | aws-sdk-s3 client + bucket lifecycle |
|
||||||
| `find_available_port` | Random free port (isolation primitive) |
|
| `find_available_port` | Random free port (isolation primitive) |
|
||||||
| `rustfs_binary_path` / `_with_features` | Locate/build the binary; honors `RUSTFS_BUILD_FEATURES` |
|
| `rustfs_binary_path` / `_with_features` | Verify this run's binary receipt and required feature subset |
|
||||||
| `requested_rustfs_build_features` / `rustfs_build_feature_enabled` | Feature-gate a test to what the binary was built with |
|
| `requested_rustfs_build_features` / `rustfs_build_feature_enabled` | Feature-gate a test to what the binary was built with |
|
||||||
| `execute_awscurl` / `awscurl_post` / `_get` / `_put` / `_delete` / `awscurl_post_sts_form_urlencoded` | Admin/STS API calls via `awscurl`; missing binaries are test failures |
|
| `execute_awscurl` / `awscurl_post` / `_get` / `_put` / `_delete` / `awscurl_post_sts_form_urlencoded` | Admin/STS API calls via `awscurl`; missing binaries are test failures |
|
||||||
| `replication_fast_env` | Env vars that shrink replication timers (from repl-4); pass to `start_rustfs_server_with_env` |
|
| `replication_fast_env` | Env vars that shrink replication timers (from repl-4); pass to `start_rustfs_server_with_env` |
|
||||||
@@ -172,7 +167,6 @@ the same profile for membership and execution with one nightly worker.
|
|||||||
| KMS suite | `e2e-full` job, merge queue + main | **Active** |
|
| KMS suite | `e2e-full` job, merge queue + main | **Active** |
|
||||||
| Direct and mixed-version rolling upgrades from pinned previous release | `e2e-upgrade.yml`, storage-sensitive PRs + release tags + weekly | **Active** |
|
| Direct and mixed-version rolling upgrades from pinned previous release | `e2e-upgrade.yml`, storage-sensitive PRs + release tags + weekly | **Active** |
|
||||||
| Cluster faults (`e2e-nightly` profile) | consolidated nightly workflow | **Active** (backlog#1149 ci-7) |
|
| Cluster faults (`e2e-nightly` profile) | consolidated nightly workflow | **Active** (backlog#1149 ci-7) |
|
||||||
| Distributed 4-node 4-disk (`e2e-distributed` profile) | `.github/workflows/e2e-distributed.yml` | **Active** (nightly / dispatch; not a merge gate) |
|
|
||||||
| Protocols (FTPS/WebDAV/SFTP) | consolidated nightly workflow, serial | **Active** (backlog#1149 ci-7) |
|
| Protocols (FTPS/WebDAV/SFTP) | consolidated nightly workflow, serial | **Active** (backlog#1149 ci-7) |
|
||||||
| Replication (fast subset) | `e2e-smoke` profile, `e2e-tests` job, every PR | **Active** (backlog#1147 repl-1) |
|
| Replication (fast subset) | `e2e-smoke` profile, `e2e-tests` job, every PR | **Active** (backlog#1147 repl-1) |
|
||||||
| Replication (slow + multi-node) | `e2e-repl-nightly` profile, consolidated nightly workflow | **Active** (backlog#1147 repl-1) |
|
| Replication (slow + multi-node) | `e2e-repl-nightly` profile, consolidated nightly workflow | **Active** (backlog#1147 repl-1) |
|
||||||
@@ -187,35 +181,26 @@ the wiring source of truth. Committed test-ID digests under
|
|||||||
**Reproduce a CI failure locally** — run the exact profile/lane:
|
**Reproduce a CI failure locally** — run the exact profile/lane:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
# Smoke (e2e-tests job) — includes the 20 fast replication tests
|
# Smoke, full, and cluster lanes share a server with fault-test hooks.
|
||||||
cargo nextest run --profile e2e-smoke -p e2e_test
|
python3 scripts/e2e_binary.py build --features e2e-test-hooks
|
||||||
# Full single-node merge/main lane
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-smoke -p e2e_test
|
||||||
cargo nextest run --profile e2e-full -p e2e_test
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-full -p e2e_test
|
||||||
# Cluster fault nightly lane
|
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-nightly -p e2e_test
|
||||||
cargo nextest run --profile e2e-nightly -p e2e_test
|
|
||||||
# 4-node 4-disk distributed lane (S3 / lock / versioning / replication / decommission / chaos / upgrade)
|
# Replication nightly uses the default server; awscurl is required for STS.
|
||||||
# Upgrade cases need RUSTFS_UPGRADE_SOURCE_BINARY; without it they fail closed.
|
python3 scripts/e2e_binary.py build
|
||||||
cargo nextest run --profile e2e-distributed -p e2e_test
|
python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-repl-nightly -p e2e_test
|
||||||
# Replication nightly lane; awscurl is required for STS paths
|
|
||||||
cargo nextest run --profile e2e-repl-nightly -p e2e_test
|
# Protocol nightly owns fixed ports.
|
||||||
# Fixed-port protocol nightly lane
|
python3 scripts/e2e_binary.py build --features ftps,webdav,sftp
|
||||||
RUSTFS_BUILD_FEATURES=ftps,webdav,sftp \
|
python3 scripts/e2e_binary.py run --features ftps,webdav,sftp -- cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
|
||||||
cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
|
|
||||||
# ILM serial lane
|
# The ILM serial lane does not use this server harness.
|
||||||
cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \
|
cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \
|
||||||
-E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))'
|
-E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))'
|
||||||
# s3s-e2e black box
|
|
||||||
./scripts/e2e-run.sh ./target/debug/rustfs /tmp/rustfs-e2e-data
|
|
||||||
```
|
```
|
||||||
|
|
||||||
**Stale binary.** Tests build the `rustfs` binary once and reuse it. To avoid
|
**Stale or unverified binary.** Re-run the matching `build` command after changing source or features, then invoke tests through `run`. A missing receipt, copied old executable, or mismatched build identity is a prerequisite failure. Bare Cargo invocations that start a server deliberately fail; unit tests that do not start a server can still run directly.
|
||||||
rebuilding while iterating on tests, `common.rs` reuses an existing binary when
|
|
||||||
running *inside* the e2e test process even if sources changed
|
|
||||||
(`can_reuse_inside_e2e`, [`src/common.rs`](src/common.rs) line 98). Downside: if
|
|
||||||
you changed **server** code, force a rebuild with
|
|
||||||
`cargo build -p rustfs` (or `touch` a source file outside the reuse window)
|
|
||||||
before re-running, or CI's freshly built artifact will diverge from your local
|
|
||||||
one.
|
|
||||||
|
|
||||||
**Port already in use / orphan processes.** A hard-killed run can leak a
|
**Port already in use / orphan processes.** A hard-killed run can leak a
|
||||||
`rustfs` child holding its port. Find and kill it:
|
`rustfs` child holding its port. Find and kill it:
|
||||||
|
|||||||
+128
-227
@@ -31,7 +31,6 @@ use rustfs_signer::constants::UNSIGNED_PAYLOAD;
|
|||||||
use rustfs_signer::sign_v4;
|
use rustfs_signer::sign_v4;
|
||||||
use s3s::Body;
|
use s3s::Body;
|
||||||
use serde_json;
|
use serde_json;
|
||||||
use std::ffi::OsStr;
|
|
||||||
use std::fs as stdfs;
|
use std::fs as stdfs;
|
||||||
use std::io::ErrorKind;
|
use std::io::ErrorKind;
|
||||||
use std::net::SocketAddr;
|
use std::net::SocketAddr;
|
||||||
@@ -44,7 +43,6 @@ use tokio::net::TcpStream;
|
|||||||
use tokio::time::sleep;
|
use tokio::time::sleep;
|
||||||
use tracing::{error, info, warn};
|
use tracing::{error, info, warn};
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
use walkdir::WalkDir;
|
|
||||||
|
|
||||||
// Common constants for all E2E tests
|
// Common constants for all E2E tests
|
||||||
pub const DEFAULT_ACCESS_KEY: &str = "rustfsadmin";
|
pub const DEFAULT_ACCESS_KEY: &str = "rustfsadmin";
|
||||||
@@ -365,59 +363,75 @@ fn resolve_rustfs_binary_path(workspace: &Path, configured_target_dir: Option<&P
|
|||||||
path
|
path
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve the RustFS binary relative to the workspace, optionally requesting build features.
|
/// Resolve the server verified by `scripts/e2e_binary.py run` for this test invocation.
|
||||||
|
/// Requested features are a required subset of the server's resolved Cargo features.
|
||||||
pub fn rustfs_binary_path_with_features(requested_features: Option<&str>) -> PathBuf {
|
pub fn rustfs_binary_path_with_features(requested_features: Option<&str>) -> PathBuf {
|
||||||
if let Some(path) = std::env::var_os("CARGO_BIN_EXE_rustfs") {
|
|
||||||
return PathBuf::from(path);
|
|
||||||
}
|
|
||||||
let requested_features = requested_features.and_then(normalize_rustfs_build_features);
|
|
||||||
|
|
||||||
let workspace = workspace_root();
|
let workspace = workspace_root();
|
||||||
let configured_target_dir = std::env::var_os("CARGO_TARGET_DIR").map(PathBuf::from);
|
let configured_target_dir = std::env::var_os("CARGO_TARGET_DIR").map(PathBuf::from);
|
||||||
let binary_path = resolve_rustfs_binary_path(&workspace, configured_target_dir.as_deref());
|
let binary_path = std::env::var_os("CARGO_BIN_EXE_rustfs")
|
||||||
|
.map(PathBuf::from)
|
||||||
|
.unwrap_or_else(|| resolve_rustfs_binary_path(&workspace, configured_target_dir.as_deref()));
|
||||||
|
let receipt_path = std::env::var_os("RUSTFS_E2E_BINARY_RECEIPT").map(PathBuf::from);
|
||||||
|
receipt_path
|
||||||
|
.ok_or_else(|| std::io::Error::new(ErrorKind::NotFound, "missing E2E run receipt"))
|
||||||
|
.and_then(|receipt| verify_e2e_binary_receipt(&receipt, &workspace, &binary_path, requested_features))
|
||||||
|
.unwrap_or_else(|error| {
|
||||||
|
panic!(
|
||||||
|
"E2E server prerequisite failed: {error}. Build with `python3 scripts/e2e_binary.py build --features <features>` and run tests with `python3 scripts/e2e_binary.py run --features <features> -- cargo nextest run ...`"
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
let features_match = binary_features_match(&binary_path, requested_features.as_deref());
|
#[derive(serde::Deserialize)]
|
||||||
let source_is_newer = workspace_sources_newer_than_binary(&binary_path);
|
#[serde(deny_unknown_fields)]
|
||||||
let can_reuse_inside_e2e = running_inside_e2e_test_binary() && requested_features.is_none() && features_match;
|
struct E2eBinaryReceipt {
|
||||||
if binary_path.is_file() && features_match && (!source_is_newer || can_reuse_inside_e2e) {
|
schema: u32,
|
||||||
if source_is_newer {
|
workspace: PathBuf,
|
||||||
warn!(
|
binary: PathBuf,
|
||||||
"RustFS binary at {:?} appears older than workspace sources; reusing it inside cargo test to avoid nested builds",
|
size: u64,
|
||||||
binary_path
|
modified_ns: u128,
|
||||||
);
|
features: Vec<String>,
|
||||||
}
|
}
|
||||||
info!("Using existing RustFS binary at {:?}", binary_path);
|
|
||||||
return binary_path;
|
fn verify_e2e_binary_receipt(
|
||||||
|
receipt_path: &Path,
|
||||||
|
workspace: &Path,
|
||||||
|
binary_path: &Path,
|
||||||
|
requested_features: Option<&str>,
|
||||||
|
) -> std::io::Result<PathBuf> {
|
||||||
|
let receipt: E2eBinaryReceipt = serde_json::from_slice(&stdfs::read(receipt_path)?)?;
|
||||||
|
let binary = binary_path.canonicalize()?;
|
||||||
|
let metadata = binary.metadata()?;
|
||||||
|
let modified_ns = metadata
|
||||||
|
.modified()?
|
||||||
|
.duration_since(std::time::UNIX_EPOCH)
|
||||||
|
.map_err(std::io::Error::other)?
|
||||||
|
.as_nanos();
|
||||||
|
// The runner hashes source and binary before/after the entire suite. Each
|
||||||
|
// nextest process checks only this invocation's path, features, and file stat.
|
||||||
|
if receipt.schema != 1
|
||||||
|
|| receipt.workspace != workspace.canonicalize()?
|
||||||
|
|| receipt.binary != binary
|
||||||
|
|| !metadata.is_file()
|
||||||
|
|| receipt.size != metadata.len()
|
||||||
|
|| receipt.modified_ns != modified_ns
|
||||||
|
{
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
ErrorKind::InvalidData,
|
||||||
|
"E2E server differs from this run's verified binary",
|
||||||
|
));
|
||||||
}
|
}
|
||||||
|
if let Some(requested) = requested_features.and_then(normalize_rustfs_build_features)
|
||||||
info!("Building RustFS binary to ensure it's up to date...");
|
&& requested
|
||||||
build_rustfs_binary(requested_features.as_deref(), &binary_path);
|
.split(',')
|
||||||
|
.any(|feature| !receipt.features.iter().any(|actual| actual == feature))
|
||||||
info!("Using RustFS binary at {:?}", binary_path);
|
{
|
||||||
binary_path
|
return Err(std::io::Error::new(
|
||||||
}
|
ErrorKind::InvalidInput,
|
||||||
|
"E2E server is missing a requested build feature",
|
||||||
fn workspace_sources_newer_than_binary(binary_path: &PathBuf) -> bool {
|
));
|
||||||
let Ok(binary_meta) = std::fs::metadata(binary_path) else {
|
}
|
||||||
return true;
|
Ok(binary)
|
||||||
};
|
|
||||||
let Ok(binary_modified) = binary_meta.modified() else {
|
|
||||||
return true;
|
|
||||||
};
|
|
||||||
|
|
||||||
let workspace = workspace_root();
|
|
||||||
let watch_roots = [
|
|
||||||
workspace.join("Cargo.toml"),
|
|
||||||
workspace.join("Cargo.lock"),
|
|
||||||
workspace.join("rustfs"),
|
|
||||||
workspace.join("crates"),
|
|
||||||
];
|
|
||||||
|
|
||||||
watch_roots.iter().any(|path| path_is_newer_than(binary_modified, path))
|
|
||||||
}
|
|
||||||
|
|
||||||
fn running_inside_e2e_test_binary() -> bool {
|
|
||||||
std::env::var("CARGO_PKG_NAME").is_ok_and(|value| value == "e2e_test")
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn requested_rustfs_build_features() -> Option<String> {
|
pub fn requested_rustfs_build_features() -> Option<String> {
|
||||||
@@ -447,96 +461,6 @@ pub fn rustfs_build_feature_enabled(requested_features: Option<&str>, required_f
|
|||||||
.any(|feature| feature.eq_ignore_ascii_case(RUSTFS_FULL_FEATURE) || feature.eq_ignore_ascii_case(required_feature))
|
.any(|feature| feature.eq_ignore_ascii_case(RUSTFS_FULL_FEATURE) || feature.eq_ignore_ascii_case(required_feature))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn rustfs_binary_features_stamp_path(binary_path: &Path) -> PathBuf {
|
|
||||||
binary_path.with_extension("features")
|
|
||||||
}
|
|
||||||
|
|
||||||
fn binary_features_match(binary_path: &Path, requested_features: Option<&str>) -> bool {
|
|
||||||
let stamp_path = rustfs_binary_features_stamp_path(binary_path);
|
|
||||||
let recorded = stdfs::read_to_string(stamp_path)
|
|
||||||
.ok()
|
|
||||||
.and_then(|value| normalize_rustfs_build_features(&value));
|
|
||||||
let requested = requested_features.and_then(normalize_rustfs_build_features);
|
|
||||||
|
|
||||||
match requested.as_deref() {
|
|
||||||
Some(features) => recorded.as_deref() == Some(features),
|
|
||||||
None => recorded.is_none(),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn path_is_newer_than(binary_modified: std::time::SystemTime, path: &Path) -> bool {
|
|
||||||
if path.is_file() {
|
|
||||||
return std::fs::metadata(path)
|
|
||||||
.and_then(|meta| meta.modified())
|
|
||||||
.map(|modified| modified > binary_modified)
|
|
||||||
.unwrap_or(false);
|
|
||||||
}
|
|
||||||
|
|
||||||
if !path.is_dir() {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
WalkDir::new(path)
|
|
||||||
.into_iter()
|
|
||||||
.filter_entry(|entry| {
|
|
||||||
let name = entry.file_name();
|
|
||||||
name != OsStr::new("target") && name != OsStr::new(".git")
|
|
||||||
})
|
|
||||||
.filter_map(Result::ok)
|
|
||||||
.filter(|entry| entry.file_type().is_file())
|
|
||||||
.any(|entry| {
|
|
||||||
std::fs::metadata(entry.path())
|
|
||||||
.and_then(|meta| meta.modified())
|
|
||||||
.map(|modified| modified > binary_modified)
|
|
||||||
.unwrap_or(false)
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Build the RustFS binary using cargo
|
|
||||||
fn build_rustfs_binary(requested_features: Option<&str>, binary_path: &Path) {
|
|
||||||
let workspace = workspace_root();
|
|
||||||
info!("Building RustFS binary from workspace: {:?}", workspace);
|
|
||||||
|
|
||||||
let _profile = if cfg!(debug_assertions) {
|
|
||||||
info!("Building in debug mode");
|
|
||||||
"dev"
|
|
||||||
} else {
|
|
||||||
info!("Building in release mode");
|
|
||||||
"release"
|
|
||||||
};
|
|
||||||
|
|
||||||
let mut cmd = Command::new("cargo");
|
|
||||||
cmd.current_dir(&workspace).args(["build", "--bin", "rustfs"]);
|
|
||||||
|
|
||||||
if let Some(features) = requested_features {
|
|
||||||
cmd.arg("--features").arg(features);
|
|
||||||
info!("Building with features: {}", features);
|
|
||||||
}
|
|
||||||
|
|
||||||
if !cfg!(debug_assertions) {
|
|
||||||
cmd.arg("--release");
|
|
||||||
}
|
|
||||||
|
|
||||||
info!(
|
|
||||||
"Executing: cargo build --bin rustfs {}",
|
|
||||||
if cfg!(debug_assertions) { "" } else { "--release" }
|
|
||||||
);
|
|
||||||
|
|
||||||
let output = cmd.output().expect("Failed to execute cargo build command");
|
|
||||||
|
|
||||||
if !output.status.success() {
|
|
||||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
|
||||||
panic!("Failed to build RustFS binary. Error: {stderr}");
|
|
||||||
}
|
|
||||||
|
|
||||||
let stamp_path = rustfs_binary_features_stamp_path(binary_path);
|
|
||||||
if let Err(err) = stdfs::write(&stamp_path, requested_features.unwrap_or_default()) {
|
|
||||||
warn!("Failed to write RustFS feature stamp {:?}: {}", stamp_path, err);
|
|
||||||
}
|
|
||||||
|
|
||||||
info!("✅ RustFS binary built successfully");
|
|
||||||
}
|
|
||||||
|
|
||||||
fn awscurl_binary_path() -> PathBuf {
|
fn awscurl_binary_path() -> PathBuf {
|
||||||
std::env::var_os("AWSCURL_PATH")
|
std::env::var_os("AWSCURL_PATH")
|
||||||
.map(PathBuf::from)
|
.map(PathBuf::from)
|
||||||
@@ -1483,9 +1407,8 @@ impl RustFSTestClusterEnvironment {
|
|||||||
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
||||||
}
|
}
|
||||||
|
|
||||||
for i in 0..self.nodes.len() {
|
for (i, node) in self.nodes.iter().enumerate() {
|
||||||
let address = self.nodes[i].address.clone();
|
self.wait_for_node_ready(&node.address, i).await?;
|
||||||
self.wait_for_node_ready(&address, i).await?;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
for node_idx in 0..self.nodes.len() {
|
for node_idx in 0..self.nodes.len() {
|
||||||
@@ -1511,8 +1434,7 @@ impl RustFSTestClusterEnvironment {
|
|||||||
let volumes_arg = self.build_volumes_arg();
|
let volumes_arg = self.build_volumes_arg();
|
||||||
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
||||||
|
|
||||||
let address = self.nodes[node_idx].address.clone();
|
self.wait_for_node_ready(&self.nodes[node_idx].address, node_idx).await?;
|
||||||
self.wait_for_node_ready(&address, node_idx).await?;
|
|
||||||
self.wait_for_node_service_ready(node_idx).await?;
|
self.wait_for_node_service_ready(node_idx).await?;
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
@@ -1561,18 +1483,8 @@ impl RustFSTestClusterEnvironment {
|
|||||||
///
|
///
|
||||||
/// Attempts to establish a TCP connection to the node's address, retries up to 60 times
|
/// Attempts to establish a TCP connection to the node's address, retries up to 60 times
|
||||||
/// with a 1-second interval between attempts. Fails if the port is unreachable after all retries.
|
/// with a 1-second interval between attempts. Fails if the port is unreachable after all retries.
|
||||||
fn node_process_exited(&mut self, idx: usize) -> Result<bool, Box<dyn std::error::Error + Send + Sync>> {
|
async fn wait_for_node_ready(&self, address: &str, idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
let Some(process) = self.nodes.get_mut(idx).and_then(|node| node.process.as_mut()) else {
|
|
||||||
return Ok(true);
|
|
||||||
};
|
|
||||||
Ok(process.try_wait()?.is_some())
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn wait_for_node_ready(&mut self, address: &str, idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
|
||||||
for attempt in 0..60 {
|
for attempt in 0..60 {
|
||||||
if self.node_process_exited(idx)? {
|
|
||||||
return Err(format!("cluster node {idx} process exited before TCP ready").into());
|
|
||||||
}
|
|
||||||
if TcpStream::connect(address).await.is_ok() {
|
if TcpStream::connect(address).await.is_ok() {
|
||||||
info!("Node {} ({}) TCP ready after {} attempts", idx, address, attempt + 1);
|
info!("Node {} ({}) TCP ready after {} attempts", idx, address, attempt + 1);
|
||||||
return Ok(());
|
return Ok(());
|
||||||
@@ -1586,13 +1498,10 @@ impl RustFSTestClusterEnvironment {
|
|||||||
///
|
///
|
||||||
/// Verifies service availability by calling the S3 `list_buckets` API against the requested node,
|
/// Verifies service availability by calling the S3 `list_buckets` API against the requested node,
|
||||||
/// retries up to 120 times with a 1-second interval between attempts.
|
/// retries up to 120 times with a 1-second interval between attempts.
|
||||||
async fn wait_for_node_service_ready(&mut self, node_idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
async fn wait_for_node_service_ready(&self, node_idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||||
let client = self.create_s3_client(node_idx)?;
|
let client = self.create_s3_client(node_idx)?;
|
||||||
|
|
||||||
for attempt in 0..120 {
|
for attempt in 0..120 {
|
||||||
if self.node_process_exited(node_idx)? {
|
|
||||||
return Err(format!("cluster node {node_idx} process exited before S3 ready").into());
|
|
||||||
}
|
|
||||||
match client.list_buckets().send().await {
|
match client.list_buckets().send().await {
|
||||||
Ok(_) => {
|
Ok(_) => {
|
||||||
info!("Cluster node {} service ready after {} attempts", node_idx, attempt + 1);
|
info!("Cluster node {} service ready after {} attempts", node_idx, attempt + 1);
|
||||||
@@ -1715,64 +1624,6 @@ impl RustFSTestClusterEnvironment {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Append a new single-node erasure pool to a stopped multi-pool cluster.
|
|
||||||
///
|
|
||||||
/// Used to simulate pool expansion on localhost: every pool already owns
|
|
||||||
/// exactly one node with `drives_per_node >= 2` (the only multi-pool layout
|
|
||||||
/// the single-host `RUSTFS_VOLUMES` syntax can express). The new node is
|
|
||||||
/// allocated a fresh port and empty drive directories; callers must
|
|
||||||
/// [`Self::start`] afterwards so every process picks up the extended
|
|
||||||
/// volumes argument. Existing data directories are left untouched.
|
|
||||||
pub async fn append_single_node_pool(&mut self) -> Result<usize, Box<dyn std::error::Error + Send + Sync>> {
|
|
||||||
if self.nodes.iter().any(|node| node.process.is_some()) {
|
|
||||||
return Err("stop the cluster before appending a pool".into());
|
|
||||||
}
|
|
||||||
if self.topology.drives_per_node < 2 {
|
|
||||||
return Err(
|
|
||||||
"append_single_node_pool requires drives_per_node >= 2 (the server parser rejects a single-drive ellipses pool)"
|
|
||||||
.into(),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut pools = self.topology.normalized_pools();
|
|
||||||
for (pool_idx, nodes) in pools.iter().enumerate() {
|
|
||||||
if nodes.len() != 1 {
|
|
||||||
return Err(format!(
|
|
||||||
"pool {pool_idx} spans {} nodes; append_single_node_pool requires one node per pool",
|
|
||||||
nodes.len()
|
|
||||||
)
|
|
||||||
.into());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
let new_idx = self.nodes.len();
|
|
||||||
let port = RustFSTestEnvironment::find_available_port().await?;
|
|
||||||
let address = format!("127.0.0.1:{port}");
|
|
||||||
let data_dirs: Vec<String> = (0..self.topology.drives_per_node)
|
|
||||||
.map(|drive| format!("{}/node{}/drive{}", self.temp_dir, new_idx, drive))
|
|
||||||
.collect();
|
|
||||||
for dir in &data_dirs {
|
|
||||||
fs::create_dir_all(dir).await?;
|
|
||||||
}
|
|
||||||
|
|
||||||
self.nodes.push(ClusterNode {
|
|
||||||
url: format!("http://{address}"),
|
|
||||||
address,
|
|
||||||
data_dir: data_dirs[0].clone(),
|
|
||||||
data_dirs,
|
|
||||||
pool_idx: pools.len(),
|
|
||||||
process: None,
|
|
||||||
});
|
|
||||||
pools.push(vec![new_idx]);
|
|
||||||
self.topology.node_count = self.nodes.len();
|
|
||||||
self.topology.pools = pools;
|
|
||||||
self.node_extra_env.push(Vec::new());
|
|
||||||
self.node_capture_log_paths.push(None);
|
|
||||||
self.volume_proxy_addresses.push(None);
|
|
||||||
|
|
||||||
Ok(new_idx)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Gracefully stop one cluster node and wait for its process to exit.
|
/// Gracefully stop one cluster node and wait for its process to exit.
|
||||||
///
|
///
|
||||||
/// This is intentionally separate from [`Self::stop_node`]: the latter is
|
/// This is intentionally separate from [`Self::stop_node`]: the latter is
|
||||||
@@ -2146,16 +1997,66 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn binary_feature_stamp_matching_uses_normalized_features() {
|
fn explicit_binary_without_run_receipt_is_rejected() {
|
||||||
let binary_path = std::env::temp_dir().join(format!("rustfs-feature-stamp-test-{}", Uuid::new_v4()));
|
const CHILD_ENV: &str = "RUSTFS_E2E_RECEIPT_TEST_CHILD";
|
||||||
let stamp_path = rustfs_binary_features_stamp_path(&binary_path);
|
if std::env::var_os(CHILD_ENV).is_some() {
|
||||||
|
rustfs_binary_path_with_features(None);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let executable = std::env::current_exe().expect("locate isolated test process");
|
||||||
|
let output = Command::new(&executable)
|
||||||
|
.args([
|
||||||
|
"--exact",
|
||||||
|
"common::tests::explicit_binary_without_run_receipt_is_rejected",
|
||||||
|
"--nocapture",
|
||||||
|
])
|
||||||
|
.env(CHILD_ENV, "1")
|
||||||
|
.env("CARGO_BIN_EXE_rustfs", &executable)
|
||||||
|
.env_remove("RUSTFS_E2E_BINARY_RECEIPT")
|
||||||
|
.output()
|
||||||
|
.expect("run the missing-receipt scenario with isolated environment variables");
|
||||||
|
assert!(!output.status.success(), "an explicit binary must not bypass run verification");
|
||||||
|
assert!(String::from_utf8_lossy(&output.stderr).contains("missing E2E run receipt"));
|
||||||
|
}
|
||||||
|
|
||||||
stdfs::write(&stamp_path, " SFTP, ftps ").expect("write feature stamp");
|
#[test]
|
||||||
assert!(binary_features_match(&binary_path, Some("sftp,ftps")));
|
fn e2e_run_receipt_rejects_replaced_binary_and_missing_features() {
|
||||||
assert!(binary_features_match(&binary_path, Some(" SFTP, FTPS ")));
|
let directory = std::env::temp_dir().join(format!("rustfs-e2e-receipt-test-{}", Uuid::new_v4()));
|
||||||
assert!(!binary_features_match(&binary_path, Some("sftp")));
|
stdfs::create_dir(&directory).expect("create receipt fixture");
|
||||||
|
let binary = directory.join("rustfs");
|
||||||
stdfs::remove_file(stamp_path).ok();
|
let receipt = directory.join("receipt.json");
|
||||||
|
stdfs::write(&binary, "server").expect("write fixture binary");
|
||||||
|
let metadata = binary.metadata().expect("stat fixture binary");
|
||||||
|
let record = serde_json::json!({
|
||||||
|
"schema": 1,
|
||||||
|
"workspace": directory.canonicalize().expect("canonical workspace"),
|
||||||
|
"binary": binary.canonicalize().expect("canonical binary"),
|
||||||
|
"size": metadata.len(),
|
||||||
|
"modified_ns": metadata.modified().expect("modified time").duration_since(std::time::UNIX_EPOCH).expect("positive timestamp").as_nanos(),
|
||||||
|
"features": ["default", "full", "ftps", "webdav", "sftp"]
|
||||||
|
});
|
||||||
|
stdfs::write(&receipt, serde_json::to_vec(&record).expect("serialize receipt")).expect("write receipt");
|
||||||
|
verify_e2e_binary_receipt(&receipt, &directory, &binary, Some("sftp,webdav")).expect("resolved feature subset");
|
||||||
|
verify_e2e_binary_receipt(&receipt, &directory, &binary, Some("full")).expect("full was actually requested");
|
||||||
|
assert_eq!(
|
||||||
|
verify_e2e_binary_receipt(&receipt, &directory, &binary, Some("rio-v2"))
|
||||||
|
.expect_err("full does not enable rio-v2")
|
||||||
|
.kind(),
|
||||||
|
ErrorKind::InvalidInput
|
||||||
|
);
|
||||||
|
let other = directory.join("old-server");
|
||||||
|
stdfs::write(&other, "server").expect("write alternate binary");
|
||||||
|
assert!(verify_e2e_binary_receipt(&receipt, &directory, &other, None).is_err());
|
||||||
|
stdfs::write(&binary, "different server").expect("replace fixture binary");
|
||||||
|
assert!(verify_e2e_binary_receipt(&receipt, &directory, &binary, None).is_err());
|
||||||
|
stdfs::remove_file(&receipt).expect("remove expired receipt");
|
||||||
|
assert_eq!(
|
||||||
|
verify_e2e_binary_receipt(&receipt, &directory, &binary, None)
|
||||||
|
.expect_err("expired receipt")
|
||||||
|
.kind(),
|
||||||
|
ErrorKind::NotFound
|
||||||
|
);
|
||||||
|
stdfs::remove_dir_all(directory).expect("remove receipt fixture");
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build a cluster environment struct in-memory (no ports, no processes) so
|
/// Build a cluster environment struct in-memory (no ports, no processes) so
|
||||||
|
|||||||
@@ -1,108 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, assert_object_bytes, bring_drive_online, put_object, retrying_get_equals,
|
|
||||||
take_drive_offline, unique_bucket, wait_for_ready,
|
|
||||||
};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use std::sync::Arc;
|
|
||||||
use std::time::Duration;
|
|
||||||
use tokio::sync::Barrier;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn kill_and_restart_node_preserves_objects() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let mut dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("killnode");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let body = vec![0x11u8; 128 * 1024];
|
|
||||||
put_object(&dist.client(0)?, &bucket, "keep.bin", body.clone()).await?;
|
|
||||||
|
|
||||||
dist.cluster.stop_node(3)?;
|
|
||||||
retrying_get_equals(&dist.client(0)?, &bucket, "keep.bin", &body, Duration::from_secs(20)).await?;
|
|
||||||
|
|
||||||
dist.cluster.start_node(3).await?;
|
|
||||||
wait_for_ready(&dist.cluster).await?;
|
|
||||||
assert_object_bytes(&dist.client(3)?, &bucket, "keep.bin", &body).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn full_cluster_restart_preserves_objects() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let mut dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("pwr");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let body = vec![0x44u8; 64 * 1024];
|
|
||||||
put_object(&dist.client(1)?, &bucket, "survive.bin", body.clone()).await?;
|
|
||||||
|
|
||||||
dist.cluster.stop();
|
|
||||||
dist.cluster.start().await?;
|
|
||||||
wait_for_ready(&dist.cluster).await?;
|
|
||||||
for node_idx in 0..dist.cluster.nodes.len() {
|
|
||||||
assert_object_bytes(&dist.client(node_idx)?, &bucket, "survive.bin", &body).await?;
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn offline_drive_then_replace_keeps_object_readable() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("baddrive");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let body = vec![0x22u8; 96 * 1024];
|
|
||||||
put_object(&dist.client(1)?, &bucket, "durable.bin", body.clone()).await?;
|
|
||||||
|
|
||||||
take_drive_offline(&dist.cluster, 0, 0)?;
|
|
||||||
retrying_get_equals(&dist.client(2)?, &bucket, "durable.bin", &body, Duration::from_secs(20)).await?;
|
|
||||||
bring_drive_online(&dist.cluster, 0, 0)?;
|
|
||||||
retrying_get_equals(&dist.client(3)?, &bucket, "durable.bin", &body, Duration::from_secs(20)).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn concurrent_gets_survive_peer_node_kill() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let mut dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("getkill");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let body = vec![0x7Au8; 96 * 1024];
|
|
||||||
put_object(&dist.client(0)?, &bucket, "steady.bin", body.clone()).await?;
|
|
||||||
|
|
||||||
let live: Vec<_> = (0..3).map(|idx| dist.client(idx)).collect::<Result<Vec<_>, _>>()?;
|
|
||||||
let start = Arc::new(Barrier::new(13));
|
|
||||||
let mut handles = Vec::new();
|
|
||||||
for idx in 0..12 {
|
|
||||||
let client = live[idx % live.len()].clone();
|
|
||||||
let bucket = bucket.clone();
|
|
||||||
let body = body.clone();
|
|
||||||
let start = start.clone();
|
|
||||||
handles.push(tokio::spawn(async move {
|
|
||||||
start.wait().await;
|
|
||||||
retrying_get_equals(&client, &bucket, "steady.bin", &body, Duration::from_secs(20)).await
|
|
||||||
}));
|
|
||||||
}
|
|
||||||
start.wait().await;
|
|
||||||
dist.cluster.stop_node(3)?;
|
|
||||||
for handle in handles {
|
|
||||||
handle.await??;
|
|
||||||
}
|
|
||||||
|
|
||||||
dist.cluster.start_node(3).await?;
|
|
||||||
wait_for_ready(&dist.cluster).await?;
|
|
||||||
assert_object_bytes(&dist.client(3)?, &bucket, "steady.bin", &body).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,57 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{DistCluster, DistLayout, TestResult, assert_object_bytes, payload_for, put_object, unique_bucket};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use std::sync::Arc;
|
|
||||||
use tokio::sync::Barrier;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_high_concurrency_puts_are_readable_from_every_node() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("conc");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let clients = Arc::new(dist.clients()?);
|
|
||||||
let barrier = Arc::new(Barrier::new(32));
|
|
||||||
|
|
||||||
let mut handles = Vec::new();
|
|
||||||
for idx in 0..32 {
|
|
||||||
let clients = clients.clone();
|
|
||||||
let barrier = barrier.clone();
|
|
||||||
let bucket = bucket.clone();
|
|
||||||
handles.push(tokio::spawn(async move {
|
|
||||||
barrier.wait().await;
|
|
||||||
let client = &clients[idx % clients.len()];
|
|
||||||
let key = format!("c/{idx:02}.bin");
|
|
||||||
let body = payload_for(&key, 16 * 1024);
|
|
||||||
put_object(client, &bucket, &key, body.clone()).await?;
|
|
||||||
Ok::<_, Box<dyn std::error::Error + Send + Sync>>((key, body))
|
|
||||||
}));
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut inventory = Vec::new();
|
|
||||||
for handle in handles {
|
|
||||||
inventory.push(handle.await??);
|
|
||||||
}
|
|
||||||
|
|
||||||
for (node_idx, client) in clients.iter().enumerate() {
|
|
||||||
for (key, body) in &inventory {
|
|
||||||
assert_object_bytes(client, &bucket, key, body)
|
|
||||||
.await
|
|
||||||
.map_err(|error| format!("node {node_idx} failed to read {key}: {error}"))?;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,67 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, assert_inventory, decommission_started_or_refused, payload_for, put_inventory_retrying,
|
|
||||||
retrying_get_equals, retrying_put, unique_bucket, wait_for_decommission_complete,
|
|
||||||
};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use std::sync::Arc;
|
|
||||||
use std::time::Duration;
|
|
||||||
use tokio::sync::Barrier;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn concurrent_puts_during_decommission_do_not_lose_baseline_or_new_objects() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("concdecom");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let baseline_client = dist.client(0)?;
|
|
||||||
let inventory = put_inventory_retrying(&baseline_client, &bucket, 10, 24 * 1024, Duration::from_secs(30)).await?;
|
|
||||||
|
|
||||||
let decommission_started = decommission_started_or_refused(&dist.cluster, 0).await?;
|
|
||||||
|
|
||||||
let clients = Arc::new(dist.clients()?);
|
|
||||||
let barrier = Arc::new(Barrier::new(16));
|
|
||||||
let mut handles = Vec::new();
|
|
||||||
for idx in 0..16 {
|
|
||||||
let clients = clients.clone();
|
|
||||||
let barrier = barrier.clone();
|
|
||||||
let bucket = bucket.clone();
|
|
||||||
handles.push(tokio::spawn(async move {
|
|
||||||
barrier.wait().await;
|
|
||||||
let client = &clients[idx % clients.len()];
|
|
||||||
let key = format!("live/{idx:02}.bin");
|
|
||||||
let body = payload_for(&key, 8 * 1024);
|
|
||||||
retrying_put(client, &bucket, &key, body.clone(), Duration::from_secs(45)).await?;
|
|
||||||
Ok::<_, Box<dyn std::error::Error + Send + Sync>>((key, body))
|
|
||||||
}));
|
|
||||||
}
|
|
||||||
|
|
||||||
let mut live_objects = Vec::new();
|
|
||||||
for handle in handles {
|
|
||||||
live_objects.push(handle.await??);
|
|
||||||
}
|
|
||||||
|
|
||||||
if decommission_started {
|
|
||||||
wait_for_decommission_complete(&dist.cluster, 0, Duration::from_secs(180)).await?;
|
|
||||||
}
|
|
||||||
|
|
||||||
let checker = dist.client(3)?;
|
|
||||||
assert_inventory(&checker, &bucket, &inventory).await?;
|
|
||||||
for (key, body) in live_objects {
|
|
||||||
retrying_get_equals(&checker, &bucket, &key, &body, Duration::from_secs(30)).await?;
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,44 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, assert_inventory, decommission_started_or_refused, put_inventory_retrying, sha256_hex,
|
|
||||||
unique_bucket, wait_for_decommission_complete,
|
|
||||||
};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn decommission_attempt_does_not_alter_object_sha256() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("integrity");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let client = dist.client(0)?;
|
|
||||||
let inventory = put_inventory_retrying(&client, &bucket, 20, 64 * 1024, Duration::from_secs(30)).await?;
|
|
||||||
let before: Vec<(String, String)> = inventory.iter().map(|(key, body)| (key.clone(), sha256_hex(body))).collect();
|
|
||||||
|
|
||||||
if decommission_started_or_refused(&dist.cluster, 0).await? {
|
|
||||||
wait_for_decommission_complete(&dist.cluster, 0, Duration::from_secs(180)).await?;
|
|
||||||
}
|
|
||||||
|
|
||||||
let after_client = dist.client(2)?;
|
|
||||||
assert_inventory(&after_client, &bucket, &inventory).await?;
|
|
||||||
for (key, expected_hash) in before {
|
|
||||||
let got = after_client.get_object().bucket(&bucket).key(&key).send().await?;
|
|
||||||
let body = got.body.collect().await?.into_bytes();
|
|
||||||
assert_eq!(sha256_hex(body.as_ref()), expected_hash, "checksum changed for {key} after decommission");
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,71 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, assert_inventory, decommission_started_or_refused, list_pools_json, put_inventory,
|
|
||||||
rebalance_started_or_refused, unique_bucket, wait_for_decommission_complete, wait_for_rebalance_idle,
|
|
||||||
};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_restart_preserves_objects_then_rebalance_attempt() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let mut dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("expand");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let client = dist.client(0)?;
|
|
||||||
let inventory = put_inventory(&client, &bucket, 12, 32 * 1024).await?;
|
|
||||||
assert_inventory(&client, &bucket, &inventory).await?;
|
|
||||||
|
|
||||||
dist.cluster.stop();
|
|
||||||
dist.cluster.start().await?;
|
|
||||||
|
|
||||||
let after_restart = dist.client(0)?;
|
|
||||||
assert_inventory(&after_restart, &bucket, &inventory).await?;
|
|
||||||
let peer = dist.client(3)?;
|
|
||||||
assert_inventory(&peer, &bucket, &inventory).await?;
|
|
||||||
|
|
||||||
if rebalance_started_or_refused(&dist.cluster).await? {
|
|
||||||
wait_for_rebalance_idle(&dist.cluster, Duration::from_secs(90)).await?;
|
|
||||||
}
|
|
||||||
assert_inventory(&peer, &bucket, &inventory).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_decommission_attempt_does_not_lose_objects() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("decom");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let client = dist.client(1)?;
|
|
||||||
let inventory = put_inventory(&client, &bucket, 16, 48 * 1024).await?;
|
|
||||||
|
|
||||||
let pools_before = list_pools_json(&dist.cluster).await?;
|
|
||||||
let pool_count = pools_before
|
|
||||||
.as_array()
|
|
||||||
.map(Vec::len)
|
|
||||||
.or_else(|| pools_before.get("pools").and_then(serde_json::Value::as_array).map(Vec::len))
|
|
||||||
.unwrap_or(1);
|
|
||||||
assert!(pool_count >= 1, "expected at least one pool before decommission: {pools_before}");
|
|
||||||
|
|
||||||
if decommission_started_or_refused(&dist.cluster, 0).await? {
|
|
||||||
wait_for_decommission_complete(&dist.cluster, 0, Duration::from_secs(180)).await?;
|
|
||||||
}
|
|
||||||
|
|
||||||
let after = dist.client(3)?;
|
|
||||||
assert_inventory(&after, &bucket, &inventory).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,149 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, assert_object_bytes, get_object_bytes, put_object, unique_bucket, wait_until,
|
|
||||||
};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use aws_sdk_s3::primitives::ByteStream;
|
|
||||||
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart};
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_multipart_and_cross_node_listing_agree() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("extra");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let client = dist.client(0)?;
|
|
||||||
|
|
||||||
let key = "multipart.bin";
|
|
||||||
let part1 = vec![0x41u8; 5 * 1024 * 1024];
|
|
||||||
let part2 = vec![0x42u8; 5 * 1024 * 1024];
|
|
||||||
let upload = client.create_multipart_upload().bucket(&bucket).key(key).send().await?;
|
|
||||||
let upload_id = upload.upload_id().ok_or("missing upload id")?.to_string();
|
|
||||||
|
|
||||||
let uploaded1 = client
|
|
||||||
.upload_part()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.upload_id(&upload_id)
|
|
||||||
.part_number(1)
|
|
||||||
.body(ByteStream::from(part1.clone()))
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
let uploaded2 = client
|
|
||||||
.upload_part()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.upload_id(&upload_id)
|
|
||||||
.part_number(2)
|
|
||||||
.body(ByteStream::from(part2.clone()))
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
client
|
|
||||||
.complete_multipart_upload()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.upload_id(&upload_id)
|
|
||||||
.multipart_upload(
|
|
||||||
CompletedMultipartUpload::builder()
|
|
||||||
.parts(
|
|
||||||
CompletedPart::builder()
|
|
||||||
.part_number(1)
|
|
||||||
.e_tag(uploaded1.e_tag().unwrap_or_default())
|
|
||||||
.build(),
|
|
||||||
)
|
|
||||||
.parts(
|
|
||||||
CompletedPart::builder()
|
|
||||||
.part_number(2)
|
|
||||||
.e_tag(uploaded2.e_tag().unwrap_or_default())
|
|
||||||
.build(),
|
|
||||||
)
|
|
||||||
.build(),
|
|
||||||
)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let mut expected = part1;
|
|
||||||
expected.extend_from_slice(&part2);
|
|
||||||
for node_idx in 0..dist.cluster.nodes.len() {
|
|
||||||
assert_object_bytes(&dist.client(node_idx)?, &bucket, key, &expected).await?;
|
|
||||||
}
|
|
||||||
|
|
||||||
put_object(&client, &bucket, "list/a", b"a".to_vec()).await?;
|
|
||||||
put_object(&dist.client(2)?, &bucket, "list/b", b"b".to_vec()).await?;
|
|
||||||
let mut seen = Vec::new();
|
|
||||||
for node_idx in 0..dist.cluster.nodes.len() {
|
|
||||||
let listed = dist
|
|
||||||
.client(node_idx)?
|
|
||||||
.list_objects_v2()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.prefix("list/")
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
let keys: Vec<String> = listed
|
|
||||||
.contents()
|
|
||||||
.iter()
|
|
||||||
.filter_map(|object| object.key().map(str::to_string))
|
|
||||||
.collect();
|
|
||||||
seen.push(keys);
|
|
||||||
}
|
|
||||||
for keys in &seen[1..] {
|
|
||||||
assert_eq!(&seen[0], keys, "list results diverged across nodes: {seen:?}");
|
|
||||||
}
|
|
||||||
|
|
||||||
let got = get_object_bytes(&dist.client(3)?, &bucket, "list/a").await?;
|
|
||||||
assert_eq!(got, b"a");
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_list_buckets_agree_across_all_nodes() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("listed");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
put_object(&dist.client(0)?, &bucket, "seed.bin", b"seed".to_vec()).await?;
|
|
||||||
|
|
||||||
for node_idx in 0..dist.cluster.nodes.len() {
|
|
||||||
let client = dist.client(node_idx)?;
|
|
||||||
let name = bucket.clone();
|
|
||||||
wait_until(
|
|
||||||
Duration::from_secs(20),
|
|
||||||
|| {
|
|
||||||
let client = client.clone();
|
|
||||||
let name = name.clone();
|
|
||||||
async move {
|
|
||||||
let listed = client.list_buckets().send().await?;
|
|
||||||
Ok(listed.buckets().iter().any(|entry| entry.name() == Some(name.as_str())))
|
|
||||||
}
|
|
||||||
},
|
|
||||||
&format!("node {node_idx} lists {bucket}"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
wait_until(
|
|
||||||
Duration::from_secs(20),
|
|
||||||
|| {
|
|
||||||
let client = dist.client(node_idx).expect("client");
|
|
||||||
let name = bucket.clone();
|
|
||||||
async move { Ok(get_object_bytes(&client, &name, "seed.bin").await.ok() == Some(b"seed".to_vec())) }
|
|
||||||
},
|
|
||||||
&format!("node {node_idx} reads seed.bin"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,933 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
//! Shared 4-node distributed e2e helpers.
|
|
||||||
//!
|
|
||||||
//! Two localhost-expressible layouts cover the suite:
|
|
||||||
//!
|
|
||||||
//! * **4×4 single pool** (`four_by_four`) — four processes, four drives each,
|
|
||||||
//! one `DistErasure` pool (16 explicit volume endpoints). This is the
|
|
||||||
//! default S3 / lock / versioning / chaos topology.
|
|
||||||
//! * **4×4 four pool** — `append_single_node_pool` exists for harness unit
|
|
||||||
//! tests. Live expand-then-restart currently hits `pool metadata recovery
|
|
||||||
//! required` on localhost DistErasure. That is a production bootstrap-proof
|
|
||||||
//! limitation this test lane does not change. Movement tests use 4×4 single
|
|
||||||
//! pool and classify decommission/rebalance product refusals (and opaque
|
|
||||||
//! 500 InternalError) as a refused move while still asserting object bytes.
|
|
||||||
//!
|
|
||||||
//! Genuine multi-node *striped* pools still need multi-host CI (backlog
|
|
||||||
//! #1313 / #1314). Site replication uses two 4-node 1-drive clusters so the
|
|
||||||
//! process count stays at eight rather than sixteen.
|
|
||||||
|
|
||||||
use crate::common::{
|
|
||||||
ClusterTopology, FAST_DATA_USAGE_SCANNER_ENV, RustFSTestClusterEnvironment, admin_request, build_test_s3_config,
|
|
||||||
local_http_client, replication_fast_env, signed_request,
|
|
||||||
};
|
|
||||||
use crate::replication_extension_test::LOOPBACK_REPLICATION_TARGET_ENV;
|
|
||||||
use aws_sdk_s3::Client;
|
|
||||||
use aws_sdk_s3::primitives::ByteStream;
|
|
||||||
use aws_sdk_s3::types::{BucketVersioningStatus, VersioningConfiguration};
|
|
||||||
use http::{Method, StatusCode};
|
|
||||||
use sha2::{Digest, Sha256};
|
|
||||||
use std::collections::BTreeMap;
|
|
||||||
use std::path::Path;
|
|
||||||
use std::time::Duration;
|
|
||||||
use tokio::time::{Instant, sleep};
|
|
||||||
use uuid::Uuid;
|
|
||||||
|
|
||||||
pub(crate) type TestResult<T = ()> = Result<T, Box<dyn std::error::Error + Send + Sync>>;
|
|
||||||
|
|
||||||
pub(crate) const NODE_COUNT: usize = 4;
|
|
||||||
pub(crate) const DRIVES_PER_NODE: usize = 4;
|
|
||||||
|
|
||||||
#[derive(Clone, Copy, Debug)]
|
|
||||||
pub(crate) enum DistLayout {
|
|
||||||
/// 4 nodes × 4 drives, one erasure pool spanning every endpoint.
|
|
||||||
FourByFour,
|
|
||||||
/// 4 nodes × 1 drive, one erasure pool (minimum 4-node 4-disk layout).
|
|
||||||
FourNodeFourDisk,
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) struct DistCluster {
|
|
||||||
pub cluster: RustFSTestClusterEnvironment,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl DistCluster {
|
|
||||||
pub async fn start(layout: DistLayout) -> TestResult<Self> {
|
|
||||||
Self::start_with_env(layout, &[]).await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn start_with_env(layout: DistLayout, extra_env: &[(&str, &str)]) -> TestResult<Self> {
|
|
||||||
let mut dist = Self::new_stopped_with_env(layout, extra_env).await?;
|
|
||||||
dist.cluster.start().await?;
|
|
||||||
Ok(dist)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Allocate ports and data dirs without spawning processes.
|
|
||||||
///
|
|
||||||
/// Upgrade tests configure capture logs, then start a pinned previous
|
|
||||||
/// binary against the same directories.
|
|
||||||
pub async fn new_stopped(layout: DistLayout) -> TestResult<Self> {
|
|
||||||
Self::new_stopped_with_env(layout, &[]).await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn new_stopped_with_env(layout: DistLayout, extra_env: &[(&str, &str)]) -> TestResult<Self> {
|
|
||||||
let topology = match layout {
|
|
||||||
DistLayout::FourByFour => ClusterTopology::single_pool_multidrive(NODE_COUNT, DRIVES_PER_NODE),
|
|
||||||
DistLayout::FourNodeFourDisk => ClusterTopology::single_pool(NODE_COUNT),
|
|
||||||
};
|
|
||||||
let mut cluster = RustFSTestClusterEnvironment::with_topology(topology).await?;
|
|
||||||
cluster.set_env("NO_PROXY", "127.0.0.1,localhost");
|
|
||||||
cluster.set_env("HTTP_PROXY", "");
|
|
||||||
cluster.set_env("HTTPS_PROXY", "");
|
|
||||||
for &(key, value) in extra_env {
|
|
||||||
cluster.set_env(key, value);
|
|
||||||
}
|
|
||||||
Ok(Self { cluster })
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Start every node with a specific `rustfs` binary, keeping the allocated
|
|
||||||
/// data directories. Used to seed an old on-disk format before upgrading.
|
|
||||||
pub async fn start_from_binary(&mut self, binary: &Path) -> TestResult {
|
|
||||||
self.cluster.start_with_binary(binary).await?;
|
|
||||||
wait_for_ready(&self.cluster).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Stop every node and bring the same data directories up on the workspace
|
|
||||||
/// binary (direct upgrade).
|
|
||||||
pub async fn restart_with_current_binary(&mut self) -> TestResult {
|
|
||||||
self.cluster.stop();
|
|
||||||
self.cluster.start().await?;
|
|
||||||
wait_for_ready(&self.cluster).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Replace one running node with the workspace binary (rolling upgrade).
|
|
||||||
pub async fn replace_node_with_current_binary(&mut self, node_idx: usize) -> TestResult {
|
|
||||||
self.cluster.stop_node(node_idx)?;
|
|
||||||
self.cluster.start_node(node_idx).await?;
|
|
||||||
wait_for_ready(&self.cluster).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn client_with_credentials(&self, node_idx: usize, access_key: &str, secret_key: &str) -> TestResult<Client> {
|
|
||||||
if node_idx >= self.cluster.nodes.len() {
|
|
||||||
return Err("node_idx is invalid".into());
|
|
||||||
}
|
|
||||||
Ok(Client::from_conf(build_test_s3_config(
|
|
||||||
&self.cluster.nodes[node_idx].url,
|
|
||||||
access_key,
|
|
||||||
secret_key,
|
|
||||||
None,
|
|
||||||
"cluster-iam-test",
|
|
||||||
)))
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn start_replication_pair() -> TestResult<(Self, Self)> {
|
|
||||||
let mut extra: Vec<(&str, &str)> = replication_fast_env();
|
|
||||||
extra.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
|
||||||
extra.extend_from_slice(FAST_DATA_USAGE_SCANNER_ENV);
|
|
||||||
let source = Self::start_with_env(DistLayout::FourNodeFourDisk, &extra).await?;
|
|
||||||
let target = Self::start_with_env(DistLayout::FourNodeFourDisk, &extra).await?;
|
|
||||||
Ok((source, target))
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn client(&self, node_idx: usize) -> TestResult<Client> {
|
|
||||||
self.cluster.create_s3_client(node_idx)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub fn clients(&self) -> TestResult<Vec<Client>> {
|
|
||||||
self.cluster.create_all_clients()
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn create_bucket(&self, bucket: &str) -> TestResult {
|
|
||||||
self.cluster.create_test_bucket(bucket).await
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn unique_bucket(prefix: &str) -> String {
|
|
||||||
let id = Uuid::new_v4().simple().to_string();
|
|
||||||
format!("{prefix}-{}", &id[..12])
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn sha256_hex(bytes: &[u8]) -> String {
|
|
||||||
let digest = Sha256::digest(bytes);
|
|
||||||
digest.iter().map(|byte| format!("{byte:02x}")).collect()
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn payload_for(key: &str, size: usize) -> Vec<u8> {
|
|
||||||
let seed = key.as_bytes();
|
|
||||||
(0..size)
|
|
||||||
.map(|idx| seed.get(idx % seed.len()).copied().unwrap_or(0) ^ (idx as u8))
|
|
||||||
.collect()
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn put_object(client: &Client, bucket: &str, key: &str, body: Vec<u8>) -> TestResult {
|
|
||||||
client
|
|
||||||
.put_object()
|
|
||||||
.bucket(bucket)
|
|
||||||
.key(key)
|
|
||||||
.body(ByteStream::from(body))
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn get_object_bytes(client: &Client, bucket: &str, key: &str) -> TestResult<Vec<u8>> {
|
|
||||||
let output = client.get_object().bucket(bucket).key(key).send().await?;
|
|
||||||
Ok(output.body.collect().await?.into_bytes().to_vec())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn assert_object_bytes(client: &Client, bucket: &str, key: &str, expected: &[u8]) -> TestResult {
|
|
||||||
let got = get_object_bytes(client, bucket, key).await?;
|
|
||||||
if got.as_slice() != expected {
|
|
||||||
return Err(format!(
|
|
||||||
"object {bucket}/{key} bytes mismatch: expected {} bytes sha256={} got {} bytes sha256={}",
|
|
||||||
expected.len(),
|
|
||||||
sha256_hex(expected),
|
|
||||||
got.len(),
|
|
||||||
sha256_hex(&got)
|
|
||||||
)
|
|
||||||
.into());
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn put_inventory(
|
|
||||||
client: &Client,
|
|
||||||
bucket: &str,
|
|
||||||
count: usize,
|
|
||||||
size: usize,
|
|
||||||
) -> TestResult<BTreeMap<String, Vec<u8>>> {
|
|
||||||
let mut inventory = BTreeMap::new();
|
|
||||||
for idx in 0..count {
|
|
||||||
let key = format!("obj-{idx:04}");
|
|
||||||
let body = payload_for(&key, size);
|
|
||||||
put_object(client, bucket, &key, body.clone()).await?;
|
|
||||||
inventory.insert(key, body);
|
|
||||||
}
|
|
||||||
Ok(inventory)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Localhost DistErasure can 500 a PUT while heal_bucket hits a pool-meta
|
|
||||||
/// write fence. Retry only those transient codes.
|
|
||||||
pub(crate) async fn put_inventory_retrying(
|
|
||||||
client: &Client,
|
|
||||||
bucket: &str,
|
|
||||||
count: usize,
|
|
||||||
size: usize,
|
|
||||||
timeout: Duration,
|
|
||||||
) -> TestResult<BTreeMap<String, Vec<u8>>> {
|
|
||||||
let mut inventory = BTreeMap::new();
|
|
||||||
for idx in 0..count {
|
|
||||||
let key = format!("obj-{idx:04}");
|
|
||||||
let body = payload_for(&key, size);
|
|
||||||
retrying_put(client, bucket, &key, body.clone(), timeout).await?;
|
|
||||||
inventory.insert(key, body);
|
|
||||||
}
|
|
||||||
Ok(inventory)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn assert_inventory(client: &Client, bucket: &str, inventory: &BTreeMap<String, Vec<u8>>) -> TestResult {
|
|
||||||
for (key, expected) in inventory {
|
|
||||||
assert_object_bytes(client, bucket, key, expected).await?;
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn enable_versioning(client: &Client, bucket: &str) -> TestResult {
|
|
||||||
client
|
|
||||||
.put_bucket_versioning()
|
|
||||||
.bucket(bucket)
|
|
||||||
.versioning_configuration(
|
|
||||||
VersioningConfiguration::builder()
|
|
||||||
.status(BucketVersioningStatus::Enabled)
|
|
||||||
.build(),
|
|
||||||
)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn wait_until<F, Fut>(timeout: Duration, mut probe: F, label: &str) -> TestResult
|
|
||||||
where
|
|
||||||
F: FnMut() -> Fut,
|
|
||||||
Fut: std::future::Future<Output = TestResult<bool>>,
|
|
||||||
{
|
|
||||||
let deadline = Instant::now() + timeout;
|
|
||||||
let mut delay = Duration::from_millis(50);
|
|
||||||
loop {
|
|
||||||
let last_error = match probe().await {
|
|
||||||
Ok(true) => return Ok(()),
|
|
||||||
Ok(false) => format!("{label} still false"),
|
|
||||||
Err(error) => error.to_string(),
|
|
||||||
};
|
|
||||||
if Instant::now() >= deadline {
|
|
||||||
return Err(format!("{label} did not become true within {timeout:?}: {last_error}").into());
|
|
||||||
}
|
|
||||||
sleep(delay).await;
|
|
||||||
delay = (delay * 2).min(Duration::from_secs(1));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn cluster_admin(
|
|
||||||
cluster: &RustFSTestClusterEnvironment,
|
|
||||||
method: Method,
|
|
||||||
path_and_query: &str,
|
|
||||||
body: Option<String>,
|
|
||||||
) -> TestResult<(StatusCode, String)> {
|
|
||||||
admin_request(
|
|
||||||
&cluster.nodes[0].url,
|
|
||||||
method,
|
|
||||||
path_and_query,
|
|
||||||
body,
|
|
||||||
&cluster.access_key,
|
|
||||||
&cluster.secret_key,
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn cluster_admin_ok(
|
|
||||||
cluster: &RustFSTestClusterEnvironment,
|
|
||||||
method: Method,
|
|
||||||
path_and_query: &str,
|
|
||||||
body: Option<String>,
|
|
||||||
) -> TestResult<String> {
|
|
||||||
let (status, response) = cluster_admin(cluster, method.clone(), path_and_query, body).await?;
|
|
||||||
if !status.is_success() {
|
|
||||||
return Err(format!("{method} {path_and_query} failed: {status} {response}").into());
|
|
||||||
}
|
|
||||||
Ok(response)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn wait_for_ready(cluster: &RustFSTestClusterEnvironment) -> TestResult {
|
|
||||||
let client = local_http_client();
|
|
||||||
for node in &cluster.nodes {
|
|
||||||
let url = format!("{}/health/ready", node.url);
|
|
||||||
wait_until(
|
|
||||||
Duration::from_secs(30),
|
|
||||||
|| {
|
|
||||||
let client = client.clone();
|
|
||||||
let url = url.clone();
|
|
||||||
async move {
|
|
||||||
match client.get(&url).send().await {
|
|
||||||
Ok(response) if response.status().is_success() => Ok(true),
|
|
||||||
_ => Ok(false),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
&format!("node {} ready", node.address),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn take_drive_offline(
|
|
||||||
cluster: &RustFSTestClusterEnvironment,
|
|
||||||
node_idx: usize,
|
|
||||||
drive_idx: usize,
|
|
||||||
) -> TestResult<String> {
|
|
||||||
let dir = cluster
|
|
||||||
.nodes
|
|
||||||
.get(node_idx)
|
|
||||||
.and_then(|node| node.data_dirs.get(drive_idx))
|
|
||||||
.ok_or("invalid node/drive index")?;
|
|
||||||
let offline = format!("{dir}.offline");
|
|
||||||
if Path::new(&offline).exists() {
|
|
||||||
return Err(format!("drive already offline: {offline}").into());
|
|
||||||
}
|
|
||||||
std::fs::rename(dir, &offline)?;
|
|
||||||
Ok(offline)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn bring_drive_online(cluster: &RustFSTestClusterEnvironment, node_idx: usize, drive_idx: usize) -> TestResult {
|
|
||||||
let dir = cluster
|
|
||||||
.nodes
|
|
||||||
.get(node_idx)
|
|
||||||
.and_then(|node| node.data_dirs.get(drive_idx))
|
|
||||||
.ok_or("invalid node/drive index")?;
|
|
||||||
let offline = format!("{dir}.offline");
|
|
||||||
if Path::new(dir).exists() {
|
|
||||||
std::fs::remove_dir_all(dir)?;
|
|
||||||
}
|
|
||||||
std::fs::rename(&offline, dir)?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn set_remote_target(
|
|
||||||
source: &RustFSTestClusterEnvironment,
|
|
||||||
source_bucket: &str,
|
|
||||||
target: &RustFSTestClusterEnvironment,
|
|
||||||
target_bucket: &str,
|
|
||||||
) -> TestResult<String> {
|
|
||||||
let body = serde_json::json!({
|
|
||||||
"endpoint": target.nodes[0].address,
|
|
||||||
"credentials": {
|
|
||||||
"accessKey": target.access_key,
|
|
||||||
"secretKey": target.secret_key
|
|
||||||
},
|
|
||||||
"targetbucket": target_bucket,
|
|
||||||
"secure": false,
|
|
||||||
"type": "replication"
|
|
||||||
});
|
|
||||||
let url = format!(
|
|
||||||
"{}/rustfs/admin/v3/set-remote-target?bucket={}",
|
|
||||||
source.nodes[0].url,
|
|
||||||
urlencoding::encode(source_bucket)
|
|
||||||
);
|
|
||||||
let response = signed_request(
|
|
||||||
Method::PUT,
|
|
||||||
&url,
|
|
||||||
&source.access_key,
|
|
||||||
&source.secret_key,
|
|
||||||
Some(body.to_string().into_bytes()),
|
|
||||||
Some("application/json"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
if response.status() != StatusCode::OK {
|
|
||||||
let status = response.status();
|
|
||||||
let body = response.text().await.unwrap_or_default();
|
|
||||||
return Err(format!("set remote target failed: {status} {body}").into());
|
|
||||||
}
|
|
||||||
Ok(serde_json::from_slice(&response.bytes().await?)?)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn put_bucket_replication(source: &RustFSTestClusterEnvironment, bucket: &str, target_arn: &str) -> TestResult {
|
|
||||||
let body = format!(
|
|
||||||
r#"<ReplicationConfiguration xmlns="http://s3.amazonaws.com/doc/2006-03-01/">
|
|
||||||
<Role></Role>
|
|
||||||
<Rule>
|
|
||||||
<ID>rule-1</ID>
|
|
||||||
<Priority>1</Priority>
|
|
||||||
<Status>Enabled</Status>
|
|
||||||
<DeleteMarkerReplication>
|
|
||||||
<Status>Enabled</Status>
|
|
||||||
</DeleteMarkerReplication>
|
|
||||||
<ExistingObjectReplication>
|
|
||||||
<Status>Enabled</Status>
|
|
||||||
</ExistingObjectReplication>
|
|
||||||
<Destination>
|
|
||||||
<Bucket>{target_arn}</Bucket>
|
|
||||||
</Destination>
|
|
||||||
</Rule>
|
|
||||||
</ReplicationConfiguration>"#
|
|
||||||
);
|
|
||||||
let url = format!("{}/{bucket}?replication", source.nodes[0].url);
|
|
||||||
let response = signed_request(
|
|
||||||
Method::PUT,
|
|
||||||
&url,
|
|
||||||
&source.access_key,
|
|
||||||
&source.secret_key,
|
|
||||||
Some(body.into_bytes()),
|
|
||||||
Some("application/xml"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
if !response.status().is_success() {
|
|
||||||
let status = response.status();
|
|
||||||
let body = response.text().await.unwrap_or_default();
|
|
||||||
return Err(format!("put bucket replication failed: {status} {body}").into());
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn wait_for_replicated_bytes(
|
|
||||||
client: &Client,
|
|
||||||
bucket: &str,
|
|
||||||
key: &str,
|
|
||||||
expected: &[u8],
|
|
||||||
timeout: Duration,
|
|
||||||
) -> TestResult {
|
|
||||||
wait_until(
|
|
||||||
timeout,
|
|
||||||
|| async {
|
|
||||||
match get_object_bytes(client, bucket, key).await {
|
|
||||||
Ok(got) if got.as_slice() == expected => Ok(true),
|
|
||||||
Ok(_) => Ok(false),
|
|
||||||
Err(error) => {
|
|
||||||
let message = error.to_string();
|
|
||||||
if message.contains("NoSuchKey") || message.contains("NotFound") {
|
|
||||||
Ok(false)
|
|
||||||
} else {
|
|
||||||
Err(error)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
&format!("replicated object {bucket}/{key}"),
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn set_bucket_quota(cluster: &RustFSTestClusterEnvironment, bucket: &str, quota_bytes: u64) -> TestResult {
|
|
||||||
wait_until(
|
|
||||||
Duration::from_secs(30),
|
|
||||||
|| async {
|
|
||||||
let (status, _) =
|
|
||||||
cluster_admin(cluster, Method::GET, &format!("/rustfs/admin/v3/quota-stats/{bucket}"), None).await?;
|
|
||||||
Ok(status.is_success() || status == StatusCode::NOT_FOUND)
|
|
||||||
},
|
|
||||||
"quota stats ready",
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
let body = serde_json::json!({ "quota": quota_bytes, "quota_type": "HARD" }).to_string();
|
|
||||||
wait_until(
|
|
||||||
Duration::from_secs(30),
|
|
||||||
|| async {
|
|
||||||
let (status, response) =
|
|
||||||
cluster_admin(cluster, Method::PUT, &format!("/rustfs/admin/v3/quota/{bucket}"), Some(body.clone())).await?;
|
|
||||||
if status.is_success() {
|
|
||||||
return Ok(true);
|
|
||||||
}
|
|
||||||
if status == StatusCode::SERVICE_UNAVAILABLE {
|
|
||||||
return Ok(false);
|
|
||||||
}
|
|
||||||
Err(format!("failed to set quota for {bucket}: {status} {response}").into())
|
|
||||||
},
|
|
||||||
"set hard quota",
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Localhost DistErasure can boot and serve S3 while refusing pool.bin
|
|
||||||
/// mutations (`pool metadata writes remain blocked` / missing fleet
|
|
||||||
/// capability proof). Single-pool 4×4 also rejects decommission/rebalance
|
|
||||||
/// with a product error. Tests must not pretend a move ran.
|
|
||||||
pub(crate) fn is_pool_meta_write_fence(body: &str) -> bool {
|
|
||||||
body.contains("pool metadata writes remain blocked")
|
|
||||||
|| body.contains("pool metadata recovery required")
|
|
||||||
|| body.contains("pool activation requires a live fleet capability proof")
|
|
||||||
|| body.contains("pool activation fleet capability proof expired")
|
|
||||||
|| body.contains("live fleet capability proof")
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Product refusals that movement tests observe. Opaque 500 InternalError stays
|
|
||||||
/// in [`classify_data_movement_http`] because admin often wraps the fence as
|
|
||||||
/// InternalError XML without the inner string. 502/503 and auth failures are
|
|
||||||
/// not refusals.
|
|
||||||
pub(crate) fn is_known_data_movement_refusal(body: &str) -> bool {
|
|
||||||
is_pool_meta_write_fence(body)
|
|
||||||
|| body.contains("NotImplemented")
|
|
||||||
|| body.contains("single pool deployments do not support")
|
|
||||||
|| body.contains("at least one active pool must remain")
|
|
||||||
}
|
|
||||||
|
|
||||||
#[derive(Debug)]
|
|
||||||
pub(crate) enum DataMovementStart {
|
|
||||||
Started,
|
|
||||||
Refused(String),
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn classify_data_movement_http(status: StatusCode, body: &str) -> Result<DataMovementStart, String> {
|
|
||||||
if status.is_success() {
|
|
||||||
return Ok(DataMovementStart::Started);
|
|
||||||
}
|
|
||||||
if is_known_data_movement_refusal(body) || status.as_u16() == 501 || status == StatusCode::INTERNAL_SERVER_ERROR {
|
|
||||||
return Ok(DataMovementStart::Refused(format!("{status} {body}")));
|
|
||||||
}
|
|
||||||
Err(format!("{status} {body}"))
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn try_start_decommission(
|
|
||||||
cluster: &RustFSTestClusterEnvironment,
|
|
||||||
pool_id: usize,
|
|
||||||
) -> TestResult<DataMovementStart> {
|
|
||||||
let path = format!("/rustfs/admin/v3/pools/decommission?pool={pool_id}&by-id=true");
|
|
||||||
let (status, response) = cluster_admin(cluster, Method::POST, &path, None).await?;
|
|
||||||
classify_data_movement_http(status, &response).map_err(|detail| format!("POST {path} failed: {detail}").into())
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Returns whether decommission actually started. A product refusal or opaque
|
|
||||||
/// 500 InternalError is not a test failure: callers still assert object bytes.
|
|
||||||
pub(crate) async fn decommission_started_or_refused(cluster: &RustFSTestClusterEnvironment, pool_id: usize) -> TestResult<bool> {
|
|
||||||
match try_start_decommission(cluster, pool_id).await? {
|
|
||||||
DataMovementStart::Started => Ok(true),
|
|
||||||
DataMovementStart::Refused(detail) => {
|
|
||||||
eprintln!("decommission POST refused; objects still asserted: {detail}");
|
|
||||||
Ok(false)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn decommission_status_json(cluster: &RustFSTestClusterEnvironment) -> TestResult<serde_json::Value> {
|
|
||||||
let body = cluster_admin_ok(cluster, Method::GET, "/rustfs/admin/v3/decommission/status", None).await?;
|
|
||||||
Ok(serde_json::from_str(&body)?)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn pool_entry(status: &serde_json::Value, pool_id: usize) -> Option<&serde_json::Value> {
|
|
||||||
if let Some(pools) = status.get("pools").and_then(serde_json::Value::as_array) {
|
|
||||||
return pools
|
|
||||||
.iter()
|
|
||||||
.find(|pool| pool.get("id").and_then(serde_json::Value::as_u64) == Some(pool_id as u64));
|
|
||||||
}
|
|
||||||
if status.get("id").and_then(serde_json::Value::as_u64) == Some(pool_id as u64) {
|
|
||||||
Some(status)
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn decommission_pool_failed(pool: &serde_json::Value) -> bool {
|
|
||||||
let info = pool.get("decommissionInfo");
|
|
||||||
let flagged = |key: &str| info.and_then(|value| value.get(key)).and_then(serde_json::Value::as_bool) == Some(true);
|
|
||||||
flagged("failed")
|
|
||||||
|| flagged("canceled")
|
|
||||||
|| pool
|
|
||||||
.get("status")
|
|
||||||
.and_then(serde_json::Value::as_str)
|
|
||||||
.is_some_and(|status| status.eq_ignore_ascii_case("failed") || status.eq_ignore_ascii_case("canceled"))
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn decommission_complete(status: &serde_json::Value, pool_id: usize) -> bool {
|
|
||||||
let Some(pool) = pool_entry(status, pool_id) else {
|
|
||||||
return false;
|
|
||||||
};
|
|
||||||
if decommission_pool_failed(pool) {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
let info_complete = pool
|
|
||||||
.get("decommissionInfo")
|
|
||||||
.and_then(|value| value.get("complete"))
|
|
||||||
.and_then(serde_json::Value::as_bool)
|
|
||||||
== Some(true);
|
|
||||||
let status_text = pool.get("status").and_then(serde_json::Value::as_str).unwrap_or("");
|
|
||||||
let pool_status = pool.get("poolStatus").and_then(serde_json::Value::as_str).unwrap_or("");
|
|
||||||
info_complete || status_text.eq_ignore_ascii_case("complete") || pool_status.eq_ignore_ascii_case("decommissioned")
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn decommission_failed(status: &serde_json::Value, pool_id: usize) -> bool {
|
|
||||||
pool_entry(status, pool_id).is_some_and(decommission_pool_failed)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// `Ok(true)` complete, `Ok(false)` still running, `Err` terminal failure.
|
|
||||||
pub(crate) fn decommission_progress(status: &serde_json::Value, pool_id: usize) -> Result<bool, String> {
|
|
||||||
if decommission_failed(status, pool_id) {
|
|
||||||
return Err(format!("decommission failed for pool {pool_id}: {status}"));
|
|
||||||
}
|
|
||||||
Ok(decommission_complete(status, pool_id))
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn wait_for_decommission_complete(
|
|
||||||
cluster: &RustFSTestClusterEnvironment,
|
|
||||||
pool_id: usize,
|
|
||||||
timeout: Duration,
|
|
||||||
) -> TestResult {
|
|
||||||
let deadline = Instant::now() + timeout;
|
|
||||||
let mut delay = Duration::from_millis(50);
|
|
||||||
let mut last_error;
|
|
||||||
loop {
|
|
||||||
last_error = match decommission_status_json(cluster).await {
|
|
||||||
Ok(status) => match decommission_progress(&status, pool_id) {
|
|
||||||
Ok(true) => return Ok(()),
|
|
||||||
Ok(false) => format!("decommission complete still false: {status}"),
|
|
||||||
Err(failed) => return Err(failed.into()),
|
|
||||||
},
|
|
||||||
Err(error) => error.to_string(),
|
|
||||||
};
|
|
||||||
if Instant::now() >= deadline {
|
|
||||||
return Err(format!("decommission complete did not become true within {timeout:?}: {last_error}").into());
|
|
||||||
}
|
|
||||||
sleep(delay).await;
|
|
||||||
delay = (delay * 2).min(Duration::from_secs(1));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn try_start_rebalance(cluster: &RustFSTestClusterEnvironment) -> TestResult<DataMovementStart> {
|
|
||||||
let path = "/rustfs/admin/v3/rebalance/start";
|
|
||||||
let (status, response) = cluster_admin(cluster, Method::POST, path, None).await?;
|
|
||||||
classify_data_movement_http(status, &response).map_err(|detail| format!("POST {path} failed: {detail}").into())
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn rebalance_started_or_refused(cluster: &RustFSTestClusterEnvironment) -> TestResult<bool> {
|
|
||||||
match try_start_rebalance(cluster).await? {
|
|
||||||
DataMovementStart::Started => Ok(true),
|
|
||||||
DataMovementStart::Refused(detail) => {
|
|
||||||
eprintln!("rebalance POST refused; objects still asserted: {detail}");
|
|
||||||
Ok(false)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn rebalance_status_json(cluster: &RustFSTestClusterEnvironment) -> TestResult<serde_json::Value> {
|
|
||||||
let body = cluster_admin_ok(cluster, Method::GET, "/rustfs/admin/v3/rebalance/status", None).await?;
|
|
||||||
Ok(serde_json::from_str(&body)?)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) fn rebalance_active(status: &serde_json::Value) -> bool {
|
|
||||||
status
|
|
||||||
.get("pools")
|
|
||||||
.and_then(serde_json::Value::as_array)
|
|
||||||
.is_some_and(|pools| {
|
|
||||||
pools.iter().any(|pool| {
|
|
||||||
let stopping = pool.get("stopping").and_then(serde_json::Value::as_bool) == Some(true);
|
|
||||||
let value = pool.get("status").and_then(serde_json::Value::as_str).unwrap_or("");
|
|
||||||
stopping
|
|
||||||
|| value.eq_ignore_ascii_case("started")
|
|
||||||
|| value.eq_ignore_ascii_case("active")
|
|
||||||
|| value.eq_ignore_ascii_case("running")
|
|
||||||
|| value.eq_ignore_ascii_case("stopping")
|
|
||||||
})
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn wait_for_rebalance_idle(cluster: &RustFSTestClusterEnvironment, timeout: Duration) -> TestResult {
|
|
||||||
wait_until(
|
|
||||||
timeout,
|
|
||||||
|| async {
|
|
||||||
match rebalance_status_json(cluster).await {
|
|
||||||
Ok(status) => Ok(!rebalance_active(&status)),
|
|
||||||
Err(error) => {
|
|
||||||
let message = error.to_string();
|
|
||||||
if message.contains("NoSuchResource") || message.contains("404") || message.contains("not started") {
|
|
||||||
Ok(true)
|
|
||||||
} else {
|
|
||||||
Err(error)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"rebalance idle",
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn list_pools_json(cluster: &RustFSTestClusterEnvironment) -> TestResult<serde_json::Value> {
|
|
||||||
let body = cluster_admin_ok(cluster, Method::GET, "/rustfs/admin/v3/pools/list", None).await?;
|
|
||||||
Ok(serde_json::from_str(&body)?)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn retrying_put(client: &Client, bucket: &str, key: &str, body: Vec<u8>, timeout: Duration) -> TestResult {
|
|
||||||
wait_until(
|
|
||||||
timeout,
|
|
||||||
|| {
|
|
||||||
let client = client.clone();
|
|
||||||
let bucket = bucket.to_string();
|
|
||||||
let key = key.to_string();
|
|
||||||
let body = body.clone();
|
|
||||||
async move {
|
|
||||||
match put_object(&client, &bucket, &key, body).await {
|
|
||||||
Ok(()) => Ok(true),
|
|
||||||
Err(error) => {
|
|
||||||
let message = error.to_string();
|
|
||||||
if message.contains("SlowDown")
|
|
||||||
|| message.contains("ServiceUnavailable")
|
|
||||||
|| message.contains("InternalError")
|
|
||||||
|| message.contains("503")
|
|
||||||
|| message.contains("500")
|
|
||||||
{
|
|
||||||
Ok(false)
|
|
||||||
} else {
|
|
||||||
Err(error)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
&format!("put {bucket}/{key} during data movement"),
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
pub(crate) async fn retrying_get_equals(
|
|
||||||
client: &Client,
|
|
||||||
bucket: &str,
|
|
||||||
key: &str,
|
|
||||||
expected: &[u8],
|
|
||||||
timeout: Duration,
|
|
||||||
) -> TestResult {
|
|
||||||
wait_until(
|
|
||||||
timeout,
|
|
||||||
|| async {
|
|
||||||
match get_object_bytes(client, bucket, key).await {
|
|
||||||
Ok(got) if got.as_slice() == expected => Ok(true),
|
|
||||||
Ok(_) => Ok(false),
|
|
||||||
Err(error) => {
|
|
||||||
let message = error.to_string();
|
|
||||||
if message.contains("NoSuchKey")
|
|
||||||
|| message.contains("SlowDown")
|
|
||||||
|| message.contains("ServiceUnavailable")
|
|
||||||
|| message.contains("InternalError")
|
|
||||||
|| message.contains("503")
|
|
||||||
|| message.contains("500")
|
|
||||||
{
|
|
||||||
Ok(false)
|
|
||||||
} else {
|
|
||||||
Err(error)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
&format!("get {bucket}/{key} during data movement"),
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn append_single_node_pool_extends_ellipses_volumes() {
|
|
||||||
let mut env =
|
|
||||||
RustFSTestClusterEnvironment::with_topology(ClusterTopology::per_node_pools(DRIVES_PER_NODE, vec![vec![0], vec![1]]))
|
|
||||||
.await
|
|
||||||
.expect("two-pool seed topology");
|
|
||||||
assert_eq!(env.rustfs_volumes_arg().split(' ').count(), 2);
|
|
||||||
|
|
||||||
let added = env.append_single_node_pool().await.expect("append third pool");
|
|
||||||
assert_eq!(added, 2);
|
|
||||||
assert_eq!(env.nodes.len(), 3);
|
|
||||||
assert_eq!(env.nodes[2].pool_idx, 2);
|
|
||||||
assert_eq!(env.nodes[2].data_dirs.len(), DRIVES_PER_NODE);
|
|
||||||
let volumes = env.rustfs_volumes_arg();
|
|
||||||
assert_eq!(volumes.split(' ').count(), 3, "expected three pool arguments, got: {volumes}");
|
|
||||||
assert!(volumes.contains("/drive{0...3}"), "expanded layout must keep drive ellipses: {volumes}");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn append_single_node_pool_rejects_striped_single_pool() {
|
|
||||||
let mut env = RustFSTestClusterEnvironment::new(4).await.expect("four-node single pool");
|
|
||||||
let err = env
|
|
||||||
.append_single_node_pool()
|
|
||||||
.await
|
|
||||||
.expect_err("a striped single pool cannot gain a localhost pool");
|
|
||||||
let message = err.to_string();
|
|
||||||
assert!(
|
|
||||||
message.contains("drives_per_node") || message.contains("one node per pool"),
|
|
||||||
"unexpected error: {message}"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(unix)]
|
|
||||||
#[tokio::test]
|
|
||||||
async fn cluster_start_fails_fast_when_node_process_exits() {
|
|
||||||
let mut dist = DistCluster::new_stopped(DistLayout::FourNodeFourDisk)
|
|
||||||
.await
|
|
||||||
.expect("stopped 4-node cluster");
|
|
||||||
let script = format!("{}/immediate-exit.sh", dist.cluster.temp_dir);
|
|
||||||
std::fs::write(&script, "#!/bin/sh\nexit 1\n").expect("write exit stub");
|
|
||||||
let mut perms = std::fs::metadata(&script).expect("stat exit stub").permissions();
|
|
||||||
std::os::unix::fs::PermissionsExt::set_mode(&mut perms, 0o755);
|
|
||||||
std::fs::set_permissions(&script, perms).expect("chmod exit stub");
|
|
||||||
|
|
||||||
let started = Instant::now();
|
|
||||||
let err = dist
|
|
||||||
.start_from_binary(Path::new(&script))
|
|
||||||
.await
|
|
||||||
.expect_err("a node that exits immediately must fail start");
|
|
||||||
let elapsed = started.elapsed();
|
|
||||||
let message = err.to_string();
|
|
||||||
assert!(
|
|
||||||
message.contains("exited before TCP ready") || message.contains("exited before S3 ready"),
|
|
||||||
"unexpected start error: {message}"
|
|
||||||
);
|
|
||||||
assert!(
|
|
||||||
elapsed < Duration::from_secs(30),
|
|
||||||
"cluster start must fail fast when a node exits, took {elapsed:?}"
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn decommission_complete_reads_pool_status_and_info_flag() {
|
|
||||||
let status = serde_json::json!({
|
|
||||||
"pools": [
|
|
||||||
{
|
|
||||||
"id": 0,
|
|
||||||
"status": "complete",
|
|
||||||
"poolStatus": "decommissioned",
|
|
||||||
"decommissionInfo": { "complete": true, "failed": false, "canceled": false }
|
|
||||||
},
|
|
||||||
{ "id": 1, "status": "none", "poolStatus": "active" }
|
|
||||||
]
|
|
||||||
});
|
|
||||||
assert!(decommission_complete(&status, 0));
|
|
||||||
assert!(!decommission_complete(&status, 1));
|
|
||||||
assert!(!decommission_failed(&status, 0));
|
|
||||||
assert!(decommission_progress(&status, 0).expect("complete pool"));
|
|
||||||
assert!(!decommission_progress(&status, 1).expect("other pool is not complete"));
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn decommission_progress_fails_closed_on_failed_flag() {
|
|
||||||
let failed = serde_json::json!({
|
|
||||||
"pools": [{
|
|
||||||
"id": 0,
|
|
||||||
"status": "failed",
|
|
||||||
"decommissionInfo": { "complete": false, "failed": true, "canceled": false }
|
|
||||||
}]
|
|
||||||
});
|
|
||||||
let err = decommission_progress(&failed, 0).expect_err("failed decommission must not look complete");
|
|
||||||
assert!(err.contains("decommission failed for pool 0"), "{err}");
|
|
||||||
assert!(!decommission_progress(&failed, 1).expect("missing pool is still running"));
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn rebalance_active_treats_started_as_in_progress() {
|
|
||||||
let started = serde_json::json!({ "pools": [{ "id": 0, "status": "Started", "stopping": false }] });
|
|
||||||
let done = serde_json::json!({ "pools": [{ "id": 0, "status": "Completed", "stopping": false }] });
|
|
||||||
assert!(rebalance_active(&started));
|
|
||||||
assert!(!rebalance_active(&done));
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn pool_meta_write_fence_matches_known_product_gates() {
|
|
||||||
assert!(is_pool_meta_write_fence(
|
|
||||||
"heal_bucket: pool metadata writes remain blocked after a recovery-required replica state"
|
|
||||||
));
|
|
||||||
assert!(is_pool_meta_write_fence(
|
|
||||||
"rebalance meta save failed: pool activation requires a live fleet capability proof"
|
|
||||||
));
|
|
||||||
assert!(is_pool_meta_write_fence("pool metadata recovery required: no durable bootstrap identity"));
|
|
||||||
assert!(!is_pool_meta_write_fence("NotImplemented: single pool cannot decommission"));
|
|
||||||
assert!(!is_pool_meta_write_fence("AccessDenied"));
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn classify_data_movement_http_observes_product_refusals_not_auth_failures() {
|
|
||||||
assert!(matches!(classify_data_movement_http(StatusCode::OK, ""), Ok(DataMovementStart::Started)));
|
|
||||||
assert!(matches!(
|
|
||||||
classify_data_movement_http(
|
|
||||||
StatusCode::BAD_REQUEST,
|
|
||||||
"failed to start decommission: single pool deployments do not support decommission"
|
|
||||||
),
|
|
||||||
Ok(DataMovementStart::Refused(_))
|
|
||||||
));
|
|
||||||
assert!(matches!(
|
|
||||||
classify_data_movement_http(
|
|
||||||
StatusCode::BAD_REQUEST,
|
|
||||||
"failed to start decommission: at least one active pool must remain after decommission start"
|
|
||||||
),
|
|
||||||
Ok(DataMovementStart::Refused(_))
|
|
||||||
));
|
|
||||||
assert!(matches!(
|
|
||||||
classify_data_movement_http(StatusCode::NOT_IMPLEMENTED, "NotImplemented"),
|
|
||||||
Ok(DataMovementStart::Refused(_))
|
|
||||||
));
|
|
||||||
assert!(matches!(
|
|
||||||
classify_data_movement_http(
|
|
||||||
StatusCode::INTERNAL_SERVER_ERROR,
|
|
||||||
"pool metadata writes remain blocked after a recovery-required replica state"
|
|
||||||
),
|
|
||||||
Ok(DataMovementStart::Refused(_))
|
|
||||||
));
|
|
||||||
assert!(matches!(
|
|
||||||
classify_data_movement_http(StatusCode::INTERNAL_SERVER_ERROR, "InternalError"),
|
|
||||||
Ok(DataMovementStart::Refused(_))
|
|
||||||
));
|
|
||||||
let denied = classify_data_movement_http(StatusCode::FORBIDDEN, "AccessDenied").expect_err("auth failure is not a refusal");
|
|
||||||
assert!(denied.contains("AccessDenied"), "{denied}");
|
|
||||||
let unavailable = classify_data_movement_http(StatusCode::SERVICE_UNAVAILABLE, "ServiceUnavailable")
|
|
||||||
.expect_err("503 is not a product refusal");
|
|
||||||
assert!(unavailable.contains("ServiceUnavailable"), "{unavailable}");
|
|
||||||
let bad_gateway =
|
|
||||||
classify_data_movement_http(StatusCode::BAD_GATEWAY, "Bad Gateway").expect_err("502 is not a product refusal");
|
|
||||||
assert!(bad_gateway.contains("502") || bad_gateway.contains("Bad Gateway"), "{bad_gateway}");
|
|
||||||
}
|
|
||||||
@@ -1,35 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
//! 4-node 4-drive distributed e2e coverage.
|
|
||||||
//!
|
|
||||||
//! Selected by `[profile.e2e-distributed]` and run from
|
|
||||||
//! `.github/workflows/e2e-distributed.yml`. Excluded from `e2e-full` because
|
|
||||||
//! each case starts four real `rustfs` processes.
|
|
||||||
|
|
||||||
mod chaos_test;
|
|
||||||
mod concurrency_stability_test;
|
|
||||||
mod concurrent_data_movement_test;
|
|
||||||
mod data_integrity_movement_test;
|
|
||||||
mod expand_decommission_rebalance_test;
|
|
||||||
mod extra_test;
|
|
||||||
mod harness;
|
|
||||||
mod object_lock_test;
|
|
||||||
mod observability_test;
|
|
||||||
mod replication_quota_test;
|
|
||||||
mod s3_basic_test;
|
|
||||||
mod s3_during_data_movement_test;
|
|
||||||
mod site_replication_test;
|
|
||||||
mod upgrade_test;
|
|
||||||
mod versioning_test;
|
|
||||||
@@ -1,111 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{DistCluster, DistLayout, TestResult, unique_bucket};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use crate::object_lock::common::{delete_object_with_bypass, put_object_with_legal_hold, put_object_with_retention};
|
|
||||||
use aws_sdk_s3::Client;
|
|
||||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
|
||||||
use aws_sdk_s3::error::SdkError;
|
|
||||||
use aws_sdk_s3::operation::delete_object::DeleteObjectError;
|
|
||||||
use aws_sdk_s3::types::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode};
|
|
||||||
use chrono::{Duration as ChronoDuration, Utc};
|
|
||||||
|
|
||||||
fn delete_denied(error: &SdkError<DeleteObjectError>, context: &str) -> TestResult {
|
|
||||||
let code = error.as_service_error().and_then(ProvideErrorMetadata::code);
|
|
||||||
if code == Some("AccessDenied") {
|
|
||||||
Ok(())
|
|
||||||
} else {
|
|
||||||
Err(format!("{context}: expected AccessDenied, got {error:?}").into())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn expect_versioned_delete_denied(
|
|
||||||
client: &Client,
|
|
||||||
bucket: &str,
|
|
||||||
key: &str,
|
|
||||||
version_id: &str,
|
|
||||||
bypass: bool,
|
|
||||||
context: &str,
|
|
||||||
) -> TestResult {
|
|
||||||
match delete_object_with_bypass(client, bucket, key, Some(version_id), bypass).await {
|
|
||||||
Ok(_) => Err(format!("{context}: DeleteObject of retained version must be denied").into()),
|
|
||||||
Err(error) => delete_denied(error.as_ref(), context),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_object_lock_worm_blocks_delete() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let client = dist.client(0)?;
|
|
||||||
let peer = dist.client(2)?;
|
|
||||||
let bucket = unique_bucket("objlock");
|
|
||||||
|
|
||||||
client
|
|
||||||
.create_bucket()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.object_lock_enabled_for_bucket(true)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let retain_until = Utc::now() + ChronoDuration::days(1);
|
|
||||||
|
|
||||||
let compliance_key = "compliance.bin";
|
|
||||||
let compliance_version = put_object_with_retention(
|
|
||||||
&client,
|
|
||||||
&bucket,
|
|
||||||
compliance_key,
|
|
||||||
b"locked-compliance",
|
|
||||||
ObjectLockRetentionMode::Compliance,
|
|
||||||
retain_until,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
// Unversioned DELETE is allowed: it only creates a delete marker. WORM
|
|
||||||
// applies to a specific version id.
|
|
||||||
let marker = peer.delete_object().bucket(&bucket).key(compliance_key).send().await?;
|
|
||||||
assert_eq!(
|
|
||||||
marker.delete_marker(),
|
|
||||||
Some(true),
|
|
||||||
"unversioned DELETE on a locked object must create a delete marker"
|
|
||||||
);
|
|
||||||
|
|
||||||
expect_versioned_delete_denied(&peer, &bucket, compliance_key, &compliance_version, false, "COMPLIANCE without bypass")
|
|
||||||
.await?;
|
|
||||||
expect_versioned_delete_denied(&peer, &bucket, compliance_key, &compliance_version, true, "COMPLIANCE with bypass").await?;
|
|
||||||
|
|
||||||
let governance_key = "governance.bin";
|
|
||||||
let governance_version = put_object_with_retention(
|
|
||||||
&client,
|
|
||||||
&bucket,
|
|
||||||
governance_key,
|
|
||||||
b"locked-governance",
|
|
||||||
ObjectLockRetentionMode::Governance,
|
|
||||||
retain_until,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
expect_versioned_delete_denied(&peer, &bucket, governance_key, &governance_version, false, "GOVERNANCE without bypass")
|
|
||||||
.await?;
|
|
||||||
delete_object_with_bypass(&peer, &bucket, governance_key, Some(&governance_version), true).await?;
|
|
||||||
|
|
||||||
let hold_key = "legal-hold.bin";
|
|
||||||
let hold_version =
|
|
||||||
put_object_with_legal_hold(&client, &bucket, hold_key, b"legal-hold", ObjectLockLegalHoldStatus::On).await?;
|
|
||||||
expect_versioned_delete_denied(&peer, &bucket, hold_key, &hold_version, false, "legal hold without bypass").await?;
|
|
||||||
expect_versioned_delete_denied(&peer, &bucket, hold_key, &hold_version, true, "legal hold with bypass").await?;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,80 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, cluster_admin, cluster_admin_ok, put_object, unique_bucket, wait_for_ready,
|
|
||||||
};
|
|
||||||
use crate::common::{init_logging, local_http_client};
|
|
||||||
use http::Method;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_health_admin_info_and_audit_list() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
wait_for_ready(&dist.cluster).await?;
|
|
||||||
|
|
||||||
let http = local_http_client();
|
|
||||||
for node in &dist.cluster.nodes {
|
|
||||||
let ready = http.get(format!("{}/health/ready", node.url)).send().await?;
|
|
||||||
assert!(ready.status().is_success(), "node {} not ready: {}", node.address, ready.status());
|
|
||||||
let live = http.get(format!("{}/health/live", node.url)).send().await;
|
|
||||||
if let Ok(response) = live {
|
|
||||||
assert!(
|
|
||||||
response.status().is_success() || response.status().as_u16() == 404,
|
|
||||||
"unexpected live probe on {}: {}",
|
|
||||||
node.address,
|
|
||||||
response.status()
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
let info = cluster_admin_ok(&dist.cluster, Method::GET, "/rustfs/admin/v3/info", None).await?;
|
|
||||||
assert!(!info.is_empty(), "admin info was empty");
|
|
||||||
let storage = cluster_admin_ok(&dist.cluster, Method::GET, "/rustfs/admin/v3/storageinfo", None).await?;
|
|
||||||
assert!(
|
|
||||||
storage.contains("disks") || storage.contains("backend") || storage.contains("info"),
|
|
||||||
"storageinfo missing expected fields: {storage}"
|
|
||||||
);
|
|
||||||
|
|
||||||
let audit = cluster_admin_ok(&dist.cluster, Method::GET, "/rustfs/admin/v3/audit/target/list", None).await?;
|
|
||||||
let trimmed = audit.trim();
|
|
||||||
if !trimmed.is_empty() && trimmed != "null" && !trimmed.starts_with('[') && !trimmed.starts_with('{') {
|
|
||||||
return Err(format!("audit target list was not machine-readable: {audit}").into());
|
|
||||||
}
|
|
||||||
|
|
||||||
// Optional surfaces: 404/400/501 are acceptable (route missing or stubbed);
|
|
||||||
// unexpected 5xx is not. A 2xx body must be non-empty.
|
|
||||||
for path in [
|
|
||||||
"/rustfs/admin/v3/log/search",
|
|
||||||
"/rustfs/admin/v4/runtime/capabilities",
|
|
||||||
"/minio/v2/metrics/cluster",
|
|
||||||
] {
|
|
||||||
let (status, body) = cluster_admin(&dist.cluster, Method::GET, path, None).await?;
|
|
||||||
assert!(
|
|
||||||
status.is_success() || status.is_client_error() || status.as_u16() == 501,
|
|
||||||
"observability path {path} returned {status}: {body}"
|
|
||||||
);
|
|
||||||
if status.is_success() {
|
|
||||||
assert!(!body.trim().is_empty(), "empty body from {path}");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
let bucket = unique_bucket("obs");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
put_object(&dist.client(0)?, &bucket, "probe.log", b"observability".to_vec()).await?;
|
|
||||||
|
|
||||||
let trace = cluster_admin_ok(&dist.cluster, Method::GET, "/rustfs/admin/v3/info", None).await?;
|
|
||||||
assert!(!trace.is_empty());
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,144 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, enable_versioning, put_bucket_replication, put_object, retrying_put, set_bucket_quota,
|
|
||||||
set_remote_target, unique_bucket, wait_for_replicated_bytes, wait_until,
|
|
||||||
};
|
|
||||||
use crate::common::{FAST_DATA_USAGE_SCANNER_ENV, init_logging};
|
|
||||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
|
||||||
use http::Method;
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
/// `Ok(true)` quota admission rejected the PUT, `Ok(false)` retry, `Err` not quota.
|
|
||||||
fn quota_over_limit_put_outcome(code: Option<&str>, message: Option<&str>) -> Result<bool, String> {
|
|
||||||
let quota_message = message.is_some_and(|text| text.starts_with("Bucket quota exceeded"));
|
|
||||||
match code {
|
|
||||||
Some("InvalidRequest" | "QuotaExceeded") if quota_message => Ok(true),
|
|
||||||
Some("SlowDown" | "ServiceUnavailable") => Ok(false),
|
|
||||||
Some("AccessDenied") => Err("AccessDenied is not a quota admission rejection".to_string()),
|
|
||||||
Some("InvalidRequest" | "QuotaExceeded") => {
|
|
||||||
Err(format!("InvalidRequest/QuotaExceeded without quota admission message: {message:?}"))
|
|
||||||
}
|
|
||||||
other => Err(format!("unexpected over-quota error code {other:?} message {message:?}")),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_bucket_replication_converges_to_peer_cluster() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let (source, target) = DistCluster::start_replication_pair().await?;
|
|
||||||
let source_bucket = unique_bucket("replsrc");
|
|
||||||
let target_bucket = unique_bucket("repldst");
|
|
||||||
source.create_bucket(&source_bucket).await?;
|
|
||||||
target.create_bucket(&target_bucket).await?;
|
|
||||||
|
|
||||||
let source_client = source.client(0)?;
|
|
||||||
let target_client = target.client(0)?;
|
|
||||||
enable_versioning(&source_client, &source_bucket).await?;
|
|
||||||
enable_versioning(&target_client, &target_bucket).await?;
|
|
||||||
|
|
||||||
let arn = set_remote_target(&source.cluster, &source_bucket, &target.cluster, &target_bucket).await?;
|
|
||||||
put_bucket_replication(&source.cluster, &source_bucket, &arn).await?;
|
|
||||||
|
|
||||||
let key = "replicated.bin";
|
|
||||||
let body = b"distributed-bucket-replication".to_vec();
|
|
||||||
put_object(&source_client, &source_bucket, key, body.clone()).await?;
|
|
||||||
wait_for_replicated_bytes(&target_client, &target_bucket, key, &body, Duration::from_secs(45)).await?;
|
|
||||||
|
|
||||||
let peer_read = target.client(3)?;
|
|
||||||
wait_for_replicated_bytes(&peer_read, &target_bucket, key, &body, Duration::from_secs(15)).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_hard_quota_rejects_over_limit_put() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start_with_env(DistLayout::FourByFour, FAST_DATA_USAGE_SCANNER_ENV).await?;
|
|
||||||
let bucket = unique_bucket("quota");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
set_bucket_quota(&dist.cluster, &bucket, 8 * 1024).await?;
|
|
||||||
|
|
||||||
let client = dist.client(1)?;
|
|
||||||
retrying_put(&client, &bucket, "small.bin", vec![0u8; 1024], Duration::from_secs(30)).await?;
|
|
||||||
wait_until(
|
|
||||||
Duration::from_secs(30),
|
|
||||||
|| async {
|
|
||||||
let (status, body) = super::harness::cluster_admin(
|
|
||||||
&dist.cluster,
|
|
||||||
Method::GET,
|
|
||||||
&format!("/rustfs/admin/v3/quota-stats/{bucket}"),
|
|
||||||
None,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
if !status.is_success() {
|
|
||||||
return Ok(false);
|
|
||||||
}
|
|
||||||
let stats: serde_json::Value = serde_json::from_str(&body).unwrap_or_default();
|
|
||||||
Ok(stats.get("current_usage").and_then(serde_json::Value::as_u64).unwrap_or(0) >= 1024)
|
|
||||||
},
|
|
||||||
"quota stats observe small object",
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let mut oversized_attempt = 0u32;
|
|
||||||
wait_until(
|
|
||||||
Duration::from_secs(30),
|
|
||||||
|| {
|
|
||||||
oversized_attempt += 1;
|
|
||||||
let key = format!("too-big-{oversized_attempt}.bin");
|
|
||||||
let client = client.clone();
|
|
||||||
let bucket = bucket.clone();
|
|
||||||
async move {
|
|
||||||
match client
|
|
||||||
.put_object()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.body(vec![0u8; 16 * 1024].into())
|
|
||||||
.send()
|
|
||||||
.await
|
|
||||||
{
|
|
||||||
Ok(_) => Ok(false),
|
|
||||||
Err(error) => {
|
|
||||||
let code = error.as_service_error().and_then(ProvideErrorMetadata::code);
|
|
||||||
let message = error.as_service_error().and_then(ProvideErrorMetadata::message);
|
|
||||||
match quota_over_limit_put_outcome(code, message) {
|
|
||||||
Ok(done) => Ok(done),
|
|
||||||
Err(detail) => Err(format!("{detail}: {error:?}").into()),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
},
|
|
||||||
"hard quota rejects oversized PUT",
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn quota_over_limit_put_outcome_requires_quota_admission() {
|
|
||||||
assert_eq!(
|
|
||||||
quota_over_limit_put_outcome(Some("InvalidRequest"), Some("Bucket quota exceeded for bucket x")),
|
|
||||||
Ok(true)
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
quota_over_limit_put_outcome(Some("QuotaExceeded"), Some("Bucket quota exceeded")),
|
|
||||||
Ok(true)
|
|
||||||
);
|
|
||||||
assert_eq!(quota_over_limit_put_outcome(Some("SlowDown"), Some("slow down")), Ok(false));
|
|
||||||
assert_eq!(quota_over_limit_put_outcome(Some("ServiceUnavailable"), Some("unavailable")), Ok(false));
|
|
||||||
assert!(quota_over_limit_put_outcome(Some("AccessDenied"), Some("Access Denied")).is_err());
|
|
||||||
assert!(quota_over_limit_put_outcome(Some("InvalidRequest"), Some("invalid argument")).is_err());
|
|
||||||
}
|
|
||||||
@@ -1,111 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/licenses/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{DistCluster, DistLayout, TestResult, assert_object_bytes, get_object_bytes, put_object, unique_bucket};
|
|
||||||
use crate::common::{init_logging, local_http_client};
|
|
||||||
use aws_sdk_s3::presigning::PresigningConfig;
|
|
||||||
use aws_sdk_s3::types::{Delete, MetadataDirective, ObjectIdentifier};
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_s3_put_get_head_list_copy_rename_delete_and_presign() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("s3basic");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
|
|
||||||
let writer = dist.client(0)?;
|
|
||||||
let reader = dist.client(3)?;
|
|
||||||
let key = "dir/object.bin";
|
|
||||||
let body = vec![0xA5u8; 256 * 1024];
|
|
||||||
put_object(&writer, &bucket, key, body.clone()).await?;
|
|
||||||
|
|
||||||
let head = reader.head_object().bucket(&bucket).key(key).send().await?;
|
|
||||||
assert_eq!(head.content_length(), Some(body.len() as i64));
|
|
||||||
assert_object_bytes(&reader, &bucket, key, &body).await?;
|
|
||||||
|
|
||||||
let ranged = reader
|
|
||||||
.get_object()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.range("bytes=0-15")
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
let ranged_body = ranged.body.collect().await?.into_bytes();
|
|
||||||
assert_eq!(ranged_body.as_ref(), &body[..16]);
|
|
||||||
|
|
||||||
let listed = reader.list_objects_v2().bucket(&bucket).prefix("dir/").send().await?;
|
|
||||||
let keys: Vec<_> = listed.contents().iter().filter_map(|object| object.key()).collect();
|
|
||||||
assert_eq!(keys, vec![key]);
|
|
||||||
|
|
||||||
let copy_key = "dir/object-copy.bin";
|
|
||||||
reader
|
|
||||||
.copy_object()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(copy_key)
|
|
||||||
.copy_source(format!("{bucket}/{key}"))
|
|
||||||
.metadata_directive(MetadataDirective::Copy)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
assert_object_bytes(&writer, &bucket, copy_key, &body).await?;
|
|
||||||
|
|
||||||
let moved_key = "dir/object-moved.bin";
|
|
||||||
writer
|
|
||||||
.copy_object()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(moved_key)
|
|
||||||
.copy_source(format!("{bucket}/{copy_key}"))
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
writer.delete_object().bucket(&bucket).key(copy_key).send().await?;
|
|
||||||
match writer.head_object().bucket(&bucket).key(copy_key).send().await {
|
|
||||||
Ok(_) => return Err("copied source still present after rename delete".into()),
|
|
||||||
Err(error) if error.as_service_error().is_some_and(|err| err.is_not_found()) => {}
|
|
||||||
Err(error) => return Err(error.into()),
|
|
||||||
}
|
|
||||||
assert_object_bytes(&reader, &bucket, moved_key, &body).await?;
|
|
||||||
|
|
||||||
let presigned = writer
|
|
||||||
.get_object()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.presigned(PresigningConfig::expires_in(Duration::from_secs(120))?)
|
|
||||||
.await?;
|
|
||||||
let response = local_http_client().get(presigned.uri().to_string()).send().await?;
|
|
||||||
assert!(response.status().is_success(), "presigned GET failed: {}", response.status());
|
|
||||||
let presigned_body = response.bytes().await?;
|
|
||||||
assert_eq!(presigned_body.as_ref(), body.as_slice());
|
|
||||||
|
|
||||||
let empty_key = "empty";
|
|
||||||
put_object(&writer, &bucket, empty_key, Vec::new()).await?;
|
|
||||||
let empty = get_object_bytes(&reader, &bucket, empty_key).await?;
|
|
||||||
assert!(empty.is_empty());
|
|
||||||
|
|
||||||
writer
|
|
||||||
.delete_objects()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.delete(
|
|
||||||
Delete::builder()
|
|
||||||
.objects(ObjectIdentifier::builder().key(key).build()?)
|
|
||||||
.objects(ObjectIdentifier::builder().key(moved_key).build()?)
|
|
||||||
.objects(ObjectIdentifier::builder().key(empty_key).build()?)
|
|
||||||
.build()?,
|
|
||||||
)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let remaining = reader.list_objects_v2().bucket(&bucket).send().await?;
|
|
||||||
assert!(remaining.contents().is_empty(), "bucket still has objects after delete");
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,82 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, assert_inventory, decommission_started_or_refused, put_inventory_retrying,
|
|
||||||
rebalance_started_or_refused, retrying_get_equals, retrying_put, unique_bucket, wait_for_decommission_complete,
|
|
||||||
};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn s3_put_get_list_succeed_during_decommission_and_rebalance() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("s3move");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let client = dist.client(0)?;
|
|
||||||
let inventory = put_inventory_retrying(&client, &bucket, 8, 16 * 1024, Duration::from_secs(30)).await?;
|
|
||||||
|
|
||||||
let decommission_started = decommission_started_or_refused(&dist.cluster, 0).await?;
|
|
||||||
let live = dist.client(2)?;
|
|
||||||
retrying_put(
|
|
||||||
&live,
|
|
||||||
&bucket,
|
|
||||||
"during-decommission.bin",
|
|
||||||
b"written-while-decommissioning".to_vec(),
|
|
||||||
Duration::from_secs(30),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
retrying_get_equals(
|
|
||||||
&live,
|
|
||||||
&bucket,
|
|
||||||
"during-decommission.bin",
|
|
||||||
b"written-while-decommissioning",
|
|
||||||
Duration::from_secs(30),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
let listed = live.list_objects_v2().bucket(&bucket).send().await?;
|
|
||||||
assert!(
|
|
||||||
listed
|
|
||||||
.contents()
|
|
||||||
.iter()
|
|
||||||
.any(|object| object.key() == Some("during-decommission.bin")),
|
|
||||||
"list during decommission missed the newly written key"
|
|
||||||
);
|
|
||||||
|
|
||||||
if decommission_started {
|
|
||||||
wait_for_decommission_complete(&dist.cluster, 0, Duration::from_secs(180)).await?;
|
|
||||||
}
|
|
||||||
assert_inventory(&live, &bucket, &inventory).await?;
|
|
||||||
|
|
||||||
let _ = rebalance_started_or_refused(&dist.cluster).await?;
|
|
||||||
retrying_put(
|
|
||||||
&live,
|
|
||||||
&bucket,
|
|
||||||
"during-rebalance.bin",
|
|
||||||
b"written-while-rebalancing".to_vec(),
|
|
||||||
Duration::from_secs(30),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
retrying_get_equals(
|
|
||||||
&live,
|
|
||||||
&bucket,
|
|
||||||
"during-rebalance.bin",
|
|
||||||
b"written-while-rebalancing",
|
|
||||||
Duration::from_secs(30),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
assert_inventory(&dist.client(1)?, &bucket, &inventory).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,87 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, TestResult, cluster_admin_ok, enable_versioning, put_object, unique_bucket, wait_for_replicated_bytes,
|
|
||||||
};
|
|
||||||
use crate::common::{init_logging, signed_request};
|
|
||||||
use http::{Method, StatusCode};
|
|
||||||
use rustfs_madmin::PeerSite;
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
async fn site_replication_add(cluster: &crate::common::RustFSTestClusterEnvironment, sites: &[PeerSite]) -> TestResult<String> {
|
|
||||||
let url = format!("{}/rustfs/admin/v3/site-replication/add?replicateILMExpiry=false", cluster.nodes[0].url);
|
|
||||||
let response = signed_request(
|
|
||||||
Method::PUT,
|
|
||||||
&url,
|
|
||||||
&cluster.access_key,
|
|
||||||
&cluster.secret_key,
|
|
||||||
Some(serde_json::to_vec(sites)?),
|
|
||||||
Some("application/json"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
if response.status() != StatusCode::OK {
|
|
||||||
let status = response.status();
|
|
||||||
let body = response.text().await.unwrap_or_default();
|
|
||||||
return Err(format!("site replication add failed: {status} {body}").into());
|
|
||||||
}
|
|
||||||
Ok(response.text().await?)
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_site_replication_replicates_object_to_peer_site() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let (site_a, site_b) = DistCluster::start_replication_pair().await?;
|
|
||||||
let bucket = unique_bucket("siterepl");
|
|
||||||
site_a.create_bucket(&bucket).await?;
|
|
||||||
site_b.create_bucket(&bucket).await?;
|
|
||||||
|
|
||||||
let client_a = site_a.client(0)?;
|
|
||||||
let client_b = site_b.client(0)?;
|
|
||||||
enable_versioning(&client_a, &bucket).await?;
|
|
||||||
enable_versioning(&client_b, &bucket).await?;
|
|
||||||
|
|
||||||
let sites = vec![
|
|
||||||
PeerSite {
|
|
||||||
name: "site-a".to_string(),
|
|
||||||
endpoint: site_a.cluster.nodes[0].url.clone(),
|
|
||||||
access_key: site_a.cluster.access_key.clone(),
|
|
||||||
secret_key: site_a.cluster.secret_key.clone(),
|
|
||||||
..Default::default()
|
|
||||||
},
|
|
||||||
PeerSite {
|
|
||||||
name: "site-b".to_string(),
|
|
||||||
endpoint: site_b.cluster.nodes[0].url.clone(),
|
|
||||||
access_key: site_b.cluster.access_key.clone(),
|
|
||||||
secret_key: site_b.cluster.secret_key.clone(),
|
|
||||||
..Default::default()
|
|
||||||
},
|
|
||||||
];
|
|
||||||
site_replication_add(&site_a.cluster, &sites).await?;
|
|
||||||
|
|
||||||
let info = cluster_admin_ok(&site_a.cluster, Method::GET, "/rustfs/admin/v3/site-replication/info", None).await?;
|
|
||||||
assert!(
|
|
||||||
info.contains("site-a") || info.contains("enabled") || info.contains("true"),
|
|
||||||
"site replication info did not show a configured peer: {info}"
|
|
||||||
);
|
|
||||||
|
|
||||||
let key = "site-object.bin";
|
|
||||||
let body = b"four-node-site-replication".to_vec();
|
|
||||||
put_object(&client_a, &bucket, key, body.clone()).await?;
|
|
||||||
wait_for_replicated_bytes(&client_b, &bucket, key, &body, Duration::from_secs(60)).await?;
|
|
||||||
|
|
||||||
let peer_b = site_b.client(3)?;
|
|
||||||
wait_for_replicated_bytes(&peer_b, &bucket, key, &body, Duration::from_secs(20)).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -1,363 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
//! 4-node upgrade coverage for historical objects and IAM AK/SK.
|
|
||||||
//!
|
|
||||||
//! Complements `upgrade_compatibility_test` (single-node SSE/multipart and
|
|
||||||
//! mixed-version listing). This module pins the distributed contract the
|
|
||||||
//! hardware upgrade chain is meant to catch: after a 4-node upgrade, objects
|
|
||||||
//! written on the previous release still read back, and IAM user credentials
|
|
||||||
//! created before the upgrade still authenticate.
|
|
||||||
//!
|
|
||||||
//! Requires `RUSTFS_UPGRADE_SOURCE_BINARY` pointing at the pinned previous
|
|
||||||
//! release. The `e2e-distributed` workflow downloads that binary; a local run
|
|
||||||
//! without it fails closed rather than skipping.
|
|
||||||
|
|
||||||
use super::harness::{
|
|
||||||
DistCluster, DistLayout, TestResult, assert_object_bytes, cluster_admin_ok, enable_versioning, get_object_bytes, put_object,
|
|
||||||
unique_bucket, wait_until,
|
|
||||||
};
|
|
||||||
use crate::common::{
|
|
||||||
AdminTransport, admin_add_canned_policy_via, admin_attach_user_policy_via, admin_create_user_via, init_logging,
|
|
||||||
};
|
|
||||||
use aws_sdk_s3::Client;
|
|
||||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
|
||||||
use std::ffi::OsString;
|
|
||||||
use std::path::{Path, PathBuf};
|
|
||||||
use std::time::Duration;
|
|
||||||
use uuid::Uuid;
|
|
||||||
|
|
||||||
const SOURCE_BINARY_ENV: &str = "RUSTFS_UPGRADE_SOURCE_BINARY";
|
|
||||||
const IAM_SECRET: &str = "UpgradeTestSecretKey1";
|
|
||||||
const WRONG_SECRET: &str = "WrongSecretKey000000";
|
|
||||||
const CREDENTIAL_TIMEOUT: Duration = Duration::from_secs(30);
|
|
||||||
|
|
||||||
struct UpgradeSeed {
|
|
||||||
history_bucket: String,
|
|
||||||
history_key: &'static str,
|
|
||||||
history_body: Vec<u8>,
|
|
||||||
versioned_bucket: String,
|
|
||||||
versioned_key: &'static str,
|
|
||||||
version1: String,
|
|
||||||
version1_body: Vec<u8>,
|
|
||||||
version2: String,
|
|
||||||
version2_body: Vec<u8>,
|
|
||||||
iam_bucket: String,
|
|
||||||
iam_key: &'static str,
|
|
||||||
iam_body: Vec<u8>,
|
|
||||||
iam_user: String,
|
|
||||||
iam_secret: &'static str,
|
|
||||||
}
|
|
||||||
|
|
||||||
fn resolve_source_binary(value: Option<OsString>) -> TestResult<PathBuf> {
|
|
||||||
let path = value.map(PathBuf::from).ok_or_else(|| {
|
|
||||||
format!(
|
|
||||||
"{SOURCE_BINARY_ENV} must point to the pinned previous release binary (the e2e-distributed workflow downloads it)"
|
|
||||||
)
|
|
||||||
})?;
|
|
||||||
if !path.is_file() {
|
|
||||||
return Err(format!("upgrade source binary does not exist: {}", path.display()).into());
|
|
||||||
}
|
|
||||||
Ok(path)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn source_binary() -> TestResult<PathBuf> {
|
|
||||||
resolve_source_binary(std::env::var_os(SOURCE_BINARY_ENV))
|
|
||||||
}
|
|
||||||
|
|
||||||
fn capture_upgrade_logs(cluster: &mut DistCluster, label: &str) -> TestResult {
|
|
||||||
let Some(log_dir) = std::env::var_os("RUSTFS_E2E_LOG_DIR") else {
|
|
||||||
return Ok(());
|
|
||||||
};
|
|
||||||
std::fs::create_dir_all(&log_dir)?;
|
|
||||||
for node_idx in 0..cluster.cluster.nodes.len() {
|
|
||||||
let path = Path::new(&log_dir).join(format!("{label}-node-{node_idx}.log"));
|
|
||||||
cluster
|
|
||||||
.cluster
|
|
||||||
.set_node_capture_log_path(node_idx, path.to_string_lossy().into_owned())?;
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn iam_rw_policy(bucket: &str) -> String {
|
|
||||||
serde_json::json!({
|
|
||||||
"Version": "2012-10-17",
|
|
||||||
"Statement": [{
|
|
||||||
"Effect": "Allow",
|
|
||||||
"Action": ["s3:*"],
|
|
||||||
"Resource": [
|
|
||||||
format!("arn:aws:s3:::{bucket}"),
|
|
||||||
format!("arn:aws:s3:::{bucket}/*")
|
|
||||||
]
|
|
||||||
}]
|
|
||||||
})
|
|
||||||
.to_string()
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn create_iam_user(dist: &DistCluster, user: &str, secret: &str, policy_name: &str, bucket: &str) -> TestResult {
|
|
||||||
let url = &dist.cluster.nodes[0].url;
|
|
||||||
let access = &dist.cluster.access_key;
|
|
||||||
let admin_secret = &dist.cluster.secret_key;
|
|
||||||
admin_create_user_via(AdminTransport::Signed, url, access, admin_secret, user, secret).await?;
|
|
||||||
admin_add_canned_policy_via(AdminTransport::Signed, url, access, admin_secret, policy_name, &iam_rw_policy(bucket)).await?;
|
|
||||||
admin_attach_user_policy_via(AdminTransport::Signed, url, access, admin_secret, policy_name, user).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn wait_for_put(client: &Client, bucket: &str, key: &str, body: Vec<u8>, label: &str) -> TestResult {
|
|
||||||
wait_until(
|
|
||||||
CREDENTIAL_TIMEOUT,
|
|
||||||
|| {
|
|
||||||
let client = client.clone();
|
|
||||||
let bucket = bucket.to_string();
|
|
||||||
let key = key.to_string();
|
|
||||||
let body = body.clone();
|
|
||||||
async move {
|
|
||||||
put_object(&client, &bucket, &key, body).await?;
|
|
||||||
Ok(true)
|
|
||||||
}
|
|
||||||
},
|
|
||||||
label,
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn wait_for_bytes(client: &Client, bucket: &str, key: &str, expected: &[u8], label: &str) -> TestResult {
|
|
||||||
wait_until(
|
|
||||||
CREDENTIAL_TIMEOUT,
|
|
||||||
|| {
|
|
||||||
let client = client.clone();
|
|
||||||
let bucket = bucket.to_string();
|
|
||||||
let key = key.to_string();
|
|
||||||
let expected = expected.to_vec();
|
|
||||||
async move {
|
|
||||||
let got = get_object_bytes(&client, &bucket, &key).await?;
|
|
||||||
Ok(got == expected)
|
|
||||||
}
|
|
||||||
},
|
|
||||||
label,
|
|
||||||
)
|
|
||||||
.await
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn seed_history_and_iam(dist: &DistCluster) -> TestResult<UpgradeSeed> {
|
|
||||||
let history_bucket = unique_bucket("upg-hist");
|
|
||||||
let versioned_bucket = unique_bucket("upg-ver");
|
|
||||||
let iam_bucket = unique_bucket("upg-iam");
|
|
||||||
dist.create_bucket(&history_bucket).await?;
|
|
||||||
dist.create_bucket(&versioned_bucket).await?;
|
|
||||||
dist.create_bucket(&iam_bucket).await?;
|
|
||||||
|
|
||||||
let root = dist.client(0)?;
|
|
||||||
enable_versioning(&root, &versioned_bucket).await?;
|
|
||||||
|
|
||||||
let history_key = "plain-history.bin";
|
|
||||||
let history_body = b"written by the previous 4-node release".to_vec();
|
|
||||||
put_object(&root, &history_bucket, history_key, history_body.clone()).await?;
|
|
||||||
|
|
||||||
let versioned_key = "versioned-history.txt";
|
|
||||||
let version1_body = b"version-one-before-upgrade".to_vec();
|
|
||||||
let version1 = root
|
|
||||||
.put_object()
|
|
||||||
.bucket(&versioned_bucket)
|
|
||||||
.key(versioned_key)
|
|
||||||
.body(aws_sdk_s3::primitives::ByteStream::from(version1_body.clone()))
|
|
||||||
.send()
|
|
||||||
.await?
|
|
||||||
.version_id()
|
|
||||||
.ok_or("first versioned PUT omitted version ID")?
|
|
||||||
.to_string();
|
|
||||||
let version2_body = b"version-two-before-upgrade".to_vec();
|
|
||||||
let version2 = root
|
|
||||||
.put_object()
|
|
||||||
.bucket(&versioned_bucket)
|
|
||||||
.key(versioned_key)
|
|
||||||
.body(aws_sdk_s3::primitives::ByteStream::from(version2_body.clone()))
|
|
||||||
.send()
|
|
||||||
.await?
|
|
||||||
.version_id()
|
|
||||||
.ok_or("second versioned PUT omitted version ID")?
|
|
||||||
.to_string();
|
|
||||||
|
|
||||||
let iam_user = format!("upg{}", &Uuid::new_v4().simple().to_string()[..8]);
|
|
||||||
let policy_name = format!("upgpol{}", &Uuid::new_v4().simple().to_string()[..8]);
|
|
||||||
create_iam_user(dist, &iam_user, IAM_SECRET, &policy_name, &iam_bucket).await?;
|
|
||||||
|
|
||||||
let iam_key = "iam-history.bin";
|
|
||||||
let iam_body = b"written with pre-upgrade IAM AK/SK".to_vec();
|
|
||||||
let iam_client = dist.client_with_credentials(1, &iam_user, IAM_SECRET)?;
|
|
||||||
wait_for_put(&iam_client, &iam_bucket, iam_key, iam_body.clone(), "IAM user PUT before upgrade").await?;
|
|
||||||
|
|
||||||
Ok(UpgradeSeed {
|
|
||||||
history_bucket,
|
|
||||||
history_key,
|
|
||||||
history_body,
|
|
||||||
versioned_bucket,
|
|
||||||
versioned_key,
|
|
||||||
version1,
|
|
||||||
version1_body,
|
|
||||||
version2,
|
|
||||||
version2_body,
|
|
||||||
iam_bucket,
|
|
||||||
iam_key,
|
|
||||||
iam_body,
|
|
||||||
iam_user,
|
|
||||||
iam_secret: IAM_SECRET,
|
|
||||||
})
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn assert_history_and_iam(dist: &DistCluster, seed: &UpgradeSeed, context: &str) -> TestResult {
|
|
||||||
let root_a = dist.client(0)?;
|
|
||||||
let root_b = dist.client(3)?;
|
|
||||||
wait_for_bytes(
|
|
||||||
&root_b,
|
|
||||||
&seed.history_bucket,
|
|
||||||
seed.history_key,
|
|
||||||
&seed.history_body,
|
|
||||||
&format!("{context}: root GET historical object"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
assert_object_bytes(&root_a, &seed.history_bucket, seed.history_key, &seed.history_body).await?;
|
|
||||||
|
|
||||||
let v1 = root_b
|
|
||||||
.get_object()
|
|
||||||
.bucket(&seed.versioned_bucket)
|
|
||||||
.key(seed.versioned_key)
|
|
||||||
.version_id(&seed.version1)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
let v1_body = v1.body.collect().await?.into_bytes();
|
|
||||||
if v1_body.as_ref() != seed.version1_body.as_slice() {
|
|
||||||
return Err(format!("{context}: version 1 bytes changed after upgrade").into());
|
|
||||||
}
|
|
||||||
let v2 = root_a
|
|
||||||
.get_object()
|
|
||||||
.bucket(&seed.versioned_bucket)
|
|
||||||
.key(seed.versioned_key)
|
|
||||||
.version_id(&seed.version2)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
let v2_body = v2.body.collect().await?.into_bytes();
|
|
||||||
if v2_body.as_ref() != seed.version2_body.as_slice() {
|
|
||||||
return Err(format!("{context}: version 2 bytes changed after upgrade").into());
|
|
||||||
}
|
|
||||||
|
|
||||||
let users = cluster_admin_ok(&dist.cluster, http::Method::GET, "/rustfs/admin/v3/list-users", None).await?;
|
|
||||||
if !users.contains(&seed.iam_user) {
|
|
||||||
return Err(format!("{context}: list-users lost IAM user {}: {users}", seed.iam_user).into());
|
|
||||||
}
|
|
||||||
|
|
||||||
let iam_on_upgraded = dist.client_with_credentials(0, &seed.iam_user, seed.iam_secret)?;
|
|
||||||
let iam_on_peer = dist.client_with_credentials(3, &seed.iam_user, seed.iam_secret)?;
|
|
||||||
wait_for_bytes(
|
|
||||||
&iam_on_upgraded,
|
|
||||||
&seed.iam_bucket,
|
|
||||||
seed.iam_key,
|
|
||||||
&seed.iam_body,
|
|
||||||
&format!("{context}: IAM GET historical object on node 0"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
wait_for_bytes(
|
|
||||||
&iam_on_peer,
|
|
||||||
&seed.iam_bucket,
|
|
||||||
seed.iam_key,
|
|
||||||
&seed.iam_body,
|
|
||||||
&format!("{context}: IAM GET historical object on node 3"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let post_key = format!("after-upgrade-{context}.txt");
|
|
||||||
let post_body = format!("{context}: written with the same IAM AK/SK after upgrade").into_bytes();
|
|
||||||
wait_for_put(
|
|
||||||
&iam_on_peer,
|
|
||||||
&seed.iam_bucket,
|
|
||||||
&post_key,
|
|
||||||
post_body.clone(),
|
|
||||||
&format!("{context}: IAM PUT after upgrade"),
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
assert_object_bytes(&iam_on_upgraded, &seed.iam_bucket, &post_key, &post_body).await?;
|
|
||||||
|
|
||||||
let bad = dist.client_with_credentials(1, &seed.iam_user, WRONG_SECRET)?;
|
|
||||||
match bad.get_object().bucket(&seed.iam_bucket).key(seed.iam_key).send().await {
|
|
||||||
Ok(_) => return Err(format!("{context}: wrong secret must not read the IAM object").into()),
|
|
||||||
Err(error) => {
|
|
||||||
let code = error.as_service_error().and_then(ProvideErrorMetadata::code);
|
|
||||||
if code == Some("SignatureDoesNotMatch")
|
|
||||||
|| code == Some("InvalidAccessKeyId")
|
|
||||||
|| code == Some("AccessDenied")
|
|
||||||
|| code == Some("InvalidArgument")
|
|
||||||
{
|
|
||||||
} else if error.raw_response().is_some_and(|response| response.status().as_u16() == 403) {
|
|
||||||
} else {
|
|
||||||
return Err(format!("{context}: wrong secret failed with unexpected error {error:?}").into());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
let post_root_key = format!("root-after-{context}.bin");
|
|
||||||
let post_root_body = format!("{context}: root write after upgrade").into_bytes();
|
|
||||||
put_object(&root_a, &seed.history_bucket, &post_root_key, post_root_body.clone()).await?;
|
|
||||||
assert_object_bytes(&root_b, &seed.history_bucket, &post_root_key, &post_root_body).await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_direct_upgrade_preserves_history_and_iam_credentials() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let previous = source_binary()?;
|
|
||||||
let mut dist = DistCluster::new_stopped(DistLayout::FourNodeFourDisk).await?;
|
|
||||||
capture_upgrade_logs(&mut dist, "direct-upgrade")?;
|
|
||||||
dist.start_from_binary(&previous).await?;
|
|
||||||
|
|
||||||
let seed = seed_history_and_iam(&dist).await?;
|
|
||||||
dist.restart_with_current_binary().await?;
|
|
||||||
assert_history_and_iam(&dist, &seed, "direct").await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_rolling_upgrade_preserves_history_and_iam_credentials() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let previous = source_binary()?;
|
|
||||||
let mut dist = DistCluster::new_stopped(DistLayout::FourNodeFourDisk).await?;
|
|
||||||
capture_upgrade_logs(&mut dist, "rolling-upgrade")?;
|
|
||||||
dist.start_from_binary(&previous).await?;
|
|
||||||
|
|
||||||
let seed = seed_history_and_iam(&dist).await?;
|
|
||||||
|
|
||||||
dist.replace_node_with_current_binary(0).await?;
|
|
||||||
assert_history_and_iam(&dist, &seed, "one-current-node").await?;
|
|
||||||
|
|
||||||
for node_idx in [1, 2] {
|
|
||||||
dist.replace_node_with_current_binary(node_idx).await?;
|
|
||||||
}
|
|
||||||
assert_history_and_iam(&dist, &seed, "one-previous-node").await?;
|
|
||||||
|
|
||||||
dist.replace_node_with_current_binary(3).await?;
|
|
||||||
assert_history_and_iam(&dist, &seed, "homogeneous-current").await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn missing_upgrade_source_binary_fails_closed() {
|
|
||||||
let err = resolve_source_binary(None).expect_err("absent env must fail closed");
|
|
||||||
assert!(err.to_string().contains(SOURCE_BINARY_ENV), "{err}");
|
|
||||||
}
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn missing_upgrade_source_binary_file_fails_closed() {
|
|
||||||
let err = resolve_source_binary(Some("/no/such/rustfs-upgrade-source".into())).expect_err("missing file must fail closed");
|
|
||||||
assert!(err.to_string().contains("does not exist"), "{err}");
|
|
||||||
}
|
|
||||||
@@ -1,88 +0,0 @@
|
|||||||
// Copyright 2026 RustFS Team
|
|
||||||
//
|
|
||||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
||||||
// you may not use this file except in compliance with the License.
|
|
||||||
// You may obtain a copy of the License at
|
|
||||||
//
|
|
||||||
// http://www.apache.org/LICENSE-2.0
|
|
||||||
//
|
|
||||||
// Unless required by applicable law or agreed to in writing, software
|
|
||||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
||||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
||||||
// See the License for the specific language governing permissions and
|
|
||||||
// limitations under the License.
|
|
||||||
|
|
||||||
use super::harness::{DistCluster, DistLayout, TestResult, enable_versioning, get_object_bytes, put_object, unique_bucket};
|
|
||||||
use crate::common::init_logging;
|
|
||||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn four_node_four_drive_versioning_put_list_get_delete_marker() -> TestResult {
|
|
||||||
init_logging();
|
|
||||||
let dist = DistCluster::start(DistLayout::FourByFour).await?;
|
|
||||||
let bucket = unique_bucket("version");
|
|
||||||
dist.create_bucket(&bucket).await?;
|
|
||||||
let writer = dist.client(0)?;
|
|
||||||
let reader = dist.client(3)?;
|
|
||||||
enable_versioning(&writer, &bucket).await?;
|
|
||||||
|
|
||||||
let key = "versioned.txt";
|
|
||||||
put_object(&writer, &bucket, key, b"v1".to_vec()).await?;
|
|
||||||
put_object(&writer, &bucket, key, b"v2".to_vec()).await?;
|
|
||||||
|
|
||||||
let versions = reader.list_object_versions().bucket(&bucket).prefix(key).send().await?;
|
|
||||||
let version_ids: Vec<String> = versions
|
|
||||||
.versions()
|
|
||||||
.iter()
|
|
||||||
.filter_map(|version| version.version_id().map(str::to_string))
|
|
||||||
.collect();
|
|
||||||
assert!(version_ids.len() >= 2, "expected at least two versions, got {version_ids:?}");
|
|
||||||
|
|
||||||
let latest = get_object_bytes(&reader, &bucket, key).await?;
|
|
||||||
assert_eq!(latest, b"v2");
|
|
||||||
|
|
||||||
let older_id = versions
|
|
||||||
.versions()
|
|
||||||
.iter()
|
|
||||||
.find(|version| version.is_latest() != Some(true))
|
|
||||||
.and_then(|version| version.version_id())
|
|
||||||
.ok_or("missing non-latest version id")?;
|
|
||||||
let older = reader
|
|
||||||
.get_object()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.version_id(older_id)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
let older_body = older.body.collect().await?.into_bytes();
|
|
||||||
assert_eq!(older_body.as_ref(), b"v1");
|
|
||||||
|
|
||||||
writer.delete_object().bucket(&bucket).key(key).send().await?;
|
|
||||||
let after_delete = reader.list_object_versions().bucket(&bucket).prefix(key).send().await?;
|
|
||||||
assert!(
|
|
||||||
!after_delete.delete_markers().is_empty(),
|
|
||||||
"delete marker missing after unversioned-style delete: {after_delete:?}"
|
|
||||||
);
|
|
||||||
|
|
||||||
let latest_after_delete = reader.get_object().bucket(&bucket).key(key).send().await;
|
|
||||||
match latest_after_delete {
|
|
||||||
Ok(_) => return Err("current version should be a delete marker".into()),
|
|
||||||
Err(error)
|
|
||||||
if error
|
|
||||||
.as_service_error()
|
|
||||||
.and_then(ProvideErrorMetadata::code)
|
|
||||||
.is_some_and(|code| code == "NoSuchKey" || code == "NotFound") => {}
|
|
||||||
Err(error) => return Err(error.into()),
|
|
||||||
}
|
|
||||||
|
|
||||||
let restored = reader
|
|
||||||
.get_object()
|
|
||||||
.bucket(&bucket)
|
|
||||||
.key(key)
|
|
||||||
.version_id(older_id)
|
|
||||||
.send()
|
|
||||||
.await?;
|
|
||||||
let restored_body = restored.body.collect().await?.into_bytes();
|
|
||||||
assert_eq!(restored_body.as_ref(), b"v1");
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
@@ -378,11 +378,6 @@ mod bucket_stats_regression_test;
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod distributed_startup_regression_test;
|
mod distributed_startup_regression_test;
|
||||||
|
|
||||||
// 4-node / 4-disk distributed Actions suite (S3, lock, versioning, replication,
|
|
||||||
// quota, observability, expand/decommission/rebalance, site replication, chaos).
|
|
||||||
#[cfg(test)]
|
|
||||||
mod distributed;
|
|
||||||
|
|
||||||
// P1 regression: tier/ILM transition (rustfs#5218, #5130, #5011, #4826, #5024)
|
// P1 regression: tier/ILM transition (rustfs#5218, #5130, #5011, #4826, #5024)
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tier_transition_regression_test;
|
mod tier_transition_regression_test;
|
||||||
|
|||||||
@@ -17,15 +17,13 @@ Use the canonical CI-equivalent protocol command in the parent
|
|||||||
For targeted debugging of the core suite only:
|
For targeted debugging of the core suite only:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
RUSTFS_BUILD_FEATURES=ftps,webdav,sftp cargo test --package e2e_test test_protocol_core_suite -- --test-threads=1 --nocapture
|
python3 scripts/e2e_binary.py build --features ftps,webdav,sftp
|
||||||
|
python3 scripts/e2e_binary.py run --features ftps,webdav,sftp -- cargo test --package e2e_test test_protocol_core_suite -- --test-threads=1 --nocapture
|
||||||
```
|
```
|
||||||
|
|
||||||
This targeted command does not cover the full `e2e-protocols` profile.
|
This targeted command does not cover the full `e2e-protocols` profile.
|
||||||
|
|
||||||
`RUSTFS_BUILD_FEATURES` controls which features the test rustfs binary is
|
`e2e_binary.py` supplies `RUSTFS_BUILD_FEATURES` from the verified server's resolved Cargo features. The protocol runner schedules only entries present in that feature list; helpers check that their required features are available without rebuilding the server.
|
||||||
built with. When this variable is set, the protocol test runner schedules
|
|
||||||
only entries whose protocol is present in the requested feature list. Leave
|
|
||||||
it unset to run every protocol entry.
|
|
||||||
`--test-threads=1` is required because every entry spawns a rustfs server
|
`--test-threads=1` is required because every entry spawns a rustfs server
|
||||||
on fixed bind ports.
|
on fixed bind ports.
|
||||||
|
|
||||||
|
|||||||
@@ -6743,6 +6743,99 @@ async fn test_site_replication_replicates_object_with_bucket_versioning_real_dua
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn test_site_replication_replays_bucket_created_during_peer_outage_real_dual_node() -> TestResult {
|
||||||
|
init_logging();
|
||||||
|
|
||||||
|
// Keep compilation outside the scenario timeout. Recovery itself waits
|
||||||
|
// for the production 30-second lightweight retry tick.
|
||||||
|
let _rustfs_binary = rustfs_binary_path();
|
||||||
|
|
||||||
|
match timeout(Duration::from_secs(150), async {
|
||||||
|
let mut site_env = replication_fast_env();
|
||||||
|
site_env.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
|
||||||
|
|
||||||
|
let mut site_a_env = RustFSTestEnvironment::new().await?;
|
||||||
|
site_a_env.start_rustfs_server_with_env(vec![], &site_env).await?;
|
||||||
|
|
||||||
|
let mut site_b_env = RustFSTestEnvironment::new().await?;
|
||||||
|
site_b_env.start_rustfs_server_without_cleanup_with_env(&site_env).await?;
|
||||||
|
|
||||||
|
let site_a_client = site_a_env.create_s3_client();
|
||||||
|
let site_b_client = site_b_env.create_s3_client();
|
||||||
|
let bucket = "site-repl-peer-outage";
|
||||||
|
let key = "after-recovery.txt";
|
||||||
|
let payload = b"site replication recovered the missed bucket".to_vec();
|
||||||
|
|
||||||
|
let add_status = site_replication_add(
|
||||||
|
&site_a_env,
|
||||||
|
&[
|
||||||
|
PeerSite {
|
||||||
|
name: "outage-site-a".to_string(),
|
||||||
|
endpoint: site_a_env.url.clone(),
|
||||||
|
access_key: site_a_env.access_key.clone(),
|
||||||
|
secret_key: site_a_env.secret_key.clone(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
PeerSite {
|
||||||
|
name: "outage-site-b".to_string(),
|
||||||
|
endpoint: site_b_env.url.clone(),
|
||||||
|
access_key: site_b_env.access_key.clone(),
|
||||||
|
secret_key: site_b_env.secret_key.clone(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
assert!(add_status.success, "unexpected site add result: {add_status:?}");
|
||||||
|
wait_for_site_replication_enabled(&site_a_env, 2).await?;
|
||||||
|
wait_for_site_replication_enabled(&site_b_env, 2).await?;
|
||||||
|
|
||||||
|
site_b_env.stop_server();
|
||||||
|
site_a_client.create_bucket().bucket(bucket).send().await?;
|
||||||
|
site_a_client.head_bucket().bucket(bucket).send().await?;
|
||||||
|
|
||||||
|
let queued = site_replication_info(&site_a_env)
|
||||||
|
.await?
|
||||||
|
.retry_stats
|
||||||
|
.ok_or("peer outage did not persist a site replication retry event")?;
|
||||||
|
assert!(queued.pending + queued.failed > 0, "peer outage retry queue was unexpectedly empty");
|
||||||
|
|
||||||
|
site_b_env.restart_server_preserving_data(vec![], &site_env).await?;
|
||||||
|
let recovery_deadline = tokio::time::Instant::now() + Duration::from_secs(75);
|
||||||
|
loop {
|
||||||
|
let bucket_recovered = site_b_client.head_bucket().bucket(bucket).send().await.is_ok();
|
||||||
|
let queue_empty = site_replication_info(&site_a_env).await?.retry_stats.is_none();
|
||||||
|
if bucket_recovered && queue_empty {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if tokio::time::Instant::now() >= recovery_deadline {
|
||||||
|
return Err(format!(
|
||||||
|
"site replication retry did not settle after peer recovery; bucket_recovered={bucket_recovered}, queue_empty={queue_empty}"
|
||||||
|
)
|
||||||
|
.into());
|
||||||
|
}
|
||||||
|
sleep(Duration::from_millis(250)).await;
|
||||||
|
}
|
||||||
|
|
||||||
|
site_a_client
|
||||||
|
.put_object()
|
||||||
|
.bucket(bucket)
|
||||||
|
.key(key)
|
||||||
|
.body(ByteStream::from(payload.clone()))
|
||||||
|
.send()
|
||||||
|
.await?;
|
||||||
|
assert_eq!(wait_for_object_on_target(&site_b_client, bucket, key).await?, payload);
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(result) => result,
|
||||||
|
Err(_) => Err("site replication peer-outage recovery timed out after 150 seconds".into()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Re-applying a site's own replication config must not disable the peer's reverse direction.
|
/// Re-applying a site's own replication config must not disable the peer's reverse direction.
|
||||||
///
|
///
|
||||||
/// `PutBucketReplication` broadcasts the config to every peer — the console's replication
|
/// `PutBucketReplication` broadcasts the config to every peer — the console's replication
|
||||||
|
|||||||
@@ -244,6 +244,7 @@ windows-sys = { workspace = true, features = [
|
|||||||
windows-sys = { workspace = true, features = ["Win32_System_Ioctl"] }
|
windows-sys = { workspace = true, features = ["Win32_System_Ioctl"] }
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
|
aws-smithy-async.workspace = true
|
||||||
tokio = { workspace = true, features = ["rt-multi-thread", "macros", "test-util", "fs"] }
|
tokio = { workspace = true, features = ["rt-multi-thread", "macros", "test-util", "fs"] }
|
||||||
criterion = { workspace = true, features = ["html_reports"] }
|
criterion = { workspace = true, features = ["html_reports"] }
|
||||||
temp-env = { workspace = true, features = ["async_closure"] }
|
temp-env = { workspace = true, features = ["async_closure"] }
|
||||||
|
|||||||
@@ -196,15 +196,16 @@ pub mod bucket {
|
|||||||
pub use crate::bucket::metadata_sys::ConfigWriteLockProbe;
|
pub use crate::bucket::metadata_sys::ConfigWriteLockProbe;
|
||||||
pub use crate::bucket::metadata_sys::{
|
pub use crate::bucket::metadata_sys::{
|
||||||
BucketMetadataMutationGuard, BucketMetadataSys, ObjectLockConfigState, acquire_bucket_metadata_transaction_lock,
|
BucketMetadataMutationGuard, BucketMetadataSys, ObjectLockConfigState, acquire_bucket_metadata_transaction_lock,
|
||||||
acquire_bucket_metadata_transaction_lock_for_incarnation, capture_bucket_metadata_incarnation, delete,
|
acquire_bucket_metadata_transaction_lock_for_incarnation, acquire_scanner_bucket_incarnation_fence,
|
||||||
delete_if_incarnation, delete_under_transaction_lock, get, get_accelerate_config, get_bucket_policy,
|
capture_bucket_metadata_incarnation, delete, delete_if_incarnation, delete_under_transaction_lock, get,
|
||||||
get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk, get_cors_config, get_durability_config,
|
get_accelerate_config, get_bucket_policy, get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk,
|
||||||
get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config, get_notification_config,
|
get_cors_config, get_durability_config, get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config,
|
||||||
get_object_lock_config, get_object_lock_config_state, get_on_demand_migration_config, get_public_access_block_config,
|
get_notification_config, get_object_lock_config, get_object_lock_config_state, get_on_demand_migration_config,
|
||||||
get_quota_config, get_replication_config, get_request_payment_config, get_sse_config, get_tagging_config,
|
get_public_access_block_config, get_quota_config, get_replication_config, get_request_payment_config, get_sse_config,
|
||||||
get_versioning_config, get_website_config, init_bucket_metadata_sys, list_bucket_targets, reload_bucket_metadata,
|
get_tagging_config, get_versioning_config, get_website_config, init_bucket_metadata_sys, list_bucket_targets,
|
||||||
remove_bucket_metadata, set_bucket_metadata, update, update_bucket_targets_under_transaction_lock,
|
reload_bucket_metadata, remove_bucket_metadata, set_bucket_metadata, update,
|
||||||
update_config_with, update_if_incarnation, update_quota_if_incarnation, update_under_transaction_lock,
|
update_bucket_targets_under_transaction_lock, update_config_with, update_if_incarnation, update_quota_if_incarnation,
|
||||||
|
update_under_transaction_lock,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -479,9 +480,11 @@ pub mod notification {
|
|||||||
#[cfg(any(test, feature = "test-util"))]
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
pub use crate::services::notification_sys::rotate_cross_pool_fence_fleet_proof_for_test;
|
pub use crate::services::notification_sys::rotate_cross_pool_fence_fleet_proof_for_test;
|
||||||
pub use crate::services::notification_sys::{
|
pub use crate::services::notification_sys::{
|
||||||
ClusterTierDailyStats, CrossPoolFenceFleetProofToken, NotificationPeerErr, NotificationSys, ScannerPublicationLeaseGrant,
|
ClusterTierDailyStats, CrossPoolFenceFleetProofToken, LegacyTransitionStateReconcileFleetProofToken, NotificationPeerErr,
|
||||||
acquire_cross_pool_fence_fleet_proof, cross_pool_fence_fleet_proof_matches, get_global_notification_sys,
|
NotificationSys, ScannerPublicationLeaseGrant, acquire_cross_pool_fence_fleet_proof,
|
||||||
new_global_notification_sys, scanner_peer_transport_error_message_is_retryable, start_remote_version_state_fleet_probe,
|
acquire_legacy_transition_state_reconcile_fleet_proof, cross_pool_fence_fleet_proof_matches, get_global_notification_sys,
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_matches, new_global_notification_sys,
|
||||||
|
scanner_peer_transport_error_message_is_retryable, start_remote_version_state_fleet_probe,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -59,7 +59,7 @@ use rustfs_utils::http::{
|
|||||||
insert_header,
|
insert_header,
|
||||||
};
|
};
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
use std::collections::HashMap;
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::error::Error;
|
use std::error::Error;
|
||||||
use std::fmt;
|
use std::fmt;
|
||||||
use std::str::FromStr as _;
|
use std::str::FromStr as _;
|
||||||
@@ -376,6 +376,11 @@ pub struct BucketTargetSys {
|
|||||||
/// [`SsecPassthroughCapability`]; reset alongside `arn_remotes_map`.
|
/// [`SsecPassthroughCapability`]; reset alongside `arn_remotes_map`.
|
||||||
ssec_passthrough_map: Arc<RwLock<HashMap<String, SsecPassthroughRecord>>>,
|
ssec_passthrough_map: Arc<RwLock<HashMap<String, SsecPassthroughRecord>>>,
|
||||||
pub targets_map: Arc<RwLock<HashMap<String, Vec<BucketTarget>>>>,
|
pub targets_map: Arc<RwLock<HashMap<String, Vec<BucketTarget>>>>,
|
||||||
|
/// Buckets whose persisted `bucket-targets.json` exists but cannot be
|
||||||
|
/// decoded (rustfs/backlog#2282). Written under the bucket's update mutex
|
||||||
|
/// alongside `targets_map`, and read before it so an unreadable
|
||||||
|
/// configuration surfaces as a typed error instead of an empty target set.
|
||||||
|
unreadable_targets: Arc<RwLock<HashSet<String>>>,
|
||||||
pub h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>,
|
pub h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>,
|
||||||
target_h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>,
|
target_h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>,
|
||||||
pub hc_client: Arc<HttpClient>,
|
pub hc_client: Arc<HttpClient>,
|
||||||
@@ -419,6 +424,7 @@ impl BucketTargetSys {
|
|||||||
arn_remotes_map: Arc::new(RwLock::new(HashMap::new())),
|
arn_remotes_map: Arc::new(RwLock::new(HashMap::new())),
|
||||||
ssec_passthrough_map: Arc::new(RwLock::new(HashMap::new())),
|
ssec_passthrough_map: Arc::new(RwLock::new(HashMap::new())),
|
||||||
targets_map: Arc::new(RwLock::new(HashMap::new())),
|
targets_map: Arc::new(RwLock::new(HashMap::new())),
|
||||||
|
unreadable_targets: Arc::new(RwLock::new(HashSet::new())),
|
||||||
h_mutex: Arc::new(RwLock::new(HashMap::new())),
|
h_mutex: Arc::new(RwLock::new(HashMap::new())),
|
||||||
target_h_mutex: Arc::new(RwLock::new(HashMap::new())),
|
target_h_mutex: Arc::new(RwLock::new(HashMap::new())),
|
||||||
hc_client: Arc::new(build_health_check_client()),
|
hc_client: Arc::new(build_health_check_client()),
|
||||||
@@ -628,30 +634,40 @@ impl BucketTargetSys {
|
|||||||
health_map.clone()
|
health_map.clone()
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn list_targets(&self, bucket: &str, arn_type: &str) -> Vec<BucketTarget> {
|
/// Targets of one bucket, or of every bucket when `bucket` is empty.
|
||||||
|
///
|
||||||
|
/// A bucket that simply has no targets yields an empty list; a bucket
|
||||||
|
/// whose persisted configuration cannot be decoded is an error, so an
|
||||||
|
/// admin listing reports the fault instead of an empty list that reads as
|
||||||
|
/// "replication is not configured" (rustfs/backlog#2282).
|
||||||
|
pub async fn list_targets(&self, bucket: &str, arn_type: &str) -> Result<Vec<BucketTarget>, BucketTargetError> {
|
||||||
let health_stats = self.target_health_stats().await;
|
let health_stats = self.target_health_stats().await;
|
||||||
let mut targets = Vec::new();
|
let mut targets = Vec::new();
|
||||||
|
|
||||||
if !bucket.is_empty() {
|
if !bucket.is_empty() {
|
||||||
if let Ok(bucket_targets) = self.list_bucket_targets(bucket).await {
|
match self.list_bucket_targets(bucket).await {
|
||||||
for mut target in bucket_targets.targets {
|
Ok(bucket_targets) => {
|
||||||
if arn_type.is_empty() || target.target_type.to_string() == arn_type {
|
for mut target in bucket_targets.targets {
|
||||||
if let Some(health) = health_stats.get(&target.arn) {
|
if arn_type.is_empty() || target.target_type.to_string() == arn_type {
|
||||||
target.total_downtime = health.offline_duration;
|
if let Some(health) = health_stats.get(&target.arn) {
|
||||||
target.online = health.online;
|
target.total_downtime = health.offline_duration;
|
||||||
target.last_online = health.last_online;
|
target.online = health.online;
|
||||||
target.latency = target::LatencyStat {
|
target.last_online = health.last_online;
|
||||||
curr: health.latency.curr,
|
target.latency = target::LatencyStat {
|
||||||
avg: health.latency.avg,
|
curr: health.latency.curr,
|
||||||
max: health.latency.peak,
|
avg: health.latency.avg,
|
||||||
};
|
max: health.latency.peak,
|
||||||
target.offline_count = health.offline_count;
|
};
|
||||||
|
target.offline_count = health.offline_count;
|
||||||
|
}
|
||||||
|
targets.push(target);
|
||||||
}
|
}
|
||||||
targets.push(target);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
Err(BucketTargetError::BucketRemoteTargetNotFound { .. }) => {}
|
||||||
|
Err(err) => return Err(err),
|
||||||
}
|
}
|
||||||
return targets;
|
return Ok(targets);
|
||||||
}
|
}
|
||||||
|
|
||||||
let targets_map = self.targets_map.read().await;
|
let targets_map = self.targets_map.read().await;
|
||||||
@@ -674,10 +690,16 @@ impl BucketTargetSys {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
targets
|
Ok(targets)
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn list_bucket_targets(&self, bucket: &str) -> Result<BucketTargets, BucketTargetError> {
|
pub async fn list_bucket_targets(&self, bucket: &str) -> Result<BucketTargets, BucketTargetError> {
|
||||||
|
if self.unreadable_targets.read().await.contains(bucket) {
|
||||||
|
return Err(BucketTargetError::BucketRemoteTargetsUnreadable {
|
||||||
|
bucket: bucket.to_string(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
let targets_map = self.targets_map.read().await;
|
let targets_map = self.targets_map.read().await;
|
||||||
if let Some(targets) = targets_map.get(bucket) {
|
if let Some(targets) = targets_map.get(bucket) {
|
||||||
Ok(BucketTargets {
|
Ok(BucketTargets {
|
||||||
@@ -690,13 +712,30 @@ impl BucketTargetSys {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Record that this bucket's persisted targets configuration exists but
|
||||||
|
/// cannot be decoded (rustfs/backlog#2282).
|
||||||
|
///
|
||||||
|
/// Any snapshot published from an earlier readable load is deliberately
|
||||||
|
/// left in place: withdrawing it would produce exactly the silent "no
|
||||||
|
/// targets configured" state this marker exists to prevent. The marker is
|
||||||
|
/// cleared by the next successful publish, which is what makes a repaired
|
||||||
|
/// configuration take effect without a restart.
|
||||||
|
pub async fn mark_targets_unreadable(&self, bucket: &str) {
|
||||||
|
let update_mutex = self.target_update_mutex(bucket).await;
|
||||||
|
let _update_guard = update_mutex.lock().await;
|
||||||
|
|
||||||
|
self.unreadable_targets.write().await.insert(bucket.to_string());
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn delete(&self, bucket: &str) {
|
pub async fn delete(&self, bucket: &str) {
|
||||||
let update_mutex = self.target_update_mutex(bucket).await;
|
let update_mutex = self.target_update_mutex(bucket).await;
|
||||||
let _update_guard = update_mutex.lock().await;
|
let _update_guard = update_mutex.lock().await;
|
||||||
|
|
||||||
// Lock order: targets_map, then arn_remotes_map, then target_h_mutex,
|
// Lock order: unreadable_targets, then targets_map, then
|
||||||
// then ssec_passthrough_map (always last; also taken standalone by the
|
// arn_remotes_map, then target_h_mutex, then ssec_passthrough_map
|
||||||
// capability accessors).
|
// (always last; also taken standalone by the capability accessors).
|
||||||
|
self.unreadable_targets.write().await.remove(bucket);
|
||||||
|
|
||||||
let mut targets_map = self.targets_map.write().await;
|
let mut targets_map = self.targets_map.write().await;
|
||||||
let mut arn_remotes_map = self.arn_remotes_map.write().await;
|
let mut arn_remotes_map = self.arn_remotes_map.write().await;
|
||||||
let mut health_map = self.target_h_mutex.write().await;
|
let mut health_map = self.target_h_mutex.write().await;
|
||||||
@@ -1093,6 +1132,11 @@ impl BucketTargetSys {
|
|||||||
/// Keeping persisted-config reads under the same mutex prevents a stale
|
/// Keeping persisted-config reads under the same mutex prevents a stale
|
||||||
/// reload from overwriting a concurrent credential rotation.
|
/// reload from overwriting a concurrent credential rotation.
|
||||||
async fn update_all_targets_locked(&self, bucket: &str, targets: Option<&BucketTargets>) {
|
async fn update_all_targets_locked(&self, bucket: &str, targets: Option<&BucketTargets>) {
|
||||||
|
// Reaching here means the persisted configuration decoded, so the
|
||||||
|
// unreadable marker (if any) is stale. Cleared before the maps below
|
||||||
|
// so `unreadable_targets` stays the outermost of this module's locks.
|
||||||
|
self.unreadable_targets.write().await.remove(bucket);
|
||||||
|
|
||||||
let mut clients = Vec::new();
|
let mut clients = Vec::new();
|
||||||
if let Some(new_targets) = targets {
|
if let Some(new_targets) = targets {
|
||||||
for target in &new_targets.targets {
|
for target in &new_targets.targets {
|
||||||
@@ -1100,9 +1144,9 @@ impl BucketTargetSys {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Lock order: targets_map, then arn_remotes_map, then target_h_mutex,
|
// Lock order: unreadable_targets (above), then targets_map, then
|
||||||
// then ssec_passthrough_map (always last; also taken standalone by the
|
// arn_remotes_map, then target_h_mutex, then ssec_passthrough_map
|
||||||
// capability accessors).
|
// (always last; also taken standalone by the capability accessors).
|
||||||
let mut targets_map = self.targets_map.write().await;
|
let mut targets_map = self.targets_map.write().await;
|
||||||
let mut arn_remotes_map = self.arn_remotes_map.write().await;
|
let mut arn_remotes_map = self.arn_remotes_map.write().await;
|
||||||
let mut health_map = self.target_h_mutex.write().await;
|
let mut health_map = self.target_h_mutex.write().await;
|
||||||
@@ -1161,6 +1205,11 @@ impl BucketTargetSys {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub async fn set(&self, bucket: &str, meta: &BucketMetadata) {
|
pub async fn set(&self, bucket: &str, meta: &BucketMetadata) {
|
||||||
|
if meta.bucket_targets_unreadable() {
|
||||||
|
self.mark_targets_unreadable(bucket).await;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
let Some(config) = &meta.bucket_target_config else {
|
let Some(config) = &meta.bucket_target_config else {
|
||||||
return;
|
return;
|
||||||
};
|
};
|
||||||
@@ -2276,6 +2325,13 @@ pub enum BucketTargetError {
|
|||||||
BucketRemoteTargetNotFound {
|
BucketRemoteTargetNotFound {
|
||||||
bucket: String,
|
bucket: String,
|
||||||
},
|
},
|
||||||
|
/// The bucket's persisted targets configuration exists but cannot be
|
||||||
|
/// decoded. Distinct from `BucketRemoteTargetNotFound`, which means the
|
||||||
|
/// bucket genuinely has no targets: callers must not degrade this one to
|
||||||
|
/// an empty target set (rustfs/backlog#2282).
|
||||||
|
BucketRemoteTargetsUnreadable {
|
||||||
|
bucket: String,
|
||||||
|
},
|
||||||
BucketRemoteArnTypeInvalid {
|
BucketRemoteArnTypeInvalid {
|
||||||
bucket: String,
|
bucket: String,
|
||||||
},
|
},
|
||||||
@@ -2309,6 +2365,9 @@ impl fmt::Display for BucketTargetError {
|
|||||||
BucketTargetError::BucketRemoteTargetNotFound { bucket } => {
|
BucketTargetError::BucketRemoteTargetNotFound { bucket } => {
|
||||||
write!(f, "Remote target not found for bucket: {bucket}")
|
write!(f, "Remote target not found for bucket: {bucket}")
|
||||||
}
|
}
|
||||||
|
BucketTargetError::BucketRemoteTargetsUnreadable { bucket } => {
|
||||||
|
write!(f, "Persisted replication target configuration is unreadable for bucket: {bucket}")
|
||||||
|
}
|
||||||
BucketTargetError::BucketRemoteArnTypeInvalid { bucket } => {
|
BucketTargetError::BucketRemoteArnTypeInvalid { bucket } => {
|
||||||
write!(f, "Invalid ARN type for bucket: {bucket}")
|
write!(f, "Invalid ARN type for bucket: {bucket}")
|
||||||
}
|
}
|
||||||
@@ -3256,7 +3315,7 @@ mod tests {
|
|||||||
}],
|
}],
|
||||||
);
|
);
|
||||||
|
|
||||||
let targets = sys.list_targets("", "").await;
|
let targets = sys.list_targets("", "").await.expect("listing every bucket's targets");
|
||||||
|
|
||||||
assert_eq!(targets.len(), 1);
|
assert_eq!(targets.len(), 1);
|
||||||
assert!(!targets[0].online);
|
assert!(!targets[0].online);
|
||||||
|
|||||||
@@ -584,33 +584,173 @@ impl ExpiryOp for FreeVersionTask {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||||
|
enum TransitionDeleteVersionPlan {
|
||||||
|
Direct { version_id_exact: bool },
|
||||||
|
ProbeLegacyUnknown,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn legacy_transition_version_state_missing(oi: &ObjectInfo) -> Result<bool, std::io::Error> {
|
||||||
|
use rustfs_utils::http::metadata_compat::{
|
||||||
|
SUFFIX_TRANSITIONED_VERSION_ID, SUFFIX_TRANSITIONED_VERSION_STATE, contains_key_str, get_consistent_str,
|
||||||
|
};
|
||||||
|
|
||||||
|
if !contains_key_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_STATE) {
|
||||||
|
let version_key_present = contains_key_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_ID);
|
||||||
|
if version_key_present {
|
||||||
|
if oi.transitioned_object.version_id.is_empty() {
|
||||||
|
let has_non_empty_version = oi.user_defined.iter().any(|(key, value)| {
|
||||||
|
rustfs_utils::http::metadata_compat::strip_internal_prefix_preserving_case(key)
|
||||||
|
.is_some_and(|suffix| suffix.eq_ignore_ascii_case(SUFFIX_TRANSITIONED_VERSION_ID))
|
||||||
|
&& !value.is_empty()
|
||||||
|
});
|
||||||
|
if !has_non_empty_version {
|
||||||
|
// MinIO writes the transitioned-versionID key with an empty value
|
||||||
|
// for unversioned tier objects. The backend probe remains the proof.
|
||||||
|
return Ok(true);
|
||||||
|
}
|
||||||
|
} else if get_consistent_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_ID)
|
||||||
|
== Some(oi.transitioned_object.version_id.as_str())
|
||||||
|
{
|
||||||
|
return Ok(true);
|
||||||
|
}
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidData,
|
||||||
|
"legacy remote tier version metadata is conflicting or malformed",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if !oi.transitioned_object.version_id.is_empty() {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidData,
|
||||||
|
"legacy remote tier version metadata is missing or inconsistent",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
return Ok(true);
|
||||||
|
}
|
||||||
|
let persisted = get_consistent_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_STATE).ok_or_else(|| {
|
||||||
|
std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidData,
|
||||||
|
"remote tier object has conflicting transition version state metadata",
|
||||||
|
)
|
||||||
|
})?;
|
||||||
|
if persisted != oi.transition_version_state.as_str() {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidData,
|
||||||
|
"remote tier object transition version state metadata changed during decoding",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
Ok(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn transition_remote_version_delete_plan(oi: &ObjectInfo) -> Result<TransitionDeleteVersionPlan, std::io::Error> {
|
||||||
|
match oi.transition_version_state {
|
||||||
|
rustfs_filemeta::TransitionVersionState::Unknown => {
|
||||||
|
if legacy_transition_version_state_missing(oi)? {
|
||||||
|
Ok(TransitionDeleteVersionPlan::ProbeLegacyUnknown)
|
||||||
|
} else {
|
||||||
|
validate_transition_remote_version(oi)
|
||||||
|
.map(|version_id_exact| TransitionDeleteVersionPlan::Direct { version_id_exact })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => validate_transition_remote_version(oi)
|
||||||
|
.map(|version_id_exact| TransitionDeleteVersionPlan::Direct { version_id_exact }),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||||
|
struct ResolvedTransitionDeleteVersion {
|
||||||
|
version_id_exact: bool,
|
||||||
|
remote_already_missing: bool,
|
||||||
|
}
|
||||||
|
|
||||||
async fn acquire_free_version_tier_lease(
|
async fn acquire_free_version_tier_lease(
|
||||||
oi: &ObjectInfo,
|
oi: &ObjectInfo,
|
||||||
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
||||||
) -> Result<(TierOperationLease, bool), std::io::Error> {
|
) -> Result<(TierOperationLease, TransitionDeleteVersionPlan), std::io::Error> {
|
||||||
let version_id_exact = validate_transition_remote_version(oi)?;
|
let delete_plan = transition_remote_version_delete_plan(oi)?;
|
||||||
let identity = tier_destination_id_from_metadata(&oi.user_defined)?
|
let identity = tier_destination_id_from_metadata(&oi.user_defined)?
|
||||||
.ok_or_else(|| std::io::Error::other("tier free-version has no durable backend identity"))?;
|
.ok_or_else(|| std::io::Error::other("tier free-version has no durable backend identity"))?;
|
||||||
let lease =
|
let lease =
|
||||||
TierConfigMgr::acquire_operation_lease_for_backend_identity(tier_config_mgr, &oi.transitioned_object.tier, identity)
|
TierConfigMgr::acquire_operation_lease_for_backend_identity(tier_config_mgr, &oi.transitioned_object.tier, identity)
|
||||||
.await
|
.await
|
||||||
.map_err(std::io::Error::other)?;
|
.map_err(std::io::Error::other)?;
|
||||||
Ok((lease, version_id_exact))
|
Ok((lease, delete_plan))
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn resolve_transition_delete_version_plan(
|
||||||
|
oi: &ObjectInfo,
|
||||||
|
lease: &TierOperationLease,
|
||||||
|
delete_plan: TransitionDeleteVersionPlan,
|
||||||
|
) -> Result<ResolvedTransitionDeleteVersion, std::io::Error> {
|
||||||
|
match delete_plan {
|
||||||
|
TransitionDeleteVersionPlan::Direct { version_id_exact } => Ok(ResolvedTransitionDeleteVersion {
|
||||||
|
version_id_exact,
|
||||||
|
remote_already_missing: false,
|
||||||
|
}),
|
||||||
|
TransitionDeleteVersionPlan::ProbeLegacyUnknown => {
|
||||||
|
let expected_version = oi.transitioned_object.version_id.as_str();
|
||||||
|
if expected_version.is_empty() {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::WouldBlock,
|
||||||
|
"remote tier cannot safely delete a legacy object without an exact version ID",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let probe = lease
|
||||||
|
.probe_transition_version(&oi.transitioned_object.name, expected_version)
|
||||||
|
.await?;
|
||||||
|
match (expected_version, probe) {
|
||||||
|
(expected, crate::services::tier::warm_backend::TransitionCandidateProbe::VersionedPresent(actual))
|
||||||
|
if expected == actual =>
|
||||||
|
{
|
||||||
|
lease.validate_remote_version_id(expected)?;
|
||||||
|
Ok(ResolvedTransitionDeleteVersion {
|
||||||
|
version_id_exact: true,
|
||||||
|
remote_already_missing: false,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
(_, crate::services::tier::warm_backend::TransitionCandidateProbe::Missing) => {
|
||||||
|
Ok(ResolvedTransitionDeleteVersion {
|
||||||
|
version_id_exact: false,
|
||||||
|
remote_already_missing: true,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
(_, crate::services::tier::warm_backend::TransitionCandidateProbe::Unsupported) => Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::Unsupported,
|
||||||
|
"remote tier cannot prove legacy transition delete state",
|
||||||
|
)),
|
||||||
|
_ => Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::WouldBlock,
|
||||||
|
"remote tier object version state is unknown",
|
||||||
|
)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn execute_resolved_transition_delete(
|
||||||
|
oi: &ObjectInfo,
|
||||||
|
lease: &TierOperationLease,
|
||||||
|
resolved: ResolvedTransitionDeleteVersion,
|
||||||
|
) -> Result<(), std::io::Error> {
|
||||||
|
if !resolved.remote_already_missing {
|
||||||
|
delete_object_from_remote_tier_with_lease_idempotent(
|
||||||
|
&oi.transitioned_object.name,
|
||||||
|
&oi.transitioned_object.version_id,
|
||||||
|
lease,
|
||||||
|
resolved.version_id_exact,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn delete_free_version_remote_object_with_lease(
|
async fn delete_free_version_remote_object_with_lease(
|
||||||
oi: &ObjectInfo,
|
oi: &ObjectInfo,
|
||||||
lease: &TierOperationLease,
|
lease: &TierOperationLease,
|
||||||
version_id_exact: bool,
|
delete_plan: TransitionDeleteVersionPlan,
|
||||||
) -> Result<(), std::io::Error> {
|
) -> Result<(), std::io::Error> {
|
||||||
delete_object_from_remote_tier_with_lease_idempotent(
|
let resolved = resolve_transition_delete_version_plan(oi, lease, delete_plan).await?;
|
||||||
&oi.transitioned_object.name,
|
execute_resolved_transition_delete(oi, lease, resolved).await
|
||||||
&oi.transitioned_object.version_id,
|
|
||||||
lease,
|
|
||||||
version_id_exact,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
Ok(())
|
|
||||||
}
|
}
|
||||||
|
|
||||||
fn free_version_physical_topology_generation(api: &ECStore) -> String {
|
fn free_version_physical_topology_generation(api: &ECStore) -> String {
|
||||||
@@ -641,6 +781,16 @@ fn free_version_remote_tuple_matches(candidate: &ObjectInfo, expected: &ObjectIn
|
|||||||
if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||||
|| expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
|| expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||||
{
|
{
|
||||||
|
let candidate_legacy_missing = legacy_transition_version_state_missing(candidate)?;
|
||||||
|
let expected_legacy_missing = legacy_transition_version_state_missing(expected)?;
|
||||||
|
if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||||
|
&& expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||||
|
&& candidate_legacy_missing
|
||||||
|
&& expected_legacy_missing
|
||||||
|
&& candidate.transitioned_object.version_id == expected.transitioned_object.version_id
|
||||||
|
{
|
||||||
|
return Ok(true);
|
||||||
|
}
|
||||||
return Err(std::io::Error::new(
|
return Err(std::io::Error::new(
|
||||||
std::io::ErrorKind::WouldBlock,
|
std::io::ErrorKind::WouldBlock,
|
||||||
"tier free-version remote version state is unknown",
|
"tier free-version remote version state is unknown",
|
||||||
@@ -716,7 +866,7 @@ async fn cleanup_free_version_exact(api: Arc<ECStore>, oi: &ObjectInfo, cancel:
|
|||||||
.acquire_bucket_lifecycle_read_lock(&oi.bucket)
|
.acquire_bucket_lifecycle_read_lock(&oi.bucket)
|
||||||
.await
|
.await
|
||||||
.map_err(std::io::Error::other)?;
|
.map_err(std::io::Error::other)?;
|
||||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, &api.tier_config_mgr()).await?;
|
let (lease, delete_plan) = acquire_free_version_tier_lease(oi, &api.tier_config_mgr()).await?;
|
||||||
let local_object = encode_dir_object(&oi.name);
|
let local_object = encode_dir_object(&oi.name);
|
||||||
let object_guards = api
|
let object_guards = api
|
||||||
.acquire_all_physical_object_write_locks("tier_free_version_cleanup", &oi.bucket, &local_object)
|
.acquire_all_physical_object_write_locks("tier_free_version_cleanup", &oi.bucket, &local_object)
|
||||||
@@ -734,16 +884,30 @@ async fn cleanup_free_version_exact(api: Arc<ECStore>, oi: &ObjectInfo, cancel:
|
|||||||
"tier free-version cleanup fence is invalid before remote delete",
|
"tier free-version cleanup fence is invalid before remote delete",
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
let resolved = tokio::select! {
|
||||||
|
_ = cancel.cancelled() => {
|
||||||
|
return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled"));
|
||||||
|
}
|
||||||
|
result = tokio::time::timeout_at(deadline, resolve_transition_delete_version_plan(oi, &lease, delete_plan)) => {
|
||||||
|
result.map_err(|_| {
|
||||||
|
std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote probe timed out")
|
||||||
|
})??
|
||||||
|
}
|
||||||
|
};
|
||||||
|
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::WouldBlock,
|
||||||
|
"tier free-version cleanup fence changed after remote probe",
|
||||||
|
));
|
||||||
|
}
|
||||||
tokio::select! {
|
tokio::select! {
|
||||||
_ = cancel.cancelled() => {
|
_ = cancel.cancelled() => {
|
||||||
return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled"));
|
return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled"));
|
||||||
}
|
}
|
||||||
result = tokio::time::timeout_at(
|
result = tokio::time::timeout_at(deadline, execute_resolved_transition_delete(oi, &lease, resolved)) => {
|
||||||
deadline,
|
result.map_err(|_| {
|
||||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact),
|
std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote delete timed out")
|
||||||
) => {
|
})??;
|
||||||
result
|
|
||||||
.map_err(|_| std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote delete timed out"))??;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
|
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
|
||||||
@@ -791,8 +955,8 @@ async fn delete_free_version_remote_object(
|
|||||||
oi: &ObjectInfo,
|
oi: &ObjectInfo,
|
||||||
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
||||||
) -> Result<(), std::io::Error> {
|
) -> Result<(), std::io::Error> {
|
||||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
let (lease, delete_plan) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
||||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await
|
delete_free_version_remote_object_with_lease(oi, &lease, delete_plan).await
|
||||||
}
|
}
|
||||||
|
|
||||||
#[allow(
|
#[allow(
|
||||||
@@ -808,8 +972,8 @@ where
|
|||||||
F: FnOnce() -> Fut,
|
F: FnOnce() -> Fut,
|
||||||
Fut: std::future::Future<Output = T>,
|
Fut: std::future::Future<Output = T>,
|
||||||
{
|
{
|
||||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
let (lease, delete_plan) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
||||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await?;
|
delete_free_version_remote_object_with_lease(oi, &lease, delete_plan).await?;
|
||||||
let result = delete_local().await;
|
let result = delete_local().await;
|
||||||
drop(lease);
|
drop(lease);
|
||||||
Ok(result)
|
Ok(result)
|
||||||
@@ -4688,6 +4852,39 @@ fn validate_transition_remote_version(oi: &ObjectInfo) -> Result<bool, std::io::
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||||
|
enum TransitionReadVersionPlan {
|
||||||
|
Direct,
|
||||||
|
ProbeLegacyUnversioned,
|
||||||
|
}
|
||||||
|
|
||||||
|
const LEGACY_TRANSITION_READ_PROBE_TIMEOUT: StdDuration = StdDuration::from_secs(30);
|
||||||
|
|
||||||
|
fn transition_remote_version_read_plan(oi: &ObjectInfo) -> Result<TransitionReadVersionPlan, std::io::Error> {
|
||||||
|
let version = oi.transitioned_object.version_id.as_str();
|
||||||
|
match oi.transition_version_state {
|
||||||
|
rustfs_filemeta::TransitionVersionState::Unknown => {
|
||||||
|
if !legacy_transition_version_state_missing(oi)? {
|
||||||
|
return validate_transition_remote_version(oi).map(|_| TransitionReadVersionPlan::Direct);
|
||||||
|
}
|
||||||
|
if version.is_empty() {
|
||||||
|
Ok(TransitionReadVersionPlan::ProbeLegacyUnversioned)
|
||||||
|
} else {
|
||||||
|
Ok(TransitionReadVersionPlan::Direct)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
rustfs_filemeta::TransitionVersionState::KnownDisabled if version.is_empty() => Ok(TransitionReadVersionPlan::Direct),
|
||||||
|
rustfs_filemeta::TransitionVersionState::SuspendedNull if version == "null" => Ok(TransitionReadVersionPlan::Direct),
|
||||||
|
rustfs_filemeta::TransitionVersionState::Exact if !version.is_empty() && version != "null" => {
|
||||||
|
Ok(TransitionReadVersionPlan::Direct)
|
||||||
|
}
|
||||||
|
_ => Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidData,
|
||||||
|
"remote tier object version state conflicts with its version ID",
|
||||||
|
)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// The resolver joins the tier manager as the second injected port this read
|
// The resolver joins the tier manager as the second injected port this read
|
||||||
// needs; grouping the request half into a struct would churn every call site of
|
// needs; grouping the request half into a struct would churn every call site of
|
||||||
// a bug fix.
|
// a bug fix.
|
||||||
@@ -4702,7 +4899,12 @@ pub(crate) async fn get_transitioned_object_reader_with_tier_manager(
|
|||||||
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
||||||
resolver: Option<&dyn ObjectEncryptionResolver>,
|
resolver: Option<&dyn ObjectEncryptionResolver>,
|
||||||
) -> Result<GetObjectReader, std::io::Error> {
|
) -> Result<GetObjectReader, std::io::Error> {
|
||||||
validate_transition_remote_version(oi)?;
|
let read_plan = transition_remote_version_read_plan(oi)?;
|
||||||
|
// Reject invalid ranges and encryption requests before a compatibility
|
||||||
|
// probe can amplify them into remote listing work.
|
||||||
|
let plan = ReadPlan::build_for_request(rs.clone(), oi, opts, h, resolver)
|
||||||
|
.await
|
||||||
|
.map_err(|err| std::io::Error::other(format!("building the read plan for {bucket}/{object} failed: {err}")))?;
|
||||||
let expected_identity = tier_destination_id_from_metadata(&oi.user_defined)?;
|
let expected_identity = tier_destination_id_from_metadata(&oi.user_defined)?;
|
||||||
let lease = match expected_identity {
|
let lease = match expected_identity {
|
||||||
Some(identity) => {
|
Some(identity) => {
|
||||||
@@ -4716,7 +4918,36 @@ pub(crate) async fn get_transitioned_object_reader_with_tier_manager(
|
|||||||
Err(err) => return Err(std::io::Error::other(err)),
|
Err(err) => return Err(std::io::Error::other(err)),
|
||||||
};
|
};
|
||||||
|
|
||||||
tgt_client.validate_remote_version_id(&oi.transitioned_object.version_id)?;
|
match read_plan {
|
||||||
|
TransitionReadVersionPlan::Direct => {
|
||||||
|
tgt_client.validate_remote_version_id(&oi.transitioned_object.version_id)?;
|
||||||
|
}
|
||||||
|
TransitionReadVersionPlan::ProbeLegacyUnversioned => {
|
||||||
|
// RUSTFS_COMPAT_TODO(backlog#2203): remove operation-time probing
|
||||||
|
// after an admin reconcile can persist every proven legacy state.
|
||||||
|
let probe = tokio::time::timeout(
|
||||||
|
LEGACY_TRANSITION_READ_PROBE_TIMEOUT,
|
||||||
|
tgt_client.probe_transition_candidate(&oi.transitioned_object.name),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|_| std::io::Error::new(std::io::ErrorKind::TimedOut, "legacy remote tier version probe timed out"))??;
|
||||||
|
match probe {
|
||||||
|
crate::services::tier::warm_backend::TransitionCandidateProbe::UnversionedPresent => {}
|
||||||
|
crate::services::tier::warm_backend::TransitionCandidateProbe::Unsupported => {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::Unsupported,
|
||||||
|
"remote tier cannot prove legacy unversioned transition state",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidData,
|
||||||
|
"remote tier object version state is unknown",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// The same read plan the local path uses, so the tier fetch is positioned in
|
// The same read plan the local path uses, so the tier fetch is positioned in
|
||||||
// the object's *stored* coordinate system and the stream is handed the same
|
// the object's *stored* coordinate system and the stream is handed the same
|
||||||
@@ -4724,9 +4955,6 @@ pub(crate) async fn get_transitioned_object_reader_with_tier_manager(
|
|||||||
// through a plaintext-coordinate range and skipping the transform is how a
|
// through a plaintext-coordinate range and skipping the transform is how a
|
||||||
// transitioned SSE object used to come back as silently corrupt bytes of the
|
// transitioned SSE object used to come back as silently corrupt bytes of the
|
||||||
// right length (rustfs/rustfs#6025).
|
// right length (rustfs/rustfs#6025).
|
||||||
let plan = ReadPlan::build_for_request(rs.clone(), oi, opts, h, resolver)
|
|
||||||
.await
|
|
||||||
.map_err(|err| std::io::Error::other(format!("building the read plan for {bucket}/{object} failed: {err}")))?;
|
|
||||||
let (off, length) = (plan.storage_offset() as i64, plan.storage_length());
|
let (off, length) = (plan.storage_offset() as i64, plan.storage_length());
|
||||||
let mut gopts = WarmBackendGetOpts::default();
|
let mut gopts = WarmBackendGetOpts::default();
|
||||||
|
|
||||||
@@ -5599,11 +5827,13 @@ mod tests {
|
|||||||
use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
|
use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
|
||||||
use crate::object_api::{ObjectInfo, ObjectOptions, PutObjReader};
|
use crate::object_api::{ObjectInfo, ObjectOptions, PutObjReader};
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
|
use crate::services::tier::test_util::MockWarmOp;
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
use crate::services::tier::test_util::register_mock_tier;
|
use crate::services::tier::test_util::register_mock_tier;
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
use crate::services::tier::tier::TierConfigMgr;
|
use crate::services::tier::tier::TierConfigMgr;
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
use crate::services::tier::warm_backend::WarmBackend as _;
|
use crate::services::tier::warm_backend::{TransitionCandidateProbe, WarmBackend as _};
|
||||||
use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause};
|
use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause};
|
||||||
use crate::set_disk::{RUSTFS_MULTIPART_BUCKET_KEY, RUSTFS_MULTIPART_OBJECT_KEY};
|
use crate::set_disk::{RUSTFS_MULTIPART_BUCKET_KEY, RUSTFS_MULTIPART_OBJECT_KEY};
|
||||||
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
|
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
|
||||||
@@ -6299,7 +6529,75 @@ mod tests {
|
|||||||
|
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn transitioned_get_rejects_unknown_version_state_before_backend_io() {
|
async fn transitioned_get_allows_legacy_unknown_exact_version_for_non_destructive_read() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
||||||
|
let backend = register_mock_tier(&manager, &tier).await;
|
||||||
|
let remote_object = format!("remote/{}", Uuid::new_v4());
|
||||||
|
let body = Bytes::from_static(b"legacy transitioned object body");
|
||||||
|
let remote_version = backend
|
||||||
|
.put(
|
||||||
|
&remote_object,
|
||||||
|
ReaderImpl::Body(body.clone()),
|
||||||
|
i64::try_from(body.len()).expect("body length should fit"),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("mock remote object should be stored");
|
||||||
|
let mut user_defined = HashMap::new();
|
||||||
|
insert_legacy_transition_version_id(&mut user_defined, &remote_version);
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
bucket: "bucket".to_string(),
|
||||||
|
name: "object".to_string(),
|
||||||
|
size: i64::try_from(body.len()).expect("body length should fit"),
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: remote_object,
|
||||||
|
version_id: remote_version,
|
||||||
|
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
|
||||||
|
tier: tier.clone(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined.into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let range = Some(crate::storage_api_contracts::range::HTTPRangeSpec {
|
||||||
|
is_suffix_length: false,
|
||||||
|
start: 7,
|
||||||
|
end: 18,
|
||||||
|
});
|
||||||
|
let mut reader = get_transitioned_object_reader_with_tier_manager(
|
||||||
|
&object_info.bucket,
|
||||||
|
&object_info.name,
|
||||||
|
&range,
|
||||||
|
&HeaderMap::new(),
|
||||||
|
&object_info,
|
||||||
|
&ObjectOptions::default(),
|
||||||
|
&manager,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("legacy unknown state should still allow a non-destructive read");
|
||||||
|
let mut got = Vec::new();
|
||||||
|
reader
|
||||||
|
.stream
|
||||||
|
.read_to_end(&mut got)
|
||||||
|
.await
|
||||||
|
.expect("transitioned reader should drain");
|
||||||
|
|
||||||
|
assert_eq!(got, &body.as_ref()[7..=18]);
|
||||||
|
assert_eq!(backend.get_count().await, 1);
|
||||||
|
assert_eq!(backend.remove_count().await, 0);
|
||||||
|
assert_eq!(
|
||||||
|
TierConfigMgr::active_operation_lease_count(&manager, &tier).await,
|
||||||
|
0,
|
||||||
|
"tier generation lease should release after EOF"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn transitioned_get_rejects_explicit_unknown_version_state_before_backend_io() {
|
||||||
let manager = TierConfigMgr::new();
|
let manager = TierConfigMgr::new();
|
||||||
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
||||||
let backend = register_mock_tier(&manager, &tier).await;
|
let backend = register_mock_tier(&manager, &tier).await;
|
||||||
@@ -6315,6 +6613,181 @@ mod tests {
|
|||||||
..Default::default()
|
..Default::default()
|
||||||
},
|
},
|
||||||
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined_with_transition_version_state(rustfs_filemeta::TransitionVersionState::Unknown).into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = match get_transitioned_object_reader_with_tier_manager(
|
||||||
|
&object_info.bucket,
|
||||||
|
&object_info.name,
|
||||||
|
&None,
|
||||||
|
&HeaderMap::new(),
|
||||||
|
&object_info,
|
||||||
|
&ObjectOptions::default(),
|
||||||
|
&manager,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(_) => panic!("explicit unknown remote version state must fail before backend IO"),
|
||||||
|
Err(err) => err,
|
||||||
|
};
|
||||||
|
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||||
|
assert_eq!(backend.op_log().await, Vec::<MockWarmOp>::new());
|
||||||
|
assert_eq!(backend.get_count().await, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn transitioned_get_rejects_present_but_invalid_legacy_version_metadata() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
||||||
|
let backend = register_mock_tier(&manager, &tier).await;
|
||||||
|
|
||||||
|
for persisted_version in [
|
||||||
|
Uuid::nil().to_string(),
|
||||||
|
"\u{fffd}".to_string(),
|
||||||
|
"bad\u{0001}version".to_string(),
|
||||||
|
] {
|
||||||
|
let mut user_defined = HashMap::new();
|
||||||
|
insert_legacy_transition_version_id(&mut user_defined, &persisted_version);
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
bucket: "bucket".to_string(),
|
||||||
|
name: "object".to_string(),
|
||||||
|
size: 1,
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: "remote/object".to_string(),
|
||||||
|
version_id: String::new(),
|
||||||
|
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
|
||||||
|
tier: tier.clone(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined.into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = match get_transitioned_object_reader_with_tier_manager(
|
||||||
|
&object_info.bucket,
|
||||||
|
&object_info.name,
|
||||||
|
&None,
|
||||||
|
&HeaderMap::new(),
|
||||||
|
&object_info,
|
||||||
|
&ObjectOptions::default(),
|
||||||
|
&manager,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(_) => panic!("present but invalid legacy version metadata must fail before backend IO"),
|
||||||
|
Err(err) => err,
|
||||||
|
};
|
||||||
|
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||||
|
}
|
||||||
|
|
||||||
|
assert_eq!(backend.op_log().await, Vec::<MockWarmOp>::new());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn transitioned_get_probes_legacy_empty_unknown_state_before_unversioned_read() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
||||||
|
let backend = register_mock_tier(&manager, &tier).await;
|
||||||
|
backend.set_put_remote_version(Some(String::new())).await;
|
||||||
|
let remote_object = format!("remote/{}", Uuid::new_v4());
|
||||||
|
let body = Bytes::from_static(b"legacy unversioned transitioned object body");
|
||||||
|
let remote_version = backend
|
||||||
|
.put(
|
||||||
|
&remote_object,
|
||||||
|
ReaderImpl::Body(body.clone()),
|
||||||
|
i64::try_from(body.len()).expect("body length should fit"),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("mock remote object should be stored");
|
||||||
|
assert!(remote_version.is_empty());
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
bucket: "bucket".to_string(),
|
||||||
|
name: "object".to_string(),
|
||||||
|
size: i64::try_from(body.len()).expect("body length should fit"),
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: remote_object.clone(),
|
||||||
|
version_id: String::new(),
|
||||||
|
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
|
||||||
|
tier: tier.clone(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: HashMap::from([("x-minio-internal-transitioned-versionID".to_string(), String::new())]).into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut reader = get_transitioned_object_reader_with_tier_manager(
|
||||||
|
&object_info.bucket,
|
||||||
|
&object_info.name,
|
||||||
|
&None,
|
||||||
|
&HeaderMap::new(),
|
||||||
|
&object_info,
|
||||||
|
&ObjectOptions::default(),
|
||||||
|
&manager,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("probe-proven legacy unversioned state should allow a non-destructive read");
|
||||||
|
let mut got = Vec::new();
|
||||||
|
reader
|
||||||
|
.stream
|
||||||
|
.read_to_end(&mut got)
|
||||||
|
.await
|
||||||
|
.expect("transitioned reader should drain");
|
||||||
|
|
||||||
|
assert_eq!(got, body.as_ref());
|
||||||
|
assert_eq!(backend.remove_count().await, 0);
|
||||||
|
assert_eq!(
|
||||||
|
backend.op_log().await,
|
||||||
|
vec![
|
||||||
|
MockWarmOp::Put {
|
||||||
|
object: remote_object.clone()
|
||||||
|
},
|
||||||
|
MockWarmOp::Probe {
|
||||||
|
object: remote_object.clone()
|
||||||
|
},
|
||||||
|
MockWarmOp::Get { object: remote_object },
|
||||||
|
]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
TierConfigMgr::active_operation_lease_count(&manager, &tier).await,
|
||||||
|
0,
|
||||||
|
"tier generation lease should release after EOF"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn transitioned_get_rejects_ambiguous_empty_unknown_state_without_backend_get() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
|
||||||
|
let backend = register_mock_tier(&manager, &tier).await;
|
||||||
|
let remote_object = format!("remote/{}", Uuid::new_v4());
|
||||||
|
backend
|
||||||
|
.set_transition_candidate_probe_override(Some(TransitionCandidateProbe::VersionedPresent(
|
||||||
|
"versioned-candidate".to_string(),
|
||||||
|
)))
|
||||||
|
.await;
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
bucket: "bucket".to_string(),
|
||||||
|
name: "object".to_string(),
|
||||||
|
size: 1,
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: remote_object.clone(),
|
||||||
|
version_id: String::new(),
|
||||||
|
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
|
||||||
|
tier,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -6330,19 +6803,28 @@ mod tests {
|
|||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
Ok(_) => panic!("unknown remote version state must fail before backend IO"),
|
Ok(_) => panic!("versioned legacy unknown state without stored version must fail before backend GET"),
|
||||||
Err(err) => err,
|
Err(err) => err,
|
||||||
};
|
};
|
||||||
|
|
||||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||||
|
assert_eq!(backend.op_log().await, vec![MockWarmOp::Probe { object: remote_object }]);
|
||||||
assert_eq!(backend.get_count().await, 0);
|
assert_eq!(backend.get_count().await, 0);
|
||||||
|
assert_eq!(backend.remove_count().await, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn free_version_delete_rejects_unknown_version_state_before_backend_io() {
|
async fn free_version_delete_rejects_explicit_unknown_before_backend_io() {
|
||||||
let manager = TierConfigMgr::new();
|
let manager = TierConfigMgr::new();
|
||||||
let backend = register_mock_tier(&manager, "WARM").await;
|
let backend = register_mock_tier(&manager, "WARM").await;
|
||||||
|
let identity = test_tier_destination_identity(&manager, "WARM").await;
|
||||||
|
let mut user_defined = user_defined_with_tier_destination_identity(identity);
|
||||||
|
rustfs_utils::http::metadata_compat::insert_str(
|
||||||
|
&mut user_defined,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
|
||||||
|
rustfs_filemeta::TransitionVersionState::Unknown.as_str().to_string(),
|
||||||
|
);
|
||||||
let object_info = ObjectInfo {
|
let object_info = ObjectInfo {
|
||||||
transitioned_object: TransitionedObject {
|
transitioned_object: TransitionedObject {
|
||||||
name: "remote/object".to_string(),
|
name: "remote/object".to_string(),
|
||||||
@@ -6351,17 +6833,251 @@ mod tests {
|
|||||||
..Default::default()
|
..Default::default()
|
||||||
},
|
},
|
||||||
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined.into(),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|
||||||
let err = super::delete_free_version_remote_object(&object_info, &manager)
|
let err = super::delete_free_version_remote_object(&object_info, &manager)
|
||||||
.await
|
.await
|
||||||
.expect_err("unknown remote version state must fail before backend IO");
|
.expect_err("explicit unknown cleanup must fail before backend IO");
|
||||||
|
|
||||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||||
|
assert!(err.to_string().contains("version state is unknown"));
|
||||||
|
assert_eq!(backend.op_log().await, Vec::<MockWarmOp>::new());
|
||||||
assert_eq!(backend.remove_count().await, 0);
|
assert_eq!(backend.remove_count().await, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
async fn test_tier_destination_identity(
|
||||||
|
manager: &Arc<tokio::sync::RwLock<TierConfigMgr>>,
|
||||||
|
tier: &str,
|
||||||
|
) -> crate::services::tier::tier::TierDestinationId {
|
||||||
|
TierConfigMgr::acquire_operation_lease(manager, tier)
|
||||||
|
.await
|
||||||
|
.expect("test tier lease should be available")
|
||||||
|
.backend_identity()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
fn user_defined_with_tier_destination_identity(
|
||||||
|
identity: crate::services::tier::tier::TierDestinationId,
|
||||||
|
) -> HashMap<String, String> {
|
||||||
|
let mut user_defined = HashMap::new();
|
||||||
|
rustfs_utils::http::metadata_compat::insert_str(
|
||||||
|
&mut user_defined,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITION_TIER_DESTINATION_ID,
|
||||||
|
rustfs_utils::crypto::hex(identity),
|
||||||
|
);
|
||||||
|
user_defined
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
fn user_defined_with_transition_version_state(state: rustfs_filemeta::TransitionVersionState) -> HashMap<String, String> {
|
||||||
|
let mut user_defined = HashMap::new();
|
||||||
|
rustfs_utils::http::metadata_compat::insert_str(
|
||||||
|
&mut user_defined,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
|
||||||
|
state.as_str().to_string(),
|
||||||
|
);
|
||||||
|
user_defined
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
fn insert_legacy_transition_version_id(user_defined: &mut HashMap<String, String>, version_id: &str) {
|
||||||
|
rustfs_utils::http::metadata_compat::insert_str(
|
||||||
|
user_defined,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_ID,
|
||||||
|
version_id.to_string(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn free_version_tuple_rejects_mixed_legacy_missing_and_explicit_unknown() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
register_mock_tier(&manager, "WARM").await;
|
||||||
|
let identity = test_tier_destination_identity(&manager, "WARM").await;
|
||||||
|
let mut legacy_metadata = user_defined_with_tier_destination_identity(identity);
|
||||||
|
insert_legacy_transition_version_id(&mut legacy_metadata, "legacy-version");
|
||||||
|
let mut explicit_metadata = legacy_metadata.clone();
|
||||||
|
rustfs_utils::http::metadata_compat::insert_str(
|
||||||
|
&mut explicit_metadata,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
|
||||||
|
rustfs_filemeta::TransitionVersionState::Unknown.as_str().to_string(),
|
||||||
|
);
|
||||||
|
let make_info = |user_defined: HashMap<String, String>| ObjectInfo {
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: "remote/object".to_string(),
|
||||||
|
version_id: "legacy-version".to_string(),
|
||||||
|
tier: "WARM".to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined.into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = super::free_version_remote_tuple_matches(&make_info(legacy_metadata), &make_info(explicit_metadata))
|
||||||
|
.expect_err("mixed legacy-missing and explicit unknown provenance must fail closed");
|
||||||
|
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::WouldBlock);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn free_version_delete_probes_exact_version_hidden_by_current_delete_marker() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
let tier = "WARM";
|
||||||
|
let backend = register_mock_tier(&manager, tier).await;
|
||||||
|
let identity = test_tier_destination_identity(&manager, tier).await;
|
||||||
|
let remote_object = format!("remote/{}", Uuid::new_v4());
|
||||||
|
let body = Bytes::from_static(b"legacy exact cleanup body");
|
||||||
|
let remote_version = backend
|
||||||
|
.put(
|
||||||
|
&remote_object,
|
||||||
|
ReaderImpl::Body(body),
|
||||||
|
i64::try_from(b"legacy exact cleanup body".len()).expect("body length should fit"),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("mock remote object should be stored");
|
||||||
|
let mut user_defined = user_defined_with_tier_destination_identity(identity);
|
||||||
|
insert_legacy_transition_version_id(&mut user_defined, &remote_version);
|
||||||
|
backend
|
||||||
|
.set_transition_candidate_probe_override(Some(TransitionCandidateProbe::Missing))
|
||||||
|
.await;
|
||||||
|
assert_eq!(
|
||||||
|
backend
|
||||||
|
.probe_transition_candidate_state(&remote_object)
|
||||||
|
.await
|
||||||
|
.expect("current remote view should be readable"),
|
||||||
|
TransitionCandidateProbe::Missing,
|
||||||
|
"a current delete marker must hide the historical data version from an unversioned probe"
|
||||||
|
);
|
||||||
|
backend.clear_op_log().await;
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: remote_object.clone(),
|
||||||
|
version_id: remote_version,
|
||||||
|
tier: tier.to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined.into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
super::delete_free_version_remote_object(&object_info, &manager)
|
||||||
|
.await
|
||||||
|
.expect("probe-proven legacy exact cleanup should delete the remote version");
|
||||||
|
super::delete_free_version_remote_object(&object_info, &manager)
|
||||||
|
.await
|
||||||
|
.expect("a retry after the exact remote version is already missing should be idempotent");
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
backend.op_log().await,
|
||||||
|
vec![
|
||||||
|
MockWarmOp::Get {
|
||||||
|
object: remote_object.clone()
|
||||||
|
},
|
||||||
|
MockWarmOp::Remove {
|
||||||
|
object: remote_object.clone()
|
||||||
|
},
|
||||||
|
MockWarmOp::Get {
|
||||||
|
object: remote_object.clone()
|
||||||
|
},
|
||||||
|
]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
backend.remove_versions().await,
|
||||||
|
vec![(remote_object, object_info.transitioned_object.version_id)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn free_version_delete_retains_legacy_unknown_unversioned_object() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
let tier = "WARM";
|
||||||
|
let backend = register_mock_tier(&manager, tier).await;
|
||||||
|
backend.set_put_remote_version(Some(String::new())).await;
|
||||||
|
let identity = test_tier_destination_identity(&manager, tier).await;
|
||||||
|
let remote_object = format!("remote/{}", Uuid::new_v4());
|
||||||
|
let body = Bytes::from_static(b"legacy unversioned cleanup body");
|
||||||
|
let remote_version = backend
|
||||||
|
.put(
|
||||||
|
&remote_object,
|
||||||
|
ReaderImpl::Body(body),
|
||||||
|
i64::try_from(b"legacy unversioned cleanup body".len()).expect("body length should fit"),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("mock remote object should be stored");
|
||||||
|
assert!(remote_version.is_empty());
|
||||||
|
backend.clear_op_log().await;
|
||||||
|
let mut user_defined = user_defined_with_tier_destination_identity(identity);
|
||||||
|
user_defined.insert("x-minio-internal-transitioned-versionID".to_string(), String::new());
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: remote_object.clone(),
|
||||||
|
version_id: String::new(),
|
||||||
|
tier: tier.to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined.into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let err = super::delete_free_version_remote_object(&object_info, &manager)
|
||||||
|
.await
|
||||||
|
.expect_err("legacy unversioned cleanup cannot exclude a versioning-state race");
|
||||||
|
|
||||||
|
assert_eq!(err.kind(), std::io::ErrorKind::WouldBlock);
|
||||||
|
assert!(backend.op_log().await.is_empty());
|
||||||
|
assert_eq!(backend.remove_count().await, 0);
|
||||||
|
assert!(backend.remove_versions().await.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn free_version_delete_does_not_remove_a_different_remote_version() {
|
||||||
|
let manager = TierConfigMgr::new();
|
||||||
|
let tier = "WARM";
|
||||||
|
let backend = register_mock_tier(&manager, tier).await;
|
||||||
|
let identity = test_tier_destination_identity(&manager, tier).await;
|
||||||
|
let remote_object = format!("remote/{}", Uuid::new_v4());
|
||||||
|
backend.set_put_remote_version(Some("different-version".to_string())).await;
|
||||||
|
backend
|
||||||
|
.put(
|
||||||
|
&remote_object,
|
||||||
|
ReaderImpl::Body(Bytes::from_static(b"different remote version")),
|
||||||
|
i64::try_from(b"different remote version".len()).expect("body length should fit"),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("different remote version should be stored");
|
||||||
|
backend.clear_op_log().await;
|
||||||
|
let mut user_defined = user_defined_with_tier_destination_identity(identity);
|
||||||
|
insert_legacy_transition_version_id(&mut user_defined, "legacy-version");
|
||||||
|
let object_info = ObjectInfo {
|
||||||
|
transitioned_object: TransitionedObject {
|
||||||
|
name: remote_object.clone(),
|
||||||
|
version_id: "legacy-version".to_string(),
|
||||||
|
tier: tier.to_string(),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||||
|
user_defined: user_defined.into(),
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
super::delete_free_version_remote_object(&object_info, &manager)
|
||||||
|
.await
|
||||||
|
.expect("a missing exact legacy version should be an idempotent cleanup success");
|
||||||
|
|
||||||
|
assert_eq!(backend.op_log().await, vec![MockWarmOp::Get { object: remote_object }]);
|
||||||
|
assert_eq!(backend.remove_count().await, 0);
|
||||||
|
assert!(backend.remove_versions().await.is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn free_version_remote_delete_requires_persisted_destination_identity() {
|
async fn free_version_remote_delete_requires_persisted_destination_identity() {
|
||||||
|
|||||||
@@ -25,6 +25,7 @@ use super::{
|
|||||||
manual_transition_job, tier_delete_journal, transition_transaction,
|
manual_transition_job, tier_delete_journal, transition_transaction,
|
||||||
};
|
};
|
||||||
use crate::error::{Error, Result};
|
use crate::error::{Error, Result};
|
||||||
|
use crate::services::tier::tier_probe_intent;
|
||||||
|
|
||||||
pub(crate) const ILM_META_PREFIX: &str = "ilm";
|
pub(crate) const ILM_META_PREFIX: &str = "ilm";
|
||||||
const ILM_META_OBJECT_PREFIX: &str = "ilm/";
|
const ILM_META_OBJECT_PREFIX: &str = "ilm/";
|
||||||
@@ -35,6 +36,7 @@ pub(crate) enum DurableIlmRecordKind {
|
|||||||
TierDeleteJournal,
|
TierDeleteJournal,
|
||||||
TierDeleteDispatchManifest,
|
TierDeleteDispatchManifest,
|
||||||
TransitionTransaction,
|
TransitionTransaction,
|
||||||
|
TierProbeIntent,
|
||||||
ManualTransitionJob,
|
ManualTransitionJob,
|
||||||
ManualTransitionScope,
|
ManualTransitionScope,
|
||||||
ManualTransitionTask,
|
ManualTransitionTask,
|
||||||
@@ -73,6 +75,12 @@ pub(crate) const TRANSITION_TRANSACTION_NAMESPACE: DurableIlmNamespace = Durable
|
|||||||
max_record_size: transition_transaction::MAX_TRANSITION_TRANSACTION_SIZE,
|
max_record_size: transition_transaction::MAX_TRANSITION_TRANSACTION_SIZE,
|
||||||
kind: DurableIlmRecordKind::TransitionTransaction,
|
kind: DurableIlmRecordKind::TransitionTransaction,
|
||||||
};
|
};
|
||||||
|
pub(crate) const TIER_PROBE_INTENT_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||||
|
name: "tier-probe-intent",
|
||||||
|
prefix: tier_probe_intent::TIER_PROBE_INTENT_RECORD_PREFIX,
|
||||||
|
max_record_size: tier_probe_intent::MAX_TIER_PROBE_INTENT_SIZE,
|
||||||
|
kind: DurableIlmRecordKind::TierProbeIntent,
|
||||||
|
};
|
||||||
pub(crate) const MANUAL_TRANSITION_JOB_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
pub(crate) const MANUAL_TRANSITION_JOB_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||||
name: "manual-transition-job",
|
name: "manual-transition-job",
|
||||||
prefix: "ilm/manual-transition/jobs",
|
prefix: "ilm/manual-transition/jobs",
|
||||||
@@ -98,11 +106,12 @@ pub(crate) const MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE: DurableIlmNamespace
|
|||||||
kind: DurableIlmRecordKind::ManualTransitionWorkerResult,
|
kind: DurableIlmRecordKind::ManualTransitionWorkerResult,
|
||||||
};
|
};
|
||||||
|
|
||||||
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 8] = [
|
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 9] = [
|
||||||
TIER_DELETE_JOURNAL_NAMESPACE,
|
TIER_DELETE_JOURNAL_NAMESPACE,
|
||||||
TIER_DELETE_JOURNAL_V6_NAMESPACE,
|
TIER_DELETE_JOURNAL_V6_NAMESPACE,
|
||||||
TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE,
|
TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE,
|
||||||
TRANSITION_TRANSACTION_NAMESPACE,
|
TRANSITION_TRANSACTION_NAMESPACE,
|
||||||
|
TIER_PROBE_INTENT_NAMESPACE,
|
||||||
MANUAL_TRANSITION_JOB_NAMESPACE,
|
MANUAL_TRANSITION_JOB_NAMESPACE,
|
||||||
MANUAL_TRANSITION_SCOPE_NAMESPACE,
|
MANUAL_TRANSITION_SCOPE_NAMESPACE,
|
||||||
MANUAL_TRANSITION_TASK_NAMESPACE,
|
MANUAL_TRANSITION_TASK_NAMESPACE,
|
||||||
@@ -200,6 +209,15 @@ pub(crate) enum DurableIlmRecordCheckpoint {
|
|||||||
revision: u64,
|
revision: u64,
|
||||||
state: transition_transaction::TransitionTransactionState,
|
state: transition_transaction::TransitionTransactionState,
|
||||||
},
|
},
|
||||||
|
TierProbeIntent {
|
||||||
|
content_sha256: String,
|
||||||
|
identity_sha256: String,
|
||||||
|
remote_version_sha256: String,
|
||||||
|
remote_version_known: bool,
|
||||||
|
owner_fence_sha256: String,
|
||||||
|
revision: u64,
|
||||||
|
state: tier_probe_intent::TierProbeIntentState,
|
||||||
|
},
|
||||||
ManualTransitionJob {
|
ManualTransitionJob {
|
||||||
content_sha256: String,
|
content_sha256: String,
|
||||||
identity_sha256: String,
|
identity_sha256: String,
|
||||||
@@ -232,6 +250,7 @@ impl DurableIlmRecordCheckpoint {
|
|||||||
| Self::TierDeleteDispatchManifest { content_sha256, .. }
|
| Self::TierDeleteDispatchManifest { content_sha256, .. }
|
||||||
| Self::TierDeleteDispatchParent { content_sha256, .. }
|
| Self::TierDeleteDispatchParent { content_sha256, .. }
|
||||||
| Self::TransitionTransaction { content_sha256, .. }
|
| Self::TransitionTransaction { content_sha256, .. }
|
||||||
|
| Self::TierProbeIntent { content_sha256, .. }
|
||||||
| Self::ManualTransitionJob { content_sha256, .. }
|
| Self::ManualTransitionJob { content_sha256, .. }
|
||||||
| Self::ManualTransitionScope { content_sha256, .. }
|
| Self::ManualTransitionScope { content_sha256, .. }
|
||||||
| Self::ManualTransitionTask { content_sha256 }
|
| Self::ManualTransitionTask { content_sha256 }
|
||||||
@@ -421,6 +440,32 @@ impl DurableIlmRecordCheckpoint {
|
|||||||
.is_some_and(|expected_revision| *next_revision == expected_revision)
|
.is_some_and(|expected_revision| *next_revision == expected_revision)
|
||||||
&& (!previous_remote_version_known || previous_remote_version == next_remote_version)
|
&& (!previous_remote_version_known || previous_remote_version == next_remote_version)
|
||||||
}
|
}
|
||||||
|
(
|
||||||
|
Self::TierProbeIntent {
|
||||||
|
identity_sha256: previous_identity,
|
||||||
|
remote_version_sha256: previous_remote_version,
|
||||||
|
remote_version_known: previous_remote_version_known,
|
||||||
|
owner_fence_sha256: previous_owner_fence,
|
||||||
|
revision: previous_revision,
|
||||||
|
state: previous_state,
|
||||||
|
..
|
||||||
|
},
|
||||||
|
Self::TierProbeIntent {
|
||||||
|
identity_sha256: next_identity,
|
||||||
|
remote_version_sha256: next_remote_version,
|
||||||
|
owner_fence_sha256: next_owner_fence,
|
||||||
|
revision: next_revision,
|
||||||
|
state: next_state,
|
||||||
|
..
|
||||||
|
},
|
||||||
|
) => {
|
||||||
|
previous_identity == next_identity
|
||||||
|
&& previous_owner_fence == next_owner_fence
|
||||||
|
&& next_revision
|
||||||
|
.checked_sub(*previous_revision)
|
||||||
|
.is_some_and(|distance| distance == 1 && tier_probe_state_reaches(*previous_state, *next_state, distance))
|
||||||
|
&& (!previous_remote_version_known || previous_remote_version == next_remote_version)
|
||||||
|
}
|
||||||
(
|
(
|
||||||
Self::ManualTransitionJob {
|
Self::ManualTransitionJob {
|
||||||
content_sha256: previous_content,
|
content_sha256: previous_content,
|
||||||
@@ -500,6 +545,14 @@ impl DurableIlmRecordCheckpoint {
|
|||||||
/// after the exact terminal ETag and terminal receipt were committed, to
|
/// after the exact terminal ETag and terminal receipt were committed, to
|
||||||
/// purge older object versions exposed by that deletion.
|
/// purge older object versions exposed by that deletion.
|
||||||
pub(crate) fn is_predecessor_of_terminal(&self, terminal: &Self) -> bool {
|
pub(crate) fn is_predecessor_of_terminal(&self, terminal: &Self) -> bool {
|
||||||
|
if let Self::TierProbeIntent { state, .. } = terminal
|
||||||
|
&& !matches!(
|
||||||
|
state,
|
||||||
|
tier_probe_intent::TierProbeIntentState::AbortedNoRemote | tier_probe_intent::TierProbeIntentState::Completed
|
||||||
|
)
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
if self == terminal || self.validate_successor(terminal).is_ok() {
|
if self == terminal || self.validate_successor(terminal).is_ok() {
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -568,6 +621,37 @@ impl DurableIlmRecordCheckpoint {
|
|||||||
}
|
}
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
(
|
||||||
|
Self::TierProbeIntent {
|
||||||
|
identity_sha256: previous_identity,
|
||||||
|
remote_version_sha256: previous_remote_version,
|
||||||
|
remote_version_known: previous_remote_version_known,
|
||||||
|
owner_fence_sha256: previous_owner_fence,
|
||||||
|
revision: previous_revision,
|
||||||
|
state: previous_state,
|
||||||
|
..
|
||||||
|
},
|
||||||
|
Self::TierProbeIntent {
|
||||||
|
identity_sha256: terminal_identity,
|
||||||
|
remote_version_sha256: terminal_remote_version,
|
||||||
|
owner_fence_sha256: terminal_owner_fence,
|
||||||
|
revision: terminal_revision,
|
||||||
|
state: terminal_state,
|
||||||
|
..
|
||||||
|
},
|
||||||
|
) => {
|
||||||
|
previous_identity == terminal_identity
|
||||||
|
&& previous_owner_fence == terminal_owner_fence
|
||||||
|
&& matches!(
|
||||||
|
terminal_state,
|
||||||
|
tier_probe_intent::TierProbeIntentState::AbortedNoRemote
|
||||||
|
| tier_probe_intent::TierProbeIntentState::Completed
|
||||||
|
)
|
||||||
|
&& terminal_revision
|
||||||
|
.checked_sub(*previous_revision)
|
||||||
|
.is_some_and(|distance| tier_probe_state_reaches(*previous_state, *terminal_state, distance))
|
||||||
|
&& (!previous_remote_version_known || previous_remote_version == terminal_remote_version)
|
||||||
|
}
|
||||||
_ => false,
|
_ => false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -606,6 +690,23 @@ fn transition_state_distance(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn tier_probe_state_reaches(
|
||||||
|
from: tier_probe_intent::TierProbeIntentState,
|
||||||
|
to: tier_probe_intent::TierProbeIntentState,
|
||||||
|
revision_distance: u64,
|
||||||
|
) -> bool {
|
||||||
|
use tier_probe_intent::TierProbeIntentState::{AbortedNoRemote, CleanupPending, Completed, UploadOutcomeUnknown, Uploaded};
|
||||||
|
|
||||||
|
match (from, to) {
|
||||||
|
(UploadOutcomeUnknown, Uploaded | CleanupPending | AbortedNoRemote) => revision_distance == 1,
|
||||||
|
(UploadOutcomeUnknown, Completed) => matches!(revision_distance, 2 | 3),
|
||||||
|
(Uploaded, CleanupPending) => revision_distance == 1,
|
||||||
|
(Uploaded, Completed) => revision_distance == 2,
|
||||||
|
(CleanupPending, Completed) => revision_distance == 1,
|
||||||
|
_ => false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn manual_job_state_reaches(
|
fn manual_job_state_reaches(
|
||||||
from: manual_transition_job::ManualTransitionJobState,
|
from: manual_transition_job::ManualTransitionJobState,
|
||||||
to: manual_transition_job::ManualTransitionJobState,
|
to: manual_transition_job::ManualTransitionJobState,
|
||||||
@@ -1082,6 +1183,42 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
|
|||||||
},
|
},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
DurableIlmRecordKind::TierProbeIntent => {
|
||||||
|
let probe_id = tier_probe_intent::tier_probe_intent_id_from_record_object_name(path)
|
||||||
|
.map_err(|err| Error::other(err.to_string()))?;
|
||||||
|
let intent =
|
||||||
|
tier_probe_intent::TierProbeIntent::decode(probe_id, data).map_err(|err| Error::other(err.to_string()))?;
|
||||||
|
let canonical =
|
||||||
|
tier_probe_intent::tier_probe_intent_record_object_name(probe_id).map_err(|err| Error::other(err.to_string()))?;
|
||||||
|
if canonical != path {
|
||||||
|
return Err(Error::other("tier probe intent path is not canonical"));
|
||||||
|
}
|
||||||
|
let identity_sha256 = checkpoint_hash(&(
|
||||||
|
intent.probe_id,
|
||||||
|
&intent.operation,
|
||||||
|
&intent.tier_name,
|
||||||
|
intent.destination_id,
|
||||||
|
&intent.probe_object,
|
||||||
|
&intent.creator_id,
|
||||||
|
intent.creator_epoch,
|
||||||
|
intent.created_at_unix_nanos,
|
||||||
|
))?;
|
||||||
|
let remote_version_sha256 = checkpoint_hash(&intent.remote_version)?;
|
||||||
|
let owner_fence_sha256 = checkpoint_hash(&intent.owner)?;
|
||||||
|
(
|
||||||
|
"probe_id",
|
||||||
|
probe_id.to_string(),
|
||||||
|
DurableIlmRecordCheckpoint::TierProbeIntent {
|
||||||
|
content_sha256,
|
||||||
|
identity_sha256,
|
||||||
|
remote_version_sha256,
|
||||||
|
remote_version_known: !intent.remote_version.is_unknown(),
|
||||||
|
owner_fence_sha256,
|
||||||
|
revision: intent.revision,
|
||||||
|
state: intent.state,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
DurableIlmRecordKind::ManualTransitionJob => {
|
DurableIlmRecordKind::ManualTransitionJob => {
|
||||||
let job_id = manual_transition_job::manual_transition_job_id_from_record_object_name(path)
|
let job_id = manual_transition_job::manual_transition_job_id_from_record_object_name(path)
|
||||||
.map_err(|err| Error::other(err.to_string()))?;
|
.map_err(|err| Error::other(err.to_string()))?;
|
||||||
@@ -1237,6 +1374,102 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn tier_probe_intent_fixture() -> tier_probe_intent::TierProbeIntent {
|
||||||
|
let probe_id = Uuid::parse_str("36e2220e-9ad2-495b-b3bc-c4d2caf70a31").expect("fixture uuid should parse");
|
||||||
|
tier_probe_intent::TierProbeIntent {
|
||||||
|
probe_id,
|
||||||
|
revision: 1,
|
||||||
|
state: tier_probe_intent::TierProbeIntentState::UploadOutcomeUnknown,
|
||||||
|
operation: tier_probe_intent::TierProbeOperationIdentity::Verify {
|
||||||
|
config_etag: "config-etag".to_string(),
|
||||||
|
backend_identity: [1; 32],
|
||||||
|
},
|
||||||
|
tier_name: "COLD-A".to_string(),
|
||||||
|
destination_id: [1; 32],
|
||||||
|
probe_object: tier_probe_intent::tier_probe_object_name(probe_id),
|
||||||
|
creator_id: "node-a".to_string(),
|
||||||
|
creator_epoch: Uuid::parse_str("76746062-c05a-40b7-9e38-d2722d7e0332").expect("fixture creator epoch should parse"),
|
||||||
|
created_at_unix_nanos: 1_780_000_000_000_000_000,
|
||||||
|
owner: tier_probe_intent::TierProbeOwnerFence {
|
||||||
|
owner_id: "node-a".to_string(),
|
||||||
|
owner_epoch: Uuid::parse_str("76746062-c05a-40b7-9e38-d2722d7e0332").expect("fixture owner epoch should parse"),
|
||||||
|
not_after_unix_nanos: 1_780_000_900_000_000_000,
|
||||||
|
},
|
||||||
|
remote_version: tier_probe_intent::TierProbeRemoteVersion::default(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn tier_probe_checkpoint(intent: &tier_probe_intent::TierProbeIntent) -> DurableIlmRecordCheckpoint {
|
||||||
|
let path =
|
||||||
|
tier_probe_intent::tier_probe_intent_record_object_name(intent.probe_id).expect("tier probe path should build");
|
||||||
|
let encoded = intent.encode().expect("tier probe intent should encode");
|
||||||
|
let namespace = classify_durable_ilm_record(&path)
|
||||||
|
.expect("tier probe namespace should classify")
|
||||||
|
.expect("tier probe intent should be durable");
|
||||||
|
assert_eq!(namespace, &TIER_PROBE_INTENT_NAMESPACE);
|
||||||
|
validate_durable_ilm_record(&path, &encoded)
|
||||||
|
.expect("tier probe intent should validate")
|
||||||
|
.checkpoint
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tier_probe_intent_checkpoint_tracks_exact_monotonic_generations() {
|
||||||
|
let initial_intent = tier_probe_intent_fixture();
|
||||||
|
let initial = tier_probe_checkpoint(&initial_intent);
|
||||||
|
|
||||||
|
let mut uploaded_intent = initial_intent;
|
||||||
|
uploaded_intent
|
||||||
|
.advance(
|
||||||
|
tier_probe_intent::TierProbeIntentState::Uploaded,
|
||||||
|
tier_probe_intent::TierProbeRemoteVersion::versioned("opaque-v1"),
|
||||||
|
)
|
||||||
|
.expect("uploaded state should advance");
|
||||||
|
let uploaded = tier_probe_checkpoint(&uploaded_intent);
|
||||||
|
initial
|
||||||
|
.validate_successor(&uploaded)
|
||||||
|
.expect("durable receipt may adopt the exact uploaded generation");
|
||||||
|
|
||||||
|
let mut cleanup_intent = uploaded_intent.clone();
|
||||||
|
cleanup_intent
|
||||||
|
.advance(
|
||||||
|
tier_probe_intent::TierProbeIntentState::CleanupPending,
|
||||||
|
uploaded_intent.remote_version.clone(),
|
||||||
|
)
|
||||||
|
.expect("cleanup state should advance");
|
||||||
|
let cleanup = tier_probe_checkpoint(&cleanup_intent);
|
||||||
|
uploaded
|
||||||
|
.validate_successor(&cleanup)
|
||||||
|
.expect("durable receipt may adopt the exact cleanup generation");
|
||||||
|
|
||||||
|
let mut completed_intent = cleanup_intent.clone();
|
||||||
|
completed_intent
|
||||||
|
.advance(tier_probe_intent::TierProbeIntentState::Completed, cleanup_intent.remote_version.clone())
|
||||||
|
.expect("completed state should advance");
|
||||||
|
let completed = tier_probe_checkpoint(&completed_intent);
|
||||||
|
cleanup
|
||||||
|
.validate_successor(&completed)
|
||||||
|
.expect("durable receipt may adopt the exact terminal generation");
|
||||||
|
assert!(
|
||||||
|
initial.is_predecessor_of_terminal(&completed),
|
||||||
|
"terminal cleanup must recognize the full acknowledged-PUT path"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
initial.validate_successor(&completed).is_err(),
|
||||||
|
"ordinary receipt advancement must not skip intermediate generations"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!initial.is_predecessor_of_terminal(&uploaded),
|
||||||
|
"a nonterminal generation must not be accepted as terminal proof"
|
||||||
|
);
|
||||||
|
|
||||||
|
let mut rebound = uploaded_intent;
|
||||||
|
rebound.owner.owner_epoch = Uuid::new_v4();
|
||||||
|
assert!(
|
||||||
|
rebound.encode().is_err(),
|
||||||
|
"dormant v1 must reject owner takeover before producing a checkpoint"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn tier_delete_dispatch_manifest_namespace_validates_monotonic_branches() {
|
fn tier_delete_dispatch_manifest_namespace_validates_monotonic_branches() {
|
||||||
use tier_delete_journal::TierDeleteDispatchManifestState::{Aborted, Aborting, Completed, DispatchAuthorized, Preparing};
|
use tier_delete_journal::TierDeleteDispatchManifestState::{Aborted, Aborting, Completed, DispatchAuthorized, Preparing};
|
||||||
|
|||||||
@@ -1170,6 +1170,7 @@ pub async fn save_manual_transition_job_record_if_current(
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(current_etag.to_string()),
|
if_match: Some(current_etag.to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -1242,6 +1243,7 @@ pub(crate) async fn save_manual_transition_worker_result_if_absent(
|
|||||||
data,
|
data,
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -1270,6 +1272,7 @@ pub(crate) async fn save_manual_transition_task_if_absent(
|
|||||||
data,
|
data,
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -1621,6 +1624,7 @@ pub async fn save_manual_transition_scope_admission_if_absent(
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -1672,6 +1676,7 @@ pub async fn save_manual_transition_scope_admission_if_current(
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(current_etag.to_string()),
|
if_match: Some(current_etag.to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
|
|||||||
@@ -1733,6 +1733,7 @@ async fn save_config_if_none_fenced(
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -1832,6 +1833,7 @@ async fn save_decommission_manifest_checkpoint_if_match(
|
|||||||
|
|
||||||
let mut opts = ObjectOptions {
|
let mut opts = ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
no_lock: true,
|
no_lock: true,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(observed_etag),
|
if_match: Some(observed_etag),
|
||||||
@@ -1960,6 +1962,7 @@ async fn save_config_if_match_fenced(
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(etag.to_string()),
|
if_match: Some(etag.to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -3780,6 +3783,7 @@ where
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -3869,6 +3873,7 @@ where
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(etag),
|
if_match: Some(etag),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -3893,6 +3898,7 @@ where
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use super::runtime_boundary as runtime_sources;
|
use super::runtime_boundary as runtime_sources;
|
||||||
use crate::bucket::lifecycle::bucket_lifecycle_ops::ExpiryOp;
|
use crate::bucket::lifecycle::bucket_lifecycle_ops::ExpiryOp;
|
||||||
@@ -72,9 +70,11 @@ static REMOTE_DELETE_BREAKER: LazyLock<Mutex<RemoteDeleteBreaker>> = LazyLock::n
|
|||||||
});
|
});
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
static REMOTE_TIER_DELETE_TEST_HOOK: std::sync::LazyLock<
|
type RemoteTierDeleteTestHook = Box<dyn Fn(&str, &str, &str) -> std::io::Result<()> + Send + Sync>;
|
||||||
std::sync::Mutex<Option<Box<dyn Fn(&str, &str, &str) -> std::io::Result<()> + Send + Sync>>>,
|
|
||||||
> = std::sync::LazyLock::new(|| std::sync::Mutex::new(None));
|
#[cfg(test)]
|
||||||
|
static REMOTE_TIER_DELETE_TEST_HOOK: std::sync::LazyLock<std::sync::Mutex<Option<RemoteTierDeleteTestHook>>> =
|
||||||
|
std::sync::LazyLock::new(|| std::sync::Mutex::new(None));
|
||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
struct RemoteDeleteBreaker {
|
struct RemoteDeleteBreaker {
|
||||||
@@ -107,7 +107,7 @@ impl RemoteDeleteBreaker {
|
|||||||
fn prune(&mut self, now: Instant) {
|
fn prune(&mut self, now: Instant) {
|
||||||
while let Some(ts) = self.failures.front().copied() {
|
while let Some(ts) = self.failures.front().copied() {
|
||||||
if now.duration_since(ts) > self.window {
|
if now.duration_since(ts) > self.window {
|
||||||
self.failures.pop_front();
|
let _ = self.failures.pop_front();
|
||||||
} else {
|
} else {
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
@@ -137,10 +137,10 @@ fn is_signer_header_error(err: &std::io::Error) -> bool {
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
if let Some(source) = err.get_ref() {
|
if let Some(source) = err.get_ref()
|
||||||
if error_chain_contains_signer_header_marker(source) {
|
&& error_chain_contains_signer_header_marker(source)
|
||||||
return true;
|
{
|
||||||
}
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
let message = err.to_string().to_ascii_lowercase();
|
let message = err.to_string().to_ascii_lowercase();
|
||||||
@@ -205,7 +205,7 @@ impl ObjSweeper {
|
|||||||
|
|
||||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub fn with_version(&mut self, vid: Option<Uuid>) -> &Self {
|
pub fn with_version(&mut self, vid: Option<Uuid>) -> &Self {
|
||||||
self.version_id = vid.clone();
|
self.version_id = vid;
|
||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -219,7 +219,7 @@ impl ObjSweeper {
|
|||||||
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
|
||||||
pub fn get_opts(&self) -> lifecycle::ObjectOpts {
|
pub fn get_opts(&self) -> lifecycle::ObjectOpts {
|
||||||
let mut opts = ObjectOpts {
|
let mut opts = ObjectOpts {
|
||||||
version_id: self.version_id.clone(),
|
version_id: self.version_id,
|
||||||
versioned: self.versioned,
|
versioned: self.versioned,
|
||||||
version_suspended: self.suspended,
|
version_suspended: self.suspended,
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -388,8 +388,8 @@ impl Jentry {
|
|||||||
impl ExpiryOp for Jentry {
|
impl ExpiryOp for Jentry {
|
||||||
fn op_hash(&self) -> u64 {
|
fn op_hash(&self) -> u64 {
|
||||||
let mut hasher = Sha256::new();
|
let mut hasher = Sha256::new();
|
||||||
hasher.update(format!("{}", self.tier_name).as_bytes());
|
hasher.update(self.tier_name.as_bytes());
|
||||||
hasher.update(format!("{}", self.obj_name).as_bytes());
|
hasher.update(self.obj_name.as_bytes());
|
||||||
xxh64::xxh64(hasher.finalize().as_slice(), XXHASH_SEED)
|
xxh64::xxh64(hasher.finalize().as_slice(), XXHASH_SEED)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -436,7 +436,7 @@ async fn delete_object_from_remote_tier_raw_with_manager(
|
|||||||
tier_name: &str,
|
tier_name: &str,
|
||||||
tier_config_mgr: &Arc<tokio::sync::RwLock<TierConfigMgr>>,
|
tier_config_mgr: &Arc<tokio::sync::RwLock<TierConfigMgr>>,
|
||||||
) -> Result<(), std::io::Error> {
|
) -> Result<(), std::io::Error> {
|
||||||
let lease = TierConfigMgr::acquire_operation_lease(&tier_config_mgr, tier_name)
|
let lease = TierConfigMgr::acquire_operation_lease(tier_config_mgr, tier_name)
|
||||||
.await
|
.await
|
||||||
.map_err(std::io::Error::other)?;
|
.map_err(std::io::Error::other)?;
|
||||||
delete_object_from_remote_tier_raw_with_lease(obj_name, rv_id, &lease, false, true).await
|
delete_object_from_remote_tier_raw_with_lease(obj_name, rv_id, &lease, false, true).await
|
||||||
|
|||||||
@@ -612,6 +612,7 @@ pub(crate) async fn save_transition_transaction_record(
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -658,6 +659,7 @@ pub(crate) async fn save_transition_transaction_record_if_current(
|
|||||||
data.clone(),
|
data.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(etag),
|
if_match: Some(etag),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
|
|||||||
@@ -477,6 +477,18 @@ impl BucketMetadata {
|
|||||||
!self.table_bucket_config_json.is_empty()
|
!self.table_bucket_config_json.is_empty()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `bucket-targets.json` is stored for this bucket but this build cannot
|
||||||
|
/// decode it.
|
||||||
|
///
|
||||||
|
/// Keeps "no replication targets configured" and "the target
|
||||||
|
/// configuration cannot be read" apart, the same distinction the
|
||||||
|
/// `fabricated` marker draws for the bucket metadata as a whole. Only
|
||||||
|
/// meaningful after [`Self::parse_all_configs`] has run; readers must fail
|
||||||
|
/// closed on `true` instead of serving an empty target set.
|
||||||
|
pub fn bucket_targets_unreadable(&self) -> bool {
|
||||||
|
!self.bucket_targets_config_json.is_empty() && self.bucket_target_config.is_none()
|
||||||
|
}
|
||||||
|
|
||||||
/// Parsed per-bucket durability override, if a valid one is stored.
|
/// Parsed per-bucket durability override, if a valid one is stored.
|
||||||
///
|
///
|
||||||
/// Absent/empty/unparsable payloads all mean "no override" (the bucket
|
/// Absent/empty/unparsable payloads all mean "no override" (the bucket
|
||||||
@@ -964,7 +976,32 @@ impl BucketMetadata {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_all_configs(&mut self) -> Result<()> {
|
/// Decode every stored sub-configuration into its typed field.
|
||||||
|
///
|
||||||
|
/// A decode failure never fails the whole load: this runs on every bucket
|
||||||
|
/// metadata read, including startup and peer reload, so one bucket's
|
||||||
|
/// corrupt sub-configuration must not make the bucket — or the node —
|
||||||
|
/// unloadable. Instead the failure is *retained*: the raw bytes stay
|
||||||
|
/// untouched and the typed field stays `None`, so `!raw.is_empty() &&
|
||||||
|
/// typed.is_none()` is the durable "exists but cannot be read" signal that
|
||||||
|
/// each accessor keys off. Which accessors must fail closed on it:
|
||||||
|
///
|
||||||
|
/// | Config | Verdict |
|
||||||
|
/// |---|---|
|
||||||
|
/// | policy | Fails closed: `get_bucket_policy` re-parses the raw JSON and propagates the error; `get_bucket_policy_raw` returns the stored bytes. |
|
||||||
|
/// | object lock | Fails closed in `object_lock_config_state_from_authoritative_metadata`; a retention decision may never be taken on a guess. |
|
||||||
|
/// | versioning | Fails closed in `get_versioning_config`; guessing Unversioned would make delete markers and version ids diverge from what is on disk. |
|
||||||
|
/// | replication | Fails closed in `get_replication_config`. |
|
||||||
|
/// | bucket targets | Fails closed in `get_bucket_targets_config`, and `sync_bucket_target_sys` marks the bucket unreadable in `BucketTargetSys` instead of publishing an empty target set (rustfs/backlog#2282). |
|
||||||
|
/// | encryption | Fails closed in `get_sse_config`: degrading to "no default encryption" stores plaintext objects the operator required to be encrypted. |
|
||||||
|
/// | public access block | Fails closed in `get_public_access_block_config`: degrading grants the anonymous access the operator asked to block. |
|
||||||
|
/// | quota | Fails closed in `get_quota_config`; the enforcement path in `quota::checker` already re-parses the raw JSON and refuses on error. |
|
||||||
|
/// | lifecycle | Safe to degrade: no rules means no expiration and no transition, so nothing is deleted or moved on the strength of an unreadable rule set. The bucket keeps serving reads and writes. |
|
||||||
|
/// | notification | Safe to degrade: events are an outbound side channel; no consumer draws a durability or authorization conclusion from their absence. |
|
||||||
|
/// | tagging | Safe to degrade: bucket tags are cost-allocation labels here; object-level tag conditions come from object metadata, not this blob. |
|
||||||
|
/// | CORS | Safe to degrade: an absent CORS configuration rejects cross-origin browser requests, which is already the restrictive direction. |
|
||||||
|
/// | logging, website, accelerate, request payment, bucket ACL | Safe to degrade: each only shapes an optional response or an optional side channel, and none of them authorizes an action or decides whether data is retained. |
|
||||||
|
pub(super) fn parse_all_configs(&mut self) -> Result<()> {
|
||||||
if let Err(e) = self.parse_policy_config() {
|
if let Err(e) = self.parse_policy_config() {
|
||||||
tracing::warn!(
|
tracing::warn!(
|
||||||
event = "bucket_metadata_parse_failed",
|
event = "bucket_metadata_parse_failed",
|
||||||
@@ -1088,20 +1125,26 @@ impl BucketMetadata {
|
|||||||
"Failed to parse bucket metadata config"
|
"Failed to parse bucket metadata config"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
// A stored targets blob that cannot be decoded must not collapse into
|
||||||
|
// the empty target set: that is indistinguishable from "no replication
|
||||||
|
// configured", so replication stops and no caller ever sees an error
|
||||||
|
// (rustfs/backlog#2282). Leaving the typed field `None` while the raw
|
||||||
|
// bytes stay non-empty is the retained parse failure every targets
|
||||||
|
// reader keys off; the bytes are preserved so the configuration is
|
||||||
|
// still recoverable.
|
||||||
|
self.bucket_target_config = None;
|
||||||
if !self.bucket_targets_config_json.is_empty() {
|
if !self.bucket_targets_config_json.is_empty() {
|
||||||
if let Err(e) = serde_json::from_slice::<BucketTargets>(&self.bucket_targets_config_json)
|
match serde_json::from_slice::<BucketTargets>(&self.bucket_targets_config_json) {
|
||||||
.map(|t| self.bucket_target_config = Some(t))
|
Ok(targets) => self.bucket_target_config = Some(targets),
|
||||||
{
|
Err(e) => tracing::error!(
|
||||||
tracing::warn!(
|
|
||||||
event = "bucket_metadata_parse_failed",
|
event = "bucket_metadata_parse_failed",
|
||||||
component = "ecstore",
|
component = "ecstore",
|
||||||
subsystem = "bucket_metadata",
|
subsystem = "bucket_metadata",
|
||||||
bucket = %self.name,
|
bucket = %self.name,
|
||||||
config = "bucket_targets",
|
config = "bucket_targets",
|
||||||
error = %e,
|
error = %e,
|
||||||
"Failed to parse bucket metadata config"
|
"Bucket replication targets are unreadable; replication for this bucket fails closed"
|
||||||
);
|
),
|
||||||
self.bucket_target_config = Some(BucketTargets::default());
|
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
self.bucket_target_config = Some(BucketTargets::default());
|
self.bucket_target_config = Some(BucketTargets::default());
|
||||||
@@ -1535,6 +1578,117 @@ mod test {
|
|||||||
assert_eq!(bucket_targets.targets[0].target_bucket, "target-bucket");
|
assert_eq!(bucket_targets.targets[0].target_bucket, "target-bucket");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// rustfs/backlog#2282: a stored targets blob this build cannot decode
|
||||||
|
/// must not become the empty target set, and must stay distinguishable
|
||||||
|
/// from a bucket that never configured a target.
|
||||||
|
#[test]
|
||||||
|
fn unreadable_bucket_targets_never_degrade_to_an_empty_target_set() {
|
||||||
|
let truncated = br#"{"targets":[{"endpoint":"s3.example.com","#.to_vec();
|
||||||
|
let mut corrupt = BucketMetadata::new("corrupt-targets");
|
||||||
|
corrupt.bucket_targets_config_json = truncated.clone();
|
||||||
|
|
||||||
|
corrupt
|
||||||
|
.parse_all_configs()
|
||||||
|
.expect("one unreadable sub-config must not fail the whole metadata load");
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
corrupt.bucket_target_config.is_none(),
|
||||||
|
"an undecodable targets blob must not produce a target set at all"
|
||||||
|
);
|
||||||
|
assert!(corrupt.bucket_targets_unreadable());
|
||||||
|
assert_eq!(
|
||||||
|
corrupt.bucket_targets_config_json, truncated,
|
||||||
|
"the raw bytes must survive so the configuration stays recoverable"
|
||||||
|
);
|
||||||
|
|
||||||
|
// The genuinely-absent case is unchanged, and the two now diverge.
|
||||||
|
let mut absent = BucketMetadata::new("no-targets");
|
||||||
|
absent.parse_all_configs().expect("absent targets parse");
|
||||||
|
assert!(
|
||||||
|
absent.bucket_target_config.as_ref().is_some_and(BucketTargets::is_empty),
|
||||||
|
"a bucket that configured no target still reads as an empty target set"
|
||||||
|
);
|
||||||
|
assert!(!absent.bucket_targets_unreadable());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `Credentials` carries no struct-level `serde(default)`, so one target
|
||||||
|
/// missing `secretKey` is a hard parse error for the whole document. That
|
||||||
|
/// must surface as "unreadable", never as "no targets configured".
|
||||||
|
#[test]
|
||||||
|
fn bucket_targets_missing_secret_key_are_unreadable_not_empty() {
|
||||||
|
let mut bm = BucketMetadata::new("missing-secret-key");
|
||||||
|
bm.bucket_targets_config_json = br#"{"targets":[{"endpoint":"s3.example.com","targetbucket":"remote","arn":"arn:rustfs:replication:us-east-1:src:1","credentials":{"accessKey":"AKIAEXAMPLE"}}]}"#.to_vec();
|
||||||
|
|
||||||
|
bm.parse_all_configs()
|
||||||
|
.expect("a rejected targets document must not fail the whole metadata load");
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
bm.bucket_targets_unreadable(),
|
||||||
|
"a targets document rejected for a missing secretKey is unreadable, not empty"
|
||||||
|
);
|
||||||
|
assert!(bm.bucket_target_config.is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The invariant every branch of `parse_all_configs` shares: a stored but
|
||||||
|
/// undecodable payload keeps its raw bytes and leaves the typed field
|
||||||
|
/// `None`, so no branch fabricates a value. What a reader may then do with
|
||||||
|
/// that state is decided per config; see the table on `parse_all_configs`.
|
||||||
|
#[test]
|
||||||
|
fn every_config_branch_retains_its_parse_failure_instead_of_defaulting() {
|
||||||
|
let malformed_xml = b"<not-a-valid-document".to_vec();
|
||||||
|
let malformed_json = b"{not-json".to_vec();
|
||||||
|
|
||||||
|
let mut bm = BucketMetadata::new("all-configs-malformed");
|
||||||
|
bm.policy_config_json = malformed_json.clone();
|
||||||
|
bm.quota_config_json = malformed_json.clone();
|
||||||
|
bm.bucket_targets_config_json = malformed_json.clone();
|
||||||
|
bm.notification_config_xml = malformed_xml.clone();
|
||||||
|
bm.lifecycle_config_xml = malformed_xml.clone();
|
||||||
|
bm.object_lock_config_xml = malformed_xml.clone();
|
||||||
|
bm.versioning_config_xml = malformed_xml.clone();
|
||||||
|
bm.encryption_config_xml = malformed_xml.clone();
|
||||||
|
bm.tagging_config_xml = malformed_xml.clone();
|
||||||
|
bm.replication_config_xml = malformed_xml.clone();
|
||||||
|
bm.cors_config_xml = malformed_xml.clone();
|
||||||
|
bm.logging_config_xml = malformed_xml.clone();
|
||||||
|
bm.website_config_xml = malformed_xml.clone();
|
||||||
|
bm.accelerate_config_xml = malformed_xml.clone();
|
||||||
|
bm.request_payment_config_xml = malformed_xml.clone();
|
||||||
|
bm.public_access_block_config_xml = malformed_xml.clone();
|
||||||
|
// `bucket_acl_config_json` is only checked for UTF-8, so only invalid
|
||||||
|
// UTF-8 exercises its failure branch.
|
||||||
|
bm.bucket_acl_config_json = vec![0xff, 0xfe];
|
||||||
|
|
||||||
|
bm.parse_all_configs()
|
||||||
|
.expect("a bucket whose every config is corrupt must still load its metadata");
|
||||||
|
|
||||||
|
let cleared: [(&str, bool); 17] = [
|
||||||
|
("policy", bm.policy_config.is_none()),
|
||||||
|
("quota", bm.quota_config.is_none()),
|
||||||
|
("bucket_targets", bm.bucket_target_config.is_none()),
|
||||||
|
("notification", bm.notification_config.is_none()),
|
||||||
|
("lifecycle", bm.lifecycle_config.is_none()),
|
||||||
|
("object_lock", bm.object_lock_config.is_none()),
|
||||||
|
("versioning", bm.versioning_config.is_none()),
|
||||||
|
("encryption", bm.sse_config.is_none()),
|
||||||
|
("tagging", bm.tagging_config.is_none()),
|
||||||
|
("replication", bm.replication_config.is_none()),
|
||||||
|
("cors", bm.cors_config.is_none()),
|
||||||
|
("logging", bm.logging_config.is_none()),
|
||||||
|
("website", bm.website_config.is_none()),
|
||||||
|
("accelerate", bm.accelerate_config.is_none()),
|
||||||
|
("request_payment", bm.request_payment_config.is_none()),
|
||||||
|
("public_access_block", bm.public_access_block_config.is_none()),
|
||||||
|
("bucket_acl", bm.bucket_acl_config.is_none()),
|
||||||
|
];
|
||||||
|
for (config, is_cleared) in cleared {
|
||||||
|
assert!(is_cleared, "{config}: a corrupt payload must not be replaced by a default");
|
||||||
|
}
|
||||||
|
|
||||||
|
assert_eq!(bm.bucket_targets_config_json, malformed_json, "raw bytes are retained");
|
||||||
|
assert_eq!(bm.lifecycle_config_xml, malformed_xml, "raw bytes are retained");
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn lifecycle_update_config_clears_parsed_config_on_delete() {
|
fn lifecycle_update_config_clears_parsed_config_on_delete() {
|
||||||
let mut bm = BucketMetadata::new("test-bucket");
|
let mut bm = BucketMetadata::new("test-bucket");
|
||||||
|
|||||||
@@ -360,6 +360,16 @@ async fn refresh_buckets_metadata_once(sys: Arc<RwLock<BucketMetadataSys>>) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn sync_bucket_target_sys(bucket: &str, bm: &BucketMetadata) {
|
async fn sync_bucket_target_sys(bucket: &str, bm: &BucketMetadata) {
|
||||||
|
if bm.bucket_targets_unreadable() {
|
||||||
|
// "The configuration cannot be read" is not "no targets configured".
|
||||||
|
// Publishing an empty snapshot here is what silently stopped
|
||||||
|
// replication (rustfs/backlog#2282): mark the bucket instead, so every
|
||||||
|
// targets reader gets a typed error, and leave any snapshot from an
|
||||||
|
// earlier readable load in place rather than withdrawing it.
|
||||||
|
BucketTargetSys::get().mark_targets_unreadable(bucket).await;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
BucketTargetSys::get()
|
BucketTargetSys::get()
|
||||||
.update_all_targets(bucket, bm.bucket_target_config.as_ref())
|
.update_all_targets(bucket, bm.bucket_target_config.as_ref())
|
||||||
.await;
|
.await;
|
||||||
@@ -645,6 +655,12 @@ pub struct BucketMetadataMutationGuard {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl BucketMetadataMutationGuard {
|
impl BucketMetadataMutationGuard {
|
||||||
|
/// Returns the storage-verified identity while both incarnation fences remain valid.
|
||||||
|
pub fn checked_bucket_incarnation(&self) -> Result<(&str, Uuid)> {
|
||||||
|
self.ensure_valid(&self.bucket)?;
|
||||||
|
Ok((&self.bucket, self.incarnation_id))
|
||||||
|
}
|
||||||
|
|
||||||
fn ensure_valid(&self, bucket: &str) -> Result<()> {
|
fn ensure_valid(&self, bucket: &str) -> Result<()> {
|
||||||
if self.bucket != bucket {
|
if self.bucket != bucket {
|
||||||
return Err(Error::other("bucket metadata mutation guard does not match bucket"));
|
return Err(Error::other("bucket metadata mutation guard does not match bucket"));
|
||||||
@@ -664,6 +680,29 @@ async fn acquire_config_write_guard_for_incarnation(
|
|||||||
sys: Arc<RwLock<BucketMetadataSys>>,
|
sys: Arc<RwLock<BucketMetadataSys>>,
|
||||||
bucket: &str,
|
bucket: &str,
|
||||||
expected_incarnation_id: Option<Uuid>,
|
expected_incarnation_id: Option<Uuid>,
|
||||||
|
) -> Result<BucketMetadataMutationGuard> {
|
||||||
|
acquire_config_write_guard_with_migration(sys, bucket, expected_incarnation_id, true).await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Scanner probes must not create an incarnation to make a capability available.
|
||||||
|
pub async fn acquire_scanner_bucket_incarnation_fence(
|
||||||
|
bucket: &str,
|
||||||
|
expected_incarnation_id: Uuid,
|
||||||
|
expected_owner_id: Uuid,
|
||||||
|
) -> Result<BucketMetadataMutationGuard> {
|
||||||
|
super::utils::check_valid_bucket_name(bucket)?;
|
||||||
|
let sys = get_bucket_metadata_sys()?;
|
||||||
|
if expected_owner_id.is_nil() || sys.read().await.api.id != expected_owner_id || expected_incarnation_id.is_nil() {
|
||||||
|
return Err(Error::other("scanner bucket incarnation owner does not match"));
|
||||||
|
}
|
||||||
|
acquire_config_write_guard_with_migration(sys, bucket, Some(expected_incarnation_id), false).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn acquire_config_write_guard_with_migration(
|
||||||
|
sys: Arc<RwLock<BucketMetadataSys>>,
|
||||||
|
bucket: &str,
|
||||||
|
expected_incarnation_id: Option<Uuid>,
|
||||||
|
migrate: bool,
|
||||||
) -> Result<BucketMetadataMutationGuard> {
|
) -> Result<BucketMetadataMutationGuard> {
|
||||||
let metadata_sys = sys.read().await.clone();
|
let metadata_sys = sys.read().await.clone();
|
||||||
let lifecycle_guard = metadata_sys.api.acquire_bucket_lifecycle_read_lock(bucket).await?;
|
let lifecycle_guard = metadata_sys.api.acquire_bucket_lifecycle_read_lock(bucket).await?;
|
||||||
@@ -671,13 +710,15 @@ async fn acquire_config_write_guard_for_incarnation(
|
|||||||
// Legacy buckets are migrated while the lifecycle fence prevents a
|
// Legacy buckets are migrated while the lifecycle fence prevents a
|
||||||
// same-name replacement. The second read under the write transaction is
|
// same-name replacement. The second read under the write transaction is
|
||||||
// the CAS source of truth for the actual rewrite.
|
// the CAS source of truth for the actual rewrite.
|
||||||
await_bucket_namespace_operation(
|
if migrate {
|
||||||
Some(&lifecycle_guard),
|
await_bucket_namespace_operation(
|
||||||
bucket,
|
Some(&lifecycle_guard),
|
||||||
"bucket config incarnation migration",
|
bucket,
|
||||||
metadata_sys.get_bucket_incarnation_id(bucket),
|
"bucket config incarnation migration",
|
||||||
)
|
metadata_sys.get_bucket_incarnation_id(bucket),
|
||||||
.await?;
|
)
|
||||||
|
.await?;
|
||||||
|
}
|
||||||
let transaction_guard = await_bucket_namespace_operation(
|
let transaction_guard = await_bucket_namespace_operation(
|
||||||
Some(&lifecycle_guard),
|
Some(&lifecycle_guard),
|
||||||
bucket,
|
bucket,
|
||||||
@@ -2118,7 +2159,9 @@ impl BucketMetadataSys {
|
|||||||
pub async fn get_public_access_block_config(&self, bucket: &str) -> Result<(PublicAccessBlockConfiguration, OffsetDateTime)> {
|
pub async fn get_public_access_block_config(&self, bucket: &str) -> Result<(PublicAccessBlockConfiguration, OffsetDateTime)> {
|
||||||
let (bm, _) = self.get_config(bucket).await?;
|
let (bm, _) = self.get_config(bucket).await?;
|
||||||
|
|
||||||
if let Some(config) = &bm.public_access_block_config {
|
if !bm.public_access_block_config_xml.is_empty() && bm.public_access_block_config.is_none() {
|
||||||
|
Err(Error::other("persisted bucket public access block configuration is invalid"))
|
||||||
|
} else if let Some(config) = &bm.public_access_block_config {
|
||||||
Ok((config.clone(), bm.public_access_block_config_updated_at))
|
Ok((config.clone(), bm.public_access_block_config_updated_at))
|
||||||
} else {
|
} else {
|
||||||
Err(Error::ConfigNotFound)
|
Err(Error::ConfigNotFound)
|
||||||
@@ -2429,7 +2472,9 @@ impl BucketMetadataSys {
|
|||||||
pub async fn get_sse_config(&self, bucket: &str) -> Result<(ServerSideEncryptionConfiguration, OffsetDateTime)> {
|
pub async fn get_sse_config(&self, bucket: &str) -> Result<(ServerSideEncryptionConfiguration, OffsetDateTime)> {
|
||||||
let (bm, _) = self.get_config(bucket).await?;
|
let (bm, _) = self.get_config(bucket).await?;
|
||||||
|
|
||||||
if let Some(config) = &bm.sse_config {
|
if !bm.encryption_config_xml.is_empty() && bm.sse_config.is_none() {
|
||||||
|
Err(Error::other("persisted bucket encryption configuration is invalid"))
|
||||||
|
} else if let Some(config) = &bm.sse_config {
|
||||||
Ok((config.clone(), bm.encryption_config_updated_at))
|
Ok((config.clone(), bm.encryption_config_updated_at))
|
||||||
} else {
|
} else {
|
||||||
Err(Error::ConfigNotFound)
|
Err(Error::ConfigNotFound)
|
||||||
@@ -2500,7 +2545,9 @@ impl BucketMetadataSys {
|
|||||||
pub async fn get_quota_config(&self, bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> {
|
pub async fn get_quota_config(&self, bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> {
|
||||||
let (bm, _) = self.get_config(bucket).await?;
|
let (bm, _) = self.get_config(bucket).await?;
|
||||||
|
|
||||||
if let Some(config) = &bm.quota_config {
|
if !bm.quota_config_json.is_empty() && bm.quota_config.is_none() {
|
||||||
|
Err(Error::other("persisted bucket quota configuration is invalid"))
|
||||||
|
} else if let Some(config) = &bm.quota_config {
|
||||||
Ok((config.clone(), bm.quota_config_updated_at))
|
Ok((config.clone(), bm.quota_config_updated_at))
|
||||||
} else {
|
} else {
|
||||||
Err(Error::ConfigNotFound)
|
Err(Error::ConfigNotFound)
|
||||||
@@ -2522,7 +2569,9 @@ impl BucketMetadataSys {
|
|||||||
pub async fn get_bucket_targets_config(&self, bucket: &str) -> Result<BucketTargets> {
|
pub async fn get_bucket_targets_config(&self, bucket: &str) -> Result<BucketTargets> {
|
||||||
let (bm, _) = self.get_config(bucket).await?;
|
let (bm, _) = self.get_config(bucket).await?;
|
||||||
|
|
||||||
if let Some(config) = &bm.bucket_target_config {
|
if bm.bucket_targets_unreadable() {
|
||||||
|
Err(Error::other("persisted bucket replication target configuration is invalid"))
|
||||||
|
} else if let Some(config) = &bm.bucket_target_config {
|
||||||
Ok(config.clone())
|
Ok(config.clone())
|
||||||
} else {
|
} else {
|
||||||
Err(Error::ConfigNotFound)
|
Err(Error::ConfigNotFound)
|
||||||
@@ -2593,6 +2642,7 @@ pub(crate) mod test_support {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::test_support::isolated_store_over_temp_disks;
|
use super::test_support::isolated_store_over_temp_disks;
|
||||||
use super::*;
|
use super::*;
|
||||||
|
use crate::bucket::bucket_target_sys::BucketTargetError;
|
||||||
use crate::bucket::metadata::{
|
use crate::bucket::metadata::{
|
||||||
BUCKET_ACCELERATE_CONFIG, BUCKET_CORS_CONFIG, BUCKET_LIFECYCLE_CONFIG, BUCKET_LOGGING_CONFIG, BUCKET_NOTIFICATION_CONFIG,
|
BUCKET_ACCELERATE_CONFIG, BUCKET_CORS_CONFIG, BUCKET_LIFECYCLE_CONFIG, BUCKET_LOGGING_CONFIG, BUCKET_NOTIFICATION_CONFIG,
|
||||||
BUCKET_POLICY_CONFIG, BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, BUCKET_REPLICATION_CONFIG, BUCKET_REQUEST_PAYMENT_CONFIG,
|
BUCKET_POLICY_CONFIG, BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, BUCKET_REPLICATION_CONFIG, BUCKET_REQUEST_PAYMENT_CONFIG,
|
||||||
@@ -2788,6 +2838,36 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The `parse_all_configs` audit (rustfs/backlog#2282): every accessor
|
||||||
|
/// whose configuration grants something — plaintext storage, anonymous
|
||||||
|
/// access, capacity, replication targets — reports a corrupt payload as
|
||||||
|
/// invalid rather than as absent, because "absent" is what grants it.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn malformed_permissive_configs_are_not_reported_as_absent() {
|
||||||
|
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
|
||||||
|
let sys = BucketMetadataSys::new(ecstore);
|
||||||
|
let bucket = "malformed-permissive-config";
|
||||||
|
let mut metadata = BucketMetadata::new(bucket);
|
||||||
|
metadata.encryption_config_xml = b"<ServerSideEncryptionConfiguration".to_vec();
|
||||||
|
metadata.public_access_block_config_xml = b"<PublicAccessBlockConfiguration".to_vec();
|
||||||
|
metadata.quota_config_json = b"{not-json".to_vec();
|
||||||
|
metadata.bucket_targets_config_json = b"{not-json".to_vec();
|
||||||
|
metadata
|
||||||
|
.parse_all_configs()
|
||||||
|
.expect("a corrupt sub-config must not fail the load");
|
||||||
|
sys.set(bucket.to_string(), Arc::new(metadata)).await;
|
||||||
|
|
||||||
|
for (config, result) in [
|
||||||
|
("encryption", sys.get_sse_config(bucket).await.err()),
|
||||||
|
("public access block", sys.get_public_access_block_config(bucket).await.err()),
|
||||||
|
("quota", sys.get_quota_config(bucket).await.err()),
|
||||||
|
("bucket targets", sys.get_bucket_targets_config(bucket).await.err()),
|
||||||
|
] {
|
||||||
|
let err = result.unwrap_or_else(|| panic!("malformed {config} metadata must not read as a value"));
|
||||||
|
assert_ne!(err, Error::ConfigNotFound, "malformed {config} metadata must not be reported as absent");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn config_states_distinguish_authoritative_absence_from_fabricated_metadata() {
|
async fn config_states_distinguish_authoritative_absence_from_fabricated_metadata() {
|
||||||
use std::sync::atomic::Ordering;
|
use std::sync::atomic::Ordering;
|
||||||
@@ -3127,6 +3207,82 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn scoped_dirty_usage_incarnation_probe_does_not_migrate_legacy_metadata() {
|
||||||
|
let (dirs, store) = isolated_store_over_temp_disks().await;
|
||||||
|
let sys = Arc::new(RwLock::new(BucketMetadataSys::new(store.clone())));
|
||||||
|
let bucket = "scoped-ack-legacy";
|
||||||
|
for dir in &dirs {
|
||||||
|
std::fs::create_dir_all(dir.path().join(bucket)).expect("create legacy bucket");
|
||||||
|
}
|
||||||
|
let mut metadata = BucketMetadata::new(bucket);
|
||||||
|
metadata.bucket_incarnation_id = Uuid::nil();
|
||||||
|
sys.read()
|
||||||
|
.await
|
||||||
|
.persist_and_set(metadata)
|
||||||
|
.await
|
||||||
|
.expect("persist legacy metadata");
|
||||||
|
assert!(
|
||||||
|
acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(Uuid::new_v4()), false)
|
||||||
|
.await
|
||||||
|
.is_err()
|
||||||
|
);
|
||||||
|
assert!(load_bucket_incarnation(store, bucket).await.expect("read sidecar").is_none());
|
||||||
|
assert!(
|
||||||
|
sys.read()
|
||||||
|
.await
|
||||||
|
.get_config_from_disk(bucket)
|
||||||
|
.await
|
||||||
|
.expect("read metadata")
|
||||||
|
.bucket_incarnation_id
|
||||||
|
.is_nil()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||||
|
#[serial]
|
||||||
|
async fn scoped_dirty_usage_incarnation_rejects_deleted_and_recreated_bucket() {
|
||||||
|
let (_dirs, store) = isolated_store_over_temp_disks().await;
|
||||||
|
init_bucket_metadata_sys(store.clone(), Vec::new()).await;
|
||||||
|
let sys = bucket_metadata_sys_of(&store.ctx).expect("metadata owner");
|
||||||
|
let bucket = "scoped-ack-recreated";
|
||||||
|
store
|
||||||
|
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("create bucket");
|
||||||
|
let old = store.bucket_incarnation_id_from_disk(bucket).await.expect("old incarnation");
|
||||||
|
let guard = acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(old), false)
|
||||||
|
.await
|
||||||
|
.expect("trusted incarnation fence");
|
||||||
|
assert_eq!(guard.checked_bucket_incarnation().expect("valid fences"), (bucket, old));
|
||||||
|
drop(guard);
|
||||||
|
store
|
||||||
|
.delete_bucket(bucket, &DeleteBucketOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("delete bucket");
|
||||||
|
assert!(
|
||||||
|
acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(old), false)
|
||||||
|
.await
|
||||||
|
.is_err()
|
||||||
|
);
|
||||||
|
store
|
||||||
|
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("recreate bucket");
|
||||||
|
let new = store.bucket_incarnation_id_from_disk(bucket).await.expect("new incarnation");
|
||||||
|
assert_ne!(old, new);
|
||||||
|
assert!(
|
||||||
|
acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(old), false)
|
||||||
|
.await
|
||||||
|
.is_err()
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
acquire_config_write_guard_with_migration(sys, bucket, Some(new), false)
|
||||||
|
.await
|
||||||
|
.is_ok()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn old_node_metadata_rewrite_cannot_replace_bucket_incarnation_sidecar() {
|
async fn old_node_metadata_rewrite_cannot_replace_bucket_incarnation_sidecar() {
|
||||||
let (dirs, ecstore) = isolated_store_over_temp_disks().await;
|
let (dirs, ecstore) = isolated_store_over_temp_disks().await;
|
||||||
@@ -4066,6 +4222,114 @@ mod tests {
|
|||||||
target_sys.delete(bucket).await;
|
target_sys.delete(bucket).await;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// rustfs/backlog#2282: an unreadable `bucket-targets.json` reaches every
|
||||||
|
/// targets reader as a typed error; it neither withdraws a snapshot a
|
||||||
|
/// previous readable load published, nor collapses into the "no targets
|
||||||
|
/// configured" state that a bucket with an absent configuration reports.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn unreadable_bucket_targets_fail_closed_and_stay_distinct_from_absent() {
|
||||||
|
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
|
||||||
|
let sys = BucketMetadataSys::new(ecstore);
|
||||||
|
let target_sys = BucketTargetSys::get();
|
||||||
|
let unreadable = "targets-unreadable";
|
||||||
|
let absent = "targets-absent";
|
||||||
|
target_sys.delete(unreadable).await;
|
||||||
|
target_sys.delete(absent).await;
|
||||||
|
|
||||||
|
// A readable load publishes this bucket's targets.
|
||||||
|
let mut readable = BucketMetadata::new(unreadable);
|
||||||
|
readable.bucket_target_config = Some(BucketTargets {
|
||||||
|
targets: vec![target(unreadable, "live")],
|
||||||
|
});
|
||||||
|
sync_bucket_target_sys(unreadable, &readable).await;
|
||||||
|
assert_eq!(
|
||||||
|
target_sys
|
||||||
|
.list_bucket_targets(unreadable)
|
||||||
|
.await
|
||||||
|
.expect("readable targets publish")
|
||||||
|
.targets
|
||||||
|
.len(),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
|
||||||
|
// The same bucket reloaded with a blob that cannot be decoded.
|
||||||
|
let mut corrupt = BucketMetadata::new(unreadable);
|
||||||
|
corrupt.bucket_targets_config_json = br#"{"targets":[{"endpoint":"#.to_vec();
|
||||||
|
corrupt
|
||||||
|
.parse_all_configs()
|
||||||
|
.expect("an unreadable targets blob must not fail the metadata load");
|
||||||
|
sys.set(unreadable.to_string(), Arc::new(corrupt)).await;
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
matches!(
|
||||||
|
target_sys.list_bucket_targets(unreadable).await,
|
||||||
|
Err(BucketTargetError::BucketRemoteTargetsUnreadable { .. })
|
||||||
|
),
|
||||||
|
"an unreadable configuration must not read as an empty or a missing target set"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
target_sys.list_targets(unreadable, "").await.is_err(),
|
||||||
|
"the admin listing must surface the fault instead of an empty list"
|
||||||
|
);
|
||||||
|
let err = sys
|
||||||
|
.get_bucket_targets_config(unreadable)
|
||||||
|
.await
|
||||||
|
.expect_err("an unreadable targets configuration must not read as a value");
|
||||||
|
assert_ne!(err, Error::ConfigNotFound, "unreadable must not be reported as absent");
|
||||||
|
|
||||||
|
// A bucket that never configured a target keeps its previous behavior.
|
||||||
|
let mut no_targets = BucketMetadata::new(absent);
|
||||||
|
no_targets.parse_all_configs().expect("absent targets parse");
|
||||||
|
sys.set(absent.to_string(), Arc::new(no_targets)).await;
|
||||||
|
assert!(
|
||||||
|
matches!(
|
||||||
|
target_sys.list_bucket_targets(absent).await,
|
||||||
|
Err(BucketTargetError::BucketRemoteTargetNotFound { .. })
|
||||||
|
),
|
||||||
|
"an absent configuration must still report as a missing target set"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
target_sys
|
||||||
|
.list_targets(absent, "")
|
||||||
|
.await
|
||||||
|
.expect("an absent configuration lists no targets")
|
||||||
|
.is_empty()
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
sys.get_bucket_targets_config(absent)
|
||||||
|
.await
|
||||||
|
.expect("an absent targets configuration still reads as an empty set")
|
||||||
|
.is_empty(),
|
||||||
|
"the absent path must keep returning an empty target set, exactly as before"
|
||||||
|
);
|
||||||
|
|
||||||
|
// One bucket's unreadable configuration does not reach another bucket.
|
||||||
|
assert!(!matches!(
|
||||||
|
target_sys.list_bucket_targets(absent).await,
|
||||||
|
Err(BucketTargetError::BucketRemoteTargetsUnreadable { .. })
|
||||||
|
));
|
||||||
|
|
||||||
|
// A repaired configuration takes effect on the next load, no restart.
|
||||||
|
let mut repaired = BucketMetadata::new(unreadable);
|
||||||
|
repaired.bucket_target_config = Some(BucketTargets {
|
||||||
|
targets: vec![target(unreadable, "repaired")],
|
||||||
|
});
|
||||||
|
sync_bucket_target_sys(unreadable, &repaired).await;
|
||||||
|
assert_eq!(
|
||||||
|
target_sys
|
||||||
|
.list_bucket_targets(unreadable)
|
||||||
|
.await
|
||||||
|
.expect("a repaired configuration clears the unreadable marker")
|
||||||
|
.targets
|
||||||
|
.len(),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
|
||||||
|
target_sys.delete(unreadable).await;
|
||||||
|
target_sys.delete(absent).await;
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial]
|
#[serial]
|
||||||
async fn metadata_reload_clears_stale_bucket_targets_when_config_is_removed() {
|
async fn metadata_reload_clears_stale_bucket_targets_when_config_is_removed() {
|
||||||
|
|||||||
@@ -684,6 +684,7 @@ async fn write_checkpoint(
|
|||||||
};
|
};
|
||||||
let opts = ObjectOptions {
|
let opts = ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(preconditions),
|
http_preconditions: Some(preconditions),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -86,7 +86,12 @@ impl BreakerVerdict {
|
|||||||
Some(SourceError::Throttled | SourceError::Timeout | SourceError::Connect(_) | SourceError::ServerError(_)) => {
|
Some(SourceError::Throttled | SourceError::Timeout | SourceError::Connect(_) | SourceError::ServerError(_)) => {
|
||||||
BreakerVerdict::Failure
|
BreakerVerdict::Failure
|
||||||
}
|
}
|
||||||
Some(SourceError::AccessDenied | SourceError::Unsupported(_) | SourceError::Other(_)) => BreakerVerdict::Neutral,
|
Some(
|
||||||
|
SourceError::AccessDenied
|
||||||
|
| SourceError::Unsupported(_)
|
||||||
|
| SourceError::InvalidPagination(_)
|
||||||
|
| SourceError::Other(_),
|
||||||
|
) => BreakerVerdict::Neutral,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -189,8 +189,8 @@ pub enum SourceListPlan {
|
|||||||
/// delimiter — the source's own roll-up boundary matches the request's.
|
/// delimiter — the source's own roll-up boundary matches the request's.
|
||||||
Page { prefix: String },
|
Page { prefix: String },
|
||||||
/// `filter.prefix` reaches past a delimiter, so every key the source could
|
/// `filter.prefix` reaches past a delimiter, so every key the source could
|
||||||
/// contribute rolls into this one common prefix. One bounded probe listing
|
/// contribute rolls into this one common prefix. Bounded probes follow
|
||||||
/// decides whether it exists; there is nothing to paginate.
|
/// empty progressing pages until a key proves existence or the source ends.
|
||||||
Folded { probe_prefix: String, common_prefix: String },
|
Folded { probe_prefix: String, common_prefix: String },
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -279,6 +279,29 @@ pub struct FetchRequest {
|
|||||||
pub token: Option<String>,
|
pub token: Option<String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Invalid pagination metadata. Opaque cursor values are never included in errors.
|
||||||
|
#[derive(Clone, Copy, Debug, PartialEq, Eq, thiserror::Error)]
|
||||||
|
pub enum ListPageError {
|
||||||
|
#[error("truncated listing has no continuation token")]
|
||||||
|
Missing,
|
||||||
|
#[error("truncated listing has an empty continuation token")]
|
||||||
|
Empty,
|
||||||
|
#[error("truncated listing repeats a continuation token")]
|
||||||
|
Repeated,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn validate_list_page(is_truncated: bool, token: Option<&str>, next_token: Option<&str>) -> Result<(), ListPageError> {
|
||||||
|
if is_truncated {
|
||||||
|
match next_token {
|
||||||
|
None => return Err(ListPageError::Missing),
|
||||||
|
Some("") => return Err(ListPageError::Empty),
|
||||||
|
Some(next) if Some(next) == token => return Err(ListPageError::Repeated),
|
||||||
|
Some(_) => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug, Default)]
|
#[derive(Debug, Default)]
|
||||||
struct SideState {
|
struct SideState {
|
||||||
start: SideCursor,
|
start: SideCursor,
|
||||||
@@ -364,6 +387,11 @@ impl ListThroughMerger {
|
|||||||
/// or `filter.prefix` excludes it.
|
/// or `filter.prefix` excludes it.
|
||||||
pub fn disable_source(&mut self) {
|
pub fn disable_source(&mut self) {
|
||||||
self.source.disabled = true;
|
self.source.disabled = true;
|
||||||
|
// A refill can fail after a valid first page. A local-only response
|
||||||
|
// must discard both that source payload and its ordering horizon.
|
||||||
|
self.source.entries.clear();
|
||||||
|
self.source.pages.clear();
|
||||||
|
self.source.more = false;
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn next_fetch(&self) -> Option<FetchRequest> {
|
pub fn next_fetch(&self) -> Option<FetchRequest> {
|
||||||
@@ -378,7 +406,13 @@ impl ListThroughMerger {
|
|||||||
/// Records one fetched page. `entries` must be sorted by `name` and already
|
/// Records one fetched page. `entries` must be sorted by `name` and already
|
||||||
/// filtered with [`Self::accepts`]; the caller keeps the matching payloads
|
/// filtered with [`Self::accepts`]; the caller keeps the matching payloads
|
||||||
/// in the same order.
|
/// in the same order.
|
||||||
pub fn push_page(&mut self, side: MergeSide, entries: Vec<ListEntryKey>, is_truncated: bool, next_token: Option<String>) {
|
pub fn push_page(
|
||||||
|
&mut self,
|
||||||
|
side: MergeSide,
|
||||||
|
entries: Vec<ListEntryKey>,
|
||||||
|
is_truncated: bool,
|
||||||
|
next_token: Option<String>,
|
||||||
|
) -> Result<(), ListPageError> {
|
||||||
let state = match side {
|
let state = match side {
|
||||||
MergeSide::Local => &mut self.local,
|
MergeSide::Local => &mut self.local,
|
||||||
MergeSide::Source => &mut self.source,
|
MergeSide::Source => &mut self.source,
|
||||||
@@ -387,15 +421,19 @@ impl ListThroughMerger {
|
|||||||
Some(last) => last.next_token.clone(),
|
Some(last) => last.next_token.clone(),
|
||||||
None => state.start.token.clone(),
|
None => state.start.token.clone(),
|
||||||
};
|
};
|
||||||
// A truncated page without a cursor cannot be continued; treating the
|
validate_list_page(is_truncated, token.as_deref(), next_token.as_deref())?;
|
||||||
// side as finished is the only alternative to looping on it forever.
|
// Also reject a cycle through an earlier page in this bounded fetch.
|
||||||
state.more = is_truncated && next_token.is_some();
|
if is_truncated && state.pages.iter().any(|page| page.token == next_token) {
|
||||||
|
return Err(ListPageError::Repeated);
|
||||||
|
}
|
||||||
|
state.more = is_truncated;
|
||||||
state.pages.push(FetchedPage {
|
state.pages.push(FetchedPage {
|
||||||
token,
|
token,
|
||||||
count: entries.len(),
|
count: entries.len(),
|
||||||
next_token: is_truncated.then_some(next_token).flatten(),
|
next_token: is_truncated.then_some(next_token).flatten(),
|
||||||
});
|
});
|
||||||
state.entries.extend(entries);
|
state.entries.extend(entries);
|
||||||
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn finish(self) -> MergeOutcome {
|
pub fn finish(self) -> MergeOutcome {
|
||||||
@@ -599,9 +637,15 @@ mod tests {
|
|||||||
let (entries, truncated, next) = reference_page(keys, prefix, delimiter, fetch.token.as_deref(), max_keys);
|
let (entries, truncated, next) = reference_page(keys, prefix, delimiter, fetch.token.as_deref(), max_keys);
|
||||||
let kept: Vec<ListEntryKey> = entries.into_iter().filter(|entry| merger.accepts(&entry.name)).collect();
|
let kept: Vec<ListEntryKey> = entries.into_iter().filter(|entry| merger.accepts(&entry.name)).collect();
|
||||||
buffers[usize::from(fetch.side == MergeSide::Source)].extend(kept.iter().cloned());
|
buffers[usize::from(fetch.side == MergeSide::Source)].extend(kept.iter().cloned());
|
||||||
merger.push_page(fetch.side, kept, truncated, next);
|
merger
|
||||||
|
.push_page(fetch.side, kept, truncated, next)
|
||||||
|
.expect("reference provider pages must advance");
|
||||||
}
|
}
|
||||||
let outcome = merger.finish();
|
let outcome = merger.finish();
|
||||||
|
assert_eq!(outcome.is_truncated, outcome.next_token.is_some());
|
||||||
|
if outcome.is_truncated {
|
||||||
|
assert_ne!(outcome.next_token, token, "every truncated merged page must make progress");
|
||||||
|
}
|
||||||
page_sizes.push(outcome.picks.len());
|
page_sizes.push(outcome.picks.len());
|
||||||
for pick in &outcome.picks {
|
for pick in &outcome.picks {
|
||||||
let entry = buffers[usize::from(pick.side == MergeSide::Source)][pick.index].clone();
|
let entry = buffers[usize::from(pick.side == MergeSide::Source)][pick.index].clone();
|
||||||
@@ -616,11 +660,25 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn expected(local: &[String], source: &[String], prefix: &str, delimiter: Option<&str>) -> Vec<ListEntryKey> {
|
fn expected(local: &[String], source: &[String], prefix: &str, delimiter: Option<&str>) -> Vec<ListEntryKey> {
|
||||||
let mut all: Vec<String> = local.iter().chain(source.iter()).cloned().collect();
|
// This oracle builds the complete namespace independently of the
|
||||||
all.sort();
|
// provider's page/marker helper and the production merger.
|
||||||
all.dedup();
|
let mut namespace = std::collections::BTreeMap::new();
|
||||||
let (entries, _, _) = reference_page(&all, prefix, delimiter, None, usize::MAX);
|
for key in local.iter().chain(source) {
|
||||||
entries
|
let Some(suffix) = key.strip_prefix(prefix) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if let Some(delimiter) = delimiter.filter(|delimiter| !delimiter.is_empty())
|
||||||
|
&& let Some((directory, _)) = suffix.split_once(delimiter)
|
||||||
|
{
|
||||||
|
namespace.insert(format!("{prefix}{directory}{delimiter}"), true);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
namespace.insert(key.clone(), false);
|
||||||
|
}
|
||||||
|
namespace
|
||||||
|
.into_iter()
|
||||||
|
.map(|(name, is_prefix)| ListEntryKey { name, is_prefix })
|
||||||
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -662,7 +720,9 @@ mod tests {
|
|||||||
token: None
|
token: None
|
||||||
})
|
})
|
||||||
);
|
);
|
||||||
merger.push_page(MergeSide::Local, vec![ListEntryKey::object("a")], false, None);
|
merger
|
||||||
|
.push_page(MergeSide::Local, vec![ListEntryKey::object("a")], false, None)
|
||||||
|
.expect("local EOF is valid");
|
||||||
assert_eq!(merger.next_fetch(), None);
|
assert_eq!(merger.next_fetch(), None);
|
||||||
let outcome = merger.finish();
|
let outcome = merger.finish();
|
||||||
assert_eq!(outcome.picks.len(), 1);
|
assert_eq!(outcome.picks.len(), 1);
|
||||||
@@ -683,12 +743,14 @@ mod tests {
|
|||||||
};
|
};
|
||||||
let mut merger = ListThroughMerger::new(1, Some(&resume));
|
let mut merger = ListThroughMerger::new(1, Some(&resume));
|
||||||
merger.disable_source();
|
merger.disable_source();
|
||||||
merger.push_page(
|
merger
|
||||||
MergeSide::Local,
|
.push_page(
|
||||||
vec![ListEntryKey::object("b"), ListEntryKey::object("c")],
|
MergeSide::Local,
|
||||||
true,
|
vec![ListEntryKey::object("b"), ListEntryKey::object("c")],
|
||||||
Some("local-2".to_string()),
|
true,
|
||||||
);
|
Some("local-2".to_string()),
|
||||||
|
)
|
||||||
|
.expect("local cursor advances");
|
||||||
let outcome = merger.finish();
|
let outcome = merger.finish();
|
||||||
assert!(outcome.is_truncated);
|
assert!(outcome.is_truncated);
|
||||||
let token = outcome.next_token.expect("truncated page carries a token");
|
let token = outcome.next_token.expect("truncated page carries a token");
|
||||||
@@ -698,6 +760,212 @@ mod tests {
|
|||||||
assert_eq!(token.local.as_deref(), Some("local-1"), "a partly read page is re-listed");
|
assert_eq!(token.local.as_deref(), Some("local-1"), "a partly read page is re-listed");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn truncated_pages_require_a_nonempty_advancing_cursor() {
|
||||||
|
for side in [MergeSide::Local, MergeSide::Source] {
|
||||||
|
for entries in [vec![], vec![ListEntryKey::object("a")]] {
|
||||||
|
for (next, expected) in [
|
||||||
|
(None, Err(ListPageError::Missing)),
|
||||||
|
(Some(""), Err(ListPageError::Empty)),
|
||||||
|
(Some("stuck"), Err(ListPageError::Repeated)),
|
||||||
|
(Some("advances"), Ok(())),
|
||||||
|
] {
|
||||||
|
let resume = ListThroughToken::new(
|
||||||
|
SideCursor {
|
||||||
|
token: Some("stuck".into()),
|
||||||
|
done: false,
|
||||||
|
},
|
||||||
|
SideCursor {
|
||||||
|
token: Some("stuck".into()),
|
||||||
|
done: false,
|
||||||
|
},
|
||||||
|
None,
|
||||||
|
);
|
||||||
|
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||||
|
let result = merger.push_page(side, entries.clone(), true, next.map(str::to_string));
|
||||||
|
assert_eq!(result, expected, "{side:?}, {entries:?}, {next:?}");
|
||||||
|
let state = if side == MergeSide::Local {
|
||||||
|
&merger.local
|
||||||
|
} else {
|
||||||
|
&merger.source
|
||||||
|
};
|
||||||
|
assert_eq!(state.pages.len(), usize::from(result.is_ok()), "invalid page must not be accepted");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn repeated_empty_cursor_is_rejected_before_an_identical_page_can_escape() {
|
||||||
|
let resume = ListThroughToken::new(
|
||||||
|
SideCursor { token: None, done: true },
|
||||||
|
SideCursor {
|
||||||
|
token: Some("stuck".into()),
|
||||||
|
done: false,
|
||||||
|
},
|
||||||
|
None,
|
||||||
|
);
|
||||||
|
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||||
|
assert_eq!(
|
||||||
|
merger.next_fetch(),
|
||||||
|
Some(FetchRequest {
|
||||||
|
side: MergeSide::Source,
|
||||||
|
token: Some("stuck".into())
|
||||||
|
})
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
merger.push_page(MergeSide::Source, vec![], true, Some("stuck".into())),
|
||||||
|
Err(ListPageError::Repeated)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn empty_pages_may_advance_within_the_fetch_budget_until_eof() {
|
||||||
|
let mut merger = ListThroughMerger::new(2, None);
|
||||||
|
merger.push_page(MergeSide::Local, vec![], false, None).expect("local EOF");
|
||||||
|
for next in ["opaque-z", "opaque-a"] {
|
||||||
|
assert_eq!(merger.next_fetch().expect("bounded source fetch").side, MergeSide::Source);
|
||||||
|
merger
|
||||||
|
.push_page(MergeSide::Source, vec![], true, Some(next.into()))
|
||||||
|
.expect("opaque cursor advances regardless of sort order");
|
||||||
|
}
|
||||||
|
assert!(merger.next_fetch().is_none(), "two source fetches exhaust the request budget");
|
||||||
|
let outcome = merger.finish();
|
||||||
|
assert!(outcome.picks.is_empty());
|
||||||
|
assert!(outcome.is_truncated);
|
||||||
|
let token = outcome.next_token.expect("empty progressing page has a cursor");
|
||||||
|
assert_eq!(token.source.as_deref(), Some("opaque-a"));
|
||||||
|
let mut merger = ListThroughMerger::new(2, Some(&token));
|
||||||
|
assert_eq!(merger.next_fetch().expect("source resumes").token.as_deref(), Some("opaque-a"));
|
||||||
|
merger
|
||||||
|
.push_page(MergeSide::Source, vec![ListEntryKey::object("result")], false, None)
|
||||||
|
.expect("source EOF");
|
||||||
|
let outcome = merger.finish();
|
||||||
|
assert_eq!(
|
||||||
|
outcome.picks,
|
||||||
|
vec![MergePick {
|
||||||
|
side: MergeSide::Source,
|
||||||
|
index: 0
|
||||||
|
}]
|
||||||
|
);
|
||||||
|
assert!(!outcome.is_truncated);
|
||||||
|
assert!(outcome.next_token.is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_cursor_cycle_inside_the_fetch_budget_is_rejected() {
|
||||||
|
let resume = ListThroughToken::new(
|
||||||
|
SideCursor { token: None, done: true },
|
||||||
|
SideCursor {
|
||||||
|
token: Some("first".into()),
|
||||||
|
done: false,
|
||||||
|
},
|
||||||
|
None,
|
||||||
|
);
|
||||||
|
let mut merger = ListThroughMerger::new(2, Some(&resume));
|
||||||
|
merger
|
||||||
|
.push_page(MergeSide::Source, vec![], true, Some("second".into()))
|
||||||
|
.expect("first page advances");
|
||||||
|
assert_eq!(
|
||||||
|
merger.push_page(MergeSide::Source, vec![], true, Some("first".into())),
|
||||||
|
Err(ListPageError::Repeated)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn source_refill_failure_discards_buffered_source_entries_and_horizon() {
|
||||||
|
let mut merger = ListThroughMerger::new(2, None);
|
||||||
|
merger
|
||||||
|
.push_page(MergeSide::Local, vec![ListEntryKey::object("z")], false, None)
|
||||||
|
.expect("local EOF");
|
||||||
|
merger
|
||||||
|
.push_page(MergeSide::Source, vec![ListEntryKey::object("a")], true, Some("stuck".into()))
|
||||||
|
.expect("first source page advances");
|
||||||
|
assert_eq!(merger.next_fetch().expect("source refill is required").token.as_deref(), Some("stuck"));
|
||||||
|
assert_eq!(
|
||||||
|
merger.push_page(MergeSide::Source, vec![], true, Some("stuck".into())),
|
||||||
|
Err(ListPageError::Repeated)
|
||||||
|
);
|
||||||
|
merger.disable_source();
|
||||||
|
let outcome = merger.finish();
|
||||||
|
assert_eq!(
|
||||||
|
outcome.picks,
|
||||||
|
vec![MergePick {
|
||||||
|
side: MergeSide::Local,
|
||||||
|
index: 0
|
||||||
|
}]
|
||||||
|
);
|
||||||
|
assert!(!outcome.is_truncated);
|
||||||
|
assert!(outcome.next_token.is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn list_through_static_namespace_boundary_matrix() {
|
||||||
|
let corpus = [
|
||||||
|
"a",
|
||||||
|
"a/",
|
||||||
|
"a/b",
|
||||||
|
"a/b/child",
|
||||||
|
"a0",
|
||||||
|
"b",
|
||||||
|
"b/leaf",
|
||||||
|
"quote\"&<",
|
||||||
|
"space key",
|
||||||
|
"z",
|
||||||
|
"é",
|
||||||
|
"中/文",
|
||||||
|
];
|
||||||
|
for count in [0, 1, 3, 4, corpus.len()] {
|
||||||
|
let keys: Vec<String> = corpus[..count].iter().map(|key| (*key).to_string()).collect();
|
||||||
|
for placement in 0..3 {
|
||||||
|
let (local, source): (Vec<_>, Vec<_>) =
|
||||||
|
keys.iter()
|
||||||
|
.enumerate()
|
||||||
|
.fold((vec![], vec![]), |(mut local, mut source), (index, key)| {
|
||||||
|
if placement != 1 || index % 2 == 0 {
|
||||||
|
local.push(key.clone());
|
||||||
|
}
|
||||||
|
if placement != 0 || index % 2 == 0 {
|
||||||
|
source.push(key.clone());
|
||||||
|
}
|
||||||
|
(local, source)
|
||||||
|
});
|
||||||
|
for prefix in ["", "a", "a/", "中/"] {
|
||||||
|
for delimiter in [None, Some("/")] {
|
||||||
|
for max_keys in [1, 3, 4] {
|
||||||
|
let oracle = expected(&local, &source, prefix, delimiter);
|
||||||
|
let (emitted, sizes) = walk(&local, &source, prefix, delimiter, max_keys);
|
||||||
|
assert_eq!(
|
||||||
|
emitted.iter().map(|(entry, _)| entry.clone()).collect::<Vec<_>>(),
|
||||||
|
oracle,
|
||||||
|
"count={count}, placement={placement}, prefix={prefix}, delimiter={delimiter:?}, max={max_keys}"
|
||||||
|
);
|
||||||
|
let expected_sizes: Vec<_> = if oracle.is_empty() {
|
||||||
|
vec![0]
|
||||||
|
} else {
|
||||||
|
oracle.chunks(max_keys).map(<[ListEntryKey]>::len).collect()
|
||||||
|
};
|
||||||
|
assert_eq!(sizes, expected_sizes, "exact max and max+1 boundaries must agree");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn list_through_large_overlap_walk_keeps_all_5300_keys() {
|
||||||
|
let source: Vec<_> = (0..5000).map(|index| format!("k{index:05}")).collect();
|
||||||
|
let local: Vec<_> = (4800..5300).map(|index| format!("k{index:05}")).collect();
|
||||||
|
let (emitted, sizes) = walk(&local, &source, "", None, 333);
|
||||||
|
assert_eq!(emitted.len(), 5300);
|
||||||
|
for (index, (entry, side)) in emitted.iter().enumerate() {
|
||||||
|
assert_eq!(entry.name, format!("k{index:05}"));
|
||||||
|
assert_eq!(*side, if index >= 4800 { MergeSide::Local } else { MergeSide::Source });
|
||||||
|
}
|
||||||
|
assert_eq!(sizes, [vec![333; 15], vec![305]].concat());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn token_round_trips_and_rejects_tampering() {
|
fn token_round_trips_and_rejects_tampering() {
|
||||||
let token = ListThroughToken::new(
|
let token = ListThroughToken::new(
|
||||||
@@ -796,7 +1064,10 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
proptest! {
|
proptest! {
|
||||||
#![proptest_config(ProptestConfig::with_cases(256))]
|
#![proptest_config(ProptestConfig {
|
||||||
|
rng_seed: proptest::test_runner::RngSeed::Fixed(0xec5706),
|
||||||
|
..ProptestConfig::with_cases(256)
|
||||||
|
})]
|
||||||
|
|
||||||
/// Full pagination of a merged listing equals the sorted, deduplicated
|
/// Full pagination of a merged listing equals the sorted, deduplicated
|
||||||
/// union of both sides, with every shared key served by local, and no
|
/// union of both sides, with every shared key served by local, and no
|
||||||
|
|||||||
@@ -25,6 +25,7 @@
|
|||||||
//! Client-supplied `If-*`, `Authorization`, `Host` and SSE-C headers are never
|
//! Client-supplied `If-*`, `Authorization`, `Host` and SSE-C headers are never
|
||||||
//! forwarded: v1 rejects SSE-C source objects outright.
|
//! forwarded: v1 rejects SSE-C source objects outright.
|
||||||
|
|
||||||
|
use super::list_through::{ListPageError, validate_list_page};
|
||||||
use crate::bucket::remote_s3_client::{
|
use crate::bucket::remote_s3_client::{
|
||||||
PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_config,
|
PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_config,
|
||||||
};
|
};
|
||||||
@@ -223,6 +224,8 @@ pub enum SourceError {
|
|||||||
ServerError(u16),
|
ServerError(u16),
|
||||||
#[error("unsupported source object: {0}")]
|
#[error("unsupported source object: {0}")]
|
||||||
Unsupported(String),
|
Unsupported(String),
|
||||||
|
#[error("invalid source listing: {0}")]
|
||||||
|
InvalidPagination(#[from] ListPageError),
|
||||||
#[error("source request failed: {0}")]
|
#[error("source request failed: {0}")]
|
||||||
Other(String),
|
Other(String),
|
||||||
}
|
}
|
||||||
@@ -245,6 +248,7 @@ impl SourceError {
|
|||||||
SourceError::Connect(_) => "connect",
|
SourceError::Connect(_) => "connect",
|
||||||
SourceError::ServerError(_) => "server_error",
|
SourceError::ServerError(_) => "server_error",
|
||||||
SourceError::Unsupported(_) => "unsupported",
|
SourceError::Unsupported(_) => "unsupported",
|
||||||
|
SourceError::InvalidPagination(_) => "invalid_pagination",
|
||||||
SourceError::Other(_) => "other",
|
SourceError::Other(_) => "other",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -714,6 +718,7 @@ impl SourceClient {
|
|||||||
..*request
|
..*request
|
||||||
})
|
})
|
||||||
.await?;
|
.await?;
|
||||||
|
validate_list_page(page.is_truncated, request.continuation_token, page.next_continuation_token.as_deref())?;
|
||||||
page.objects = page
|
page.objects = page
|
||||||
.objects
|
.objects
|
||||||
.into_iter()
|
.into_iter()
|
||||||
@@ -800,11 +805,6 @@ impl SourceBackend for S3SourceBackend {
|
|||||||
|
|
||||||
let is_truncated = output.is_truncated.unwrap_or(false);
|
let is_truncated = output.is_truncated.unwrap_or(false);
|
||||||
let next_continuation_token = output.next_continuation_token;
|
let next_continuation_token = output.next_continuation_token;
|
||||||
if is_truncated && next_continuation_token.is_none() {
|
|
||||||
return Err(SourceError::Other(
|
|
||||||
"source reported a truncated listing without a continuation token".to_string(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let objects = output
|
let objects = output
|
||||||
.contents
|
.contents
|
||||||
.unwrap_or_default()
|
.unwrap_or_default()
|
||||||
@@ -1274,7 +1274,9 @@ mod tests {
|
|||||||
<CommonPrefixes><Prefix>data/photos/</Prefix></CommonPrefixes>
|
<CommonPrefixes><Prefix>data/photos/</Prefix></CommonPrefixes>
|
||||||
<CommonPrefixes><Prefix>outside/</Prefix></CommonPrefixes>
|
<CommonPrefixes><Prefix>outside/</Prefix></CommonPrefixes>
|
||||||
</ListBucketResult>"#;
|
</ListBucketResult>"#;
|
||||||
let (client, requests) = scripted_client(&spec(Some("data/")), vec![ok(Vec::new(), body), ok(Vec::new(), body)]).await;
|
let next_body = body.replace("data/opaque", "data/next");
|
||||||
|
let (client, requests) =
|
||||||
|
scripted_client(&spec(Some("data/")), vec![ok(Vec::new(), body), ok(Vec::new(), &next_body)]).await;
|
||||||
let first = client
|
let first = client
|
||||||
.list_page(&SourceListRequest {
|
.list_page(&SourceListRequest {
|
||||||
prefix: Some("photos/"),
|
prefix: Some("photos/"),
|
||||||
@@ -1336,7 +1338,104 @@ mod tests {
|
|||||||
.list_objects_v2(None, None, 10)
|
.list_objects_v2(None, None, 10)
|
||||||
.await
|
.await
|
||||||
.expect_err("truncated page without token is corrupt");
|
.expect_err("truncated page without token is corrupt");
|
||||||
assert!(matches!(err, SourceError::Other(_)), "{err:?}");
|
assert!(matches!(err, SourceError::InvalidPagination(ListPageError::Missing)), "{err:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn list_page_validates_s3_cursor_progress_before_mapping_entries() {
|
||||||
|
for contents in ["", "<Contents><Key>data/a</Key><Size>1</Size></Contents>"] {
|
||||||
|
for (truncated, next, expected) in [
|
||||||
|
(true, None, Some(ListPageError::Missing)),
|
||||||
|
(true, Some(""), Some(ListPageError::Empty)),
|
||||||
|
(true, Some("stuck"), Some(ListPageError::Repeated)),
|
||||||
|
(true, Some("opaque-next"), None),
|
||||||
|
(false, None, None),
|
||||||
|
(false, Some("stuck"), None),
|
||||||
|
] {
|
||||||
|
let next_xml = next
|
||||||
|
.map(|next| format!("<NextContinuationToken>{next}</NextContinuationToken>"))
|
||||||
|
.unwrap_or_default();
|
||||||
|
let body = format!(
|
||||||
|
"<ListBucketResult xmlns=\"http://s3.amazonaws.com/doc/2006-03-01/\"><IsTruncated>{truncated}</IsTruncated>{next_xml}{contents}</ListBucketResult>"
|
||||||
|
);
|
||||||
|
let (client, requests) = scripted_client(&spec(Some("data/")), vec![ok(Vec::new(), &body)]).await;
|
||||||
|
let result = client
|
||||||
|
.list_page(&SourceListRequest {
|
||||||
|
continuation_token: Some("stuck"),
|
||||||
|
max_keys: 2,
|
||||||
|
..Default::default()
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
match expected {
|
||||||
|
Some(expected) => {
|
||||||
|
let error = result.expect_err("malformed pagination must fail at the provider boundary");
|
||||||
|
assert!(
|
||||||
|
matches!(&error, SourceError::InvalidPagination(actual) if *actual == expected),
|
||||||
|
"{error:?}"
|
||||||
|
);
|
||||||
|
assert_eq!(error.class_label(), "invalid_pagination");
|
||||||
|
assert!(!error.is_retryable());
|
||||||
|
assert!(!error.to_string().contains("stuck"), "errors must not echo opaque tokens");
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
let page = result.expect("progressing empty/nonempty pages and EOF are valid");
|
||||||
|
assert_eq!(page.is_truncated, truncated);
|
||||||
|
assert_eq!(page.next_continuation_token.as_deref(), next);
|
||||||
|
assert_eq!(page.objects.len(), usize::from(!contents.is_empty()));
|
||||||
|
if let Some(object) = page.objects.first() {
|
||||||
|
assert_eq!(object.key, "a");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let requests = recorded(&requests);
|
||||||
|
assert_eq!(requests.len(), 1, "invalid pagination must not be retried");
|
||||||
|
assert!(requests[0].uri.contains("continuation-token=stuck"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
struct ListOnlyBackend(SourcePage);
|
||||||
|
|
||||||
|
#[async_trait::async_trait]
|
||||||
|
impl SourceBackend for ListOnlyBackend {
|
||||||
|
async fn list(&self, request: &SourceListRequest<'_>) -> Result<SourcePage, SourceError> {
|
||||||
|
assert_eq!(request.continuation_token, Some("stuck"), "opaque cursors reach every provider unchanged");
|
||||||
|
Ok(self.0.clone())
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn head(&self, _key: &str) -> Result<SourceHead, SourceError> {
|
||||||
|
panic!("unexpected HEAD in list test")
|
||||||
|
}
|
||||||
|
async fn get(&self, _key: &str, _range: Option<&HTTPRangeSpec>) -> Result<SourceGet, SourceError> {
|
||||||
|
panic!("unexpected GET in list test")
|
||||||
|
}
|
||||||
|
async fn tagging(&self, _key: &str) -> Result<HashMap<String, String>, SourceError> {
|
||||||
|
panic!("unexpected tagging in list test")
|
||||||
|
}
|
||||||
|
async fn probe(&self) -> Result<(), SourceError> {
|
||||||
|
panic!("unexpected probe in list test")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn list_page_validates_non_s3_provider_cursors_at_the_common_boundary() {
|
||||||
|
for (next, expected) in [
|
||||||
|
(None, ListPageError::Missing),
|
||||||
|
(Some(""), ListPageError::Empty),
|
||||||
|
(Some("stuck"), ListPageError::Repeated),
|
||||||
|
] {
|
||||||
|
let mut client = prefix_client(Some("data/".into()));
|
||||||
|
client.backend = Box::new(ListOnlyBackend(SourcePage {
|
||||||
|
is_truncated: true,
|
||||||
|
next_continuation_token: next.map(str::to_string),
|
||||||
|
..Default::default()
|
||||||
|
}));
|
||||||
|
let error = client
|
||||||
|
.list_objects_v2(None, Some("stuck"), 2)
|
||||||
|
.await
|
||||||
|
.expect_err("all providers must advance pagination");
|
||||||
|
assert!(matches!(error, SourceError::InvalidPagination(actual) if actual == expected));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
const TAGGING_BODY: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
|
const TAGGING_BODY: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
|||||||
@@ -177,7 +177,7 @@ impl From<&SourceError> for PullFailureReason {
|
|||||||
SourceError::Connect(_) => PullFailureReason::SourceConnect,
|
SourceError::Connect(_) => PullFailureReason::SourceConnect,
|
||||||
SourceError::ServerError(_) => PullFailureReason::SourceServerError,
|
SourceError::ServerError(_) => PullFailureReason::SourceServerError,
|
||||||
SourceError::Unsupported(_) => PullFailureReason::SourceUnsupported,
|
SourceError::Unsupported(_) => PullFailureReason::SourceUnsupported,
|
||||||
SourceError::Other(_) => PullFailureReason::SourceOther,
|
SourceError::InvalidPagination(_) | SourceError::Other(_) => PullFailureReason::SourceOther,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -652,9 +652,10 @@ async fn build_aws_s3_http_client_from_tls_path() -> Option<SharedHttpClient> {
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
use aws_smithy_async::time::TimeSource;
|
||||||
use aws_smithy_runtime_api::http::StatusCode as SmithyStatusCode;
|
use aws_smithy_runtime_api::http::StatusCode as SmithyStatusCode;
|
||||||
use std::sync::Mutex;
|
use std::sync::Mutex;
|
||||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering};
|
||||||
|
|
||||||
fn spec(endpoint: &str, secure: bool) -> RemoteS3EndpointSpec {
|
fn spec(endpoint: &str, secure: bool) -> RemoteS3EndpointSpec {
|
||||||
RemoteS3EndpointSpec {
|
RemoteS3EndpointSpec {
|
||||||
@@ -824,6 +825,174 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
struct ClockSkewTimeSource(Arc<AtomicU64>);
|
||||||
|
|
||||||
|
impl TimeSource for ClockSkewTimeSource {
|
||||||
|
fn now(&self) -> SystemTime {
|
||||||
|
SystemTime::UNIX_EPOCH + Duration::from_secs(self.0.load(Ordering::SeqCst))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
struct ClockSkewConnector {
|
||||||
|
request_headers: RecordedHeaders,
|
||||||
|
error_code: &'static str,
|
||||||
|
skew_seconds: i64,
|
||||||
|
clock: ClockSkewTimeSource,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn recorded_header<'a>(headers: &'a [(String, String)], name: &str) -> &'a str {
|
||||||
|
headers
|
||||||
|
.iter()
|
||||||
|
.find(|(key, _)| key.eq_ignore_ascii_case(name))
|
||||||
|
.map(|(_, value)| value.as_str())
|
||||||
|
.unwrap_or_else(|| panic!("signed request must contain {name}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn signing_time(headers: &[(String, String)]) -> chrono::NaiveDateTime {
|
||||||
|
chrono::NaiveDateTime::parse_from_str(recorded_header(headers, "x-amz-date"), "%Y%m%dT%H%M%SZ")
|
||||||
|
.expect("SDK signing timestamp must use the SigV4 format")
|
||||||
|
}
|
||||||
|
|
||||||
|
impl SmithyHttpConnector for ClockSkewConnector {
|
||||||
|
fn call(&self, request: HttpRequest) -> HttpConnectorFuture {
|
||||||
|
let mut headers = self.request_headers.lock().expect("clock skew request capture lock");
|
||||||
|
assert!(headers.len() < 3, "clock skew fixture must not exceed two GET attempts and one HEAD");
|
||||||
|
headers.push(
|
||||||
|
request
|
||||||
|
.headers()
|
||||||
|
.iter()
|
||||||
|
.map(|(key, value)| (key.to_string(), value.to_string()))
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
|
let server_time = chrono::DateTime::<chrono::Utc>::from(self.clock.now()).naive_utc()
|
||||||
|
+ chrono::Duration::seconds(self.skew_seconds);
|
||||||
|
let (status, body) = if headers.len() == 1 {
|
||||||
|
(
|
||||||
|
403,
|
||||||
|
format!("<Error><Code>{}</Code><Message>Clock skew fixture</Message></Error>", self.error_code),
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
(200, String::new())
|
||||||
|
};
|
||||||
|
let response = http::Response::builder()
|
||||||
|
.status(status)
|
||||||
|
.header("date", server_time.format("%a, %d %b %Y %H:%M:%S GMT").to_string())
|
||||||
|
.header("content-type", "application/xml")
|
||||||
|
.header("content-length", body.len())
|
||||||
|
.body(SdkBody::from(body))
|
||||||
|
.expect("clock skew fixture response");
|
||||||
|
HttpConnectorFuture::ready(Ok(HttpResponse::try_from(response).expect("Smithy fixture response")))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn clock_skew_client(
|
||||||
|
error_code: &'static str,
|
||||||
|
skew_seconds: i64,
|
||||||
|
retry: RemoteS3RetryPolicy,
|
||||||
|
) -> (S3Client, RecordedHeaders, ClockSkewTimeSource) {
|
||||||
|
let headers: RecordedHeaders = Arc::new(Mutex::new(Vec::new()));
|
||||||
|
let clock = ClockSkewTimeSource(Arc::new(AtomicU64::new(1_700_000_000)));
|
||||||
|
let connector = SharedHttpConnector::new(ClockSkewConnector {
|
||||||
|
request_headers: Arc::clone(&headers),
|
||||||
|
error_code,
|
||||||
|
skew_seconds,
|
||||||
|
clock: clock.clone(),
|
||||||
|
});
|
||||||
|
let mut spec = spec("s3.example.com", true);
|
||||||
|
spec.retry = retry;
|
||||||
|
let config = build_remote_s3_config(&spec)
|
||||||
|
.await
|
||||||
|
.expect("clock skew fixture uses the production outbound configuration")
|
||||||
|
.http_client(http_client_fn(move |_settings, _components| connector.clone()))
|
||||||
|
.time_source(clock.clone())
|
||||||
|
.build();
|
||||||
|
(S3Client::from_conf(config), headers, clock)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
async fn remote_s3_clock_skew_retries_resign_and_seed_next_operation() {
|
||||||
|
for error_code in ["RequestTimeTooSkewed", "SignatureDoesNotMatch"] {
|
||||||
|
for skew_seconds in [-600, 600] {
|
||||||
|
let (client, headers, clock) = clock_skew_client(error_code, skew_seconds, REPLICATION_TARGET_RETRY_POLICY).await;
|
||||||
|
let initial = chrono::DateTime::<chrono::Utc>::from(clock.now()).naive_utc();
|
||||||
|
client
|
||||||
|
.get_object()
|
||||||
|
.bucket("bucket")
|
||||||
|
.key("object")
|
||||||
|
.send()
|
||||||
|
.await
|
||||||
|
.expect("clock skew GET must retry successfully");
|
||||||
|
assert_eq!(
|
||||||
|
headers.lock().expect("captured requests").len(),
|
||||||
|
2,
|
||||||
|
"{error_code}: GET needs exactly one retry"
|
||||||
|
);
|
||||||
|
clock.0.fetch_add(17, Ordering::SeqCst);
|
||||||
|
// SDK signing time is independent of Tokio's retry/scheduler clock.
|
||||||
|
tokio::time::advance(Duration::from_secs(61)).await;
|
||||||
|
client
|
||||||
|
.head_bucket()
|
||||||
|
.bucket("bucket")
|
||||||
|
.send()
|
||||||
|
.await
|
||||||
|
.expect("subsequent HEAD must use the client's cached skew");
|
||||||
|
let headers = headers.lock().expect("captured signed requests");
|
||||||
|
assert_eq!(headers.len(), 3, "subsequent operation must succeed on its first attempt");
|
||||||
|
assert_eq!(signing_time(&headers[0]), initial, "the first attempt must use the injected clock");
|
||||||
|
assert_eq!(
|
||||||
|
signing_time(&headers[1]),
|
||||||
|
initial + chrono::Duration::seconds(skew_seconds),
|
||||||
|
"{error_code}: retry must apply the measured offset exactly"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
signing_time(&headers[2]),
|
||||||
|
initial + chrono::Duration::seconds(skew_seconds + 17),
|
||||||
|
"{error_code}: the next operation must apply cached skew to the advanced signing clock"
|
||||||
|
);
|
||||||
|
let signature = |index: usize| {
|
||||||
|
recorded_header(&headers[index], "authorization")
|
||||||
|
.rsplit_once("Signature=")
|
||||||
|
.expect("SigV4 authorization contains a signature")
|
||||||
|
.1
|
||||||
|
};
|
||||||
|
assert_ne!(
|
||||||
|
signature(0),
|
||||||
|
signature(1),
|
||||||
|
"{error_code}: retry must be signed again after adjusting its date"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test(start_paused = true)]
|
||||||
|
async fn remote_s3_clock_skew_respects_one_attempt_policy() {
|
||||||
|
use aws_smithy_types::error::metadata::ProvideErrorMetadata;
|
||||||
|
|
||||||
|
for error_code in ["RequestTimeTooSkewed", "SignatureDoesNotMatch"] {
|
||||||
|
for retry in [
|
||||||
|
RemoteS3RetryPolicy::Disabled,
|
||||||
|
RemoteS3RetryPolicy::Standard { max_attempts: 1 },
|
||||||
|
] {
|
||||||
|
let (client, headers, _clock) = clock_skew_client(error_code, 600, retry).await;
|
||||||
|
let error = client
|
||||||
|
.get_object()
|
||||||
|
.bucket("bucket")
|
||||||
|
.key("object")
|
||||||
|
.send()
|
||||||
|
.await
|
||||||
|
.expect_err("clock skew must not override the caller's one-attempt budget");
|
||||||
|
assert_eq!(error.as_service_error().and_then(ProvideErrorMetadata::code), Some(error_code));
|
||||||
|
assert_eq!(
|
||||||
|
headers.lock().expect("captured requests").len(),
|
||||||
|
1,
|
||||||
|
"{error_code}: {retry:?} must send exactly one request"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn path_style_auto_and_path_force_path_style() {
|
fn path_style_auto_and_path_force_path_style() {
|
||||||
assert!(PathStyle::Auto.force_path_style());
|
assert!(PathStyle::Auto.force_path_style());
|
||||||
|
|||||||
@@ -46,7 +46,7 @@ use super::replication_storage_boundary::{
|
|||||||
HTTPPreconditions, ObjectInfo, ObjectOptions, ObjectToDelete, ReplicationDeletedObject, ReplicationObjectIO,
|
HTTPPreconditions, ObjectInfo, ObjectOptions, ObjectToDelete, ReplicationDeletedObject, ReplicationObjectIO,
|
||||||
ReplicationStorage,
|
ReplicationStorage,
|
||||||
};
|
};
|
||||||
use super::replication_target_boundary::{ReplicationTargetStore, replication_object_is_ssec_encrypted};
|
use super::replication_target_boundary::{BucketTargetError, ReplicationTargetStore, replication_object_is_ssec_encrypted};
|
||||||
use super::replication_versioning_boundary::ReplicationVersioningStore;
|
use super::replication_versioning_boundary::ReplicationVersioningStore;
|
||||||
use super::runtime_boundary as runtime_sources;
|
use super::runtime_boundary as runtime_sources;
|
||||||
use futures_util::stream::{self, StreamExt};
|
use futures_util::stream::{self, StreamExt};
|
||||||
@@ -3084,6 +3084,23 @@ pub async fn queue_replication_heal(bucket: &str, oi: ObjectInfo, retry_count: u
|
|||||||
|
|
||||||
let tgts = match ReplicationTargetStore::list_bucket_targets(bucket).await {
|
let tgts = match ReplicationTargetStore::list_bucket_targets(bucket).await {
|
||||||
Ok(targets) => Some(targets),
|
Ok(targets) => Some(targets),
|
||||||
|
// A bucket whose persisted target configuration cannot be decoded has
|
||||||
|
// an unknown target set, not an empty one: scheduling against `None`
|
||||||
|
// here would drop every heal for it without a trace
|
||||||
|
// (rustfs/backlog#2282). Report it missed so the object is retried
|
||||||
|
// once the configuration is readable again.
|
||||||
|
Err(BucketTargetError::BucketRemoteTargetsUnreadable { .. }) => {
|
||||||
|
warn!(
|
||||||
|
event = EVENT_REPLICATION_CONFIG_LOOKUP_SKIPPED,
|
||||||
|
component = LOG_COMPONENT_ECSTORE,
|
||||||
|
subsystem = LOG_SUBSYSTEM_REPLICATION,
|
||||||
|
bucket,
|
||||||
|
reason = "target_config_unreadable",
|
||||||
|
"Bucket replication targets are unreadable; replication heal queue fails closed"
|
||||||
|
);
|
||||||
|
|
||||||
|
return ReplicationQueueAdmission::Missed;
|
||||||
|
}
|
||||||
Err(err) => {
|
Err(err) => {
|
||||||
debug!(
|
debug!(
|
||||||
event = EVENT_REPLICATION_CONFIG_LOOKUP_SKIPPED,
|
event = EVENT_REPLICATION_CONFIG_LOOKUP_SKIPPED,
|
||||||
|
|||||||
@@ -15,7 +15,8 @@
|
|||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
|
|
||||||
use crate::bucket::bucket_target_sys::{BucketTargetError, BucketTargetSys};
|
pub(crate) use crate::bucket::bucket_target_sys::BucketTargetError;
|
||||||
|
use crate::bucket::bucket_target_sys::BucketTargetSys;
|
||||||
use aws_sdk_s3::operation::head_object::HeadObjectOutput;
|
use aws_sdk_s3::operation::head_object::HeadObjectOutput;
|
||||||
use aws_sdk_s3::types::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode};
|
use aws_sdk_s3::types::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode};
|
||||||
use http::HeaderMap;
|
use http::HeaderMap;
|
||||||
|
|||||||
@@ -28,7 +28,7 @@
|
|||||||
|
|
||||||
use async_trait::async_trait;
|
use async_trait::async_trait;
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
use std::collections::HashMap;
|
use std::collections::BTreeMap;
|
||||||
use std::fmt;
|
use std::fmt;
|
||||||
use std::sync::{Arc, OnceLock};
|
use std::sync::{Arc, OnceLock};
|
||||||
|
|
||||||
@@ -81,8 +81,8 @@ impl SealScope {
|
|||||||
/// The encryption context handed to the sealer. Keys are stable: they are
|
/// The encryption context handed to the sealer. Keys are stable: they are
|
||||||
/// part of the on-disk contract, because a ciphertext only decrypts under
|
/// part of the on-disk contract, because a ciphertext only decrypts under
|
||||||
/// the same context.
|
/// the same context.
|
||||||
pub fn encryption_context(&self) -> HashMap<String, String> {
|
pub fn encryption_context(&self) -> BTreeMap<String, String> {
|
||||||
HashMap::from([
|
BTreeMap::from([
|
||||||
("rustfs:store".to_string(), self.store.as_str().to_string()),
|
("rustfs:store".to_string(), self.store.as_str().to_string()),
|
||||||
("rustfs:owner".to_string(), self.owner.clone()),
|
("rustfs:owner".to_string(), self.owner.clone()),
|
||||||
("rustfs:field".to_string(), self.field.to_string()),
|
("rustfs:field".to_string(), self.field.to_string()),
|
||||||
@@ -201,12 +201,18 @@ pub async fn unseal_secret(sealed: &SealedCredential, scope: &SealScope) -> Resu
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
use parking_lot::Mutex;
|
use parking_lot::Mutex;
|
||||||
|
use std::collections::BTreeMap;
|
||||||
|
|
||||||
|
fn encode_context(context: &BTreeMap<String, String>) -> String {
|
||||||
|
let ordered = context.iter().collect::<BTreeMap<_, _>>();
|
||||||
|
serde_json::to_string(&ordered).expect("context serializes")
|
||||||
|
}
|
||||||
|
|
||||||
/// Stands in for the KMS-backed sealer: records the context it was called
|
/// Stands in for the KMS-backed sealer: records the context it was called
|
||||||
/// with, and refuses a ciphertext presented under a different one.
|
/// with, and refuses a ciphertext presented under a different one.
|
||||||
#[derive(Default)]
|
#[derive(Default)]
|
||||||
struct FakeSealer {
|
struct FakeSealer {
|
||||||
sealed_contexts: Mutex<Vec<HashMap<String, String>>>,
|
sealed_contexts: Mutex<Vec<BTreeMap<String, String>>>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[async_trait]
|
#[async_trait]
|
||||||
@@ -214,7 +220,7 @@ mod tests {
|
|||||||
async fn seal(&self, plaintext: &str, scope: &SealScope) -> Result<SealedCredential, SealedCredentialError> {
|
async fn seal(&self, plaintext: &str, scope: &SealScope) -> Result<SealedCredential, SealedCredentialError> {
|
||||||
let context = scope.encryption_context();
|
let context = scope.encryption_context();
|
||||||
self.sealed_contexts.lock().push(context.clone());
|
self.sealed_contexts.lock().push(context.clone());
|
||||||
let mut bound = serde_json::to_string(&context).expect("context serializes");
|
let mut bound = encode_context(&context);
|
||||||
bound.push('|');
|
bound.push('|');
|
||||||
bound.push_str(plaintext);
|
bound.push_str(plaintext);
|
||||||
Ok(SealedCredential {
|
Ok(SealedCredential {
|
||||||
@@ -231,7 +237,7 @@ mod tests {
|
|||||||
.decode_to_vec(sealed.ct.as_bytes())
|
.decode_to_vec(sealed.ct.as_bytes())
|
||||||
.map_err(|err| SealedCredentialError::Malformed(err.to_string()))?;
|
.map_err(|err| SealedCredentialError::Malformed(err.to_string()))?;
|
||||||
let bound = String::from_utf8(raw).map_err(|err| SealedCredentialError::Malformed(err.to_string()))?;
|
let bound = String::from_utf8(raw).map_err(|err| SealedCredentialError::Malformed(err.to_string()))?;
|
||||||
let expected = serde_json::to_string(&scope.encryption_context()).expect("context serializes");
|
let expected = encode_context(&scope.encryption_context());
|
||||||
bound
|
bound
|
||||||
.strip_prefix(&expected)
|
.strip_prefix(&expected)
|
||||||
.and_then(|rest| rest.strip_prefix('|'))
|
.and_then(|rest| rest.strip_prefix('|'))
|
||||||
|
|||||||
@@ -30,6 +30,7 @@ use rustfs_protos::{
|
|||||||
ChannelClass, create_new_channel, get_channel_for_class,
|
ChannelClass, create_new_channel, get_channel_for_class,
|
||||||
proto_gen::node_service::{
|
proto_gen::node_service::{
|
||||||
heal_control_service_client::HealControlServiceClient, node_service_client::NodeServiceClient,
|
heal_control_service_client::HealControlServiceClient, node_service_client::NodeServiceClient,
|
||||||
|
scanner_control_service_client::ScannerControlServiceClient,
|
||||||
tier_mutation_control_service_client::TierMutationControlServiceClient,
|
tier_mutation_control_service_client::TierMutationControlServiceClient,
|
||||||
},
|
},
|
||||||
};
|
};
|
||||||
@@ -60,6 +61,24 @@ pub async fn node_service_time_out_client(
|
|||||||
node_service_time_out_client_for_class(addr, interceptor, ChannelClass::Control).await
|
node_service_time_out_client_for_class(addr, interceptor, ChannelClass::Control).await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn scanner_control_time_out_client(
|
||||||
|
addr: &str,
|
||||||
|
interceptor: TonicInterceptor,
|
||||||
|
) -> crate::error::Result<ScannerControlServiceClient<InterceptedService<AuthenticatedChannel, TonicInterceptor>>> {
|
||||||
|
let interceptor = interceptor.with_rpc_audience(addr)?;
|
||||||
|
let channel = match runtime_sources::cached_node_channel(addr).await {
|
||||||
|
Some(channel) => channel,
|
||||||
|
None => create_new_channel(addr)
|
||||||
|
.await
|
||||||
|
.map_err(|err| crate::error::Error::other(err.to_string()))?,
|
||||||
|
};
|
||||||
|
let channel = ReplayScopeChannel::new(channel, interceptor.replay_scope_audience());
|
||||||
|
let limit = rustfs_protos::scoped_dirty_usage::SCOPED_DIRTY_USAGE_MAX_REQUEST_BYTES as usize;
|
||||||
|
Ok(ScannerControlServiceClient::with_interceptor(channel, interceptor)
|
||||||
|
.max_decoding_message_size(limit)
|
||||||
|
.max_encoding_message_size(limit))
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn heal_control_time_out_client(
|
pub async fn heal_control_time_out_client(
|
||||||
addr: &str,
|
addr: &str,
|
||||||
interceptor: TonicInterceptor,
|
interceptor: TonicInterceptor,
|
||||||
|
|||||||
@@ -2050,6 +2050,53 @@ impl PeerRestClient {
|
|||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Probe only: scoped ACK production requires a durable per-bucket proof.
|
||||||
|
pub async fn scanner_scoped_dirty_usage_capability(
|
||||||
|
&self,
|
||||||
|
owner_id: String,
|
||||||
|
instance_id: String,
|
||||||
|
entries: Vec<rustfs_protos::proto_gen::node_service::ScannerScopedDirtyUsageEntry>,
|
||||||
|
) -> Result<bool> {
|
||||||
|
use rustfs_protos::scoped_dirty_usage::*;
|
||||||
|
let payload = rustfs_protos::proto_gen::node_service::ScannerScopedDirtyUsageAckRequest {
|
||||||
|
challenge: Uuid::new_v4().as_bytes().to_vec().into(),
|
||||||
|
protocol_version: SCOPED_DIRTY_USAGE_PROTOCOL_VERSION,
|
||||||
|
owner_id,
|
||||||
|
instance_id,
|
||||||
|
scope: SCOPED_DIRTY_USAGE_BUCKET_SCOPE,
|
||||||
|
probe_only: true,
|
||||||
|
entries,
|
||||||
|
};
|
||||||
|
let canonical = canonical_scoped_dirty_usage_request(&payload).map_err(|err| Error::other(err.to_string()))?;
|
||||||
|
self.finalize_result(
|
||||||
|
async {
|
||||||
|
let mut client = super::client::scanner_control_time_out_client(
|
||||||
|
&self.grid_host,
|
||||||
|
TonicInterceptor::Signature(gen_tonic_signature_interceptor()),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
let mut request = Request::new(payload.clone());
|
||||||
|
set_tonic_canonical_body_digest(&mut request, &canonical)?;
|
||||||
|
let response = client.scanner_scoped_dirty_usage_ack(request).await?.into_inner();
|
||||||
|
let body = canonical_scoped_dirty_usage_response(&canonical, &response)
|
||||||
|
.map_err(|_| Error::other("scoped dirty usage capability response is too large"))?;
|
||||||
|
verify_tonic_rpc_response_proof(&body, response.response_proof.as_ref())?;
|
||||||
|
if response.protocol_version != SCOPED_DIRTY_USAGE_PROTOCOL_VERSION
|
||||||
|
|| response.owner_id != payload.owner_id
|
||||||
|
|| response.instance_id != payload.instance_id
|
||||||
|
|| response.max_entries != SCOPED_DIRTY_USAGE_MAX_ENTRIES
|
||||||
|
|| response.max_request_bytes != SCOPED_DIRTY_USAGE_MAX_REQUEST_BYTES
|
||||||
|
|| response.cleared != 0
|
||||||
|
{
|
||||||
|
return Err(Error::other("scoped dirty usage capability response does not match request"));
|
||||||
|
}
|
||||||
|
Ok(response.supported)
|
||||||
|
}
|
||||||
|
.await,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn acknowledge_scanner_dirty_usage(&self, instance_id: String, generation: u64) -> Result<ScannerPeerActivity> {
|
pub async fn acknowledge_scanner_dirty_usage(&self, instance_id: String, generation: u64) -> Result<ScannerPeerActivity> {
|
||||||
let result = self
|
let result = self
|
||||||
.scanner_activity_request_with_protocol(instance_id.clone(), generation, SCANNER_ACTIVITY_PROTOCOL_VERSION)
|
.scanner_activity_request_with_protocol(instance_id.clone(), generation, SCANNER_ACTIVITY_PROTOCOL_VERSION)
|
||||||
|
|||||||
@@ -5493,6 +5493,7 @@ where
|
|||||||
fence.ensure_held()?;
|
fence.ensure_held()?;
|
||||||
let mut opts = ObjectOptions {
|
let mut opts = ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
no_lock: true,
|
no_lock: true,
|
||||||
http_preconditions: Some(pool_meta_cas_preconditions(token, object)?),
|
http_preconditions: Some(pool_meta_cas_preconditions(token, object)?),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -14412,6 +14413,7 @@ impl ECStore {
|
|||||||
encoded.clone(),
|
encoded.clone(),
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -14566,6 +14568,7 @@ impl ECStore {
|
|||||||
encoded,
|
encoded,
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(http_preconditions),
|
http_preconditions: Some(http_preconditions),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
},
|
},
|
||||||
@@ -14957,6 +14960,7 @@ impl ECStore {
|
|||||||
encoded,
|
encoded,
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(etag),
|
if_match: Some(etag),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
|
|||||||
@@ -317,6 +317,22 @@ impl DiskStoreRenameDataExt for LocalDiskWrapper {
|
|||||||
dst_path: &str,
|
dst_path: &str,
|
||||||
external_guard: Option<Arc<dyn Send + Sync>>,
|
external_guard: Option<Arc<dyn Send + Sync>>,
|
||||||
) -> Result<RenameDataResp> {
|
) -> Result<RenameDataResp> {
|
||||||
|
self.rename_data_observed(src_volume, src_path, fi, dst_volume, dst_path, external_guard)
|
||||||
|
.await
|
||||||
|
.result
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl LocalDiskWrapper {
|
||||||
|
pub(in crate::disk) async fn rename_data_observed(
|
||||||
|
&self,
|
||||||
|
src_volume: &str,
|
||||||
|
src_path: &str,
|
||||||
|
fi: &FileInfo,
|
||||||
|
dst_volume: &str,
|
||||||
|
dst_path: &str,
|
||||||
|
external_guard: Option<Arc<dyn Send + Sync>>,
|
||||||
|
) -> super::RenameDataObservation {
|
||||||
let operation = self.clone();
|
let operation = self.clone();
|
||||||
let src_volume = src_volume.to_owned();
|
let src_volume = src_volume.to_owned();
|
||||||
let src_path = src_path.to_owned();
|
let src_path = src_path.to_owned();
|
||||||
@@ -333,22 +349,35 @@ impl DiskStoreRenameDataExt for LocalDiskWrapper {
|
|||||||
} else {
|
} else {
|
||||||
get_max_timeout_duration()
|
get_max_timeout_duration()
|
||||||
};
|
};
|
||||||
run_owned_mutation(external_guard, move || async move {
|
let observed = run_owned_mutation(external_guard, move || async move {
|
||||||
operation
|
let mut preflight_rejection = None;
|
||||||
|
let result = operation
|
||||||
.track_disk_health_mutation(
|
.track_disk_health_mutation(
|
||||||
"rename_data",
|
"rename_data",
|
||||||
DiskMetricMutation::Write,
|
DiskMetricMutation::Write,
|
||||||
|| async {
|
|| async {
|
||||||
operation
|
// Preserve the former DiskAPI future's single boxing boundary.
|
||||||
.disk
|
let observed =
|
||||||
.rename_data_borrowed(&src_volume, &src_path, &fi, &dst_volume, &dst_path)
|
Box::pin(
|
||||||
.await
|
operation
|
||||||
|
.disk
|
||||||
|
.rename_data_observed(&src_volume, &src_path, &fi, &dst_volume, &dst_path),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
preflight_rejection = observed.preflight_rejection;
|
||||||
|
observed.result
|
||||||
},
|
},
|
||||||
timeout_duration,
|
timeout_duration,
|
||||||
)
|
)
|
||||||
.await
|
.await;
|
||||||
|
// Health tracking must observe the real disk error, not an Ok tuple.
|
||||||
|
Ok(super::RenameDataObservation {
|
||||||
|
result,
|
||||||
|
preflight_rejection,
|
||||||
|
})
|
||||||
})
|
})
|
||||||
.await
|
.await;
|
||||||
|
observed.unwrap_or_else(|error| super::RenameDataObservation::unknown(Err(error)))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2588,6 +2617,46 @@ mod tests {
|
|||||||
assert_eq!(wrapper.metrics_snapshot().api_calls.get("unknown"), Some(&1));
|
assert_eq!(wrapper.metrics_snapshot().api_calls.get("unknown"), Some(&1));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn rename_preflight_evidence_preserves_health_errors_and_owned_reply() {
|
||||||
|
for source_exists in [false, true] {
|
||||||
|
for guarded in [false, true] {
|
||||||
|
let dir = tempfile::tempdir().expect("temp dir should be created");
|
||||||
|
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be valid UTF-8"))
|
||||||
|
.expect("endpoint should parse");
|
||||||
|
let disk = Arc::new(LocalDisk::new(&endpoint, false).await.expect("local disk should be created"));
|
||||||
|
if source_exists {
|
||||||
|
disk.make_volume("source").await.expect("source volume should exist");
|
||||||
|
}
|
||||||
|
let wrapper = LocalDiskWrapper::new(disk, false);
|
||||||
|
let drops = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||||
|
let external_guard = guarded.then(|| Arc::new(DropProbe(Arc::clone(&drops))) as Arc<dyn Send + Sync>);
|
||||||
|
let mut file_info = FileInfo::new("object", 1, 0);
|
||||||
|
file_info.mod_time = Some(::time::OffsetDateTime::now_utc());
|
||||||
|
file_info.erasure.index = 1;
|
||||||
|
let observed = wrapper
|
||||||
|
.rename_data_observed("source", "object", &file_info, "missing-destination", "object", external_guard)
|
||||||
|
.await;
|
||||||
|
assert!(observed.rejected_before_publication(), "normal access rejection must carry proof");
|
||||||
|
assert!(matches!(observed.result, Err(DiskError::VolumeNotFound)));
|
||||||
|
let snapshot = wrapper.metrics_snapshot();
|
||||||
|
assert_eq!(snapshot.api_calls.get("rename_data"), Some(&1));
|
||||||
|
assert_eq!(snapshot.total_writes, 0, "health tracking must not observe the rejection as Ok");
|
||||||
|
assert_eq!(drops.load(Ordering::SeqCst), usize::from(guarded));
|
||||||
|
|
||||||
|
wrapper.health.force_runtime_state_for_test(RuntimeDriveHealthState::Offline);
|
||||||
|
let observed = wrapper
|
||||||
|
.rename_data_observed("source", "object", &file_info, "missing-destination", "object", None)
|
||||||
|
.await;
|
||||||
|
assert!(!observed.rejected_before_publication(), "wrapper errors carry no local preflight proof");
|
||||||
|
assert!(matches!(observed.result, Err(DiskError::FaultyDisk)));
|
||||||
|
let snapshot = wrapper.metrics_snapshot();
|
||||||
|
assert_eq!(snapshot.total_errors_availability, 1);
|
||||||
|
assert_eq!(snapshot.total_writes, 0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn local_disk_health_wrapper_counts_returned_availability_errors() {
|
async fn local_disk_health_wrapper_counts_returned_availability_errors() {
|
||||||
let dir = tempfile::tempdir().expect("temp dir should be created");
|
let dir = tempfile::tempdir().expect("temp dir should be created");
|
||||||
|
|||||||
+202
-1143
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -75,6 +75,25 @@ use time::OffsetDateTime;
|
|||||||
use tokio::io::{AsyncRead, AsyncWrite};
|
use tokio::io::{AsyncRead, AsyncWrite};
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// Local preflight evidence stays outside DiskAPI and the RPC response format.
|
||||||
|
pub(crate) struct RenameDataObservation {
|
||||||
|
pub(crate) result: Result<RenameDataResp>,
|
||||||
|
preflight_rejection: Option<local::LocalRenamePreflightRejection>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RenameDataObservation {
|
||||||
|
fn unknown(result: Result<RenameDataResp>) -> Self {
|
||||||
|
Self {
|
||||||
|
result,
|
||||||
|
preflight_rejection: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn rejected_before_publication(&self) -> bool {
|
||||||
|
self.result.is_err() && self.preflight_rejection.is_some()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
const QUOTA_MUTATION_FENCE_PREFIX: &str = "tmp/quota-mutation-fences/";
|
const QUOTA_MUTATION_FENCE_PREFIX: &str = "tmp/quota-mutation-fences/";
|
||||||
pub(crate) const QUOTA_MUTATION_FENCE_METADATA_SUFFIX: &str = "quota-mutation-fence-token";
|
pub(crate) const QUOTA_MUTATION_FENCE_METADATA_SUFFIX: &str = "quota-mutation-fence-token";
|
||||||
|
|
||||||
@@ -711,6 +730,36 @@ impl Disk {
|
|||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn rename_data_borrowed_with_fence_observed(
|
||||||
|
&self,
|
||||||
|
src_volume: &str,
|
||||||
|
src_path: &str,
|
||||||
|
fi: &FileInfo,
|
||||||
|
dst_volume: &str,
|
||||||
|
dst_path: &str,
|
||||||
|
scanner_publication_lease_token: Option<Uuid>,
|
||||||
|
) -> RenameDataObservation {
|
||||||
|
match self {
|
||||||
|
Disk::Local(local_disk) => {
|
||||||
|
local_disk
|
||||||
|
.rename_data_observed(src_volume, src_path, fi, dst_volume, dst_path, None)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
Disk::Remote(remote_disk) => RenameDataObservation::unknown(
|
||||||
|
remote_disk
|
||||||
|
.rename_data_borrowed_with_fence(
|
||||||
|
src_volume,
|
||||||
|
src_path,
|
||||||
|
fi,
|
||||||
|
dst_volume,
|
||||||
|
dst_path,
|
||||||
|
scanner_publication_lease_token,
|
||||||
|
)
|
||||||
|
.await,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn rename_data_borrowed_with_fence(
|
pub(crate) async fn rename_data_borrowed_with_fence(
|
||||||
&self,
|
&self,
|
||||||
src_volume: &str,
|
src_volume: &str,
|
||||||
|
|||||||
@@ -870,6 +870,18 @@ impl TierFreeVersionReceiptSink {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Internal PUT completion boundary; this does not change fsync or write quorum.
|
||||||
|
#[doc(hidden)]
|
||||||
|
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum WriteCompletion {
|
||||||
|
/// Return at write quorum when the commit owner can retain its guards.
|
||||||
|
#[default]
|
||||||
|
Quorum,
|
||||||
|
/// Drain the rename fan-out before returning. Minority failures still heal
|
||||||
|
/// after a successful quorum commit; this does not require every disk to succeed.
|
||||||
|
TailDrained,
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Default, Clone)]
|
#[derive(Default, Clone)]
|
||||||
pub struct ObjectOptions {
|
pub struct ObjectOptions {
|
||||||
// Use the maximum parity (N/2), used when saving server configuration files
|
// Use the maximum parity (N/2), used when saving server configuration files
|
||||||
@@ -896,6 +908,10 @@ pub struct ObjectOptions {
|
|||||||
/// Persisted bucket incarnation observed before authorization.
|
/// Persisted bucket incarnation observed before authorization.
|
||||||
pub expected_bucket_incarnation_id: Option<Uuid>,
|
pub expected_bucket_incarnation_id: Option<Uuid>,
|
||||||
pub no_lock: bool,
|
pub no_lock: bool,
|
||||||
|
/// Control-plane writers that immediately read or CAS the same namespace
|
||||||
|
/// key use TailDrained without changing namespace lock ownership.
|
||||||
|
#[doc(hidden)]
|
||||||
|
pub write_completion: WriteCompletion,
|
||||||
/// True when an upper layer already holds the object read lock before
|
/// True when an upper layer already holds the object read lock before
|
||||||
/// forwarding a no_lock read to the set layer.
|
/// forwarding a no_lock read to the set layer.
|
||||||
pub metadata_cache_safe: bool,
|
pub metadata_cache_safe: bool,
|
||||||
|
|||||||
@@ -62,12 +62,27 @@ const REMOTE_VERSION_STATE_PROOF_TTL: Duration = Duration::from_secs(30);
|
|||||||
const CROSS_POOL_FENCE_SUPPORTED_VERSION: u32 = 2;
|
const CROSS_POOL_FENCE_SUPPORTED_VERSION: u32 = 2;
|
||||||
const TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION: u32 = 3;
|
const TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION: u32 = 3;
|
||||||
const DECOMMISSION_TARGET_FENCE_POLICY_SUPPORTED_VERSION: u32 = 4;
|
const DECOMMISSION_TARGET_FENCE_POLICY_SUPPORTED_VERSION: u32 = 4;
|
||||||
|
// Keep this synchronized with the version served by node_service. Including
|
||||||
|
// the local member in the minimum prevents an older coordinator from
|
||||||
|
// self-authorizing a policy implemented only by newer remote peers.
|
||||||
|
const LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION: u32 = 4;
|
||||||
|
/// Version 5 is reserved for a fleet whose every metadata writer preserves
|
||||||
|
/// explicit transition version state and destination identity, and implements
|
||||||
|
/// conditional per-generation `xl.meta` writes with strong readback. The node
|
||||||
|
/// service must not advertise this version until the conditional writer from
|
||||||
|
/// rustfs/backlog#684 is available.
|
||||||
|
const LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION: u32 = 5;
|
||||||
type CrossPoolFencePolicyResult = Result<BTreeMap<String, Uuid>>;
|
type CrossPoolFencePolicyResult = Result<BTreeMap<String, Uuid>>;
|
||||||
|
|
||||||
fn cross_pool_fence_policy_results(
|
fn cross_pool_fence_policy_results(
|
||||||
peer_epochs: BTreeMap<String, Uuid>,
|
peer_epochs: BTreeMap<String, Uuid>,
|
||||||
minimum_version: u32,
|
minimum_version: u32,
|
||||||
) -> (CrossPoolFencePolicyResult, CrossPoolFencePolicyResult, CrossPoolFencePolicyResult) {
|
) -> (
|
||||||
|
CrossPoolFencePolicyResult,
|
||||||
|
CrossPoolFencePolicyResult,
|
||||||
|
CrossPoolFencePolicyResult,
|
||||||
|
CrossPoolFencePolicyResult,
|
||||||
|
) {
|
||||||
let journal_result = if minimum_version >= TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION {
|
let journal_result = if minimum_version >= TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION {
|
||||||
Ok(peer_epochs.clone())
|
Ok(peer_epochs.clone())
|
||||||
} else {
|
} else {
|
||||||
@@ -78,7 +93,18 @@ fn cross_pool_fence_policy_results(
|
|||||||
} else {
|
} else {
|
||||||
Err(Error::other("decommission target fence policy capability version is unsupported"))
|
Err(Error::other("decommission target fence policy capability version is unsupported"))
|
||||||
};
|
};
|
||||||
(Ok(peer_epochs), journal_result, decommission_target_fence_result)
|
let legacy_transition_state_reconcile_result =
|
||||||
|
if minimum_version >= LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION {
|
||||||
|
Ok(peer_epochs.clone())
|
||||||
|
} else {
|
||||||
|
Err(Error::other("legacy transition state reconcile policy capability version is unsupported"))
|
||||||
|
};
|
||||||
|
(
|
||||||
|
Ok(peer_epochs),
|
||||||
|
journal_result,
|
||||||
|
decommission_target_fence_result,
|
||||||
|
legacy_transition_state_reconcile_result,
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Debug)]
|
#[derive(Clone, Debug)]
|
||||||
@@ -252,10 +278,21 @@ pub(crate) struct TierDeleteJournalFleetProofToken {
|
|||||||
_permit: FleetCapabilityProofPermit,
|
_permit: FleetCapabilityProofPermit,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Effect-window authority for one legacy transition-state reconciliation.
|
||||||
|
///
|
||||||
|
/// The token intentionally cannot be cloned. Its permit keeps the admitted
|
||||||
|
/// fleet generation alive until the caller finishes the final strong
|
||||||
|
/// readback, while revocation makes every later validation fail immediately.
|
||||||
|
pub struct LegacyTransitionStateReconcileFleetProofToken {
|
||||||
|
token: FleetCapabilityProofToken,
|
||||||
|
_permit: FleetCapabilityProofPermit,
|
||||||
|
}
|
||||||
|
|
||||||
static REMOTE_VERSION_STATE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
static REMOTE_VERSION_STATE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||||
static CROSS_POOL_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
static CROSS_POOL_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||||
static TIER_DELETE_JOURNAL_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
static TIER_DELETE_JOURNAL_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||||
static DECOMMISSION_TARGET_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
static DECOMMISSION_TARGET_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||||
|
static LEGACY_TRANSITION_STATE_RECONCILE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||||
static REMOTE_VERSION_STATE_PROBE_TOPOLOGY: OnceLock<String> = OnceLock::new();
|
static REMOTE_VERSION_STATE_PROBE_TOPOLOGY: OnceLock<String> = OnceLock::new();
|
||||||
|
|
||||||
fn cross_pool_fence_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
|
fn cross_pool_fence_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
|
||||||
@@ -274,6 +311,10 @@ fn decommission_target_fence_fleet_proof_slot() -> &'static std::sync::RwLock<Fl
|
|||||||
DECOMMISSION_TARGET_FENCE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
|
DECOMMISSION_TARGET_FENCE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn legacy_transition_state_reconcile_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
|
||||||
|
LEGACY_TRANSITION_STATE_RECONCILE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
|
||||||
|
}
|
||||||
|
|
||||||
fn revoke_fleet_capability_proof_state(state: &mut FleetCapabilityProofState) {
|
fn revoke_fleet_capability_proof_state(state: &mut FleetCapabilityProofState) {
|
||||||
if let Some(proof) = state.proof.take() {
|
if let Some(proof) = state.proof.take() {
|
||||||
proof.generation.revoke();
|
proof.generation.revoke();
|
||||||
@@ -444,6 +485,125 @@ pub(crate) fn tier_delete_journal_topology_generation(proof: &TierDeleteJournalF
|
|||||||
stable_tier_delete_journal_topology_generation(&proof.token.topology_fingerprint)
|
stable_tier_delete_journal_topology_generation(&proof.token.topology_fingerprint)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Acquire one non-cloneable authority that must span the complete reconcile
|
||||||
|
/// effect window, including its final strong readback.
|
||||||
|
pub async fn acquire_legacy_transition_state_reconcile_fleet_proof() -> Option<LegacyTransitionStateReconcileFleetProofToken> {
|
||||||
|
let expected_topology = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get()?;
|
||||||
|
let proof = {
|
||||||
|
let state = legacy_transition_state_reconcile_fleet_proof_slot()
|
||||||
|
.read()
|
||||||
|
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, expected_topology, Instant::now())?
|
||||||
|
};
|
||||||
|
let observed_peer_epochs = observe_legacy_transition_state_reconcile_fleet(expected_topology).await?;
|
||||||
|
let state = legacy_transition_state_reconcile_fleet_proof_slot()
|
||||||
|
.read()
|
||||||
|
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
|
||||||
|
&state,
|
||||||
|
&proof,
|
||||||
|
expected_topology,
|
||||||
|
&observed_peer_epochs,
|
||||||
|
Instant::now(),
|
||||||
|
)
|
||||||
|
.then_some(proof)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn acquire_legacy_transition_state_reconcile_fleet_proof_from(
|
||||||
|
state: &FleetCapabilityProofState,
|
||||||
|
expected_topology: &str,
|
||||||
|
now: Instant,
|
||||||
|
) -> Option<LegacyTransitionStateReconcileFleetProofToken> {
|
||||||
|
let token = acquire_fleet_capability_proof_from(state, expected_topology, now)?;
|
||||||
|
let permit = state.proof.as_ref()?.generation.try_acquire()?;
|
||||||
|
Some(LegacyTransitionStateReconcileFleetProofToken { token, _permit: permit })
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn observe_legacy_transition_state_reconcile_fleet(expected_topology: &str) -> Option<BTreeMap<String, Uuid>> {
|
||||||
|
let notification_sys = get_global_notification_sys()?;
|
||||||
|
let (peer_epochs, minimum_version) = timeout(
|
||||||
|
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
|
||||||
|
notification_sys.probe_cross_pool_fence_fleet(expected_topology),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.ok()?
|
||||||
|
.ok()?;
|
||||||
|
let (_, _, _, reconcile_result) = cross_pool_fence_policy_results(peer_epochs, minimum_version);
|
||||||
|
reconcile_result.ok()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Revalidate the exact fleet generation captured by a reconcile token with a
|
||||||
|
/// fresh synchronous observation. Callers must await this before each
|
||||||
|
/// conditional metadata write and after the final strong readback.
|
||||||
|
pub async fn legacy_transition_state_reconcile_fleet_proof_matches(
|
||||||
|
proof: &LegacyTransitionStateReconcileFleetProofToken,
|
||||||
|
) -> bool {
|
||||||
|
let Some(expected_topology) = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get() else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_matches_with_observer(
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_slot(),
|
||||||
|
proof,
|
||||||
|
expected_topology,
|
||||||
|
|| observe_legacy_transition_state_reconcile_fleet(expected_topology),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn legacy_transition_state_reconcile_fleet_proof_matches_with_observer<F, Fut>(
|
||||||
|
slot: &std::sync::RwLock<FleetCapabilityProofState>,
|
||||||
|
proof: &LegacyTransitionStateReconcileFleetProofToken,
|
||||||
|
expected_topology: &str,
|
||||||
|
observe: F,
|
||||||
|
) -> bool
|
||||||
|
where
|
||||||
|
F: FnOnce() -> Fut,
|
||||||
|
Fut: Future<Output = Option<BTreeMap<String, Uuid>>>,
|
||||||
|
{
|
||||||
|
{
|
||||||
|
let state = slot.read().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||||
|
if !legacy_transition_state_reconcile_fleet_proof_matches_at(&state, proof, expected_topology, Instant::now()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let Some(observed_peer_epochs) = observe().await else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
let state = slot.read().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
|
||||||
|
&state,
|
||||||
|
proof,
|
||||||
|
expected_topology,
|
||||||
|
&observed_peer_epochs,
|
||||||
|
Instant::now(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn legacy_transition_state_reconcile_fleet_proof_matches_at(
|
||||||
|
state: &FleetCapabilityProofState,
|
||||||
|
proof: &LegacyTransitionStateReconcileFleetProofToken,
|
||||||
|
expected_topology: &str,
|
||||||
|
now: Instant,
|
||||||
|
) -> bool {
|
||||||
|
proof._permit.generation.is_accepting()
|
||||||
|
&& fleet_capability_proof_matches_at(state, &proof.token, expected_topology, now)
|
||||||
|
&& state
|
||||||
|
.proof
|
||||||
|
.as_ref()
|
||||||
|
.is_some_and(|current| Arc::ptr_eq(¤t.generation, &proof._permit.generation))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
|
||||||
|
state: &FleetCapabilityProofState,
|
||||||
|
proof: &LegacyTransitionStateReconcileFleetProofToken,
|
||||||
|
expected_topology: &str,
|
||||||
|
observed_peer_epochs: &BTreeMap<String, Uuid>,
|
||||||
|
now: Instant,
|
||||||
|
) -> bool {
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_matches_at(state, proof, expected_topology, now)
|
||||||
|
&& proof.token.peer_epochs.as_ref() == observed_peer_epochs
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(all(test, feature = "test-util"))]
|
#[cfg(all(test, feature = "test-util"))]
|
||||||
pub(crate) fn tier_delete_journal_fleet_proof_has_inflight_for_test() -> bool {
|
pub(crate) fn tier_delete_journal_fleet_proof_has_inflight_for_test() -> bool {
|
||||||
let state = tier_delete_journal_fleet_proof_slot()
|
let state = tier_delete_journal_fleet_proof_slot()
|
||||||
@@ -766,6 +926,7 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
|||||||
cross_pool_fence_fleet_proof_slot(),
|
cross_pool_fence_fleet_proof_slot(),
|
||||||
tier_delete_journal_fleet_proof_slot(),
|
tier_delete_journal_fleet_proof_slot(),
|
||||||
decommission_target_fence_fleet_proof_slot(),
|
decommission_target_fence_fleet_proof_slot(),
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_slot(),
|
||||||
] {
|
] {
|
||||||
mark_fleet_capability_topology_conflict(slot);
|
mark_fleet_capability_topology_conflict(slot);
|
||||||
}
|
}
|
||||||
@@ -798,11 +959,12 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
|||||||
.unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))),
|
.unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))),
|
||||||
None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")),
|
None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")),
|
||||||
};
|
};
|
||||||
let (fence_result, journal_result, decommission_target_fence_result) = match fence_probe {
|
let (fence_result, journal_result, decommission_target_fence_result, reconcile_result) = match fence_probe {
|
||||||
Ok((peer_epochs, minimum_version)) => cross_pool_fence_policy_results(peer_epochs, minimum_version),
|
Ok((peer_epochs, minimum_version)) => cross_pool_fence_policy_results(peer_epochs, minimum_version),
|
||||||
Err(err) => {
|
Err(err) => {
|
||||||
let message = err.to_string();
|
let message = err.to_string();
|
||||||
(
|
(
|
||||||
|
Err(Error::other(message.clone())),
|
||||||
Err(Error::other(message.clone())),
|
Err(Error::other(message.clone())),
|
||||||
Err(Error::other(message.clone())),
|
Err(Error::other(message.clone())),
|
||||||
Err(Error::other(message)),
|
Err(Error::other(message)),
|
||||||
@@ -818,6 +980,7 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
|||||||
revoke_fleet_capability_proof(cross_pool_fence_fleet_proof_slot());
|
revoke_fleet_capability_proof(cross_pool_fence_fleet_proof_slot());
|
||||||
revoke_fleet_capability_proof(tier_delete_journal_fleet_proof_slot());
|
revoke_fleet_capability_proof(tier_delete_journal_fleet_proof_slot());
|
||||||
revoke_fleet_capability_proof(decommission_target_fence_fleet_proof_slot());
|
revoke_fleet_capability_proof(decommission_target_fence_fleet_proof_slot());
|
||||||
|
revoke_fleet_capability_proof(legacy_transition_state_reconcile_fleet_proof_slot());
|
||||||
} else if let Some(err) = publish_fleet_capability_probe_result(
|
} else if let Some(err) = publish_fleet_capability_probe_result(
|
||||||
remote_version_state_fleet_proof_slot(),
|
remote_version_state_fleet_proof_slot(),
|
||||||
&topology_fingerprint,
|
&topology_fingerprint,
|
||||||
@@ -880,6 +1043,24 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
|||||||
"notification capability probe"
|
"notification capability probe"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
if !topology_conflict
|
||||||
|
&& let Some(err) = publish_fleet_capability_probe_result(
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_slot(),
|
||||||
|
&topology_fingerprint,
|
||||||
|
reconcile_result,
|
||||||
|
Instant::now(),
|
||||||
|
)
|
||||||
|
{
|
||||||
|
debug!(
|
||||||
|
event = EVENT_NOTIFICATION_CAPABILITY_PROBE,
|
||||||
|
component = LOG_COMPONENT_ECSTORE,
|
||||||
|
subsystem = LOG_SUBSYSTEM_NOTIFICATION,
|
||||||
|
capability = "legacy_transition_state_reconcile_v1",
|
||||||
|
state = "failed_closed",
|
||||||
|
error = %err,
|
||||||
|
"notification capability probe"
|
||||||
|
);
|
||||||
|
}
|
||||||
sleep(REMOTE_VERSION_STATE_PROBE_INTERVAL).await;
|
sleep(REMOTE_VERSION_STATE_PROBE_INTERVAL).await;
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
@@ -959,7 +1140,7 @@ impl NotificationSys {
|
|||||||
client.probe_cross_pool_fence(topology_fingerprint.to_string()).await
|
client.probe_cross_pool_fence(topology_fingerprint.to_string()).await
|
||||||
});
|
});
|
||||||
let mut peer_epochs = BTreeMap::new();
|
let mut peer_epochs = BTreeMap::new();
|
||||||
let mut minimum_version = u32::MAX;
|
let mut minimum_version = LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION;
|
||||||
for result in join_all(probes).await {
|
for result in join_all(probes).await {
|
||||||
let (peer, version, epoch) = result?;
|
let (peer, version, epoch) = result?;
|
||||||
if version < CROSS_POOL_FENCE_SUPPORTED_VERSION {
|
if version < CROSS_POOL_FENCE_SUPPORTED_VERSION {
|
||||||
@@ -968,11 +1149,6 @@ impl NotificationSys {
|
|||||||
minimum_version = minimum_version.min(version);
|
minimum_version = minimum_version.min(version);
|
||||||
insert_remote_version_state_peer(&mut peer_epochs, peer, epoch)?;
|
insert_remote_version_state_peer(&mut peer_epochs, peer, epoch)?;
|
||||||
}
|
}
|
||||||
// A single-node deployment has no remote member to lower the local
|
|
||||||
// policy version advertised by this binary.
|
|
||||||
if minimum_version == u32::MAX {
|
|
||||||
minimum_version = DECOMMISSION_TARGET_FENCE_POLICY_SUPPORTED_VERSION;
|
|
||||||
}
|
|
||||||
Ok((peer_epochs, minimum_version))
|
Ok((peer_epochs, minimum_version))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -3190,20 +3366,36 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn cross_pool_policy_versions_authorize_only_their_supported_protocols() {
|
fn cross_pool_policy_versions_authorize_only_their_supported_protocols() {
|
||||||
let peers = BTreeMap::from([("node-b:9000".to_string(), Uuid::new_v4())]);
|
let peers = BTreeMap::from([("node-b:9000".to_string(), Uuid::new_v4())]);
|
||||||
let (generic_v2, journal_v2, decommission_v2) = cross_pool_fence_policy_results(peers.clone(), 2);
|
let (generic_v2, journal_v2, decommission_v2, reconcile_v2) = cross_pool_fence_policy_results(peers.clone(), 2);
|
||||||
assert!(generic_v2.is_ok(), "v2 remains valid for existing cross-pool fencing");
|
assert!(generic_v2.is_ok(), "v2 remains valid for existing cross-pool fencing");
|
||||||
assert!(journal_v2.is_err(), "a mixed v2/v3 fleet must fail closed for journal-v6 deletion");
|
assert!(journal_v2.is_err(), "a mixed v2/v3 fleet must fail closed for journal-v6 deletion");
|
||||||
assert!(decommission_v2.is_err(), "v2 cannot authorize the sticky per-target decommission fence");
|
assert!(decommission_v2.is_err(), "v2 cannot authorize the sticky per-target decommission fence");
|
||||||
|
assert!(reconcile_v2.is_err(), "v2 cannot authorize legacy transition-state reconciliation");
|
||||||
|
|
||||||
let (generic_v3, journal_v3, decommission_v3) = cross_pool_fence_policy_results(peers.clone(), 3);
|
let (generic_v3, journal_v3, decommission_v3, reconcile_v3) = cross_pool_fence_policy_results(peers.clone(), 3);
|
||||||
assert!(generic_v3.is_ok());
|
assert!(generic_v3.is_ok());
|
||||||
assert!(journal_v3.is_ok(), "an all-v3 fleet may authorize journal-v6 deletion");
|
assert!(journal_v3.is_ok(), "an all-v3 fleet may authorize journal-v6 deletion");
|
||||||
assert!(decommission_v3.is_err(), "v3 members do not understand the per-target decommission fence");
|
assert!(decommission_v3.is_err(), "v3 members do not understand the per-target decommission fence");
|
||||||
|
assert!(reconcile_v3.is_err());
|
||||||
|
|
||||||
let (generic_v4, journal_v4, decommission_v4) = cross_pool_fence_policy_results(peers, 4);
|
let (generic_v4, journal_v4, decommission_v4, reconcile_v4) =
|
||||||
|
cross_pool_fence_policy_results(peers.clone(), LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION);
|
||||||
assert!(generic_v4.is_ok());
|
assert!(generic_v4.is_ok());
|
||||||
assert!(journal_v4.is_ok());
|
assert!(journal_v4.is_ok());
|
||||||
assert!(decommission_v4.is_ok(), "an all-v4 fleet may create sticky per-target reservations");
|
assert!(decommission_v4.is_ok(), "an all-v4 fleet may create sticky per-target reservations");
|
||||||
|
assert!(
|
||||||
|
reconcile_v4.is_err(),
|
||||||
|
"the current local policy lacks the conditional xl.meta writer required by reconcile"
|
||||||
|
);
|
||||||
|
|
||||||
|
let (generic_v5, journal_v5, decommission_v5, reconcile_v5) = cross_pool_fence_policy_results(peers, 5);
|
||||||
|
assert!(generic_v5.is_ok());
|
||||||
|
assert!(journal_v5.is_ok());
|
||||||
|
assert!(decommission_v5.is_ok());
|
||||||
|
assert!(
|
||||||
|
reconcile_v5.is_ok(),
|
||||||
|
"only an all-v5 fleet preserves destination identity and conditional reconcile writes"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -3458,6 +3650,234 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn legacy_transition_state_reconcile_admits_only_compatible_single_and_multi_node_fleets() {
|
||||||
|
let now = Instant::now();
|
||||||
|
for peers in [
|
||||||
|
BTreeMap::new(),
|
||||||
|
BTreeMap::from([("peer-a".to_string(), Uuid::new_v4()), ("peer-b".to_string(), Uuid::new_v4())]),
|
||||||
|
] {
|
||||||
|
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||||
|
let (_, _, _, result) =
|
||||||
|
cross_pool_fence_policy_results(peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
|
||||||
|
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", result, now).is_none());
|
||||||
|
|
||||||
|
let admitted = {
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("an all-compatible fleet should admit reconciliation")
|
||||||
|
};
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
assert!(legacy_transition_state_reconcile_fleet_proof_matches_at(
|
||||||
|
&state,
|
||||||
|
&admitted,
|
||||||
|
"topology-a",
|
||||||
|
now,
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn legacy_transition_state_reconcile_restart_drains_concurrent_effect_windows() {
|
||||||
|
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||||
|
let now = Instant::now();
|
||||||
|
let original_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||||
|
let (_, _, _, original_result) =
|
||||||
|
cross_pool_fence_policy_results(original_peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
|
||||||
|
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", original_result, now).is_none());
|
||||||
|
|
||||||
|
let (first, second) = {
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
(
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("the first reconcile writer should be admitted"),
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("the second reconcile writer should be admitted"),
|
||||||
|
)
|
||||||
|
};
|
||||||
|
|
||||||
|
let restarted_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||||
|
let (_, _, _, restarted_result) =
|
||||||
|
cross_pool_fence_policy_results(restarted_peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
|
||||||
|
let blocked =
|
||||||
|
publish_fleet_capability_probe_result(&slot, "topology-a", restarted_result, now + Duration::from_millis(1))
|
||||||
|
.expect("a restarted member must revoke the old generation and wait for both writers");
|
||||||
|
assert!(blocked.to_string().contains("previous generation to drain"));
|
||||||
|
{
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
assert!(state.proof.is_none());
|
||||||
|
assert!(state.draining_generation.is_some());
|
||||||
|
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
|
||||||
|
&state,
|
||||||
|
&first,
|
||||||
|
"topology-a",
|
||||||
|
now + Duration::from_millis(1),
|
||||||
|
));
|
||||||
|
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
|
||||||
|
&state,
|
||||||
|
&second,
|
||||||
|
"topology-a",
|
||||||
|
now + Duration::from_millis(1),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
drop(first);
|
||||||
|
let (_, _, _, still_blocked_result) =
|
||||||
|
cross_pool_fence_policy_results(restarted_peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
|
||||||
|
assert!(
|
||||||
|
publish_fleet_capability_probe_result(&slot, "topology-a", still_blocked_result, now + Duration::from_millis(2),)
|
||||||
|
.is_some(),
|
||||||
|
"one remaining writer must keep the successor generation closed"
|
||||||
|
);
|
||||||
|
|
||||||
|
drop(second);
|
||||||
|
let (_, _, _, admitted_result) =
|
||||||
|
cross_pool_fence_policy_results(restarted_peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
|
||||||
|
assert!(
|
||||||
|
publish_fleet_capability_probe_result(&slot, "topology-a", admitted_result, now + Duration::from_millis(3),)
|
||||||
|
.is_none(),
|
||||||
|
"the restarted generation may publish only after every old writer drains"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn legacy_transition_state_reconcile_fresh_observation_closes_the_polling_window() {
|
||||||
|
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||||
|
let now = Instant::now();
|
||||||
|
let original_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||||
|
let (_, _, _, original_result) =
|
||||||
|
cross_pool_fence_policy_results(original_peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
|
||||||
|
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", original_result, now).is_none());
|
||||||
|
let admitted = {
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("the original fleet should admit reconciliation")
|
||||||
|
};
|
||||||
|
|
||||||
|
let restarted_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
assert!(
|
||||||
|
legacy_transition_state_reconcile_fleet_proof_matches_at(&state, &admitted, "topology-a", now),
|
||||||
|
"the periodic cache has not observed the restart yet"
|
||||||
|
);
|
||||||
|
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
|
||||||
|
&state,
|
||||||
|
&admitted,
|
||||||
|
"topology-a",
|
||||||
|
&restarted_peers,
|
||||||
|
now,
|
||||||
|
));
|
||||||
|
|
||||||
|
let (_, _, _, downgraded) = cross_pool_fence_policy_results(original_peers, 4);
|
||||||
|
assert!(
|
||||||
|
downgraded.is_err(),
|
||||||
|
"a synchronous observation of a downgraded peer must fail before any cached proof can authorize a write"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn legacy_transition_state_reconcile_invalid_token_skips_fleet_observation() {
|
||||||
|
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||||
|
let now = Instant::now();
|
||||||
|
let peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||||
|
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(peers), now).is_none());
|
||||||
|
let admitted = {
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("the original fleet should admit reconciliation")
|
||||||
|
};
|
||||||
|
revoke_fleet_capability_proof(&slot);
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
!legacy_transition_state_reconcile_fleet_proof_matches_with_observer(&slot, &admitted, "topology-a", || async {
|
||||||
|
panic!("an invalid local generation must not trigger a fleet observation");
|
||||||
|
},)
|
||||||
|
.await
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn legacy_transition_state_reconcile_membership_and_topology_changes_revoke_authority() {
|
||||||
|
let now = Instant::now();
|
||||||
|
for replacement in [
|
||||||
|
BTreeMap::from([("peer-a".to_string(), Uuid::new_v4()), ("peer-b".to_string(), Uuid::new_v4())]),
|
||||||
|
BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]),
|
||||||
|
] {
|
||||||
|
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||||
|
let original = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||||
|
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(original), now).is_none());
|
||||||
|
let admitted = {
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("the original fleet should admit reconciliation")
|
||||||
|
};
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
publish_fleet_capability_probe_result(&slot, "topology-a", Ok(replacement), now + Duration::from_millis(1),)
|
||||||
|
.is_some(),
|
||||||
|
"membership or process-epoch replacement must wait for the admitted writer"
|
||||||
|
);
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
|
||||||
|
&state,
|
||||||
|
&admitted,
|
||||||
|
"topology-a",
|
||||||
|
now + Duration::from_millis(1),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||||
|
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(BTreeMap::new()), now).is_none());
|
||||||
|
let admitted = {
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("the original topology should admit reconciliation")
|
||||||
|
};
|
||||||
|
mark_fleet_capability_topology_conflict(&slot);
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
assert!(state.topology_conflict);
|
||||||
|
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
|
||||||
|
&state,
|
||||||
|
&admitted,
|
||||||
|
"topology-a",
|
||||||
|
now,
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn legacy_transition_state_reconcile_capability_downgrade_fails_closed() {
|
||||||
|
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||||
|
let now = Instant::now();
|
||||||
|
let peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||||
|
let (_, _, _, compatible_result) =
|
||||||
|
cross_pool_fence_policy_results(peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
|
||||||
|
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", compatible_result, now).is_none());
|
||||||
|
let admitted = {
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
|
||||||
|
.expect("v5 should admit reconciliation")
|
||||||
|
};
|
||||||
|
|
||||||
|
let (_, _, _, downgraded_result) =
|
||||||
|
cross_pool_fence_policy_results(peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION - 1);
|
||||||
|
let err = publish_fleet_capability_probe_result(&slot, "topology-a", downgraded_result, now + Duration::from_millis(1))
|
||||||
|
.expect("a v4 member must revoke reconcile authority");
|
||||||
|
assert!(err.to_string().contains("reconcile policy capability version is unsupported"));
|
||||||
|
let state = slot.read().expect("reconcile proof slot should not poison");
|
||||||
|
assert!(state.proof.is_none());
|
||||||
|
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
|
||||||
|
&state,
|
||||||
|
&admitted,
|
||||||
|
"topology-a",
|
||||||
|
now + Duration::from_millis(1),
|
||||||
|
));
|
||||||
|
assert!(
|
||||||
|
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now + Duration::from_millis(1),)
|
||||||
|
.is_none(),
|
||||||
|
"a downgraded fleet must remain inspect-only"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn remote_version_state_fleet_proof_conflict_revokes_atomic_snapshot() {
|
fn remote_version_state_fleet_proof_conflict_revokes_atomic_snapshot() {
|
||||||
let now = Instant::now();
|
let now = Instant::now();
|
||||||
@@ -3539,6 +3959,57 @@ mod tests {
|
|||||||
assert!(err.to_string().contains("incomplete"));
|
assert!(err.to_string().contains("incomplete"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn legacy_transition_state_reconcile_probe_rejects_missing_or_unreachable_members() {
|
||||||
|
let missing = NotificationSys {
|
||||||
|
peer_clients: Vec::new(),
|
||||||
|
all_peer_clients: vec![None],
|
||||||
|
peer_topology_hosts: vec!["peer-a".to_string()],
|
||||||
|
peer_admin_caches: Vec::new(),
|
||||||
|
tier_config_reload_workers: Default::default(),
|
||||||
|
};
|
||||||
|
let missing_err = missing
|
||||||
|
.probe_cross_pool_fence_fleet("topology-a")
|
||||||
|
.await
|
||||||
|
.expect_err("a missing member slot must prevent reconcile capability proof");
|
||||||
|
assert!(missing_err.to_string().contains("incomplete"));
|
||||||
|
|
||||||
|
let unreachable = NotificationSys {
|
||||||
|
peer_clients: vec![None],
|
||||||
|
all_peer_clients: vec![None, None],
|
||||||
|
peer_topology_hosts: vec!["peer-a".to_string()],
|
||||||
|
peer_admin_caches: vec![Mutex::new(PeerAdminCache::new())],
|
||||||
|
tier_config_reload_workers: Default::default(),
|
||||||
|
};
|
||||||
|
let unreachable_err = unreachable
|
||||||
|
.probe_cross_pool_fence_fleet("topology-a")
|
||||||
|
.await
|
||||||
|
.expect_err("an unreachable member must prevent reconcile capability proof");
|
||||||
|
assert!(unreachable_err.to_string().contains("unreachable"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn legacy_transition_state_reconcile_single_node_stays_closed_before_local_cas_support() {
|
||||||
|
let notification_sys = NotificationSys {
|
||||||
|
peer_clients: Vec::new(),
|
||||||
|
all_peer_clients: vec![None],
|
||||||
|
peer_topology_hosts: Vec::new(),
|
||||||
|
peer_admin_caches: Vec::new(),
|
||||||
|
tier_config_reload_workers: Default::default(),
|
||||||
|
};
|
||||||
|
let (peers, minimum_version) = notification_sys
|
||||||
|
.probe_cross_pool_fence_fleet("topology-a")
|
||||||
|
.await
|
||||||
|
.expect("a single-node capability probe should complete");
|
||||||
|
assert!(peers.is_empty());
|
||||||
|
assert_eq!(minimum_version, LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION);
|
||||||
|
let (_, _, _, reconcile_result) = cross_pool_fence_policy_results(peers, minimum_version);
|
||||||
|
assert!(
|
||||||
|
reconcile_result.is_err(),
|
||||||
|
"the current node must not self-authorize reconcile before the conditional writer lands"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
fn build_props(endpoint: &str) -> ServerProperties {
|
fn build_props(endpoint: &str) -> ServerProperties {
|
||||||
ServerProperties {
|
ServerProperties {
|
||||||
endpoint: endpoint.to_string(),
|
endpoint: endpoint.to_string(),
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ pub mod tier_gen;
|
|||||||
pub mod tier_handlers;
|
pub mod tier_handlers;
|
||||||
pub(crate) mod tier_mutation_intent;
|
pub(crate) mod tier_mutation_intent;
|
||||||
pub mod tier_mutation_peer;
|
pub mod tier_mutation_peer;
|
||||||
|
pub(crate) mod tier_probe_intent;
|
||||||
pub mod warm_backend;
|
pub mod warm_backend;
|
||||||
pub mod warm_backend_aliyun;
|
pub mod warm_backend_aliyun;
|
||||||
pub mod warm_backend_azure;
|
pub mod warm_backend_azure;
|
||||||
|
|||||||
@@ -701,7 +701,7 @@ impl WarmBackend for MockWarmBackend {
|
|||||||
Ok(version)
|
Ok(version)
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn get(&self, object: &str, _rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> {
|
async fn get(&self, object: &str, rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> {
|
||||||
self.precondition().await?;
|
self.precondition().await?;
|
||||||
let barrier = self.inner.get_barrier.lock().await.take();
|
let barrier = self.inner.get_barrier.lock().await.take();
|
||||||
if let Some(barrier) = barrier {
|
if let Some(barrier) = barrier {
|
||||||
@@ -719,6 +719,9 @@ impl WarmBackend for MockWarmBackend {
|
|||||||
let Some(stored) = objects.get(object) else {
|
let Some(stored) = objects.get(object) else {
|
||||||
return Err(std::io::Error::new(std::io::ErrorKind::NotFound, "mock object not found"));
|
return Err(std::io::Error::new(std::io::ErrorKind::NotFound, "mock object not found"));
|
||||||
};
|
};
|
||||||
|
if !rv.is_empty() && stored.remote_version_id != rv {
|
||||||
|
return Err(std::io::Error::new(std::io::ErrorKind::NotFound, "NoSuchVersion"));
|
||||||
|
}
|
||||||
let bytes = &stored.bytes;
|
let bytes = &stored.bytes;
|
||||||
|
|
||||||
let start = opts.start_offset.max(0) as usize;
|
let start = opts.start_offset.max(0) as usize;
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use serde::{Deserialize, Deserializer, Serialize, Serializer, de};
|
use serde::{Deserialize, Deserializer, Serialize, Serializer, de};
|
||||||
|
|
||||||
@@ -145,7 +143,7 @@ mod tests {
|
|||||||
|
|
||||||
assert_eq!(creds.access_key, "access");
|
assert_eq!(creds.access_key, "access");
|
||||||
assert_eq!(creds.secret_key, "secret");
|
assert_eq!(creds.secret_key, "secret");
|
||||||
assert_eq!(creds.creds_json.as_slice(), &service_account[..]);
|
assert_eq!(creds.creds_json.as_slice(), service_account);
|
||||||
|
|
||||||
let wire = serde_json::to_value(&creds).expect("madmin tier credentials should encode");
|
let wire = serde_json::to_value(&creds).expect("madmin tier credentials should encode");
|
||||||
assert_eq!(wire["access"], "access");
|
assert_eq!(wire["access"], "access");
|
||||||
@@ -162,7 +160,7 @@ mod tests {
|
|||||||
.expect("the former RustFS field names and byte-array encoding should remain readable");
|
.expect("the former RustFS field names and byte-array encoding should remain readable");
|
||||||
assert_eq!(legacy.access_key, "legacy-access");
|
assert_eq!(legacy.access_key, "legacy-access");
|
||||||
assert_eq!(legacy.secret_key, "legacy-secret");
|
assert_eq!(legacy.secret_key, "legacy-secret");
|
||||||
assert_eq!(legacy.creds_json.as_slice(), &service_account[..]);
|
assert_eq!(legacy.creds_json.as_slice(), service_account);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -460,6 +460,7 @@ where
|
|||||||
data,
|
data,
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_none_match: Some("*".to_string()),
|
if_none_match: Some("*".to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -556,6 +557,7 @@ where
|
|||||||
data,
|
data,
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
max_parity: true,
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
http_preconditions: Some(HTTPPreconditions {
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
if_match: Some(current_etag.to_string()),
|
if_match: Some(current_etag.to_string()),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -40,6 +40,7 @@ use rustfs_s3_client::credentials::{Credentials, SignatureType, Static, Value};
|
|||||||
use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionCore};
|
use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionCore};
|
||||||
use rustfs_s3_client::{
|
use rustfs_s3_client::{
|
||||||
admin_handler_utils::AdminError,
|
admin_handler_utils::AdminError,
|
||||||
|
api_error_response::to_error_response,
|
||||||
api_put_object::{AdvancedPutOptions, PutObjectOptions},
|
api_put_object::{AdvancedPutOptions, PutObjectOptions},
|
||||||
transition_api::{ReadCloser, ReaderImpl},
|
transition_api::{ReadCloser, ReaderImpl},
|
||||||
};
|
};
|
||||||
@@ -48,11 +49,14 @@ use rustfs_utils::egress::validate_outbound_url;
|
|||||||
use rustfs_utils::http::headers::{
|
use rustfs_utils::http::headers::{
|
||||||
CACHE_CONTROL, CONTENT_DISPOSITION, CONTENT_ENCODING, CONTENT_LANGUAGE, CONTENT_TYPE, EXPIRES, HeaderExt as _,
|
CACHE_CONTROL, CONTENT_DISPOSITION, CONTENT_ENCODING, CONTENT_LANGUAGE, CONTENT_TYPE, EXPIRES, HeaderExt as _,
|
||||||
};
|
};
|
||||||
use s3s::dto::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode, ReplicationStatus};
|
|
||||||
use s3s::header::{
|
use s3s::header::{
|
||||||
X_AMZ_OBJECT_LOCK_LEGAL_HOLD, X_AMZ_OBJECT_LOCK_MODE, X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE, X_AMZ_REPLICATION_STATUS,
|
X_AMZ_OBJECT_LOCK_LEGAL_HOLD, X_AMZ_OBJECT_LOCK_MODE, X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE, X_AMZ_REPLICATION_STATUS,
|
||||||
X_AMZ_STORAGE_CLASS,
|
X_AMZ_STORAGE_CLASS,
|
||||||
};
|
};
|
||||||
|
use s3s::{
|
||||||
|
S3ErrorCode,
|
||||||
|
dto::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode, ReplicationStatus},
|
||||||
|
};
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
@@ -141,6 +145,42 @@ pub trait WarmBackend {
|
|||||||
async fn probe_transition_candidate(&self, _object: &str) -> Result<TransitionCandidateProbe, std::io::Error> {
|
async fn probe_transition_candidate(&self, _object: &str) -> Result<TransitionCandidateProbe, std::io::Error> {
|
||||||
Ok(TransitionCandidateProbe::Unsupported)
|
Ok(TransitionCandidateProbe::Unsupported)
|
||||||
}
|
}
|
||||||
|
async fn probe_transition_version(
|
||||||
|
&self,
|
||||||
|
object: &str,
|
||||||
|
remote_version_id: &str,
|
||||||
|
) -> Result<TransitionCandidateProbe, std::io::Error> {
|
||||||
|
if remote_version_id.is_empty() {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidInput,
|
||||||
|
"an exact tier probe requires a remote version ID",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
self.validate_remote_version_id(remote_version_id)?;
|
||||||
|
match self
|
||||||
|
.get(
|
||||||
|
object,
|
||||||
|
remote_version_id,
|
||||||
|
WarmBackendGetOpts {
|
||||||
|
start_offset: 0,
|
||||||
|
length: 1,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(_) => Ok(TransitionCandidateProbe::VersionedPresent(remote_version_id.to_string())),
|
||||||
|
Err(err) if matches!(to_error_response(&err).code, S3ErrorCode::InvalidRange) => {
|
||||||
|
Ok(TransitionCandidateProbe::VersionedPresent(remote_version_id.to_string()))
|
||||||
|
}
|
||||||
|
Err(err)
|
||||||
|
if err.kind() == std::io::ErrorKind::NotFound
|
||||||
|
|| matches!(to_error_response(&err).code, S3ErrorCode::NoSuchKey | S3ErrorCode::NoSuchVersion) =>
|
||||||
|
{
|
||||||
|
Ok(TransitionCandidateProbe::Missing)
|
||||||
|
}
|
||||||
|
Err(err) => Err(err),
|
||||||
|
}
|
||||||
|
}
|
||||||
async fn in_use(&self) -> Result<bool, std::io::Error>;
|
async fn in_use(&self) -> Result<bool, std::io::Error>;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -437,6 +477,17 @@ impl WarmBackend for MeteredWarmBackend {
|
|||||||
Self::record(TierRequestOperation::Probe, result)
|
Self::record(TierRequestOperation::Probe, result)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn probe_transition_version(
|
||||||
|
&self,
|
||||||
|
object: &str,
|
||||||
|
remote_version_id: &str,
|
||||||
|
) -> Result<TransitionCandidateProbe, std::io::Error> {
|
||||||
|
Self::record(
|
||||||
|
TierRequestOperation::Probe,
|
||||||
|
self.inner.probe_transition_version(object, remote_version_id).await,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
async fn in_use(&self) -> Result<bool, std::io::Error> {
|
async fn in_use(&self) -> Result<bool, std::io::Error> {
|
||||||
Self::record(TierRequestOperation::InUse, self.inner.in_use().await)
|
Self::record(TierRequestOperation::InUse, self.inner.in_use().await)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::future::Future;
|
use std::future::Future;
|
||||||
@@ -146,11 +144,11 @@ pub struct WarmBackendGCS {
|
|||||||
|
|
||||||
impl WarmBackendGCS {
|
impl WarmBackendGCS {
|
||||||
pub async fn new(conf: &TierGCS, tier: &str) -> Result<Self, std::io::Error> {
|
pub async fn new(conf: &TierGCS, tier: &str) -> Result<Self, std::io::Error> {
|
||||||
if conf.creds == "" {
|
if conf.creds.is_empty() {
|
||||||
return Err(std::io::Error::other("both access and secret keys are required"));
|
return Err(std::io::Error::other("both access and secret keys are required"));
|
||||||
}
|
}
|
||||||
|
|
||||||
if conf.bucket == "" {
|
if conf.bucket.is_empty() {
|
||||||
return Err(std::io::Error::other("no bucket name was provided"));
|
return Err(std::io::Error::other("no bucket name was provided"));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -195,11 +193,11 @@ impl WarmBackendGCS {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub fn get_dest(&self, object: &str) -> String {
|
pub fn get_dest(&self, object: &str) -> String {
|
||||||
let mut dest_obj = object.to_string();
|
if self.prefix.is_empty() {
|
||||||
if self.prefix != "" {
|
object.to_string()
|
||||||
dest_obj = format!("{}/{}", &self.prefix, object);
|
} else {
|
||||||
|
format!("{}/{}", self.prefix, object)
|
||||||
}
|
}
|
||||||
return dest_obj;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -223,7 +221,7 @@ impl WarmBackend for WarmBackendGCS {
|
|||||||
let bucket = gcs_bucket_resource_name(&self.bucket);
|
let bucket = gcs_bucket_resource_name(&self.bucket);
|
||||||
let Ok(res) = Box::pin(
|
let Ok(res) = Box::pin(
|
||||||
self.client
|
self.client
|
||||||
.write_object(&bucket, &self.get_dest(object), Bytes::from(d))
|
.write_object(&bucket, self.get_dest(object), Bytes::from(d))
|
||||||
.send_buffered(),
|
.send_buffered(),
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
@@ -240,7 +238,7 @@ impl WarmBackend for WarmBackendGCS {
|
|||||||
|
|
||||||
async fn get(&self, object: &str, rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> {
|
async fn get(&self, object: &str, rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> {
|
||||||
let bucket = gcs_bucket_resource_name(&self.bucket);
|
let bucket = gcs_bucket_resource_name(&self.bucket);
|
||||||
let mut req = self.client.read_object(&bucket, &self.get_dest(object));
|
let mut req = self.client.read_object(&bucket, self.get_dest(object));
|
||||||
let mut max_response_bytes = None;
|
let mut max_response_bytes = None;
|
||||||
if let Some(generation) = parse_generation(rv)? {
|
if let Some(generation) = parse_generation(rv)? {
|
||||||
req = req.set_generation(generation);
|
req = req.set_generation(generation);
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
|||||||
@@ -529,6 +529,10 @@ mod tests {
|
|||||||
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>",
|
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>",
|
||||||
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 66\r\nConnection: close\r\n\r\n<Error><Code>NoSuchObject</Code><Message>missing</Message></Error>",
|
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 66\r\nConnection: close\r\n\r\n<Error><Code>NoSuchObject</Code><Message>missing</Message></Error>",
|
||||||
"HTTP/1.1 403 Forbidden\r\nContent-Type: application/xml\r\nContent-Length: 65\r\nConnection: close\r\n\r\n<Error><Code>AccessDenied</Code><Message>denied</Message></Error>",
|
"HTTP/1.1 403 Forbidden\r\nContent-Type: application/xml\r\nContent-Length: 65\r\nConnection: close\r\n\r\n<Error><Code>AccessDenied</Code><Message>denied</Message></Error>",
|
||||||
|
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>",
|
||||||
|
"HTTP/1.1 416 Range Not Satisfiable\r\nContent-Type: application/xml\r\nContent-Length: 72\r\nConnection: close\r\n\r\n<Error><Code>InvalidRange</Code><Message>empty version</Message></Error>",
|
||||||
|
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 67\r\nConnection: close\r\n\r\n<Error><Code>NoSuchVersion</Code><Message>missing</Message></Error>",
|
||||||
|
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>",
|
||||||
];
|
];
|
||||||
let mut requests = Vec::new();
|
let mut requests = Vec::new();
|
||||||
for response in responses {
|
for response in responses {
|
||||||
@@ -622,15 +626,52 @@ mod tests {
|
|||||||
.await
|
.await
|
||||||
.expect_err("an authorization failure must not be mistaken for a missing key");
|
.expect_err("an authorization failure must not be mistaken for a missing key");
|
||||||
assert_eq!(to_error_response(&err).code, S3ErrorCode::AccessDenied);
|
assert_eq!(to_error_response(&err).code, S3ErrorCode::AccessDenied);
|
||||||
|
assert_eq!(
|
||||||
|
backend
|
||||||
|
.probe_transition_candidate("delete-marker-hidden")
|
||||||
|
.await
|
||||||
|
.expect("a current delete marker should hide the data version"),
|
||||||
|
TransitionCandidateProbe::Missing
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
backend
|
||||||
|
.probe_transition_version("delete-marker-hidden", "historical-version")
|
||||||
|
.await
|
||||||
|
.expect("the stored historical version should be probed exactly"),
|
||||||
|
TransitionCandidateProbe::VersionedPresent("historical-version".to_string())
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
backend
|
||||||
|
.probe_transition_version("delete-marker-hidden", "missing-version")
|
||||||
|
.await
|
||||||
|
.expect("a missing exact version should be classified"),
|
||||||
|
TransitionCandidateProbe::Missing
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
backend
|
||||||
|
.probe_transition_version("missing-object", "historical-version")
|
||||||
|
.await
|
||||||
|
.expect("a missing key for an exact version probe should be classified"),
|
||||||
|
TransitionCandidateProbe::Missing
|
||||||
|
);
|
||||||
|
|
||||||
let requests = fixture.await.expect("candidate fixture should join");
|
let requests = fixture.await.expect("candidate fixture should join");
|
||||||
for request in requests {
|
for request in &requests[..6] {
|
||||||
let request = request.to_ascii_lowercase();
|
let request = request.to_ascii_lowercase();
|
||||||
assert!(request.starts_with("get /bucket/"), "candidate discovery must use object GET");
|
assert!(request.starts_with("get /bucket/"), "candidate discovery must use object GET");
|
||||||
assert!(request.contains("\r\nrange: bytes=0-0\r\n"));
|
assert!(request.contains("\r\nrange: bytes=0-0\r\n"));
|
||||||
assert!(!request.contains("?versioning"));
|
assert!(!request.contains("?versioning"));
|
||||||
assert!(!request.contains("?versions"));
|
assert!(!request.contains("?versions"));
|
||||||
}
|
}
|
||||||
|
for request in &requests[6..] {
|
||||||
|
let request = request.to_ascii_lowercase();
|
||||||
|
assert!(request.starts_with("get /bucket/"), "exact discovery must use object GET");
|
||||||
|
assert!(request.contains("\r\nrange: bytes=0-0\r\n"));
|
||||||
|
}
|
||||||
|
assert!(!requests[5].to_ascii_lowercase().contains("versionid="));
|
||||||
|
assert!(requests[6].to_ascii_lowercase().contains("?versionid=historical-version"));
|
||||||
|
assert!(requests[7].to_ascii_lowercase().contains("?versionid=missing-version"));
|
||||||
|
assert!(requests[8].to_ascii_lowercase().contains("?versionid=historical-version"));
|
||||||
}
|
}
|
||||||
|
|
||||||
fn list_versions(versions: &[(&str, &str)], delete_markers: &[(&str, &str)], is_truncated: bool) -> ListVersionsResult {
|
fn list_versions(versions: &[(&str, &str)], delete_markers: &[(&str, &str)], is_truncated: bool) -> ListVersionsResult {
|
||||||
|
|||||||
@@ -15,8 +15,6 @@
|
|||||||
#![allow(unused_variables)]
|
#![allow(unused_variables)]
|
||||||
#![allow(unused_mut)]
|
#![allow(unused_mut)]
|
||||||
#![allow(unused_assignments)]
|
#![allow(unused_assignments)]
|
||||||
#![allow(unused_must_use)]
|
|
||||||
#![allow(clippy::all)]
|
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,385 @@
|
|||||||
|
// Copyright 2024 RustFS Team
|
||||||
|
//
|
||||||
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
// you may not use this file except in compliance with the License.
|
||||||
|
// You may obtain a copy of the License at
|
||||||
|
//
|
||||||
|
// http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
//
|
||||||
|
// Unless required by applicable law or agreed to in writing, software
|
||||||
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
// See the License for the specific language governing permissions and
|
||||||
|
// limitations under the License.
|
||||||
|
|
||||||
|
//! Pure metadata quorum and early-stop decisions for `SetDisks` reads.
|
||||||
|
//!
|
||||||
|
//! Disk scheduling, coalescing, cancellation, and late shard materialization
|
||||||
|
//! remain with their existing owners; this module only classifies observations.
|
||||||
|
|
||||||
|
use crate::diagnostics::get::{
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA, GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER,
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_ERROR, GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM,
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_NOT_FOUND, GET_METADATA_EARLY_STOP_REASON_UNSAFE_REQUEST,
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_VALID_QUORUM, GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM,
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_VERSION_NOT_FOUND,
|
||||||
|
};
|
||||||
|
use crate::disk::error::DiskError;
|
||||||
|
use crate::disk::error_reduce::OBJECT_OP_IGNORED_ERRS;
|
||||||
|
use crate::set_disk::file_info_is_valid_for_metadata;
|
||||||
|
use rustfs_filemeta::FileInfo;
|
||||||
|
|
||||||
|
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
|
||||||
|
pub(in crate::set_disk) struct MetadataEarlyStopDecision {
|
||||||
|
pub(in crate::set_disk) reason: &'static str,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone, Debug)]
|
||||||
|
pub(in crate::set_disk) struct MetadataQuorumAccumulator {
|
||||||
|
pub(in crate::set_disk) total_disks: usize,
|
||||||
|
pub(in crate::set_disk) default_parity_count: usize,
|
||||||
|
pub(in crate::set_disk) allow_early_stop: bool,
|
||||||
|
pub(in crate::set_disk) valid_responses: usize,
|
||||||
|
pub(in crate::set_disk) not_found_responses: usize,
|
||||||
|
pub(in crate::set_disk) version_not_found_responses: usize,
|
||||||
|
pub(in crate::set_disk) ignored_errors: usize,
|
||||||
|
pub(in crate::set_disk) hard_errors: usize,
|
||||||
|
pub(in crate::set_disk) candidate: Option<FileInfo>,
|
||||||
|
pub(in crate::set_disk) candidate_votes: usize,
|
||||||
|
// Bitset of shard indexes whose metadata matches the candidate. Erasure
|
||||||
|
// layouts are capped at 16 shards, so this stays allocation-free on the
|
||||||
|
// GET metadata hot path.
|
||||||
|
candidate_shard_mask: u16,
|
||||||
|
pub(in crate::set_disk) conflicting_metadata: bool,
|
||||||
|
pub(in crate::set_disk) delete_marker_seen: bool,
|
||||||
|
pub(in crate::set_disk) delete_marker_candidates: Vec<(FileInfo, usize)>,
|
||||||
|
pub(in crate::set_disk) delete_marker_votes: usize,
|
||||||
|
pub(in crate::set_disk) requested_version_id: String,
|
||||||
|
pub(in crate::set_disk) matching_version_votes: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl MetadataQuorumAccumulator {
|
||||||
|
pub(in crate::set_disk) fn new(total_disks: usize, default_parity_count: usize, allow_early_stop: bool) -> Self {
|
||||||
|
Self {
|
||||||
|
total_disks,
|
||||||
|
default_parity_count,
|
||||||
|
allow_early_stop,
|
||||||
|
valid_responses: 0,
|
||||||
|
not_found_responses: 0,
|
||||||
|
version_not_found_responses: 0,
|
||||||
|
ignored_errors: 0,
|
||||||
|
hard_errors: 0,
|
||||||
|
candidate: None,
|
||||||
|
candidate_votes: 0,
|
||||||
|
candidate_shard_mask: 0,
|
||||||
|
conflicting_metadata: false,
|
||||||
|
delete_marker_seen: false,
|
||||||
|
delete_marker_candidates: Vec::new(),
|
||||||
|
delete_marker_votes: 0,
|
||||||
|
requested_version_id: String::new(),
|
||||||
|
matching_version_votes: 0,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn with_requested_version_id(mut self, version_id: &str) -> Self {
|
||||||
|
self.requested_version_id = version_id.to_string();
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn observe_file_info(&mut self, file_info: &FileInfo) {
|
||||||
|
self.observe_file_info_with_index(None, file_info);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn observe_file_info_at(&mut self, disk_index: usize, file_info: &FileInfo) {
|
||||||
|
self.observe_file_info_with_index(Some(disk_index), file_info);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn observe_file_info_with_index(&mut self, disk_index: Option<usize>, file_info: &FileInfo) {
|
||||||
|
if !file_info_is_valid_for_metadata(file_info) {
|
||||||
|
self.hard_errors = self.hard_errors.saturating_add(1);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
self.valid_responses = self.valid_responses.saturating_add(1);
|
||||||
|
|
||||||
|
// Track version match for versioned requests
|
||||||
|
if !self.requested_version_id.is_empty()
|
||||||
|
&& let Some(ref vid) = file_info.version_id
|
||||||
|
&& vid.to_string() == self.requested_version_id
|
||||||
|
{
|
||||||
|
self.matching_version_votes = self.matching_version_votes.saturating_add(1);
|
||||||
|
}
|
||||||
|
|
||||||
|
if file_info.is_canonical_delete_marker() {
|
||||||
|
self.delete_marker_seen = true;
|
||||||
|
if let Some((_, votes)) = self
|
||||||
|
.delete_marker_candidates
|
||||||
|
.iter_mut()
|
||||||
|
.find(|(candidate, _)| metadata_early_stop_candidate_matches(candidate, file_info))
|
||||||
|
{
|
||||||
|
*votes = votes.saturating_add(1);
|
||||||
|
} else {
|
||||||
|
self.delete_marker_candidates.push((file_info.clone(), 1));
|
||||||
|
}
|
||||||
|
self.delete_marker_votes = self
|
||||||
|
.delete_marker_candidates
|
||||||
|
.iter()
|
||||||
|
.map(|(_, votes)| *votes)
|
||||||
|
.max()
|
||||||
|
.unwrap_or_default();
|
||||||
|
self.conflicting_metadata |= self.delete_marker_candidates.len() > 1;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
match &self.candidate {
|
||||||
|
Some(candidate) if metadata_early_stop_candidate_matches(candidate, file_info) => {
|
||||||
|
self.candidate_votes = self.candidate_votes.saturating_add(1);
|
||||||
|
if let Some(disk_index) = disk_index
|
||||||
|
&& let Some(bit) = Self::candidate_shard_bit(candidate, file_info, disk_index)
|
||||||
|
{
|
||||||
|
self.candidate_shard_mask |= bit;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Some(_) => {
|
||||||
|
self.conflicting_metadata = true;
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
self.candidate = Some(file_info.clone());
|
||||||
|
self.candidate_votes = 1;
|
||||||
|
if let Some(disk_index) = disk_index
|
||||||
|
&& let Some(bit) = Self::candidate_shard_bit(file_info, file_info, disk_index)
|
||||||
|
{
|
||||||
|
self.candidate_shard_mask |= bit;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn candidate_shard_bit(candidate: &FileInfo, file_info: &FileInfo, disk_index: usize) -> Option<u16> {
|
||||||
|
let &erasure_index = candidate.erasure.distribution.get(disk_index)?;
|
||||||
|
if erasure_index == 0 || erasure_index > u16::BITS as usize || file_info.erasure.index != erasure_index {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(1u16 << (erasure_index - 1))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn candidate_has_read_reserve(&self) -> bool {
|
||||||
|
self.candidate_read_reserve_target()
|
||||||
|
.is_some_and(|required| self.candidate_shard_mask.count_ones() as usize >= required)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn candidate_read_reserve_target(&self) -> Option<usize> {
|
||||||
|
let candidate = self.candidate.as_ref()?;
|
||||||
|
Some(
|
||||||
|
candidate
|
||||||
|
.erasure
|
||||||
|
.data_blocks
|
||||||
|
.saturating_add(usize::from(candidate.erasure.parity_blocks > 0)),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn observe_error(&mut self, err: &DiskError) {
|
||||||
|
match err {
|
||||||
|
DiskError::FileNotFound | DiskError::VolumeNotFound => {
|
||||||
|
self.not_found_responses = self.not_found_responses.saturating_add(1);
|
||||||
|
}
|
||||||
|
DiskError::FileVersionNotFound => {
|
||||||
|
self.version_not_found_responses = self.version_not_found_responses.saturating_add(1);
|
||||||
|
}
|
||||||
|
_ if is_metadata_fanout_ignored_error(err) => {
|
||||||
|
self.ignored_errors = self.ignored_errors.saturating_add(1);
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
self.hard_errors = self.hard_errors.saturating_add(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn early_stop_decision(&self) -> Option<MetadataEarlyStopDecision> {
|
||||||
|
if !self.allow_early_stop {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
if self.delete_marker_votes >= self.default_write_quorum() {
|
||||||
|
return Some(MetadataEarlyStopDecision {
|
||||||
|
reason: GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
if self.conflicting_metadata
|
||||||
|
|| self.delete_marker_seen
|
||||||
|
|| self.not_found_responses > 0
|
||||||
|
|| self.version_not_found_responses > 0
|
||||||
|
|| self.hard_errors > 0
|
||||||
|
{
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
if self
|
||||||
|
.candidate
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|candidate| self.candidate_latest_quorum(candidate))
|
||||||
|
.is_some_and(|latest_quorum| self.candidate_votes >= latest_quorum)
|
||||||
|
{
|
||||||
|
return Some(MetadataEarlyStopDecision {
|
||||||
|
reason: GET_METADATA_EARLY_STOP_REASON_VALID_QUORUM,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check if a versioned request can early-stop because the requested
|
||||||
|
/// version_id has reached quorum across disks.
|
||||||
|
pub(in crate::set_disk) fn version_early_stop_decision(&self) -> Option<MetadataEarlyStopDecision> {
|
||||||
|
if !self.allow_early_stop {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
if self.requested_version_id.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
if self.conflicting_metadata
|
||||||
|
|| self.delete_marker_seen
|
||||||
|
|| self.not_found_responses > 0
|
||||||
|
|| self.version_not_found_responses > 0
|
||||||
|
|| self.hard_errors > 0
|
||||||
|
{
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
if self.matching_version_votes >= self.read_quorum_for_version() {
|
||||||
|
return Some(MetadataEarlyStopDecision {
|
||||||
|
reason: GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn can_still_reach_early_stop_with_pending(&self, pending: usize) -> bool {
|
||||||
|
if !self.allow_early_stop {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if self.delete_marker_votes.saturating_add(pending) >= self.default_write_quorum() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
if self.conflicting_metadata
|
||||||
|
|| self.delete_marker_seen
|
||||||
|
|| self.not_found_responses > 0
|
||||||
|
|| self.version_not_found_responses > 0
|
||||||
|
|| self.hard_errors > 0
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if !self.requested_version_id.is_empty()
|
||||||
|
&& self.matching_version_votes.saturating_add(pending) >= self.read_quorum_for_version()
|
||||||
|
{
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
match &self.candidate {
|
||||||
|
Some(candidate) => self
|
||||||
|
.candidate_latest_quorum(candidate)
|
||||||
|
.is_some_and(|latest_quorum| self.candidate_votes.saturating_add(pending) >= latest_quorum),
|
||||||
|
None => pending >= self.default_write_quorum(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compute the read quorum threshold for version-aware early-stop.
|
||||||
|
/// Uses `total_disks / 2` (like `missing_response_quorum`) when
|
||||||
|
/// `default_parity_count` is set, otherwise requires all disks.
|
||||||
|
pub(in crate::set_disk) fn read_quorum_for_version(&self) -> usize {
|
||||||
|
self.missing_response_quorum()
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn final_miss_reason(&self) -> &'static str {
|
||||||
|
if !self.allow_early_stop {
|
||||||
|
return GET_METADATA_EARLY_STOP_REASON_UNSAFE_REQUEST;
|
||||||
|
}
|
||||||
|
if self.conflicting_metadata {
|
||||||
|
return GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA;
|
||||||
|
}
|
||||||
|
if self.delete_marker_seen {
|
||||||
|
return GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER;
|
||||||
|
}
|
||||||
|
let missing_response_quorum = self.missing_response_quorum();
|
||||||
|
if self.version_not_found_responses >= missing_response_quorum {
|
||||||
|
return GET_METADATA_EARLY_STOP_REASON_VERSION_NOT_FOUND;
|
||||||
|
}
|
||||||
|
if self.not_found_responses >= missing_response_quorum {
|
||||||
|
return GET_METADATA_EARLY_STOP_REASON_NOT_FOUND;
|
||||||
|
}
|
||||||
|
if self.hard_errors > 0 {
|
||||||
|
return GET_METADATA_EARLY_STOP_REASON_ERROR;
|
||||||
|
}
|
||||||
|
if self.ignored_errors > 0 {
|
||||||
|
return GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM;
|
||||||
|
}
|
||||||
|
GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn candidate_latest_quorum(&self, candidate: &FileInfo) -> Option<usize> {
|
||||||
|
if self.default_parity_count == 0 {
|
||||||
|
return Some(self.total_disks);
|
||||||
|
}
|
||||||
|
if candidate.is_canonical_delete_marker() || candidate.size == 0 || candidate.erasure.parity_blocks >= self.total_disks {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let data_blocks = candidate.erasure.data_blocks;
|
||||||
|
Some(if data_blocks == candidate.erasure.parity_blocks {
|
||||||
|
data_blocks.saturating_add(1)
|
||||||
|
} else {
|
||||||
|
data_blocks
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn default_write_quorum(&self) -> usize {
|
||||||
|
if self.default_parity_count == 0 || self.default_parity_count >= self.total_disks {
|
||||||
|
return self.total_disks;
|
||||||
|
}
|
||||||
|
let data_blocks = self.total_disks.saturating_sub(self.default_parity_count);
|
||||||
|
if data_blocks == self.default_parity_count {
|
||||||
|
data_blocks.saturating_add(1)
|
||||||
|
} else {
|
||||||
|
data_blocks
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn missing_response_quorum(&self) -> usize {
|
||||||
|
if self.default_parity_count == 0 || self.default_parity_count >= self.total_disks {
|
||||||
|
self.total_disks
|
||||||
|
} else {
|
||||||
|
self.total_disks / 2
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn metadata_early_stop_candidate_matches(left: &FileInfo, right: &FileInfo) -> bool {
|
||||||
|
left.volume == right.volume
|
||||||
|
&& left.name == right.name
|
||||||
|
&& left.version_id == right.version_id
|
||||||
|
&& left.is_latest == right.is_latest
|
||||||
|
&& left.deleted == right.deleted
|
||||||
|
&& left.mark_deleted == right.mark_deleted
|
||||||
|
&& left.transition_status == right.transition_status
|
||||||
|
&& left.transitioned_objname == right.transitioned_objname
|
||||||
|
&& left.transition_tier == right.transition_tier
|
||||||
|
&& left.transition_version_id == right.transition_version_id
|
||||||
|
&& left.transition_version == right.transition_version
|
||||||
|
&& left.transition_version_state == right.transition_version_state
|
||||||
|
&& left.expire_restored == right.expire_restored
|
||||||
|
&& left.size == right.size
|
||||||
|
&& left.mod_time == right.mod_time
|
||||||
|
&& left.mode == right.mode
|
||||||
|
&& left.written_by_version == right.written_by_version
|
||||||
|
&& left.metadata == right.metadata
|
||||||
|
&& left.replication_state_internal == right.replication_state_internal
|
||||||
|
&& left.parts == right.parts
|
||||||
|
&& left.checksum == right.checksum
|
||||||
|
&& left.versioned == right.versioned
|
||||||
|
&& left.num_versions == right.num_versions
|
||||||
|
&& left.successor_mod_time == right.successor_mod_time
|
||||||
|
&& left.data_dir == right.data_dir
|
||||||
|
&& left.erasure.algorithm == right.erasure.algorithm
|
||||||
|
&& left.erasure.data_blocks == right.erasure.data_blocks
|
||||||
|
&& left.erasure.parity_blocks == right.erasure.parity_blocks
|
||||||
|
&& left.erasure.block_size == right.erasure.block_size
|
||||||
|
&& left.erasure.distribution == right.erasure.distribution
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(in crate::set_disk) fn is_metadata_fanout_ignored_error(err: &DiskError) -> bool {
|
||||||
|
OBJECT_OP_IGNORED_ERRS.iter().any(|ignored| ignored == err)
|
||||||
|
}
|
||||||
@@ -18,3 +18,4 @@
|
|||||||
//! duplicating read/write/erasure logic.
|
//! duplicating read/write/erasure logic.
|
||||||
|
|
||||||
pub(crate) mod io_primitives;
|
pub(crate) mod io_primitives;
|
||||||
|
mod metadata_quorum;
|
||||||
|
|||||||
@@ -876,7 +876,7 @@ pub use ops::multipart::{MultipartCommitBarrier, MultipartCommitPause};
|
|||||||
pub(crate) use ops::object::DeleteObjectCommitBarrier;
|
pub(crate) use ops::object::DeleteObjectCommitBarrier;
|
||||||
#[cfg(any(test, feature = "test-util"))]
|
#[cfg(any(test, feature = "test-util"))]
|
||||||
pub(crate) use ops::object::TransitionCleanupStoreBarrier as SetDiskTransitionCleanupStoreBarrier;
|
pub(crate) use ops::object::TransitionCleanupStoreBarrier as SetDiskTransitionCleanupStoreBarrier;
|
||||||
#[cfg(test)]
|
#[cfg(all(test, feature = "test-util"))]
|
||||||
pub(crate) use ops::object::TransitionUploadedCommitBarrier as SetDiskTransitionUploadedCommitBarrier;
|
pub(crate) use ops::object::TransitionUploadedCommitBarrier as SetDiskTransitionUploadedCommitBarrier;
|
||||||
pub(crate) use ops::object::body_cache_plaintext_len;
|
pub(crate) use ops::object::body_cache_plaintext_len;
|
||||||
#[cfg(all(test, feature = "test-util"))]
|
#[cfg(all(test, feature = "test-util"))]
|
||||||
|
|||||||
@@ -299,11 +299,11 @@ use crate::error::is_err_invalid_upload_id;
|
|||||||
use crate::object_api::{GetObjectBodySource, get_object_body_cache_hook_suppressed};
|
use crate::object_api::{GetObjectBodySource, get_object_body_cache_hook_suppressed};
|
||||||
use crate::object_api::{
|
use crate::object_api::{
|
||||||
NamespaceLockFence, ReplicationStatusWritebackCondition, ReplicationStatusWritebackMode,
|
NamespaceLockFence, ReplicationStatusWritebackCondition, ReplicationStatusWritebackMode,
|
||||||
SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY,
|
SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, WriteCompletion,
|
||||||
};
|
};
|
||||||
use crate::services::notification_sys::RemoteVersionStateFleetProofToken;
|
use crate::services::notification_sys::RemoteVersionStateFleetProofToken;
|
||||||
use crate::services::tier::tier::{TierConfigMgr, TierDestinationId, TierOperationLease, tier_destination_id_from_metadata};
|
use crate::services::tier::tier::{TierConfigMgr, TierDestinationId, TierOperationLease, tier_destination_id_from_metadata};
|
||||||
use crate::set_disk::core::io_primitives::{RenameTailCleanup, finish_rename_tail_heal};
|
use crate::set_disk::core::io_primitives::{RenameRollbackReceipt, RenameTailCleanup, finish_rename_tail_heal};
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
use crate::storage_api_contracts::namespace::NamespaceLocking;
|
use crate::storage_api_contracts::namespace::NamespaceLocking;
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -3548,6 +3548,7 @@ impl SetDisks {
|
|||||||
(None, None, None)
|
(None, None, None)
|
||||||
};
|
};
|
||||||
let mut tmp_cleanup_owned = false;
|
let mut tmp_cleanup_owned = false;
|
||||||
|
let rollback_receipt = RenameRollbackReceipt::default();
|
||||||
let operation = async {
|
let operation = async {
|
||||||
let erasure = Arc::new(erasure_from_file_info(&fi, false)?);
|
let erasure = Arc::new(erasure_from_file_info(&fi, false)?);
|
||||||
|
|
||||||
@@ -4256,6 +4257,7 @@ impl SetDisks {
|
|||||||
let commit_bucket = bucket.to_owned();
|
let commit_bucket = bucket.to_owned();
|
||||||
let commit_object = object.to_owned();
|
let commit_object = object.to_owned();
|
||||||
let commit_tmp_dir = tmp_dir.clone();
|
let commit_tmp_dir = tmp_dir.clone();
|
||||||
|
let commit_rollback_receipt = rollback_receipt.clone();
|
||||||
let commit_object_lock_guard = object_lock_guard.take();
|
let commit_object_lock_guard = object_lock_guard.take();
|
||||||
let commit_decommission_object_lock_guard = decommission_object_lock_guard.take();
|
let commit_decommission_object_lock_guard = decommission_object_lock_guard.take();
|
||||||
let commit_publication_guard = publication_commit_guard.take();
|
let commit_publication_guard = publication_commit_guard.take();
|
||||||
@@ -4266,13 +4268,17 @@ impl SetDisks {
|
|||||||
// complete rename fan-out drains. Keep this path synchronous so
|
// complete rename fan-out drains. Keep this path synchronous so
|
||||||
// its terminal state is known before the coordinator releases
|
// its terminal state is known before the coordinator releases
|
||||||
// remote leases.
|
// remote leases.
|
||||||
let commit_allows_early_ack = !(opts.data_movement && opts.has_decommission_capacity_reservation())
|
let commit_owns_namespace_guard = commit_object_lock_guard.is_some()
|
||||||
&& (commit_object_lock_guard.is_some()
|
|| commit_decommission_object_lock_guard.is_some()
|
||||||
|| commit_decommission_object_lock_guard.is_some()
|
|| commit_publication_guard.is_some();
|
||||||
|| commit_publication_guard.is_some())
|
let commit_allows_early_ack = opts.write_completion == WriteCompletion::Quorum
|
||||||
|
&& !(opts.data_movement && opts.has_decommission_capacity_reservation())
|
||||||
|
&& commit_owns_namespace_guard
|
||||||
&& commit_scanner_publication_scope.is_none();
|
&& commit_scanner_publication_scope.is_none();
|
||||||
|
// Full-tail callers also transfer owned guards to the coordinator:
|
||||||
|
// cancelling their ACK waiter must not cancel an in-flight rename.
|
||||||
let detach_commit_owner = commit_scanner_publication_scope.is_some()
|
let detach_commit_owner = commit_scanner_publication_scope.is_some()
|
||||||
|| commit_allows_early_ack
|
|| commit_owns_namespace_guard
|
||||||
|| commit_bucket_lifecycle_guard.is_some()
|
|| commit_bucket_lifecycle_guard.is_some()
|
||||||
|| quota_mutation_fence;
|
|| quota_mutation_fence;
|
||||||
let commit_write_path_label = write_path.metric_label();
|
let commit_write_path_label = write_path.metric_label();
|
||||||
@@ -4452,7 +4458,8 @@ impl SetDisks {
|
|||||||
write_quorum,
|
write_quorum,
|
||||||
commit_scanner_publication_lease_tokens.as_ref(),
|
commit_scanner_publication_lease_tokens.as_ref(),
|
||||||
)
|
)
|
||||||
.with_publication_scope(commit_scanner_publication_scope.clone()),
|
.with_publication_scope(commit_scanner_publication_scope.clone())
|
||||||
|
.with_rollback_receipt(commit_rollback_receipt.clone()),
|
||||||
)
|
)
|
||||||
.await;
|
.await;
|
||||||
if let Some(scope) = commit_scanner_publication_scope.as_ref() {
|
if let Some(scope) = commit_scanner_publication_scope.as_ref() {
|
||||||
@@ -4585,6 +4592,11 @@ impl SetDisks {
|
|||||||
let rename_commit = match rename_result {
|
let rename_commit = match rename_result {
|
||||||
Ok(commit) => commit,
|
Ok(commit) => commit,
|
||||||
Err(err) => {
|
Err(err) => {
|
||||||
|
if commit_rollback_receipt.is_incomplete() {
|
||||||
|
// Incomplete undo retains the staging source and
|
||||||
|
// rollback backup for recovery; cleanup is unsafe.
|
||||||
|
return Err(err.into());
|
||||||
|
}
|
||||||
if let Err(cleanup_err) = commit_set.delete_all(RUSTFS_META_TMP_BUCKET, &commit_tmp_dir).await {
|
if let Err(cleanup_err) = commit_set.delete_all(RUSTFS_META_TMP_BUCKET, &commit_tmp_dir).await {
|
||||||
warn!(tmp_dir = %commit_tmp_dir, error = ?cleanup_err, "failed to cleanup put_object temporary data");
|
warn!(tmp_dir = %commit_tmp_dir, error = ?cleanup_err, "failed to cleanup put_object temporary data");
|
||||||
} else if issue3031_diag_enabled() {
|
} else if issue3031_diag_enabled() {
|
||||||
@@ -4617,9 +4629,8 @@ impl SetDisks {
|
|||||||
request.object_version_id = committed_version_id
|
request.object_version_id = committed_version_id
|
||||||
.or_else(|| commit_version_suspended.then(Uuid::nil))
|
.or_else(|| commit_version_suspended.then(Uuid::nil))
|
||||||
.map(|version_id| version_id.to_string());
|
.map(|version_id| version_id.to_string());
|
||||||
tokio::spawn(async move {
|
let heal_set = commit_set.clone();
|
||||||
let _ = rustfs_heal_contracts::heal_channel::send_heal_request(request).await;
|
tokio::spawn(async move { heal_set.submit_rename_tail_heal(request).await });
|
||||||
});
|
|
||||||
}
|
}
|
||||||
|
|
||||||
let rename_stage_elapsed = rename_stage_start.elapsed();
|
let rename_stage_elapsed = rename_stage_start.elapsed();
|
||||||
@@ -4885,7 +4896,7 @@ impl SetDisks {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
} else {
|
} else if !rollback_receipt.is_incomplete() {
|
||||||
// Failure path (quorum loss / rollback): keep the cleanup inline so
|
// Failure path (quorum loss / rollback): keep the cleanup inline so
|
||||||
// a failed PUT never returns while its tmp shards are still on disk
|
// a failed PUT never returns while its tmp shards are still on disk
|
||||||
// (state-residue hardening tracked by backlog#864 / backlog#898).
|
// (state-residue hardening tracked by backlog#864 / backlog#898).
|
||||||
@@ -17494,27 +17505,69 @@ mod put_object_tmp_cleanup_tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(capacity_dirty_scope)]
|
||||||
async fn put_object_failure_cleans_tmp_workspace_inline() {
|
async fn put_object_failure_cleans_tmp_workspace_inline() {
|
||||||
let (temp_dirs, _disk_stores, set_disks) = hermetic_set_disks(4).await;
|
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async {
|
||||||
|
for write_completion in [WriteCompletion::Quorum, WriteCompletion::TailDrained] {
|
||||||
|
let (temp_dirs, _disk_stores, set_disks) = hermetic_set_disks(4).await;
|
||||||
|
let bucket = "tmp-clean-missing-bucket";
|
||||||
|
let object = "orphan-object";
|
||||||
|
let barrier = PutObjectCommitBarrier::install(bucket, object, PutObjectCommitPause::BeforeNamespace);
|
||||||
|
let writer = Arc::clone(&set_disks);
|
||||||
|
let put = tokio::spawn(async move {
|
||||||
|
let mut reader = PutObjReader::from_vec(vec![9u8; TEST_OBJECT_SIZE]);
|
||||||
|
writer
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
});
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
|
||||||
|
.await
|
||||||
|
.expect("missing-bucket PUT must stage before rename");
|
||||||
|
let staged = non_trash_tmp_entries(&temp_dirs).await;
|
||||||
|
assert_eq!(staged.len(), 4, "every disk must have a staged workspace before rejection");
|
||||||
|
for workspace in staged {
|
||||||
|
let mut entries = tokio::fs::read_dir(&workspace)
|
||||||
|
.await
|
||||||
|
.expect("staged workspace should be readable");
|
||||||
|
let mut shards = 0;
|
||||||
|
while let Some(entry) = entries.next_entry().await.expect("staged data directory should be readable") {
|
||||||
|
if entry.file_type().await.expect("staged entry type").is_dir() {
|
||||||
|
let part = tokio::fs::metadata(entry.path().join("part.1"))
|
||||||
|
.await
|
||||||
|
.expect("staging must contain an actual erasure shard");
|
||||||
|
assert!(part.len() > 0, "the shard must be written before the missing-bucket failure");
|
||||||
|
shards += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert_eq!(shards, 1);
|
||||||
|
}
|
||||||
|
assert!(temp_dirs.iter().all(|dir| !dir.path().join(bucket).exists()));
|
||||||
|
barrier.release();
|
||||||
|
let err = tokio::time::timeout(Duration::from_secs(30), put)
|
||||||
|
.await
|
||||||
|
.expect("missing-bucket PUT must finish")
|
||||||
|
.expect("PUT task should join")
|
||||||
|
.expect_err("put_object into a missing bucket volume must fail");
|
||||||
|
assert!(matches!(err, StorageError::VolumeNotFound), "original disk error expected: {err}");
|
||||||
|
|
||||||
// The bucket volume is never created, so the shards are written into
|
// No polling: known pre-publication rejection must clean staging
|
||||||
// the tmp workspace and the commit fails at rename_data with a quorum
|
// inline, before PUT returns (backlog#864 / backlog#898).
|
||||||
// error — exercising the failure-path cleanup.
|
let leftovers = non_trash_tmp_entries(&temp_dirs).await;
|
||||||
let mut reader = PutObjReader::from_vec(vec![9u8; TEST_OBJECT_SIZE]);
|
assert!(
|
||||||
let err = set_disks
|
leftovers.is_empty(),
|
||||||
.put_object("tmp-clean-missing-bucket", "orphan-object", &mut reader, &ObjectOptions::default())
|
"failed PUT must not leave tmp shards behind, leftovers: {leftovers:?}, err: {err}"
|
||||||
.await
|
);
|
||||||
.expect_err("put_object into a missing bucket volume must fail");
|
}
|
||||||
|
})
|
||||||
// No polling: the failure path must clean the tmp workspace inline,
|
.await;
|
||||||
// before put_object returns (backlog#864 / backlog#898 hardening).
|
|
||||||
let leftovers = non_trash_tmp_entries(&temp_dirs).await;
|
|
||||||
assert!(
|
|
||||||
leftovers.is_empty(),
|
|
||||||
"failed PUT must not leave tmp shards behind, leftovers: {leftovers:?}, err: {err}"
|
|
||||||
);
|
|
||||||
|
|
||||||
drop(temp_dirs);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
@@ -18157,6 +18210,354 @@ mod put_object_tmp_cleanup_tests {
|
|||||||
.await;
|
.await;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async fn make_completion_test_bucket(disks: &[DiskStore], bucket: &str) {
|
||||||
|
for disk in disks {
|
||||||
|
disk.make_volume(bucket)
|
||||||
|
.await
|
||||||
|
.expect("completion test bucket should be created");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Observe the actual metadata quorum while the remaining rename is parked.
|
||||||
|
/// A completed task count alone can race tasks that have not started yet.
|
||||||
|
async fn wait_for_paused_tail_metadata_quorum(disks: &[DiskStore], bucket: &str, object: &str) {
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), async {
|
||||||
|
loop {
|
||||||
|
let mut committed = 0;
|
||||||
|
for disk in disks {
|
||||||
|
match disk.read_version("", bucket, object, "", &ReadOptions::default()).await {
|
||||||
|
Ok(_) => committed += 1,
|
||||||
|
Err(DiskError::FileNotFound | DiskError::FileVersionNotFound) => {}
|
||||||
|
Err(err) => panic!("unexpected metadata error while observing {bucket}/{object}: {err}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if committed == 3 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
tokio::task::yield_now().await;
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.expect("three disks must publish metadata while the fourth rename remains paused");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(capacity_dirty_scope)]
|
||||||
|
async fn tail_drained_put_waits_for_tail_and_allows_immediate_cas() {
|
||||||
|
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async {
|
||||||
|
for size in [4096, 1024 * 1024] {
|
||||||
|
let (_dirs, disks, set) = hermetic_set_disks(4).await;
|
||||||
|
let bucket = "put-full-tail-cas";
|
||||||
|
let object = "full-tail-cas-object";
|
||||||
|
make_completion_test_bucket(&disks, bucket).await;
|
||||||
|
let tasks = rename_fanout_barrier::observe_tasks(object);
|
||||||
|
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
|
||||||
|
let writer = Arc::clone(&set);
|
||||||
|
let put = tokio::spawn(async move {
|
||||||
|
let mut reader = PutObjReader::from_vec(vec![b'1'; size]);
|
||||||
|
writer
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
});
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
|
||||||
|
.await
|
||||||
|
.expect("full-tail PUT must reach the rename barrier");
|
||||||
|
wait_for_paused_tail_metadata_quorum(&disks, bucket, object).await;
|
||||||
|
assert!(!put.is_finished(), "full-tail PUT must remain pending after metadata quorum");
|
||||||
|
let mut lock_probe = Box::pin(set.acquire_write_lock_diag("full_tail_probe", bucket, object));
|
||||||
|
assert!(
|
||||||
|
futures::poll!(lock_probe.as_mut()).is_pending(),
|
||||||
|
"the owned namespace guard must remain held"
|
||||||
|
);
|
||||||
|
barrier.release();
|
||||||
|
let written = tokio::time::timeout(Duration::from_secs(30), put)
|
||||||
|
.await
|
||||||
|
.expect("full-tail PUT should finish after release")
|
||||||
|
.expect("full-tail PUT task should join")
|
||||||
|
.expect("full-tail PUT must commit");
|
||||||
|
assert_eq!(tasks.running(), 0, "full-tail response must follow every rename task");
|
||||||
|
drop(
|
||||||
|
tokio::time::timeout(Duration::from_secs(5), lock_probe)
|
||||||
|
.await
|
||||||
|
.expect("same-key lock should be available on return")
|
||||||
|
.expect("same-key lock probe should succeed"),
|
||||||
|
);
|
||||||
|
for disk in &disks {
|
||||||
|
disk.read_version("", bucket, object, "", &ReadOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("successful full-tail PUT must publish on every healthy disk");
|
||||||
|
}
|
||||||
|
drop(barrier);
|
||||||
|
let mut replacement = PutObjReader::from_vec(b"cas successor".to_vec());
|
||||||
|
set.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut replacement,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
http_preconditions: Some(HTTPPreconditions {
|
||||||
|
if_match: written.etag,
|
||||||
|
..Default::default()
|
||||||
|
}),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("immediate same-key CAS must acquire the namespace guard");
|
||||||
|
let mut read = set
|
||||||
|
.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("CAS successor must be immediately readable");
|
||||||
|
let mut body = Vec::new();
|
||||||
|
read.stream.read_to_end(&mut body).await.expect("successor body must drain");
|
||||||
|
assert_eq!(body, b"cas successor");
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(capacity_dirty_scope)]
|
||||||
|
async fn tail_drained_put_preserves_quorum_success_and_heals_failed_tail() {
|
||||||
|
let (_dirs, disks, set) = hermetic_set_disks(4).await;
|
||||||
|
let bucket = "put-full-tail-heal";
|
||||||
|
let object = "full-tail-heal-object";
|
||||||
|
make_completion_test_bucket(&disks, bucket).await;
|
||||||
|
let mut heals = set.capture_test_rename_tail_heals();
|
||||||
|
let tasks = rename_fanout_barrier::observe_tasks(object);
|
||||||
|
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
|
||||||
|
let _fault = rename_fault_injection::fail_rename_on(object, &[0]);
|
||||||
|
let writer = Arc::clone(&set);
|
||||||
|
let put = tokio::spawn(async move {
|
||||||
|
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
|
||||||
|
writer
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
});
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
|
||||||
|
.await
|
||||||
|
.expect("failed tail must first reach the rename barrier");
|
||||||
|
wait_for_paused_tail_metadata_quorum(&disks, bucket, object).await;
|
||||||
|
assert!(!put.is_finished(), "committed quorum must still wait for the failing tail");
|
||||||
|
barrier.release();
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), put)
|
||||||
|
.await
|
||||||
|
.expect("failed tail should drain")
|
||||||
|
.expect("PUT task should join")
|
||||||
|
.expect("a minority tail error must not negate committed quorum");
|
||||||
|
assert_eq!(tasks.running(), 0);
|
||||||
|
let heal = tokio::time::timeout(Duration::from_secs(30), heals.recv())
|
||||||
|
.await
|
||||||
|
.expect("failed tail must schedule heal")
|
||||||
|
.expect("heal capture must remain connected");
|
||||||
|
assert_eq!(heal.bucket, bucket);
|
||||||
|
assert_eq!(heal.object_prefix.as_deref(), Some(object));
|
||||||
|
let info = set
|
||||||
|
.get_object_info(bucket, object, &ObjectOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("committed object must remain readable despite the failed tail");
|
||||||
|
assert_eq!(info.size, TEST_OBJECT_SIZE as i64);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(capacity_dirty_scope)]
|
||||||
|
async fn tail_drained_put_rejects_quorum_minus_one() {
|
||||||
|
let (_dirs, disks, set) = hermetic_set_disks(4).await;
|
||||||
|
let bucket = "put-full-tail-no-quorum";
|
||||||
|
let object = "full-tail-no-quorum-object";
|
||||||
|
make_completion_test_bucket(&disks, bucket).await;
|
||||||
|
let _fault = rename_fault_injection::fail_rename_on(object, &[0, 1]);
|
||||||
|
let tasks = rename_fanout_barrier::observe_tasks(object);
|
||||||
|
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
|
||||||
|
let err = set
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect_err("draining two successful disks cannot satisfy write quorum three");
|
||||||
|
assert!(
|
||||||
|
matches!(err, Error::ErasureWriteQuorum | Error::InsufficientWriteQuorum(_, _)),
|
||||||
|
"original quorum error expected: {err}"
|
||||||
|
);
|
||||||
|
assert_eq!(tasks.running(), 0, "failed fan-out and rollback must complete before return");
|
||||||
|
assert!(
|
||||||
|
set.get_object_info(bucket, object, &ObjectOptions::default()).await.is_err(),
|
||||||
|
"failed fresh write must not become visible"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(capacity_dirty_scope)]
|
||||||
|
async fn put_incomplete_rollback_preserves_staging_and_old_version_backup() {
|
||||||
|
use crate::set_disk::core::io_primitives::rollback_fault_injection;
|
||||||
|
|
||||||
|
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async {
|
||||||
|
for write_completion in [WriteCompletion::Quorum, WriteCompletion::TailDrained] {
|
||||||
|
for fault in [
|
||||||
|
rollback_fault_injection::Fault::Io,
|
||||||
|
rollback_fault_injection::Fault::VolumeNotFoundAfterRename,
|
||||||
|
] {
|
||||||
|
let (dirs, disks, set) = hermetic_set_disks(4).await;
|
||||||
|
let bucket = "put-incomplete-undo";
|
||||||
|
let object = "incomplete-undo-object";
|
||||||
|
make_completion_test_bucket(&disks, bucket).await;
|
||||||
|
let mut old_reader = PutObjReader::from_vec(vec![b'0'; TEST_OBJECT_SIZE]);
|
||||||
|
set.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut old_reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("old generation should be completely committed");
|
||||||
|
wait_for_tmp_workspace_to_drain(&dirs, "old PUT must leave no unrelated staging").await;
|
||||||
|
let old = disks[0]
|
||||||
|
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("old metadata must be readable");
|
||||||
|
let old_data_dir = old.data_dir.expect("non-inline old version needs a data directory");
|
||||||
|
let tasks = rename_fanout_barrier::observe_tasks(object);
|
||||||
|
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
|
||||||
|
let _rename_fault = rename_fault_injection::fail_rename_on(object, &[2, 3]);
|
||||||
|
let _undo_fault = rollback_fault_injection::arm(object, 0, fault);
|
||||||
|
let writer = Arc::clone(&set);
|
||||||
|
let put = tokio::spawn(async move {
|
||||||
|
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
|
||||||
|
writer
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
});
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
|
||||||
|
.await
|
||||||
|
.expect("overwrite must enter the actual rename fan-out before failure injection");
|
||||||
|
barrier.release();
|
||||||
|
let err = tokio::time::timeout(Duration::from_secs(30), put)
|
||||||
|
.await
|
||||||
|
.expect("incomplete undo must return without hanging")
|
||||||
|
.expect("PUT task should join")
|
||||||
|
.expect_err("two renamed disks cannot satisfy write quorum three");
|
||||||
|
assert!(
|
||||||
|
matches!(err, Error::ErasureWriteQuorum | Error::InsufficientWriteQuorum(_, _)),
|
||||||
|
"original quorum error expected: {err}"
|
||||||
|
);
|
||||||
|
assert_eq!(tasks.running(), 0, "every rename and undo task must be reaped before return");
|
||||||
|
let leftovers = non_trash_tmp_entries(&dirs).await;
|
||||||
|
assert!(!leftovers.is_empty(), "incomplete undo must retain the new staging source for recovery");
|
||||||
|
let backups = dirs
|
||||||
|
.iter()
|
||||||
|
.filter(|dir| {
|
||||||
|
dir.path()
|
||||||
|
.join(bucket)
|
||||||
|
.join(object)
|
||||||
|
.join(old_data_dir.to_string())
|
||||||
|
.join(crate::disk::STORAGE_FORMAT_FILE_BACKUP)
|
||||||
|
.exists()
|
||||||
|
})
|
||||||
|
.count();
|
||||||
|
assert_eq!(backups, 1, "exactly the failed undo disk must retain its old-version backup");
|
||||||
|
// The remaining three disks still serve the old generation;
|
||||||
|
// the failed minority must never become an acknowledged write.
|
||||||
|
let mut read = set
|
||||||
|
.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("old generation must remain readable after incomplete rollback");
|
||||||
|
let mut body = Vec::new();
|
||||||
|
read.stream
|
||||||
|
.read_to_end(&mut body)
|
||||||
|
.await
|
||||||
|
.expect("old generation should stream");
|
||||||
|
assert_eq!(body, vec![b'0'; TEST_OBJECT_SIZE]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(capacity_dirty_scope)]
|
||||||
|
async fn tail_drained_put_owned_commit_survives_waiter_cancellation() {
|
||||||
|
let (dirs, disks, set) = hermetic_set_disks(4).await;
|
||||||
|
let bucket = RUSTFS_META_BUCKET;
|
||||||
|
let object = "full-tail-cancelled-receipt";
|
||||||
|
// Internal config writes do not own a bucket lifecycle guard. The object
|
||||||
|
// guard alone must keep the full-tail coordinator alive after cancellation.
|
||||||
|
let tasks = rename_fanout_barrier::observe_tasks(object);
|
||||||
|
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
|
||||||
|
let writer = Arc::clone(&set);
|
||||||
|
let put = tokio::spawn(async move {
|
||||||
|
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
|
||||||
|
writer
|
||||||
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
});
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
|
||||||
|
.await
|
||||||
|
.expect("cancelled receipt must first reach the rename barrier");
|
||||||
|
wait_for_paused_tail_metadata_quorum(&disks, bucket, object).await;
|
||||||
|
put.abort();
|
||||||
|
assert!(put.await.expect_err("ACK waiter should cancel").is_cancelled());
|
||||||
|
let mut lock_probe = Box::pin(set.acquire_write_lock_diag("cancelled_full_tail_probe", bucket, object));
|
||||||
|
assert!(
|
||||||
|
futures::poll!(lock_probe.as_mut()).is_pending(),
|
||||||
|
"owned coordinator must retain the namespace guard after waiter cancellation"
|
||||||
|
);
|
||||||
|
barrier.release();
|
||||||
|
drop(
|
||||||
|
tokio::time::timeout(Duration::from_secs(30), lock_probe)
|
||||||
|
.await
|
||||||
|
.expect("cancelled coordinator must eventually release its guard")
|
||||||
|
.expect("post-commit lock probe should succeed"),
|
||||||
|
);
|
||||||
|
assert_eq!(tasks.running(), 0, "cancelled coordinator must reap every rename task");
|
||||||
|
for disk in &disks {
|
||||||
|
disk.read_version("", bucket, object, "", &ReadOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("caller cancellation must not interrupt committed receipt materialization");
|
||||||
|
}
|
||||||
|
wait_for_tmp_workspace_to_drain(&dirs, "cancelled full-tail commit should release staging ownership").await;
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial_test::serial(capacity_dirty_scope)]
|
#[serial_test::serial(capacity_dirty_scope)]
|
||||||
async fn no_lock_put_waits_for_rename_tail_under_outer_guard() {
|
async fn no_lock_put_waits_for_rename_tail_under_outer_guard() {
|
||||||
@@ -18184,6 +18585,7 @@ mod put_object_tmp_cleanup_tests {
|
|||||||
&mut reader,
|
&mut reader,
|
||||||
&ObjectOptions {
|
&ObjectOptions {
|
||||||
no_lock: true,
|
no_lock: true,
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
..Default::default()
|
..Default::default()
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
@@ -18209,7 +18611,18 @@ mod put_object_tmp_cleanup_tests {
|
|||||||
put.await
|
put.await
|
||||||
.expect("no-lock PUT task should join")
|
.expect("no-lock PUT task should join")
|
||||||
.expect("no-lock PUT should commit after the rename tail releases");
|
.expect("no-lock PUT should commit after the rename tail releases");
|
||||||
|
let mut lock_probe = Box::pin(set_disks.acquire_write_lock_diag("borrowed_full_tail_probe", bucket, object));
|
||||||
|
assert!(
|
||||||
|
futures::poll!(lock_probe.as_mut()).is_pending(),
|
||||||
|
"full-tail PUT must not release the caller's outer guard"
|
||||||
|
);
|
||||||
drop(outer_guard);
|
drop(outer_guard);
|
||||||
|
drop(
|
||||||
|
tokio::time::timeout(Duration::from_secs(5), lock_probe)
|
||||||
|
.await
|
||||||
|
.expect("outer owner releasing its guard should unblock the probe")
|
||||||
|
.expect("post-outer-guard probe should succeed"),
|
||||||
|
);
|
||||||
})
|
})
|
||||||
.await;
|
.await;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ use super::{
|
|||||||
};
|
};
|
||||||
use crate::bucket::lifecycle::lifecycle::{TRANSITION_COMPLETE, TRANSITION_PENDING, TransitionOptions, expected_expiry_time};
|
use crate::bucket::lifecycle::lifecycle::{TRANSITION_COMPLETE, TRANSITION_PENDING, TransitionOptions, expected_expiry_time};
|
||||||
use crate::ecstore_validation_blackbox::make_local_set_disks;
|
use crate::ecstore_validation_blackbox::make_local_set_disks;
|
||||||
|
use crate::object_api::WriteCompletion;
|
||||||
use crate::services::tier::test_util::register_mock_tier;
|
use crate::services::tier::test_util::register_mock_tier;
|
||||||
use crate::storage_api_contracts::bucket::BucketOperations;
|
use crate::storage_api_contracts::bucket::BucketOperations;
|
||||||
use crate::storage_api_contracts::object::{ObjectIO as _, ObjectOperations as _};
|
use crate::storage_api_contracts::object::{ObjectIO as _, ObjectOperations as _};
|
||||||
@@ -25,19 +26,24 @@ use rustfs_filemeta::{RestoreStatusOps as _, parse_restore_obj_status};
|
|||||||
use tokio::io::AsyncReadExt;
|
use tokio::io::AsyncReadExt;
|
||||||
|
|
||||||
async fn prime_metadata_generation(set_disks: &SetDisks, bucket: &str, object: &str) -> GetObjectMetadataCacheKey {
|
async fn prime_metadata_generation(set_disks: &SetDisks, bucket: &str, object: &str) -> GetObjectMetadataCacheKey {
|
||||||
set_disks
|
tokio::time::timeout(Duration::from_secs(30), async {
|
||||||
.get_object_fileinfo(bucket, object, &ObjectOptions::default(), true, false)
|
loop {
|
||||||
.await
|
set_disks
|
||||||
.expect("object metadata should resolve");
|
.get_object_fileinfo(bucket, object, &ObjectOptions::default(), true, false)
|
||||||
let generation = set_disks
|
.await
|
||||||
.get_object_metadata_cache_generation(bucket, object)
|
.expect("object metadata should resolve");
|
||||||
.expect("metadata generation should be active");
|
let generation = set_disks
|
||||||
let key = GetObjectMetadataCacheKey::new(bucket, object, generation);
|
.get_object_metadata_cache_generation(bucket, object)
|
||||||
assert!(
|
.expect("metadata generation should be active");
|
||||||
set_disks.get_object_metadata_cache.get(&key).await.is_some(),
|
let key = GetObjectMetadataCacheKey::new(bucket, object, generation);
|
||||||
"metadata read should publish the generation under test"
|
if set_disks.get_object_metadata_cache.get(&key).await.is_some() {
|
||||||
);
|
return key;
|
||||||
key
|
}
|
||||||
|
tokio::task::yield_now().await;
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.expect("metadata read should publish the generation under test")
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn assert_generation_reclaimed(set_disks: &SetDisks, key: &GetObjectMetadataCacheKey) {
|
async fn assert_generation_reclaimed(set_disks: &SetDisks, key: &GetObjectMetadataCacheKey) {
|
||||||
@@ -60,8 +66,17 @@ async fn transition_and_restore_reclaim_prior_metadata_generations() {
|
|||||||
.await
|
.await
|
||||||
.expect("bucket should be created");
|
.expect("bucket should be created");
|
||||||
let mut reader = PutObjReader::from_vec(payload.clone());
|
let mut reader = PutObjReader::from_vec(payload.clone());
|
||||||
|
// Cache priming must not race a quorum-acknowledged PUT's remaining rename tail.
|
||||||
let original = set_disks
|
let original = set_disks
|
||||||
.put_object(bucket, object, &mut reader, &ObjectOptions::default())
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.expect("source object should be written");
|
.expect("source object should be written");
|
||||||
let source_generation = prime_metadata_generation(&set_disks, bucket, object).await;
|
let source_generation = prime_metadata_generation(&set_disks, bucket, object).await;
|
||||||
@@ -164,8 +179,17 @@ async fn prepared_snapshot_transition_duplicate_and_late_get_use_committed_remot
|
|||||||
.await
|
.await
|
||||||
.expect("bucket should be created");
|
.expect("bucket should be created");
|
||||||
let mut reader = PutObjReader::from_vec(payload.clone());
|
let mut reader = PutObjReader::from_vec(payload.clone());
|
||||||
|
// Cache priming must not race a quorum-acknowledged PUT's remaining rename tail.
|
||||||
let original = set_disks
|
let original = set_disks
|
||||||
.put_object(bucket, object, &mut reader, &ObjectOptions::default())
|
.put_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&mut reader,
|
||||||
|
&ObjectOptions {
|
||||||
|
write_completion: WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.expect("source object should be written");
|
.expect("source object should be written");
|
||||||
|
|
||||||
|
|||||||
@@ -864,6 +864,11 @@ mod tests {
|
|||||||
save_tier_mutation_intent_record, save_tier_mutation_intent_record_if_current,
|
save_tier_mutation_intent_record, save_tier_mutation_intent_record_if_current,
|
||||||
},
|
},
|
||||||
tier_mutation_peer::{TierMutationPeerError, TierMutationPeerState, handle_tier_mutation_peer_request},
|
tier_mutation_peer::{TierMutationPeerError, TierMutationPeerState, handle_tier_mutation_peer_request},
|
||||||
|
tier_probe_intent::{
|
||||||
|
TierProbeIntent, TierProbeIntentState, TierProbeOperationIdentity, TierProbeOwnerFence, TierProbeRemoteVersion,
|
||||||
|
delete_tier_probe_intent_record_if_current, load_tier_probe_intent_record,
|
||||||
|
save_tier_probe_intent_record_if_absent, save_tier_probe_intent_record_if_current,
|
||||||
|
},
|
||||||
warm_backend::{TransitionCandidateProbe, WarmBackend},
|
warm_backend::{TransitionCandidateProbe, WarmBackend},
|
||||||
},
|
},
|
||||||
set_disk::SetDiskTransitionUploadedCommitBarrier as TransitionUploadedCommitBarrier,
|
set_disk::SetDiskTransitionUploadedCommitBarrier as TransitionUploadedCommitBarrier,
|
||||||
@@ -2974,6 +2979,33 @@ mod tests {
|
|||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
const DECOMMISSION_TEST_FAULT_STAGE_TIERED: &str = "decommission_tiered_object";
|
const DECOMMISSION_TEST_FAULT_STAGE_TIERED: &str = "decommission_tiered_object";
|
||||||
|
|
||||||
|
fn decommission_retry_fault_hook(
|
||||||
|
bucket: &str,
|
||||||
|
object: &str,
|
||||||
|
faults: Arc<AtomicUsize>,
|
||||||
|
) -> crate::core::pools::DecommissionTestFaultDecision {
|
||||||
|
let target_bucket = bucket.to_string();
|
||||||
|
let target_object = object.to_string();
|
||||||
|
Arc::new(move |stage, bucket, object, _attempt, succeeded| {
|
||||||
|
if !succeeded
|
||||||
|
|| stage != DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT
|
||||||
|
|| bucket != target_bucket
|
||||||
|
|| object != target_object
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Entry retries reset the local attempt; real copy errors can skip
|
||||||
|
// successful attempts. Only injected faults spend this global budget.
|
||||||
|
faults
|
||||||
|
.fetch_update(Ordering::SeqCst, Ordering::SeqCst, |faults| {
|
||||||
|
(faults < crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS.saturating_sub(1))
|
||||||
|
.then_some(faults.saturating_add(1))
|
||||||
|
})
|
||||||
|
.is_ok()
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
async fn seed_decommission_source(
|
async fn seed_decommission_source(
|
||||||
store: &Arc<crate::store::ECStore>,
|
store: &Arc<crate::store::ECStore>,
|
||||||
bucket: &str,
|
bucket: &str,
|
||||||
@@ -5115,6 +5147,33 @@ mod tests {
|
|||||||
shutdown.cancel();
|
shutdown.cancel();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decommission_retry_fault_budget_counts_successes_across_attempt_changes() {
|
||||||
|
for attempts in [[1, 2, 3], [1, 1, 2], [1, 3, 3]] {
|
||||||
|
let faults = Arc::new(AtomicUsize::new(0));
|
||||||
|
let hook = decommission_retry_fault_hook("bucket", "object", Arc::clone(&faults));
|
||||||
|
|
||||||
|
for (stage, bucket, object, succeeded) in [
|
||||||
|
("other-stage", "bucket", "object", true),
|
||||||
|
(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "other-bucket", "object", true),
|
||||||
|
(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "bucket", "other-object", true),
|
||||||
|
(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "bucket", "object", false),
|
||||||
|
] {
|
||||||
|
assert!(!hook(stage, bucket, object, 1, succeeded));
|
||||||
|
}
|
||||||
|
assert_eq!(faults.load(Ordering::SeqCst), 0, "unrelated or failed copies must not consume faults");
|
||||||
|
|
||||||
|
for (index, attempt) in attempts.into_iter().enumerate() {
|
||||||
|
assert_eq!(
|
||||||
|
hook(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "bucket", "object", attempt, true),
|
||||||
|
index < 2,
|
||||||
|
"attempts={attempts:?}, index={index}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
assert_eq!(faults.load(Ordering::SeqCst), 2, "attempts={attempts:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[serial_test::serial(storage_class_env)]
|
#[serial_test::serial(storage_class_env)]
|
||||||
fn decommission_entry_retries_source_changed_without_canceling_other_bucket() {
|
fn decommission_entry_retries_source_changed_without_canceling_other_bucket() {
|
||||||
@@ -5209,31 +5268,8 @@ mod tests {
|
|||||||
));
|
));
|
||||||
|
|
||||||
let ordinary_faults = Arc::new(AtomicUsize::new(0));
|
let ordinary_faults = Arc::new(AtomicUsize::new(0));
|
||||||
let ordinary_faults_for_hook = Arc::clone(&ordinary_faults);
|
let fault_hook = decommission_retry_fault_hook(&other_bucket, other_object, Arc::clone(&ordinary_faults));
|
||||||
let fault_bucket = other_bucket.clone();
|
let _fault_guard = crate::core::pools::DecommissionTestFaultGuard::install(fault_hook);
|
||||||
let _fault_guard = crate::core::pools::DecommissionTestFaultGuard::install(Arc::new(
|
|
||||||
move |stage, bucket, object, attempt, succeeded| {
|
|
||||||
let candidate = succeeded
|
|
||||||
&& stage == DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT
|
|
||||||
&& bucket == fault_bucket.as_str()
|
|
||||||
&& object == other_object;
|
|
||||||
if !candidate {
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Keep the fault budget global across any
|
|
||||||
// entry-level re-list; its inner attempt counter
|
|
||||||
// restarts after SourceChanged.
|
|
||||||
ordinary_faults_for_hook
|
|
||||||
.fetch_update(Ordering::SeqCst, Ordering::SeqCst, |faults| {
|
|
||||||
let next_fault = faults.saturating_add(1);
|
|
||||||
(faults < crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS.saturating_sub(1)
|
|
||||||
&& attempt == next_fault)
|
|
||||||
.then_some(next_fault)
|
|
||||||
})
|
|
||||||
.is_ok()
|
|
||||||
},
|
|
||||||
));
|
|
||||||
|
|
||||||
let rx = CancellationToken::new();
|
let rx = CancellationToken::new();
|
||||||
let source_changed_exhaustions = Arc::new(AtomicUsize::new(0));
|
let source_changed_exhaustions = Arc::new(AtomicUsize::new(0));
|
||||||
@@ -8040,10 +8076,15 @@ mod tests {
|
|||||||
);
|
);
|
||||||
assert!(com::read_config(store.pools[0].clone(), &second_page_path).await.is_ok());
|
assert!(com::read_config(store.pools[0].clone(), &second_page_path).await.is_ok());
|
||||||
|
|
||||||
com::save_config(store.pools[target_pool_idx].clone(), &second_page_path, receipt_bytes.clone())
|
let full_tail = ObjectOptions {
|
||||||
|
max_parity: true,
|
||||||
|
write_completion: crate::object_api::WriteCompletion::TailDrained,
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
com::save_config_with_opts(store.pools[target_pool_idx].clone(), &second_page_path, receipt_bytes.clone(), &full_tail)
|
||||||
.await
|
.await
|
||||||
.expect("second page receipt should restore");
|
.expect("second page receipt should restore");
|
||||||
com::save_config(store.pools[target_pool_idx].clone(), &second_page_path, b"{corrupt".to_vec())
|
com::save_config_with_opts(store.pools[target_pool_idx].clone(), &second_page_path, b"{corrupt".to_vec(), &full_tail)
|
||||||
.await
|
.await
|
||||||
.expect("second page receipt should corrupt deterministically");
|
.expect("second page receipt should corrupt deterministically");
|
||||||
let corrupt = store
|
let corrupt = store
|
||||||
@@ -11570,6 +11611,7 @@ mod tests {
|
|||||||
pool_index: usize,
|
pool_index: usize,
|
||||||
bucket: &str,
|
bucket: &str,
|
||||||
object: &str,
|
object: &str,
|
||||||
|
minio_unversioned: bool,
|
||||||
) {
|
) {
|
||||||
for disk_index in 0..4 {
|
for disk_index in 0..4 {
|
||||||
let metadata_path =
|
let metadata_path =
|
||||||
@@ -11603,6 +11645,11 @@ mod tests {
|
|||||||
] {
|
] {
|
||||||
rustfs_utils::http::metadata_compat::remove_bytes(&mut object_meta.meta_sys, suffix);
|
rustfs_utils::http::metadata_compat::remove_bytes(&mut object_meta.meta_sys, suffix);
|
||||||
}
|
}
|
||||||
|
if minio_unversioned {
|
||||||
|
object_meta
|
||||||
|
.meta_sys
|
||||||
|
.insert("x-minio-internal-transitioned-versionID".to_string(), Vec::new());
|
||||||
|
}
|
||||||
*shallow = rustfs_filemeta::FileMetaShallowVersion::try_from(version)
|
*shallow = rustfs_filemeta::FileMetaShallowVersion::try_from(version)
|
||||||
.expect("legacy transitioned version should re-encode");
|
.expect("legacy transitioned version should re-encode");
|
||||||
}
|
}
|
||||||
@@ -11613,6 +11660,152 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
async fn read_store_body(
|
||||||
|
store: &Arc<crate::store::ECStore>,
|
||||||
|
bucket: &str,
|
||||||
|
object: &str,
|
||||||
|
range: Option<HTTPRangeSpec>,
|
||||||
|
opts: &ObjectOptions,
|
||||||
|
) -> Vec<u8> {
|
||||||
|
let mut reader = store
|
||||||
|
.get_object_reader(bucket, object, range, HeaderMap::new(), opts)
|
||||||
|
.await
|
||||||
|
.expect("object reader should open");
|
||||||
|
let mut body = Vec::new();
|
||||||
|
reader.stream.read_to_end(&mut body).await.expect("object body should drain");
|
||||||
|
body
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(storage_class_env)]
|
||||||
|
async fn legacy_unknown_unversioned_transition_supports_head_get_and_range_without_backfill() {
|
||||||
|
let temp_dir = tempfile::tempdir().expect("create legacy unknown unversioned store dir");
|
||||||
|
let (ctx, store, _shutdown) =
|
||||||
|
without_storage_class_env(build_isolated_test_store(temp_dir.path(), "legacy-unknown-unversioned-read", &[4])).await;
|
||||||
|
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
|
||||||
|
let tier_name = "LEGACY-UNKNOWN-UNVERSIONED-READ";
|
||||||
|
let backend = register_mock_tier(&ctx.tier_config_mgr(), tier_name).await;
|
||||||
|
backend.set_put_remote_version(Some(String::new())).await;
|
||||||
|
let bucket = "legacy-unknown-unversioned-read-bucket";
|
||||||
|
let object = "object.bin";
|
||||||
|
let payload = b"legacy unversioned remote tier object remains readable".repeat(1024);
|
||||||
|
store
|
||||||
|
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("legacy source bucket should be created");
|
||||||
|
let mut reader = PutObjReader::from_vec(payload.clone());
|
||||||
|
let source = store
|
||||||
|
.put_object(bucket, object, &mut reader, &ObjectOptions::default())
|
||||||
|
.await
|
||||||
|
.expect("legacy source should be written");
|
||||||
|
store
|
||||||
|
.transition_object(
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
&ObjectOptions {
|
||||||
|
transition: TransitionOptions {
|
||||||
|
status: TRANSITION_PENDING.to_string(),
|
||||||
|
tier: tier_name.to_string(),
|
||||||
|
etag: source.etag.clone().expect("legacy source should have an etag"),
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
mod_time: source.mod_time,
|
||||||
|
..Default::default()
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.expect("legacy source should transition");
|
||||||
|
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 0, bucket, object, true).await;
|
||||||
|
backend.clear_op_log().await;
|
||||||
|
|
||||||
|
let opts = ObjectOptions {
|
||||||
|
metadata_cache_safe: false,
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let head = store
|
||||||
|
.get_object_info(bucket, object, &opts)
|
||||||
|
.await
|
||||||
|
.expect("legacy transitioned HEAD should use local metadata");
|
||||||
|
assert_eq!(head.transition_version_state, rustfs_filemeta::TransitionVersionState::Unknown);
|
||||||
|
assert!(head.transitioned_object.version_id.is_empty());
|
||||||
|
assert_eq!(
|
||||||
|
head.user_defined
|
||||||
|
.get("x-minio-internal-transitioned-versionID")
|
||||||
|
.map(String::as_str),
|
||||||
|
Some(""),
|
||||||
|
"the MinIO empty version-key provenance must survive xl.meta decoding"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!rustfs_utils::http::metadata_compat::contains_key_str(
|
||||||
|
&head.user_defined,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
|
||||||
|
),
|
||||||
|
"the compatibility read must not synthesize version-state metadata"
|
||||||
|
);
|
||||||
|
|
||||||
|
let full_body = read_store_body(&store, bucket, object, None, &opts).await;
|
||||||
|
assert_eq!(full_body, payload);
|
||||||
|
|
||||||
|
let range = HTTPRangeSpec {
|
||||||
|
is_suffix_length: false,
|
||||||
|
start: 7,
|
||||||
|
end: 38,
|
||||||
|
};
|
||||||
|
let ranged_body = read_store_body(&store, bucket, object, Some(range), &opts).await;
|
||||||
|
assert_eq!(ranged_body, &payload[7..=38]);
|
||||||
|
|
||||||
|
let after_read = store.pools[0]
|
||||||
|
.get_disks_by_key(object)
|
||||||
|
.load_file_info_versions_exact(bucket, object)
|
||||||
|
.await
|
||||||
|
.expect("legacy metadata should remain readable after GET")
|
||||||
|
.expect("legacy object metadata should remain on disk")
|
||||||
|
.versions
|
||||||
|
.into_iter()
|
||||||
|
.find(|version| version.transition_status == rustfs_filemeta::TRANSITION_COMPLETE)
|
||||||
|
.expect("legacy transitioned source should remain visible after GET");
|
||||||
|
assert_eq!(after_read.transition_version_state, rustfs_filemeta::TransitionVersionState::Unknown);
|
||||||
|
assert!(after_read.transition_version.is_none());
|
||||||
|
assert!(after_read.transition_version_id.is_none());
|
||||||
|
assert_eq!(
|
||||||
|
after_read
|
||||||
|
.metadata
|
||||||
|
.get("x-minio-internal-transitioned-versionID")
|
||||||
|
.map(String::as_str),
|
||||||
|
Some(""),
|
||||||
|
"the MinIO empty version-key provenance must remain after GET and Range GET"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!rustfs_utils::http::metadata_compat::contains_key_str(
|
||||||
|
&after_read.metadata,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
|
||||||
|
),
|
||||||
|
"the compatibility read must remain side-effect free"
|
||||||
|
);
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
backend.op_log().await,
|
||||||
|
vec![
|
||||||
|
MockWarmOp::Probe {
|
||||||
|
object: after_read.transitioned_objname.clone(),
|
||||||
|
},
|
||||||
|
MockWarmOp::Get {
|
||||||
|
object: after_read.transitioned_objname.clone(),
|
||||||
|
},
|
||||||
|
MockWarmOp::Probe {
|
||||||
|
object: after_read.transitioned_objname.clone(),
|
||||||
|
},
|
||||||
|
MockWarmOp::Get {
|
||||||
|
object: after_read.transitioned_objname,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
"legacy reads should probe before each unversioned GET and never mutate local metadata"
|
||||||
|
);
|
||||||
|
assert_eq!(backend.remove_count().await, 0);
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial_test::serial(storage_class_env)]
|
#[serial_test::serial(storage_class_env)]
|
||||||
@@ -11653,7 +11846,7 @@ mod tests {
|
|||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
.expect("legacy source should transition");
|
.expect("legacy source should transition");
|
||||||
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 0, bucket, object).await;
|
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 0, bucket, object, false).await;
|
||||||
let legacy = store.pools[0]
|
let legacy = store.pools[0]
|
||||||
.get_disks_by_key(object)
|
.get_disks_by_key(object)
|
||||||
.load_file_info_versions_exact(bucket, object)
|
.load_file_info_versions_exact(bucket, object)
|
||||||
@@ -12794,7 +12987,7 @@ mod tests {
|
|||||||
.expect("merge-loser source should transition");
|
.expect("merge-loser source should transition");
|
||||||
copy_test_xlmeta_between_pools(temp_dir.path(), 0, 1, bucket, object).await;
|
copy_test_xlmeta_between_pools(temp_dir.path(), 0, 1, bucket, object).await;
|
||||||
}
|
}
|
||||||
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 1, bucket, "legacy/item.bin").await;
|
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 1, bucket, "legacy/item.bin", false).await;
|
||||||
backend.set_remove_failure(true);
|
backend.set_remove_failure(true);
|
||||||
store.pools[1]
|
store.pools[1]
|
||||||
.delete_object(bucket, "hidden/item.bin", ObjectOptions::default())
|
.delete_object(bucket, "hidden/item.bin", ObjectOptions::default())
|
||||||
@@ -16861,6 +17054,10 @@ mod tests {
|
|||||||
.find(|version| version.version_id == history.version_id)
|
.find(|version| version.version_id == history.version_id)
|
||||||
.expect("transitioned history should exist");
|
.expect("transitioned history should exist");
|
||||||
transitioned.transition_version_state = rustfs_filemeta::TransitionVersionState::Unknown;
|
transitioned.transition_version_state = rustfs_filemeta::TransitionVersionState::Unknown;
|
||||||
|
rustfs_utils::http::metadata_compat::remove_str(
|
||||||
|
&mut transitioned.metadata,
|
||||||
|
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
|
||||||
|
);
|
||||||
metadata
|
metadata
|
||||||
.add_version(transitioned)
|
.add_version(transitioned)
|
||||||
.expect("unknown state should replace the transitioned version");
|
.expect("unknown state should replace the transitioned version");
|
||||||
@@ -17196,6 +17393,147 @@ mod tests {
|
|||||||
assert!(matches!(err, Error::ConfigNotFound));
|
assert!(matches!(err, Error::ConfigNotFound));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "test-util")]
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial_test::serial(storage_class_env)]
|
||||||
|
async fn tier_probe_intent_store_enforces_create_cas_and_terminal_delete_preconditions() {
|
||||||
|
let temp_dir = tempfile::tempdir().expect("create temp store dir");
|
||||||
|
let (_ctx, store, _shutdown) =
|
||||||
|
without_storage_class_env(build_isolated_test_store(temp_dir.path(), "tier-probe-intent-cas", &[4])).await;
|
||||||
|
let probe_id = uuid::Uuid::new_v4();
|
||||||
|
let creator_epoch = uuid::Uuid::new_v4();
|
||||||
|
let initial = TierProbeIntent {
|
||||||
|
probe_id,
|
||||||
|
revision: 1,
|
||||||
|
state: TierProbeIntentState::UploadOutcomeUnknown,
|
||||||
|
operation: TierProbeOperationIdentity::Verify {
|
||||||
|
config_etag: "config-etag".to_string(),
|
||||||
|
backend_identity: [1; 32],
|
||||||
|
},
|
||||||
|
tier_name: "COLD-A".to_string(),
|
||||||
|
destination_id: [1; 32],
|
||||||
|
probe_object: format!("rustfs-tier-probe-{probe_id}"),
|
||||||
|
creator_id: "node-a".to_string(),
|
||||||
|
creator_epoch,
|
||||||
|
created_at_unix_nanos: 1_780_000_000_000_000_000,
|
||||||
|
owner: TierProbeOwnerFence {
|
||||||
|
owner_id: "node-a".to_string(),
|
||||||
|
owner_epoch: creator_epoch,
|
||||||
|
not_after_unix_nanos: 1_780_000_900_000_000_000,
|
||||||
|
},
|
||||||
|
remote_version: TierProbeRemoteVersion::default(),
|
||||||
|
};
|
||||||
|
|
||||||
|
save_tier_probe_intent_record_if_absent(store.clone(), &initial)
|
||||||
|
.await
|
||||||
|
.expect("initial probe intent should persist with create-only semantics");
|
||||||
|
let duplicate = save_tier_probe_intent_record_if_absent(store.clone(), &initial)
|
||||||
|
.await
|
||||||
|
.expect_err("duplicate create must fail closed");
|
||||||
|
assert!(matches!(duplicate, Error::PreconditionFailed));
|
||||||
|
|
||||||
|
let observed_initial = load_tier_probe_intent_record(store.clone(), probe_id)
|
||||||
|
.await
|
||||||
|
.expect("initial probe intent should load with an ETag");
|
||||||
|
assert_eq!(observed_initial.intent(), &initial);
|
||||||
|
|
||||||
|
let nonterminal_delete = delete_tier_probe_intent_record_if_current(store.clone(), &observed_initial)
|
||||||
|
.await
|
||||||
|
.expect_err("nonterminal evidence must not be deleted");
|
||||||
|
assert!(nonterminal_delete.to_string().contains("must be terminal"));
|
||||||
|
|
||||||
|
let mut fabricated_current_intent = initial.clone();
|
||||||
|
fabricated_current_intent.tier_name = "COLD-B".to_string();
|
||||||
|
let mut fabricated_successor = fabricated_current_intent.clone();
|
||||||
|
fabricated_successor
|
||||||
|
.advance(
|
||||||
|
TierProbeIntentState::Uploaded,
|
||||||
|
TierProbeRemoteVersion::versioned(uuid::Uuid::new_v4().to_string()),
|
||||||
|
)
|
||||||
|
.expect("fabricated successor should be internally valid");
|
||||||
|
let fabricated_current = observed_initial.with_intent_for_test(fabricated_current_intent.clone());
|
||||||
|
let crossed_cas = save_tier_probe_intent_record_if_current(store.clone(), &fabricated_current, &fabricated_successor)
|
||||||
|
.await
|
||||||
|
.expect_err("a live ETag must not authorize a different caller record");
|
||||||
|
assert!(matches!(crossed_cas, Error::PreconditionFailed));
|
||||||
|
assert_eq!(
|
||||||
|
load_tier_probe_intent_record(store.clone(), probe_id)
|
||||||
|
.await
|
||||||
|
.expect("crossed CAS must retain the authoritative record")
|
||||||
|
.intent(),
|
||||||
|
&initial
|
||||||
|
);
|
||||||
|
|
||||||
|
let mut fabricated_terminal_intent = fabricated_current_intent;
|
||||||
|
fabricated_terminal_intent
|
||||||
|
.advance(TierProbeIntentState::AbortedNoRemote, TierProbeRemoteVersion::default())
|
||||||
|
.expect("fabricated terminal should be internally valid");
|
||||||
|
let fabricated_terminal = observed_initial.with_intent_for_test(fabricated_terminal_intent);
|
||||||
|
let crossed_delete = delete_tier_probe_intent_record_if_current(store.clone(), &fabricated_terminal)
|
||||||
|
.await
|
||||||
|
.expect_err("a live ETag must not delete for a different caller record");
|
||||||
|
assert!(matches!(crossed_delete, Error::PreconditionFailed));
|
||||||
|
assert_eq!(
|
||||||
|
load_tier_probe_intent_record(store.clone(), probe_id)
|
||||||
|
.await
|
||||||
|
.expect("crossed delete must retain the authoritative record")
|
||||||
|
.intent(),
|
||||||
|
&initial
|
||||||
|
);
|
||||||
|
|
||||||
|
let remote_version = TierProbeRemoteVersion::versioned(uuid::Uuid::new_v4().to_string());
|
||||||
|
let mut uploaded = observed_initial.intent().clone();
|
||||||
|
uploaded
|
||||||
|
.advance(TierProbeIntentState::Uploaded, remote_version.clone())
|
||||||
|
.expect("known PUT result should advance");
|
||||||
|
save_tier_probe_intent_record_if_current(store.clone(), &observed_initial, &uploaded)
|
||||||
|
.await
|
||||||
|
.expect("the matching initial ETag should admit one successor");
|
||||||
|
|
||||||
|
let stale_cas = save_tier_probe_intent_record_if_current(store.clone(), &observed_initial, &uploaded)
|
||||||
|
.await
|
||||||
|
.expect_err("a consumed ETag must not overwrite the current generation");
|
||||||
|
assert!(matches!(stale_cas, Error::PreconditionFailed));
|
||||||
|
|
||||||
|
let observed_uploaded = load_tier_probe_intent_record(store.clone(), probe_id)
|
||||||
|
.await
|
||||||
|
.expect("uploaded generation should load");
|
||||||
|
assert_eq!(observed_uploaded.intent(), &uploaded);
|
||||||
|
let mut cleanup = observed_uploaded.intent().clone();
|
||||||
|
cleanup
|
||||||
|
.advance(TierProbeIntentState::CleanupPending, remote_version.clone())
|
||||||
|
.expect("known candidate should become cleanup-pending");
|
||||||
|
save_tier_probe_intent_record_if_current(store.clone(), &observed_uploaded, &cleanup)
|
||||||
|
.await
|
||||||
|
.expect("cleanup generation should persist by exact ETag");
|
||||||
|
|
||||||
|
let observed_cleanup = load_tier_probe_intent_record(store.clone(), probe_id)
|
||||||
|
.await
|
||||||
|
.expect("cleanup generation should load");
|
||||||
|
let mut completed = observed_cleanup.intent().clone();
|
||||||
|
completed
|
||||||
|
.advance(TierProbeIntentState::Completed, remote_version)
|
||||||
|
.expect("exact cleanup should become terminal");
|
||||||
|
save_tier_probe_intent_record_if_current(store.clone(), &observed_cleanup, &completed)
|
||||||
|
.await
|
||||||
|
.expect("terminal generation should persist by exact ETag");
|
||||||
|
|
||||||
|
let stale_terminal = observed_cleanup.with_intent_for_test(completed.clone());
|
||||||
|
let stale_delete = delete_tier_probe_intent_record_if_current(store.clone(), &stale_terminal)
|
||||||
|
.await
|
||||||
|
.expect_err("a stale ETag must not delete terminal evidence");
|
||||||
|
assert!(matches!(stale_delete, Error::PreconditionFailed));
|
||||||
|
|
||||||
|
let observed_completed = load_tier_probe_intent_record(store.clone(), probe_id)
|
||||||
|
.await
|
||||||
|
.expect("terminal generation should remain after stale delete");
|
||||||
|
assert_eq!(observed_completed.intent(), &completed);
|
||||||
|
delete_tier_probe_intent_record_if_current(store.clone(), &observed_completed)
|
||||||
|
.await
|
||||||
|
.expect("the exact terminal ETag should delete the record");
|
||||||
|
assert!(matches!(load_tier_probe_intent_record(store, probe_id).await, Err(Error::ConfigNotFound)));
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(feature = "test-util")]
|
#[cfg(feature = "test-util")]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
#[serial_test::serial(storage_class_env)]
|
#[serial_test::serial(storage_class_env)]
|
||||||
|
|||||||
@@ -425,7 +425,7 @@ pub(crate) mod init_format;
|
|||||||
pub(crate) mod list_objects;
|
pub(crate) mod list_objects;
|
||||||
mod multipart;
|
mod multipart;
|
||||||
mod object;
|
mod object;
|
||||||
#[cfg(any(test, feature = "test-util"))]
|
#[cfg(feature = "test-util")]
|
||||||
pub use object::DeleteAfterObjectLockSnapshotBarrier;
|
pub use object::DeleteAfterObjectLockSnapshotBarrier;
|
||||||
pub(crate) use object::{
|
pub(crate) use object::{
|
||||||
DecommissionFixedReadAnchor, ObjectLockDiagGuard, RemoteTuplePublicationCommitGuard, RemoteTuplePublicationFence,
|
DecommissionFixedReadAnchor, ObjectLockDiagGuard, RemoteTuplePublicationCommitGuard, RemoteTuplePublicationFence,
|
||||||
|
|||||||
@@ -297,6 +297,20 @@ fn transitioned_version_from_bytes(value: Option<&[u8]>, state: TransitionVersio
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn transition_version_metadata_value(raw: &[u8], decoded: Option<&str>) -> String {
|
||||||
|
decoded.map(str::to_owned).unwrap_or_else(|| {
|
||||||
|
if raw.is_empty() {
|
||||||
|
String::new()
|
||||||
|
} else {
|
||||||
|
String::from_utf8_lossy(raw).into_owned()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn is_transition_version_metadata_key(key: &str) -> bool {
|
||||||
|
strip_internal_prefix_preserving_case(key).is_some_and(|suffix| suffix.eq_ignore_ascii_case(SUFFIX_TRANSITIONED_VERSION_ID))
|
||||||
|
}
|
||||||
|
|
||||||
fn validate_transition_version_state(state: TransitionVersionState, version: Option<&str>) -> Result<()> {
|
fn validate_transition_version_state(state: TransitionVersionState, version: Option<&str>) -> Result<()> {
|
||||||
let valid = match state {
|
let valid = match state {
|
||||||
TransitionVersionState::Unknown | TransitionVersionState::KnownDisabled => version.is_none(),
|
TransitionVersionState::Unknown | TransitionVersionState::KnownDisabled => version.is_none(),
|
||||||
@@ -366,14 +380,26 @@ impl<'a> DerivedInternalMetadata<'a> {
|
|||||||
}
|
}
|
||||||
*slot = Some(value.as_slice());
|
*slot = Some(value.as_slice());
|
||||||
}
|
}
|
||||||
|
fn merge_consistent<'a>(canonical: Option<&'a [u8]>, legacy: Option<&'a [u8]>) -> Result<Option<&'a [u8]>> {
|
||||||
|
if let (Some(canonical), Some(legacy)) = (canonical, legacy)
|
||||||
|
&& canonical != legacy
|
||||||
|
{
|
||||||
|
return Err(Error::FileCorrupt);
|
||||||
|
}
|
||||||
|
Ok(canonical.or(legacy))
|
||||||
|
}
|
||||||
|
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
checksum: canonical.checksum.or(legacy.checksum),
|
checksum: canonical.checksum.or(legacy.checksum),
|
||||||
part_checksums: canonical.part_checksums.or(legacy.part_checksums),
|
part_checksums: canonical.part_checksums.or(legacy.part_checksums),
|
||||||
transition_status: canonical.transition_status.or(legacy.transition_status),
|
transition_status: merge_consistent(canonical.transition_status, legacy.transition_status)?,
|
||||||
transitioned_object: canonical.transitioned_object.or(legacy.transitioned_object),
|
transitioned_object: merge_consistent(canonical.transitioned_object, legacy.transitioned_object)?,
|
||||||
transitioned_version: canonical.transitioned_version.or(legacy.transitioned_version),
|
transitioned_version: merge_consistent(canonical.transitioned_version, legacy.transitioned_version)?,
|
||||||
transitioned_version_state: canonical.transitioned_version_state.or(legacy.transitioned_version_state),
|
transitioned_version_state: merge_consistent(
|
||||||
transition_tier: canonical.transition_tier.or(legacy.transition_tier),
|
canonical.transitioned_version_state,
|
||||||
|
legacy.transitioned_version_state,
|
||||||
|
)?,
|
||||||
|
transition_tier: merge_consistent(canonical.transition_tier, legacy.transition_tier)?,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -438,8 +464,14 @@ impl FileInfo {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn set_transition_version_state(meta_sys: &mut HashMap<String, Vec<u8>>, state: TransitionVersionState) {
|
fn set_transition_version_state(
|
||||||
if state == TransitionVersionState::Unknown {
|
meta_sys: &mut HashMap<String, Vec<u8>>,
|
||||||
|
state: TransitionVersionState,
|
||||||
|
source_metadata: &HashMap<String, String>,
|
||||||
|
) {
|
||||||
|
if state == TransitionVersionState::Unknown
|
||||||
|
&& !rustfs_utils::http::metadata_compat::contains_key_str(source_metadata, SUFFIX_TRANSITIONED_VERSION_STATE)
|
||||||
|
{
|
||||||
remove_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE);
|
remove_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE);
|
||||||
} else {
|
} else {
|
||||||
insert_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE, state.as_str().as_bytes().to_vec());
|
insert_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE, state.as_str().as_bytes().to_vec());
|
||||||
@@ -2643,6 +2675,11 @@ impl MetaObject {
|
|||||||
if derived_metadata.transitioned_version_state.is_some() {
|
if derived_metadata.transitioned_version_state.is_some() {
|
||||||
validate_transition_version_state(transition_version_state, transition_version.as_deref())?;
|
validate_transition_version_state(transition_version_state, transition_version.as_deref())?;
|
||||||
}
|
}
|
||||||
|
for (key, value) in &self.meta_sys {
|
||||||
|
if is_transition_version_metadata_key(key) {
|
||||||
|
metadata.insert(key.to_owned(), transition_version_metadata_value(value, transition_version.as_deref()));
|
||||||
|
}
|
||||||
|
}
|
||||||
let transition_version_id = transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok());
|
let transition_version_id = transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok());
|
||||||
let transition_tier = derived_metadata
|
let transition_tier = derived_metadata
|
||||||
.transition_tier
|
.transition_tier
|
||||||
@@ -2689,7 +2726,7 @@ impl MetaObject {
|
|||||||
} else {
|
} else {
|
||||||
remove_bytes(&mut self.meta_sys, SUFFIX_TRANSITIONED_VERSION_ID);
|
remove_bytes(&mut self.meta_sys, SUFFIX_TRANSITIONED_VERSION_ID);
|
||||||
}
|
}
|
||||||
set_transition_version_state(&mut self.meta_sys, fi.transition_version_state);
|
set_transition_version_state(&mut self.meta_sys, fi.transition_version_state, &fi.metadata);
|
||||||
insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER, fi.transition_tier.as_bytes().to_vec());
|
insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER, fi.transition_tier.as_bytes().to_vec());
|
||||||
if let Some(destination_id) = get_str(&fi.metadata, SUFFIX_TRANSITION_TIER_DESTINATION_ID) {
|
if let Some(destination_id) = get_str(&fi.metadata, SUFFIX_TRANSITION_TIER_DESTINATION_ID) {
|
||||||
insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER_DESTINATION_ID, destination_id.into_bytes());
|
insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER_DESTINATION_ID, destination_id.into_bytes());
|
||||||
@@ -2830,7 +2867,7 @@ impl From<FileInfo> for MetaObject {
|
|||||||
insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version);
|
insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version);
|
||||||
}
|
}
|
||||||
if !value.transition_status.is_empty() {
|
if !value.transition_status.is_empty() {
|
||||||
set_transition_version_state(&mut meta_sys, value.transition_version_state);
|
set_transition_version_state(&mut meta_sys, value.transition_version_state, &value.metadata);
|
||||||
}
|
}
|
||||||
|
|
||||||
if !value.transition_tier.is_empty() {
|
if !value.transition_tier.is_empty() {
|
||||||
@@ -2985,6 +3022,12 @@ impl MetaDeleteMarker {
|
|||||||
fi.transition_version_state = transition_version_state_from_bytes(derived_metadata.transitioned_version_state)?;
|
fi.transition_version_state = transition_version_state_from_bytes(derived_metadata.transitioned_version_state)?;
|
||||||
fi.transition_version =
|
fi.transition_version =
|
||||||
transitioned_version_from_bytes(derived_metadata.transitioned_version, fi.transition_version_state);
|
transitioned_version_from_bytes(derived_metadata.transitioned_version, fi.transition_version_state);
|
||||||
|
for (key, value) in &self.meta_sys {
|
||||||
|
if is_transition_version_metadata_key(key) {
|
||||||
|
fi.metadata
|
||||||
|
.insert(key.to_owned(), transition_version_metadata_value(value, fi.transition_version.as_deref()));
|
||||||
|
}
|
||||||
|
}
|
||||||
fi.transition_version_id = fi.transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok());
|
fi.transition_version_id = fi.transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok());
|
||||||
if derived_metadata.transitioned_version_state.is_some() {
|
if derived_metadata.transitioned_version_state.is_some() {
|
||||||
validate_transition_version_state(fi.transition_version_state, fi.transition_version.as_deref())?;
|
validate_transition_version_state(fi.transition_version_state, fi.transition_version.as_deref())?;
|
||||||
@@ -3152,7 +3195,7 @@ impl From<FileInfo> for MetaDeleteMarker {
|
|||||||
insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version);
|
insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version);
|
||||||
}
|
}
|
||||||
if !value.transition_status.is_empty() || value.tier_free_version() {
|
if !value.transition_status.is_empty() || value.tier_free_version() {
|
||||||
set_transition_version_state(&mut meta_sys, value.transition_version_state);
|
set_transition_version_state(&mut meta_sys, value.transition_version_state, &value.metadata);
|
||||||
}
|
}
|
||||||
if !value.transition_tier.is_empty() {
|
if !value.transition_tier.is_empty() {
|
||||||
insert_bytes(&mut meta_sys, SUFFIX_TRANSITION_TIER, value.transition_tier.as_bytes().to_vec());
|
insert_bytes(&mut meta_sys, SUFFIX_TRANSITION_TIER, value.transition_tier.as_bytes().to_vec());
|
||||||
@@ -4574,6 +4617,7 @@ mod tests {
|
|||||||
.into_fileinfo("b", "k", false)
|
.into_fileinfo("b", "k", false)
|
||||||
.expect("into_fileinfo");
|
.expect("into_fileinfo");
|
||||||
assert_eq!(fi.transition_version_id, None);
|
assert_eq!(fi.transition_version_id, None);
|
||||||
|
assert_eq!(get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID), Some(String::new()));
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -4585,6 +4629,10 @@ mod tests {
|
|||||||
.into_fileinfo("b", "k", false)
|
.into_fileinfo("b", "k", false)
|
||||||
.expect("into_fileinfo");
|
.expect("into_fileinfo");
|
||||||
assert_eq!(fi.transition_version_id, None);
|
assert_eq!(fi.transition_version_id, None);
|
||||||
|
assert!(
|
||||||
|
get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID).is_some_and(|value| !value.is_empty()),
|
||||||
|
"nil UUID bytes must remain distinguishable from an empty MinIO version"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -4598,6 +4646,7 @@ mod tests {
|
|||||||
assert_eq!(fi.transition_version_id, Some(id));
|
assert_eq!(fi.transition_version_id, Some(id));
|
||||||
assert_eq!(fi.transition_version, Some(id.to_string()));
|
assert_eq!(fi.transition_version, Some(id.to_string()));
|
||||||
assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown);
|
assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown);
|
||||||
|
assert_eq!(get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID), Some(id.to_string()));
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -4637,6 +4686,36 @@ mod tests {
|
|||||||
assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown);
|
assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn meta_object_transition_version_state_explicit_unknown_is_not_legacy_missing() {
|
||||||
|
let mut metadata = HashMap::new();
|
||||||
|
rustfs_utils::http::metadata_compat::insert_str(
|
||||||
|
&mut metadata,
|
||||||
|
SUFFIX_TRANSITIONED_VERSION_STATE,
|
||||||
|
TransitionVersionState::Unknown.as_str().to_string(),
|
||||||
|
);
|
||||||
|
let fi = FileInfo {
|
||||||
|
transition_status: "complete".to_string(),
|
||||||
|
transition_version_state: TransitionVersionState::Unknown,
|
||||||
|
metadata,
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
|
||||||
|
let object = MetaObject::from(fi);
|
||||||
|
assert_eq!(
|
||||||
|
get_consistent_bytes(&object.meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE),
|
||||||
|
Some(b"unknown".as_slice())
|
||||||
|
);
|
||||||
|
let decoded = object
|
||||||
|
.into_fileinfo("b", "k", false)
|
||||||
|
.expect("explicit unknown state should decode");
|
||||||
|
assert_eq!(decoded.transition_version_state, TransitionVersionState::Unknown);
|
||||||
|
assert_eq!(
|
||||||
|
rustfs_utils::http::metadata_compat::get_consistent_str(&decoded.metadata, SUFFIX_TRANSITIONED_VERSION_STATE,),
|
||||||
|
Some("unknown")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn meta_object_transition_version_state_exact_round_trips_dual_keys() {
|
fn meta_object_transition_version_state_exact_round_trips_dual_keys() {
|
||||||
let id = sample_version_id();
|
let id = sample_version_id();
|
||||||
@@ -4753,6 +4832,10 @@ mod tests {
|
|||||||
.expect("invalid transition version bytes must not fail the object read");
|
.expect("invalid transition version bytes must not fail the object read");
|
||||||
assert_eq!(fi.transition_version_id, None);
|
assert_eq!(fi.transition_version_id, None);
|
||||||
assert_eq!(fi.transition_version, None);
|
assert_eq!(fi.transition_version, None);
|
||||||
|
assert!(
|
||||||
|
get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID).is_some_and(|value| !value.is_empty()),
|
||||||
|
"invalid raw bytes must remain distinguishable from an empty MinIO version"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -4795,6 +4878,10 @@ mod tests {
|
|||||||
.into_fileinfo("b", "k", false)
|
.into_fileinfo("b", "k", false)
|
||||||
.expect("nil tier version should remain an absent remote version");
|
.expect("nil tier version should remain an absent remote version");
|
||||||
assert_eq!(fi.transition_version_id, None);
|
assert_eq!(fi.transition_version_id, None);
|
||||||
|
assert!(
|
||||||
|
get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID).is_some_and(|value| !value.is_empty()),
|
||||||
|
"nil UUID bytes must remain distinguishable from an empty MinIO version"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -4812,6 +4899,7 @@ mod tests {
|
|||||||
.expect("legacy binary UUID tier version should decode");
|
.expect("legacy binary UUID tier version should decode");
|
||||||
assert_eq!(fi.transition_version_id, Some(id));
|
assert_eq!(fi.transition_version_id, Some(id));
|
||||||
assert_eq!(fi.transition_version, Some(id.to_string()));
|
assert_eq!(fi.transition_version, Some(id.to_string()));
|
||||||
|
assert_eq!(get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID), Some(id.to_string()));
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -4910,6 +4998,23 @@ mod tests {
|
|||||||
assert_eq!(err, Error::FileCorrupt);
|
assert_eq!(err, Error::FileCorrupt);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn meta_object_transition_version_state_mixed_case_alias_conflict_fails_closed() {
|
||||||
|
let sys = HashMap::from([
|
||||||
|
(
|
||||||
|
format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_TRANSITIONED_VERSION_STATE}"),
|
||||||
|
b"unknown".to_vec(),
|
||||||
|
),
|
||||||
|
("X-Minio-Internal-transitioned-version-state".to_string(), b"exact".to_vec()),
|
||||||
|
]);
|
||||||
|
|
||||||
|
let err = make_meta_object_with_sys(sys)
|
||||||
|
.into_fileinfo("b", "k", false)
|
||||||
|
.expect_err("mixed-case transition state aliases must agree");
|
||||||
|
|
||||||
|
assert_eq!(err, Error::FileCorrupt);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn version_header_sorts_before_prefers_object_over_delete_marker_on_equal_mod_time() {
|
fn version_header_sorts_before_prefers_object_over_delete_marker_on_equal_mod_time() {
|
||||||
let object = FileMetaVersionHeader {
|
let object = FileMetaVersionHeader {
|
||||||
|
|||||||
@@ -45,6 +45,11 @@ use tracing::{debug, error, info, warn};
|
|||||||
use super::{DiskError, Endpoint, HealDiskExt as _, local_disk_map_read};
|
use super::{DiskError, Endpoint, HealDiskExt as _, local_disk_map_read};
|
||||||
|
|
||||||
const KEEP_HEAL_TASK_STATUS_DURATION: Duration = Duration::from_secs(10 * 60);
|
const KEEP_HEAL_TASK_STATUS_DURATION: Duration = Duration::from_secs(10 * 60);
|
||||||
|
// Each cache includes alias tokens in its count and byte budget. Eviction
|
||||||
|
// removes every token sharing a snapshot; neither cache retains repair state.
|
||||||
|
const MAX_COMPLETED_HEAL_TOKENS: usize = 1024;
|
||||||
|
const MAX_COMPLETED_HEAL_BYTES: usize = 64 * 1024 * 1024;
|
||||||
|
const MAX_COMPLETED_HEAL_RESULT_BYTES: usize = 1024 * 1024;
|
||||||
const DISPLACED_HEAL_REASON: &str = "reason=displaced; retry_hint=submit_again";
|
const DISPLACED_HEAL_REASON: &str = "reason=displaced; retry_hint=submit_again";
|
||||||
const LOG_COMPONENT_HEAL: &str = "heal";
|
const LOG_COMPONENT_HEAL: &str = "heal";
|
||||||
const LOG_SUBSYSTEM_DISK_SCANNER: &str = "disk_scanner";
|
const LOG_SUBSYSTEM_DISK_SCANNER: &str = "disk_scanner";
|
||||||
@@ -180,6 +185,8 @@ fn record_displaced_terminal(
|
|||||||
request: &HealRequest,
|
request: &HealRequest,
|
||||||
) -> Arc<CompletedHealStatus> {
|
) -> Arc<CompletedHealStatus> {
|
||||||
let terminal = Arc::new(CompletedHealStatus {
|
let terminal = Arc::new(CompletedHealStatus {
|
||||||
|
progress: None,
|
||||||
|
retained_bytes: std::sync::OnceLock::new(),
|
||||||
heal_type: request.heal_type.clone(),
|
heal_type: request.heal_type.clone(),
|
||||||
status: HealTaskStatus::Failed {
|
status: HealTaskStatus::Failed {
|
||||||
error: format!("heal task displaced by a higher-priority request ({DISPLACED_HEAL_REASON})"),
|
error: format!("heal task displaced by a higher-priority request ({DISPLACED_HEAL_REASON})"),
|
||||||
@@ -193,6 +200,7 @@ fn record_displaced_terminal(
|
|||||||
let mut terminals = lock_displaced_terminals(registry);
|
let mut terminals = lock_displaced_terminals(registry);
|
||||||
prune_completed_heal_statuses(&mut terminals);
|
prune_completed_heal_statuses(&mut terminals);
|
||||||
terminals.insert(request.id.clone(), Arc::clone(&terminal));
|
terminals.insert(request.id.clone(), Arc::clone(&terminal));
|
||||||
|
prune_completed_heal_statuses(&mut terminals);
|
||||||
terminal
|
terminal
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -209,9 +217,15 @@ async fn remove_displaced_task_aliases(
|
|||||||
.collect::<Vec<_>>();
|
.collect::<Vec<_>>();
|
||||||
let mut displaced_terminals = lock_displaced_terminals(terminals);
|
let mut displaced_terminals = lock_displaced_terminals(terminals);
|
||||||
prune_completed_heal_statuses(&mut displaced_terminals);
|
prune_completed_heal_statuses(&mut displaced_terminals);
|
||||||
for alias_id in alias_ids {
|
if displaced_terminals
|
||||||
displaced_terminals.insert(alias_id, Arc::clone(terminal));
|
.get(task_id)
|
||||||
|
.is_some_and(|current| Arc::ptr_eq(current, terminal))
|
||||||
|
{
|
||||||
|
for alias_id in alias_ids {
|
||||||
|
displaced_terminals.insert(alias_id, Arc::clone(terminal));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
prune_completed_heal_statuses(&mut displaced_terminals);
|
||||||
aliases.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
|
aliases.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -222,6 +236,36 @@ async fn remove_task_aliases_for_task(registry: &Arc<Mutex<HashMap<String, HealT
|
|||||||
.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
|
.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Callers hold active ownership until publication. Lock order is active ->
|
||||||
|
// retrying (when needed) -> aliases -> completed; queries release aliases
|
||||||
|
// before looking up active state. Publishing aliases before removing their
|
||||||
|
// mapping keeps both an already-resolved token and a new lookup valid.
|
||||||
|
async fn publish_completed_heal(
|
||||||
|
completed_heals: &Mutex<HashMap<String, Arc<CompletedHealStatus>>>,
|
||||||
|
task_aliases: &Mutex<HashMap<String, HealTaskAlias>>,
|
||||||
|
task_id: &str,
|
||||||
|
completed: CompletedHealStatus,
|
||||||
|
terminal: bool,
|
||||||
|
) {
|
||||||
|
let completed = Arc::new(completed);
|
||||||
|
completed.retained_bytes();
|
||||||
|
let mut aliases = task_aliases.lock().await;
|
||||||
|
let mut retained = completed_heals.lock().await;
|
||||||
|
if let Some(previous) = retained.get(task_id).cloned() {
|
||||||
|
for entry in retained.values_mut().filter(|entry| Arc::ptr_eq(entry, &previous)) {
|
||||||
|
*entry = Arc::clone(&completed);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
retained.insert(task_id.to_owned(), Arc::clone(&completed));
|
||||||
|
if terminal {
|
||||||
|
for (alias_id, _) in aliases.iter().filter(|(_, alias)| alias.task_id == task_id) {
|
||||||
|
retained.insert(alias_id.clone(), Arc::clone(&completed));
|
||||||
|
}
|
||||||
|
aliases.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
|
||||||
|
}
|
||||||
|
prune_completed_heal_statuses(&mut retained);
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct HealTaskReport {
|
pub struct HealTaskReport {
|
||||||
pub status: HealTaskStatus,
|
pub status: HealTaskStatus,
|
||||||
@@ -268,7 +312,7 @@ fn completed_task_report(completed: &CompletedHealStatus, since: Option<u64>) ->
|
|||||||
let result_items = match since {
|
let result_items = match since {
|
||||||
None => completed.seqed_items.iter().map(|(_, item)| item.clone()).collect(),
|
None => completed.seqed_items.iter().map(|(_, item)| item.clone()).collect(),
|
||||||
Some(cursor) => {
|
Some(cursor) => {
|
||||||
if cursor + 1 < completed.min_seq {
|
if cursor.saturating_add(1) < completed.min_seq {
|
||||||
lagged = true;
|
lagged = true;
|
||||||
}
|
}
|
||||||
completed
|
completed
|
||||||
@@ -283,7 +327,7 @@ fn completed_task_report(completed: &CompletedHealStatus, since: Option<u64>) ->
|
|||||||
status: completed.status.clone(),
|
status: completed.status.clone(),
|
||||||
result_items,
|
result_items,
|
||||||
result_items_truncated: completed.result_items_truncated || lagged,
|
result_items_truncated: completed.result_items_truncated || lagged,
|
||||||
progress: None,
|
progress: completed.progress.clone(),
|
||||||
next_seq: completed.next_seq,
|
next_seq: completed.next_seq,
|
||||||
min_seq: completed.min_seq,
|
min_seq: completed.min_seq,
|
||||||
}
|
}
|
||||||
@@ -1847,14 +1891,14 @@ impl HealManager {
|
|||||||
|
|
||||||
pub async fn get_task_progress(&self, task_id: &str) -> Result<HealProgress> {
|
pub async fn get_task_progress(&self, task_id: &str) -> Result<HealProgress> {
|
||||||
let canonical_task_id = self.canonical_task_id(task_id).await;
|
let canonical_task_id = self.canonical_task_id(task_id).await;
|
||||||
let active_heals = self.active_heals.lock().await;
|
let progress = match self.lookup_task_state(&canonical_task_id, None).await {
|
||||||
if let Some(task) = active_heals.get(&canonical_task_id) {
|
TaskStateLookup::Active(task) => Some(task.get_progress().await),
|
||||||
Ok(task.get_progress().await)
|
TaskStateLookup::Completed(completed) => completed.progress.clone(),
|
||||||
} else {
|
_ => None,
|
||||||
Err(Error::TaskNotFound {
|
};
|
||||||
task_id: task_id.to_string(),
|
progress.ok_or_else(|| Error::TaskNotFound {
|
||||||
})
|
task_id: task_id.to_string(),
|
||||||
}
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Cancel task
|
/// Cancel task
|
||||||
@@ -1864,6 +1908,8 @@ impl HealManager {
|
|||||||
let mut active_heals = self.active_heals.lock().await;
|
let mut active_heals = self.active_heals.lock().await;
|
||||||
if let Some(task) = active_heals.get(&canonical_task_id) {
|
if let Some(task) = active_heals.get(&canonical_task_id) {
|
||||||
task.cancel().await?;
|
task.cancel().await?;
|
||||||
|
let completed = CompletedHealStatus::snapshot(task, HealTaskStatus::Cancelled).await;
|
||||||
|
publish_completed_heal(&self.completed_heals, &self.task_aliases, &canonical_task_id, completed, true).await;
|
||||||
active_heals.remove(&canonical_task_id);
|
active_heals.remove(&canonical_task_id);
|
||||||
publish_active_heal_count(&active_heals);
|
publish_active_heal_count(&active_heals);
|
||||||
info!(
|
info!(
|
||||||
@@ -1940,6 +1986,8 @@ impl HealManager {
|
|||||||
for task_id in &task_ids {
|
for task_id in &task_ids {
|
||||||
if let Some(task) = active_heals.get(task_id) {
|
if let Some(task) = active_heals.get(task_id) {
|
||||||
task.cancel().await?;
|
task.cancel().await?;
|
||||||
|
let completed = CompletedHealStatus::snapshot(task, HealTaskStatus::Cancelled).await;
|
||||||
|
publish_completed_heal(&self.completed_heals, &self.task_aliases, task_id, completed, true).await;
|
||||||
}
|
}
|
||||||
active_heals.remove(task_id);
|
active_heals.remove(task_id);
|
||||||
cancelled += 1;
|
cancelled += 1;
|
||||||
|
|||||||
@@ -82,6 +82,8 @@ pub(super) enum QueuePushOutcome {
|
|||||||
pub(super) struct CompletedHealStatus {
|
pub(super) struct CompletedHealStatus {
|
||||||
pub(super) heal_type: HealType,
|
pub(super) heal_type: HealType,
|
||||||
pub(super) status: HealTaskStatus,
|
pub(super) status: HealTaskStatus,
|
||||||
|
pub(super) progress: Option<HealProgress>,
|
||||||
|
pub(super) retained_bytes: std::sync::OnceLock<usize>,
|
||||||
pub(super) result_items_truncated: bool,
|
pub(super) result_items_truncated: bool,
|
||||||
pub(super) completed_at: SystemTime,
|
pub(super) completed_at: SystemTime,
|
||||||
/// Sequence-stamped retained window, archived with the completion so
|
/// Sequence-stamped retained window, archived with the completion so
|
||||||
@@ -92,6 +94,133 @@ pub(super) struct CompletedHealStatus {
|
|||||||
pub(super) min_seq: u64,
|
pub(super) min_seq: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl CompletedHealStatus {
|
||||||
|
// Account for owned capacities, including nested drive arrays. Aliases
|
||||||
|
// conservatively charge the shared allocation again, keeping both token
|
||||||
|
// count and retained payload bounded without a second ownership index.
|
||||||
|
pub(super) fn retained_bytes(&self) -> usize {
|
||||||
|
*self.retained_bytes.get_or_init(|| self.measure_retained_bytes())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn measure_retained_bytes(&self) -> usize {
|
||||||
|
let mut bytes = size_of::<Self>();
|
||||||
|
let mut add = |amount: usize| bytes = bytes.saturating_add(amount);
|
||||||
|
match &self.heal_type {
|
||||||
|
HealType::Cluster => {}
|
||||||
|
HealType::Bucket { bucket } => add(bucket.capacity()),
|
||||||
|
HealType::Object {
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id,
|
||||||
|
}
|
||||||
|
| HealType::ECDecode {
|
||||||
|
bucket,
|
||||||
|
object,
|
||||||
|
version_id,
|
||||||
|
} => {
|
||||||
|
add(bucket.capacity());
|
||||||
|
add(object.capacity());
|
||||||
|
add(version_id.as_ref().map_or(0, String::capacity));
|
||||||
|
}
|
||||||
|
HealType::Prefix { bucket, prefix } => {
|
||||||
|
add(bucket.capacity());
|
||||||
|
add(prefix.capacity());
|
||||||
|
}
|
||||||
|
HealType::Metadata { bucket, object } => {
|
||||||
|
add(bucket.capacity());
|
||||||
|
add(object.capacity());
|
||||||
|
}
|
||||||
|
HealType::ErasureSet { buckets, set_disk_id } => {
|
||||||
|
add(buckets.capacity().saturating_mul(size_of::<String>()));
|
||||||
|
for bucket in buckets {
|
||||||
|
add(bucket.capacity());
|
||||||
|
}
|
||||||
|
add(set_disk_id.capacity());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if let HealTaskStatus::Failed { error } | HealTaskStatus::Retrying { error, .. } = &self.status {
|
||||||
|
add(error.capacity());
|
||||||
|
}
|
||||||
|
add(self
|
||||||
|
.progress
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|progress| progress.current_object.as_ref())
|
||||||
|
.map_or(0, String::capacity));
|
||||||
|
add(self.seqed_items.capacity().saturating_mul(size_of::<(u64, HealResultItem)>()));
|
||||||
|
for (_, item) in &self.seqed_items {
|
||||||
|
add(Self::result_item_heap_bytes(item));
|
||||||
|
}
|
||||||
|
bytes
|
||||||
|
}
|
||||||
|
|
||||||
|
fn result_item_heap_bytes(item: &HealResultItem) -> usize {
|
||||||
|
let mut bytes = 0usize;
|
||||||
|
let mut add = |amount: usize| bytes = bytes.saturating_add(amount);
|
||||||
|
for value in [
|
||||||
|
&item.heal_item_type,
|
||||||
|
&item.bucket,
|
||||||
|
&item.object,
|
||||||
|
&item.version_id,
|
||||||
|
&item.detail,
|
||||||
|
] {
|
||||||
|
add(value.capacity());
|
||||||
|
}
|
||||||
|
for infos in [&item.before, &item.after] {
|
||||||
|
add(infos
|
||||||
|
.drives
|
||||||
|
.capacity()
|
||||||
|
.saturating_mul(size_of::<rustfs_madmin::heal_commands::HealDriveInfo>()));
|
||||||
|
for drive in &infos.drives {
|
||||||
|
add(drive.uuid.capacity());
|
||||||
|
add(drive.endpoint.capacity());
|
||||||
|
add(drive.state.capacity());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
bytes
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn bound_result_window(&mut self) {
|
||||||
|
let mut bytes = 0usize;
|
||||||
|
let retained = self
|
||||||
|
.seqed_items
|
||||||
|
.iter()
|
||||||
|
.rev()
|
||||||
|
.take_while(|(_, item)| {
|
||||||
|
bytes = bytes
|
||||||
|
.saturating_add(size_of::<(u64, HealResultItem)>())
|
||||||
|
.saturating_add(Self::result_item_heap_bytes(item));
|
||||||
|
bytes <= MAX_COMPLETED_HEAL_RESULT_BYTES
|
||||||
|
})
|
||||||
|
.count();
|
||||||
|
let truncated = retained < self.seqed_items.len();
|
||||||
|
if truncated {
|
||||||
|
self.seqed_items.drain(..self.seqed_items.len() - retained);
|
||||||
|
self.seqed_items.shrink_to_fit();
|
||||||
|
self.min_seq = self.seqed_items.first().map_or(self.next_seq, |(seq, _)| *seq);
|
||||||
|
self.result_items_truncated = true;
|
||||||
|
self.retained_bytes.take();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) async fn snapshot(task: &HealTask, status: HealTaskStatus) -> Self {
|
||||||
|
let seqed_items = task.get_seqed_result_items().await;
|
||||||
|
let (next_seq, min_seq) = task.result_seq_cursors();
|
||||||
|
let mut snapshot = Self {
|
||||||
|
heal_type: task.heal_type.clone(),
|
||||||
|
status,
|
||||||
|
progress: Some(task.get_progress().await),
|
||||||
|
retained_bytes: std::sync::OnceLock::new(),
|
||||||
|
result_items_truncated: task.result_items_truncated(),
|
||||||
|
completed_at: SystemTime::now(),
|
||||||
|
seqed_items,
|
||||||
|
next_seq,
|
||||||
|
min_seq,
|
||||||
|
};
|
||||||
|
snapshot.bound_result_window();
|
||||||
|
snapshot
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub(super) struct HealTaskAlias {
|
pub(super) struct HealTaskAlias {
|
||||||
pub(super) task_id: String,
|
pub(super) task_id: String,
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user