Compare commits

..

3 Commits

Author SHA1 Message Date
Zhengchao An 72fd7339c9 test(utils): allow ephemeral port reuse (#6122)
* test(utils): allow ephemeral port reuse

* test(kms): allow any ciphertext prefix
2026-08-15 08:32:10 +08:00
Zhengchao An 71e83aeec4 fix(ci): pin Docker images to release source (#6121) 2026-08-15 07:13:37 +08:00
唐小鸭 9138c24571 fix(site-replication): lift a rejoined site's restarted edit counter over stale marks (#6119)
fix(site-replication): lift a rejoined site's restarted edit counter over stale fence marks

A site removed while unreachable (unilateral removal: the receiver never
dropped it from its peer map, so parse_site_replication_state's load-time
mark pruning never fired) that later rejoins recreates its state object
and restarts edit_generation at zero. The receiver's surviving high-water
mark then silently fences out every stamped delivery from that origin —
peer edits and the add finalize fan-out alike are acked without applying
— until the restarted counter catches up.

Allocate the generation as a hybrid logical clock instead:
max(wall clock in unix nanoseconds, previous + 1), still inside the state
transaction under the distributed state-object lock. Every value a
lifetime hands out is capped by the wall clock at its own allocation, so
a recreated lifetime's first allocation exceeds them all and clears the
stale mark, while a pre-removal delivery still in flight stays below the
new floor and remains correctly fenced. previous+1 keeps allocations
strictly increasing across same-tick allocations and mid-lifetime clock
regressions.

Nothing changes on the wire or in the persisted schema: editGeneration
stays the single fence param and edit_generation the single counter
field, so pre-hybrid receivers get the fix as soon as the sender
upgrades, old binaries preserve the field across rolling up/downgrades,
and marks recorded by plain-counter receivers (small values) are cleared
by any wall-clock allocation. A clock that regresses across a
delete/recreate degrades to a fence that self-heals once real time
passes the previous lifetime's last allocation, and introduces no
rollback window beyond what the plain counter already had.

An epoch-based design (editEpoch wire param + per-origin epoch marks)
was built first and rejected under adversarial review: old binaries
rewriting the state object drop the unknown epoch fields, which both
disarms the fix mid-rolling-upgrade and — because epoch adoption lowers
the generation mark — reopens the pre-restart rollback the fence exists
to prevent; a backwards clock also fences an origin permanently instead
of self-healing. The hybrid clock has none of these modes.
2026-08-15 01:50:35 +08:00
5 changed files with 250 additions and 15 deletions
+25 -1
View File
@@ -94,6 +94,7 @@ jobs:
short_sha: ${{ steps.check.outputs.short_sha }}
is_prerelease: ${{ steps.check.outputs.is_prerelease }}
create_latest: ${{ steps.check.outputs.create_latest }}
source_ref: ${{ steps.check.outputs.source_ref }}
steps:
- name: Checkout repository
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
@@ -118,6 +119,7 @@ jobs:
short_sha=""
is_prerelease=false
create_latest=false
source_ref="$GITHUB_SHA"
if [[ "${{ github.event_name }}" == "workflow_run" ]]; then
# Triggered by build workflow completion
@@ -137,6 +139,7 @@ jobs:
# Extract version info from commit message or use commit SHA
# Use Git to generate consistent short SHA (ensures uniqueness like build.yml)
short_sha=$(git rev-parse --short "$HEAD_SHA")
source_ref="$HEAD_SHA"
# Determine build type based on triggering workflow event and ref
triggering_event="$TRIGGERING_EVENT"
@@ -261,6 +264,23 @@ jobs:
echo "⚠️ Only release versions (latest, v1.0.0, 1.0.0) and prereleases (v1.0.0-alpha1, 1.0.0-beta2) are supported"
;;
esac
if [[ "$should_build" == true && "$input_version" != "latest" ]]; then
tag_ref="refs/tags/$input_version"
if ! git ls-remote --exit-code origin "$tag_ref" >/dev/null 2>&1; then
if [[ "$input_version" == v* ]]; then
tag_ref="refs/tags/${input_version#v}"
else
tag_ref="refs/tags/v$input_version"
fi
fi
if ! git ls-remote --exit-code origin "$tag_ref" >/dev/null 2>&1; then
echo "❌ Release tag not found for Docker build: $input_version"
exit 1
fi
source_ref="$tag_ref"
fi
fi
{
@@ -271,6 +291,7 @@ jobs:
echo "short_sha=$short_sha"
echo "is_prerelease=$is_prerelease"
echo "create_latest=$create_latest"
echo "source_ref=$source_ref"
} >> "$GITHUB_OUTPUT"
echo "🐳 Docker Build Summary:"
@@ -281,6 +302,7 @@ jobs:
echo " - Short SHA: $short_sha"
echo " - Is prerelease: $is_prerelease"
echo " - Create latest: $create_latest"
echo " - Source ref: $source_ref"
# Build multi-arch Docker images
# Strategy: Build images using pre-built binaries from dl.rustfs.com
@@ -308,6 +330,7 @@ jobs:
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
ref: ${{ needs.build-check.outputs.source_ref }}
- name: Login to Docker Hub
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3
@@ -397,7 +420,8 @@ jobs:
LABELS="org.opencontainers.image.title=RustFS"
LABELS="$LABELS,org.opencontainers.image.description=RustFS distributed object storage system"
LABELS="$LABELS,org.opencontainers.image.version=$VERSION"
LABELS="$LABELS,org.opencontainers.image.revision=${{ github.sha }}"
SOURCE_REVISION="$(git rev-parse HEAD)"
LABELS="$LABELS,org.opencontainers.image.revision=$SOURCE_REVISION"
LABELS="$LABELS,org.opencontainers.image.source=${{ github.server_url }}/${{ github.repository }}"
LABELS="$LABELS,org.opencontainers.image.created=$(date -u +'%Y-%m-%dT%H:%M:%SZ')"
LABELS="$LABELS,org.opencontainers.image.build-type=$BUILD_TYPE"
-2
View File
@@ -225,8 +225,6 @@ async fn nothing_readable_leaves_the_bundle_unwrapped() {
"artifact {} carries the raw on-disk record",
artifact.path
);
// A cheap structural check too: an encrypted payload is not JSON.
assert_ne!(payload.first(), Some(&b'{'), "artifact {} looks like plaintext JSON", artifact.path);
}
// The manifest itself is not encrypted, so assert directly that it carries
-3
View File
@@ -659,9 +659,6 @@ mod test {
// Port should be in valid range (u16 max is always <= 65535)
assert!(port1 > 0);
assert!(port2 > 0);
// Different calls should typically return different ports
assert_ne!(port1, port2);
}
#[test]
+218 -9
View File
@@ -1067,9 +1067,13 @@ fn parse_site_replication_state(data: &[u8]) -> S3Result<SiteReplicationState> {
state.peers = normalize_peer_map_by_identity(state.peers);
// A peer-edit high-water mark only fences a CURRENT peer. A site that
// leaves drops below two peers, which clears its own state object and
// restarts its generation counter at zero — a mark left over from the
// previous membership would then reject every edit it sends after it
// rejoins. Dropping departed origins on load also keeps the map bounded.
// restarts its generation counter — a mark left over from the previous
// membership must not reject the edits it sends after it rejoins. This
// pruning covers departures THIS site observed; an origin removed
// unilaterally elsewhere stays in this peer map with its mark, and the
// wall-clock floor in `next_peer_edit_generation` is what lifts its
// restarted counter over that mark. Dropping departed origins on load
// also keeps the map bounded.
state
.applied_edit_generations
.retain(|origin, _| state.peers.contains_key(origin));
@@ -5935,11 +5939,51 @@ fn summarize_peer_error_detail(detail: &str) -> String {
summary
}
/// Allocate the next peer-edit generation. Called inside the state
/// transaction, so the counter is handed out under the distributed
/// state-object lock and two nodes of this site can never take the same one.
/// The wall clock in unix nanoseconds, clamped into u64. A pre-1970 (or
/// post-2554) clock yields 0, which makes the hybrid allocation below
/// degrade to the plain `previous + 1` counter — monotone, never panicking.
fn edit_generation_wall_clock() -> u64 {
u64::try_from(OffsetDateTime::now_utc().unix_timestamp_nanos()).unwrap_or(0)
}
/// Allocate the next peer-edit generation as a hybrid logical clock:
/// `max(wall clock in unix nanoseconds, previous + 1)`. Called inside the
/// state transaction, so the value is handed out under the distributed
/// state-object lock and two nodes of this site can never take the same one
/// (`previous + 1` keeps the sequence strictly increasing even when two
/// allocations land in one clock tick, and keeps it monotone on a node
/// whose clock stepped backwards mid-lifetime).
///
/// The wall-clock floor is what survives the counter's death. A site
/// removed while unreachable — the receiver never dropped it from its peer
/// map, so the load-time mark pruning in `parse_site_replication_state`
/// never fired — that later rejoins recreates its state object with the
/// counter back at zero. A plain counter would then hand out generations
/// below the receiver's stale high-water mark and every delivery would be
/// silently fenced until the counter caught up. Jumping to wall time clears
/// that mark: every value the deleted lifetime handed out was capped by the
/// wall clock at its own allocation (or by a prior lifetime's cap, applied
/// inductively), so the recreated lifetime's first allocation exceeds them
/// all — while a pre-removal delivery still in flight stays below the new
/// floor and remains correctly fenced. Marks recorded by pre-hybrid
/// receivers (small plain-counter values) sit far below any wall-clock
/// value, so a restarted origin passes those too — the fix needs only the
/// sender upgraded, nothing on the wire or in the receiver changed.
///
/// A wall clock that regresses across a delete/recreate (the recreating
/// node's clock behind the clock that fed the previous lifetime) mints
/// below the stale mark and the origin stays fenced — but only until real
/// time passes the previous lifetime's last allocation, because every later
/// allocation takes the wall-clock floor again. Bounded by the skew,
/// self-healing, and no rollback window beyond the plain counter's: a
/// delivery applies only at or above the receiver's mark, so the one
/// cross-lifetime interleaving that can apply stale content — a
/// pre-removal delivery whose generation lands above everything the
/// regressed new lifetime has minted — required the same straggler landing
/// above the mark under the plain counter, where the recreated counter's
/// low restart made it strictly easier to hit.
fn next_peer_edit_generation(state: &mut SiteReplicationState) -> u64 {
state.edit_generation = state.edit_generation.saturating_add(1);
state.edit_generation = edit_generation_wall_clock().max(state.edit_generation.saturating_add(1));
state.edit_generation
}
@@ -13244,6 +13288,104 @@ mod tests {
assert!(!peer_edit_delivery_is_stale(&reloaded, "origin-site", 1));
}
/// The unilateral-removal rejoin gap the hybrid clock closes. The origin
/// was removed while unreachable, but THIS site never dropped it from
/// its peer map, so the load-time mark pruning never fired and the mark
/// from the previous membership survives. The origin's recreated state
/// object restarts its counter, and with a plain `previous + 1` counter
/// every delivery it sent — generations 1, 2, … below the stale mark —
/// would be silently acked-and-dropped until the counter caught up. The
/// wall-clock floor in `next_peer_edit_generation` lifts the restarted
/// counter over every value the deleted lifetime handed out. Reverting
/// the allocation to the plain counter (dropping the wall-clock max)
/// turns the not-stale assertion red.
#[test]
fn hybrid_generation_unfences_a_rejoined_origin_whose_counter_restarted() {
// First lifetime of the origin's state object: two allocations, both
// capped by the wall clock at their own allocation.
let mut first_life = SiteReplicationState::default();
let straggler = next_peer_edit_generation(&mut first_life);
let last_applied = next_peer_edit_generation(&mut first_life);
assert!(last_applied > straggler, "allocations must be strictly increasing");
// The receiver applied up to `last_applied` and keeps the origin in
// its peer map across the unilateral removal — reloading must keep
// the mark, which is exactly why pruning cannot cover this case.
let mut receiver = SiteReplicationState::default();
receiver.peers.insert(
"origin-site".to_string(),
PeerInfo {
deployment_id: "origin-site".to_string(),
..peer("origin", "https://origin.example:9000")
},
);
record_applied_peer_edit_generation(&mut receiver, "origin-site", last_applied);
let mut receiver = parse_site_replication_state(&serde_json::to_vec(&receiver).expect("serialize")).expect("reload");
assert_eq!(receiver.applied_edit_generations.get("origin-site"), Some(&last_applied));
// The origin rejoins with a RECREATED state object: counter back at
// zero. The wall-clock floor must lift its first allocation over the
// previous lifetime's mark…
let mut second_life = SiteReplicationState::default();
let restarted = next_peer_edit_generation(&mut second_life);
assert!(
!peer_edit_delivery_is_stale(&receiver, "origin-site", restarted),
"the recreated lifetime's first allocation ({restarted}) must not be fenced by the previous lifetime's mark ({last_applied})"
);
record_applied_peer_edit_generation(&mut receiver, "origin-site", restarted);
// …while a pre-removal delivery still in flight stays below the new
// floor and remains correctly fenced — the rollback the fence exists
// to reject.
assert!(
peer_edit_delivery_is_stale(&receiver, "origin-site", straggler),
"a pre-removal in-flight delivery ({straggler}) must stay fenced after the rejoin"
);
}
/// Marks recorded before the hybrid clock existed are small plain-counter
/// values, far below any wall-clock allocation: a restarted origin passes
/// them as soon as the SENDER runs the hybrid clock — nothing changes on
/// the wire or in the receiver, so pre-hybrid receivers get the fix too.
/// The other direction is unchanged: among plain-counter values the
/// generation order still fences the delivery that lost the race.
#[test]
fn hybrid_generation_passes_marks_recorded_by_plain_counter_receivers() {
let mut receiver = SiteReplicationState::default();
record_applied_peer_edit_generation(&mut receiver, "origin-site", 57);
assert!(peer_edit_delivery_is_stale(&receiver, "origin-site", 56));
assert!(!peer_edit_delivery_is_stale(&receiver, "origin-site", 57));
let mut rejoined = SiteReplicationState::default();
let restarted = next_peer_edit_generation(&mut rejoined);
assert!(
!peer_edit_delivery_is_stale(&receiver, "origin-site", restarted),
"a wall-clock allocation ({restarted}) must clear a plain-counter mark (57)"
);
}
/// The `previous + 1` half of the hybrid clock: allocations stay strictly
/// increasing even when the wall clock cannot move them forward — two
/// allocations inside one clock tick, or a clock that stepped backwards
/// mid-lifetime (a counter already ahead of the wall clock advances by
/// exactly one per allocation instead of jumping back). Dropping the
/// `previous + 1` half (allocating bare wall time) turns this red.
#[test]
fn hybrid_generation_is_strictly_increasing_when_the_clock_stalls() {
let mut state = SiteReplicationState {
// A counter far ahead of any wall clock this test will see.
edit_generation: u64::MAX / 2,
..Default::default()
};
assert_eq!(next_peer_edit_generation(&mut state), u64::MAX / 2 + 1);
assert_eq!(next_peer_edit_generation(&mut state), u64::MAX / 2 + 2);
// Saturation pins at the ceiling instead of wrapping; the equal-value
// escape (`applied > generation` is false for equal) keeps deliveries
// applying rather than fencing the origin out.
state.edit_generation = u64::MAX;
assert_eq!(next_peer_edit_generation(&mut state), u64::MAX);
}
#[test]
fn test_retry_stats_for_state_counts_pending_and_failed() {
let state = SiteReplicationState {
@@ -16044,10 +16186,77 @@ mod tests {
generations.len(),
"two nodes took the same edit generation, so their deliveries cannot be ordered: {generations:?}"
);
// The hybrid clock allocates `max(wall nanos, previous + 1)` — the
// persisted counter is the largest allocation, and the `+ 1` half
// keeps allocations distinct even inside one clock tick.
assert_eq!(
Some(&load_site_replication_state().await.expect("reload").edit_generation),
unique.last(),
"the persisted counter must be the largest allocation handed out"
);
}
/// The unilateral-removal rejoin, end to end across the state object's
/// real lifecycle: dropping below two peers clears the object (the
/// counter dies with it), and the recreated object's first allocation —
/// raced by two nodes — must clear the previous lifetime's values via
/// the wall-clock floor, so a receiver still holding the old mark
/// accepts the restarted counter instead of fencing it.
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
#[serial]
async fn test_recreated_state_object_allocates_over_the_previous_lifetimes_mark() {
publish_ready_iam_context().await;
let seed = || SiteReplicationState {
peers: ["site-a", "site-b"]
.into_iter()
.map(|name| (name.to_string(), peer(name, &format!("https://{name}.example:9000"))))
.collect(),
..Default::default()
};
save_site_replication_state(&seed()).await.expect("seed state");
let straggler = update_site_replication_state(|state| Ok(next_peer_edit_generation(state)))
.await
.expect("first-life allocation");
let last_applied = update_site_replication_state(|state| Ok(next_peer_edit_generation(state)))
.await
.expect("first-life allocation");
// A receiver that never dropped this site from its peer map holds
// this mark across the removal.
let mut receiver = SiteReplicationState::default();
record_applied_peer_edit_generation(&mut receiver, "origin-site", last_applied);
// Unilateral removal: the site drops below two peers, which clears
// its state object and the counter with it.
let mut departed = seed();
departed.peers.remove("site-b");
save_site_replication_state(&departed).await.expect("clear state");
assert_eq!(
load_site_replication_state().await.expect("reload").edit_generation,
generations.len() as u64,
"the persisted counter must account for every allocation"
0,
"clearing the state object must take the counter with it"
);
// Rejoin recreates the state object; two nodes race the first
// allocation of the new life.
save_site_replication_state(&seed()).await.expect("recreate state");
let node_a = tokio::spawn(update_site_replication_state(|state| Ok(next_peer_edit_generation(state))));
let node_b = tokio::spawn(update_site_replication_state(|state| Ok(next_peer_edit_generation(state))));
let generation_a = node_a.await.expect("node a task").expect("node a allocation");
let generation_b = node_b.await.expect("node b task").expect("node b allocation");
assert_ne!(generation_a, generation_b, "racing allocations must stay distinct");
// The receiver's stale mark must not fence the restarted counter…
let restarted = generation_a.min(generation_b);
assert!(
!peer_edit_delivery_is_stale(&receiver, "origin-site", restarted),
"the recreated life's first allocation ({restarted}) must clear the previous life's mark ({last_applied})"
);
record_applied_peer_edit_generation(&mut receiver, "origin-site", restarted);
// …while the cleared life's in-flight leftovers stay fenced.
assert!(
peer_edit_delivery_is_stale(&receiver, "origin-site", straggler),
"a pre-removal in-flight delivery ({straggler}) must stay fenced after the rejoin"
);
}
@@ -195,6 +195,13 @@ IFS= read -r -d '' expected_docker_automatic_guard <<'EOF' || true
EOF
expected_docker_automatic_guard=${expected_docker_automatic_guard%$'\n'}
require_job_if "$docker_workflow" "build-check" "$expected_docker_automatic_guard"
require_line "$docker_workflow" ' source_ref: ${{ steps.check.outputs.source_ref }}' "Docker source ref output"
require_line "$docker_workflow" ' source_ref="$HEAD_SHA"' "automatic Docker source ref"
require_line "$docker_workflow" ' source_ref="$tag_ref"' "manual Docker source ref"
require_line "$docker_workflow" ' ref: ${{ needs.build-check.outputs.source_ref }}' "Docker release source checkout"
require_line "$docker_workflow" ' SOURCE_REVISION="$(git rev-parse HEAD)"' "Docker source revision resolution"
require_line "$docker_workflow" ' LABELS="$LABELS,org.opencontainers.image.revision=$SOURCE_REVISION"' "Docker revision label"
require_absent "$docker_workflow" 'org.opencontainers.image.revision=${{ github.sha }}' "Docker revision must not use the workflow branch SHA"
docker_manual_guard=$(awk '
$0 == " *-preview*)" { in_preview = 1 }