mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-05 19:55:37 +00:00
feat(ecstore): add the on-demand migration backfill job (#7087)
* feat(ecstore): add on-demand migration backfill job core Add the background backfill job for on-demand migration (rustfs/backlog#2159): a durable checkpoint under buckets/<bucket>/on-demand-migration-backfill.json saved by If-Match compare-and-set every 1000 keys or 10 s, a 60 s owner lease renewed by every save, a recovery pass that takes over expired leases (or jobs this node owned before a restart) and cancels jobs whose config changed, and a main loop over the source ListObjectsV2 pages with the skip_existing policy, dry runs, bounded outstanding pulls and wait-on-full enqueueing. The pull queue gains per-job completion reports so the job can count pulled/failed keys (hashes only), and pull permits become two-tier so an online miss is never queued behind a backfill pull. * feat(admin): expose on-demand migration backfill job Wire the ODM-12 backfill job (rustfs/backlog#2159) to its operators: POST /v3/on-demand-migration/{bucket}/backfill?op=start|cancel and GET .../backfill return the checkpoint document, GET .../status gains a backfill summary, and the recovery loop plus the process-wide runner are installed at startup. Backfill control reuses Set/GetBucketOnDemandMigrationAction and is recorded in the route policy, the registration matrix and the admin route snapshot. Add the rustfs-madmin wire types and client methods with golden fixtures shared by the server tests, the backfill_* metric descriptors and their collector, and three e2e scenarios: a full backfill across list pages, cancellation, and resuming from the persisted continuation token after a server restart.
This commit is contained in:
@@ -0,0 +1,254 @@
|
||||
// Copyright 2024 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
//! On-demand migration backfill job scenarios (ODM-12, rustfs/backlog#2159):
|
||||
//! a full backfill of a small-object source, cancellation, and resuming from
|
||||
//! the persisted continuation token after a server restart.
|
||||
|
||||
use super::common::{BackfillOp, BackfillRequest, ODM_SERVER_ENV, OdmSourceSpec, OdmTestEnv, SeedObject};
|
||||
use crate::fake_s3_target::Operation;
|
||||
use bytes::Bytes;
|
||||
use std::time::Duration;
|
||||
|
||||
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
|
||||
|
||||
const SOURCE_BUCKET: &str = "odm-backfill-source";
|
||||
const LOCAL_BUCKET: &str = "odm-backfill-local";
|
||||
const KEY_PREFIX: &str = "cold/";
|
||||
/// Keys per scenario. The fake source retains at most 4,096 object versions
|
||||
/// and 4,096 journal entries, and one pull is a HEAD plus a GET, so a
|
||||
/// scenario that asserts on the journal stays below ~2,000 keys. The job
|
||||
/// lists 1,000 keys per page, so this still spans several pages and exercises
|
||||
/// the continuation token, which is what the scenarios are about.
|
||||
const SEEDED_KEYS: usize = 1500;
|
||||
|
||||
fn key(i: usize) -> String {
|
||||
format!("{KEY_PREFIX}{i:05}")
|
||||
}
|
||||
|
||||
/// Content that identifies the key so a mis-stored object is caught.
|
||||
fn body(i: usize) -> Bytes {
|
||||
Bytes::from(format!("object-{i:05}-payload"))
|
||||
}
|
||||
|
||||
fn seed(env: &OdmTestEnv, count: usize) {
|
||||
let objects: Vec<SeedObject> = (0..count).map(|i| SeedObject::new(key(i), body(i))).collect();
|
||||
let etags = env.seed_source(SOURCE_BUCKET, &objects);
|
||||
assert_eq!(etags.len(), count);
|
||||
}
|
||||
|
||||
async fn configure(env: &OdmTestEnv, spec: &OdmSourceSpec) -> TestResult {
|
||||
let response = env.configure_source(LOCAL_BUCKET, spec).await?;
|
||||
assert_eq!(response.status, 200, "configure: {}", response.body);
|
||||
// The probe issued a one-key listing; count only the job's traffic.
|
||||
env.source.take_requests();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn source_lists(env: &OdmTestEnv) -> Vec<Option<String>> {
|
||||
env.source
|
||||
.requests()
|
||||
.into_iter()
|
||||
.filter(|record| record.operation == Operation::ListObjectsV2)
|
||||
.map(|record| record.continuation_token)
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn source_gets(env: &OdmTestEnv) -> usize {
|
||||
env.source
|
||||
.requests()
|
||||
.into_iter()
|
||||
.filter(|record| record.operation == Operation::GetObject)
|
||||
.count()
|
||||
}
|
||||
|
||||
fn counter(job: &serde_json::Value, name: &str) -> u64 {
|
||||
job[name].as_u64().unwrap_or_else(|| panic!("{name} missing in {job}"))
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn backfill_pulls_every_source_object_across_list_pages() -> TestResult {
|
||||
const COUNT: usize = SEEDED_KEYS;
|
||||
let env = OdmTestEnv::start().await?;
|
||||
env.source.create_bucket(SOURCE_BUCKET);
|
||||
env.rustfs.create_test_bucket(LOCAL_BUCKET).await?;
|
||||
seed(&env, COUNT);
|
||||
configure(&env, &env.fake_source_spec(SOURCE_BUCKET)).await?;
|
||||
|
||||
let started = env.start_backfill(LOCAL_BUCKET, BackfillRequest::default()).await?;
|
||||
assert_eq!(started.status, 200, "start: {}", started.body);
|
||||
let job = started.json()?["job"].clone();
|
||||
assert_eq!(job["state"], "running");
|
||||
assert_eq!(job["skip_existing"], "always");
|
||||
assert_eq!(job["dry_run"], false);
|
||||
let job_id = job["job_id"].as_str().expect("job id").to_string();
|
||||
|
||||
// A second start while the job holds its lease is a conflict.
|
||||
let again = env
|
||||
.backfill(LOCAL_BUCKET, BackfillOp::Start(BackfillRequest::default()))
|
||||
.await?;
|
||||
assert_eq!(again.status, 409, "second start: {}", again.body);
|
||||
assert!(again.body.contains("OnDemandMigrationBackfillRunning"), "{}", again.body);
|
||||
|
||||
let done = env
|
||||
.wait_for_backfill(LOCAL_BUCKET, Duration::from_secs(240), |job| job["state"] == "completed")
|
||||
.await?;
|
||||
assert_eq!(done["job_id"], job_id.as_str());
|
||||
assert_eq!(counter(&done, "listed"), COUNT as u64);
|
||||
assert_eq!(counter(&done, "enqueued"), COUNT as u64);
|
||||
assert_eq!(counter(&done, "pulled"), COUNT as u64);
|
||||
assert_eq!(counter(&done, "failed"), 0);
|
||||
assert_eq!(counter(&done, "skipped_existing"), 0);
|
||||
assert!(done["continuation_token"].is_null(), "a finished job carries no cursor");
|
||||
assert_eq!(done["last_key"], key(COUNT - 1));
|
||||
assert!(done["failed_keys"].as_array().is_some_and(Vec::is_empty));
|
||||
let expected_bytes: u64 = (0..COUNT).map(|i| body(i).len() as u64).sum();
|
||||
assert_eq!(counter(&done, "bytes"), expected_bytes);
|
||||
|
||||
assert_eq!(env.local_key_count(LOCAL_BUCKET, KEY_PREFIX).await?, COUNT);
|
||||
for i in [0, 999, 1000, 1200, COUNT - 1] {
|
||||
env.assert_local_present(LOCAL_BUCKET, &key(i), &body(i)).await;
|
||||
}
|
||||
|
||||
let lists = source_lists(&env);
|
||||
assert_eq!(lists.len(), COUNT.div_ceil(1000), "{COUNT} keys at 1000 per page: {lists:?}");
|
||||
assert!(lists[0].is_none(), "the first page starts without a cursor");
|
||||
assert!(lists[1..].iter().all(Option::is_some), "every later page carries the cursor");
|
||||
assert_eq!(source_gets(&env), COUNT, "every object is fetched exactly once");
|
||||
|
||||
// The status endpoint summarises the same job.
|
||||
let status = env.status(LOCAL_BUCKET).await?;
|
||||
assert_eq!(status.status, 200);
|
||||
let summary = status.json()?["backfill"].clone();
|
||||
assert_eq!(summary["job_id"], job_id.as_str());
|
||||
assert_eq!(summary["state"], "completed");
|
||||
assert_eq!(counter(&summary, "pulled"), COUNT as u64);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn backfill_cancel_stops_enqueueing_and_persists_cancelled() -> TestResult {
|
||||
const COUNT: usize = SEEDED_KEYS;
|
||||
let env = OdmTestEnv::start().await?;
|
||||
env.source.create_bucket(SOURCE_BUCKET);
|
||||
env.rustfs.create_test_bucket(LOCAL_BUCKET).await?;
|
||||
seed(&env, COUNT);
|
||||
let mut spec = env.fake_source_spec(SOURCE_BUCKET);
|
||||
spec.policy.max_concurrent_pulls = 1;
|
||||
configure(&env, &spec).await?;
|
||||
|
||||
// Cancelling before any job exists is a 404, not a silent success.
|
||||
let nothing = env.backfill(LOCAL_BUCKET, BackfillOp::Cancel).await?;
|
||||
assert_eq!(nothing.status, 404, "cancel without a job: {}", nothing.body);
|
||||
assert!(nothing.body.contains("NoSuchBackfillJob"), "{}", nothing.body);
|
||||
let unread = env.backfill(LOCAL_BUCKET, BackfillOp::Status).await?;
|
||||
assert_eq!(unread.status, 404, "status without a job: {}", unread.body);
|
||||
|
||||
let started = env.start_backfill(LOCAL_BUCKET, BackfillRequest::default()).await?;
|
||||
assert_eq!(started.status, 200, "start: {}", started.body);
|
||||
env.wait_for_backfill(LOCAL_BUCKET, Duration::from_secs(60), |job| {
|
||||
job["state"] == "running" && counter(job, "enqueued") > 0
|
||||
})
|
||||
.await?;
|
||||
|
||||
let cancelled = env.backfill(LOCAL_BUCKET, BackfillOp::Cancel).await?;
|
||||
assert_eq!(cancelled.status, 200, "cancel: {}", cancelled.body);
|
||||
let job = cancelled.json()?["job"].clone();
|
||||
assert_eq!(job["state"], "cancelled");
|
||||
let enqueued_at_cancel = counter(&job, "enqueued");
|
||||
assert!(enqueued_at_cancel < COUNT as u64, "the job was cancelled mid-way: {job}");
|
||||
|
||||
// Nothing is queued after the cancel: the checkpoint and the source
|
||||
// traffic both stop moving once the few in-flight pulls drain.
|
||||
tokio::time::sleep(Duration::from_secs(2)).await;
|
||||
let persisted = env.backfill_job(LOCAL_BUCKET).await?.expect("checkpoint kept for inspection");
|
||||
assert_eq!(persisted["state"], "cancelled");
|
||||
assert_eq!(counter(&persisted, "enqueued"), enqueued_at_cancel);
|
||||
let gets_after_drain = source_gets(&env);
|
||||
tokio::time::sleep(Duration::from_secs(1)).await;
|
||||
assert_eq!(source_gets(&env), gets_after_drain, "no source GET after the cancel drained");
|
||||
assert!(env.local_key_count(LOCAL_BUCKET, KEY_PREFIX).await? < COUNT);
|
||||
|
||||
// Cancel is idempotent and the status endpoint reports the final state.
|
||||
let again = env.backfill(LOCAL_BUCKET, BackfillOp::Cancel).await?;
|
||||
assert_eq!(again.status, 200, "second cancel: {}", again.body);
|
||||
assert_eq!(again.json()?["job"]["state"], "cancelled");
|
||||
let status = env.status(LOCAL_BUCKET).await?;
|
||||
assert_eq!(status.json()?["backfill"]["state"], "cancelled");
|
||||
|
||||
// A cancelled job releases the bucket: a new job can start.
|
||||
let restarted = env.start_backfill(LOCAL_BUCKET, BackfillRequest::default()).await?;
|
||||
assert_eq!(restarted.status, 200, "restart after cancel: {}", restarted.body);
|
||||
assert_ne!(restarted.json()?["job"]["job_id"], job["job_id"]);
|
||||
let _ = env.backfill(LOCAL_BUCKET, BackfillOp::Cancel).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn backfill_resumes_from_continuation_token_after_restart() -> TestResult {
|
||||
const COUNT: usize = SEEDED_KEYS;
|
||||
let mut env = OdmTestEnv::start().await?;
|
||||
env.source.create_bucket(SOURCE_BUCKET);
|
||||
env.rustfs.create_test_bucket(LOCAL_BUCKET).await?;
|
||||
seed(&env, COUNT);
|
||||
let mut spec = env.fake_source_spec(SOURCE_BUCKET);
|
||||
spec.policy.max_concurrent_pulls = 2;
|
||||
configure(&env, &spec).await?;
|
||||
|
||||
let started = env.start_backfill(LOCAL_BUCKET, BackfillRequest::default()).await?;
|
||||
assert_eq!(started.status, 200, "start: {}", started.body);
|
||||
let job_id = started.json()?["job"]["job_id"].as_str().expect("job id").to_string();
|
||||
|
||||
// Wait for the first page to be committed (cursor persisted), then kill
|
||||
// the server while the job is still running.
|
||||
let mid = env
|
||||
.wait_for_backfill(LOCAL_BUCKET, Duration::from_secs(120), |job| {
|
||||
job["state"] == "running" && job["continuation_token"].is_string()
|
||||
})
|
||||
.await?;
|
||||
assert!(counter(&mid, "listed") >= 1000 && counter(&mid, "listed") < COUNT as u64, "{mid}");
|
||||
env.source.take_requests();
|
||||
env.rustfs.restart_server_preserving_data(vec![], ODM_SERVER_ENV).await?;
|
||||
|
||||
let done = env
|
||||
.wait_for_backfill(LOCAL_BUCKET, Duration::from_secs(300), |job| job["state"] == "completed")
|
||||
.await?;
|
||||
assert_eq!(done["job_id"], job_id.as_str(), "the same job continues after the restart");
|
||||
assert_eq!(counter(&done, "failed"), 0);
|
||||
assert!(
|
||||
counter(&done, "listed") >= COUNT as u64,
|
||||
"the resumed job listed the rest (the interrupted page is listed twice): {done}"
|
||||
);
|
||||
// Keys pulled before the crash are re-listed and skipped, never re-pulled;
|
||||
// a pull whose report died with the old process is counted neither way,
|
||||
// so only the lower bound and the queue accounting are exact.
|
||||
assert!(counter(&done, "pulled") + counter(&done, "skipped_existing") >= COUNT as u64, "{done}");
|
||||
assert!(counter(&done, "pulled") <= counter(&done, "enqueued"), "{done}");
|
||||
assert_eq!(env.local_key_count(LOCAL_BUCKET, KEY_PREFIX).await?, COUNT);
|
||||
for i in [0, 500, 999, 1000, COUNT - 1] {
|
||||
env.assert_local_present(LOCAL_BUCKET, &key(i), &body(i)).await;
|
||||
}
|
||||
|
||||
let lists = source_lists(&env);
|
||||
assert!(!lists.is_empty(), "the resumed job listed the source");
|
||||
assert!(
|
||||
lists.iter().all(Option::is_some),
|
||||
"after the restart every source listing carries a continuation-token: {lists:?}"
|
||||
);
|
||||
assert!(
|
||||
lists.len() <= COUNT.div_ceil(1000),
|
||||
"the listing did not start over from the first page: {lists:?}"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
@@ -40,8 +40,12 @@ pub type BoxError = Box<dyn std::error::Error + Send + Sync>;
|
||||
/// turns it on so scenario tests exercise the feature without repeating it.
|
||||
pub const ODM_MODULE_SWITCH_ENV: &str = "RUSTFS_ON_DEMAND_MIGRATION_ENABLED";
|
||||
/// The source client shares the replication egress guard, which rejects the
|
||||
/// fake source's loopback endpoint unless this switch is set.
|
||||
/// fake source's loopback endpoint unless this switch is set. This is the
|
||||
/// documented harness opt-in; see
|
||||
/// `docs/operations/outbound-connection-policy.md`.
|
||||
pub const ALLOW_LOOPBACK_SOURCE_ENV: &str = "RUSTFS_REPLICATION_ALLOW_LOOPBACK_TARGET";
|
||||
/// Server environment every ODM scenario starts (and restarts) with.
|
||||
pub const ODM_SERVER_ENV: &[(&str, &str)] = &[(ODM_MODULE_SWITCH_ENV, "true"), (ALLOW_LOOPBACK_SOURCE_ENV, "true")];
|
||||
/// Admin route prefix; the bucket name is appended as one path segment.
|
||||
pub const ODM_ADMIN_ROUTE: &str = "/rustfs/admin/v3/on-demand-migration";
|
||||
/// Region the fake source is addressed with (it accepts any SigV4 region).
|
||||
@@ -324,7 +328,7 @@ impl OdmTestEnv {
|
||||
pub async fn start_with(options: OdmEnvOptions<'_>) -> Result<Self, BoxError> {
|
||||
let source = FakeS3Target::start_with_options(options.source).await?;
|
||||
let mut rustfs = RustFSTestEnvironment::new().await?;
|
||||
let mut env: Vec<(&str, &str)> = vec![(ODM_MODULE_SWITCH_ENV, "true"), (ALLOW_LOOPBACK_SOURCE_ENV, "true")];
|
||||
let mut env: Vec<(&str, &str)> = ODM_SERVER_ENV.to_vec();
|
||||
for (name, value) in &options.env {
|
||||
match env.iter_mut().find(|(existing, _)| existing == name) {
|
||||
Some(entry) => entry.1 = value,
|
||||
@@ -412,6 +416,81 @@ impl OdmTestEnv {
|
||||
}
|
||||
}
|
||||
|
||||
/// `POST .../{bucket}/backfill?op=start`, retried while the server
|
||||
/// answers `OnDemandMigrationDisabled`: the bucket state is built
|
||||
/// asynchronously right after the config `PUT`, so an immediate start can
|
||||
/// race it. Any other answer is returned as is.
|
||||
pub async fn start_backfill(&self, bucket: &str, request: BackfillRequest) -> Result<AdminResponse, BoxError> {
|
||||
let deadline = Instant::now() + Duration::from_secs(15);
|
||||
loop {
|
||||
let response = self.backfill(bucket, BackfillOp::Start(request.clone())).await?;
|
||||
let state_not_ready = response.status == 400 && response.body.contains("OnDemandMigrationDisabled");
|
||||
if !state_not_ready || Instant::now() >= deadline {
|
||||
return Ok(response);
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(100)).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// The `job` document of `GET .../{bucket}/backfill`, `None` on 404.
|
||||
pub async fn backfill_job(&self, bucket: &str) -> Result<Option<serde_json::Value>, BoxError> {
|
||||
let response = self.backfill(bucket, BackfillOp::Status).await?;
|
||||
match response.status {
|
||||
200 => Ok(Some(response.json()?["job"].clone())),
|
||||
404 => Ok(None),
|
||||
status => Err(format!("GET backfill answered {status}: {}", response.body).into()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Polls the backfill checkpoint until `accept` returns true or `timeout`
|
||||
/// elapses (an error naming the last observed document).
|
||||
pub async fn wait_for_backfill(
|
||||
&self,
|
||||
bucket: &str,
|
||||
timeout: Duration,
|
||||
accept: impl Fn(&serde_json::Value) -> bool,
|
||||
) -> Result<serde_json::Value, BoxError> {
|
||||
let deadline = Instant::now() + timeout;
|
||||
let mut last = serde_json::Value::Null;
|
||||
loop {
|
||||
if let Some(job) = self.backfill_job(bucket).await? {
|
||||
if accept(&job) {
|
||||
return Ok(job);
|
||||
}
|
||||
last = job;
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return Err(
|
||||
format!("backfill of {bucket} did not reach the expected state within {timeout:?}; last: {last}").into(),
|
||||
);
|
||||
}
|
||||
tokio::time::sleep(Duration::from_millis(100)).await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Number of keys under `prefix` listed by the RustFS under test (local
|
||||
/// state only, no migration side effects).
|
||||
pub async fn local_key_count(&self, bucket: &str, prefix: &str) -> Result<usize, BoxError> {
|
||||
let mut count = 0;
|
||||
let mut token: Option<String> = None;
|
||||
loop {
|
||||
let page = self
|
||||
.client
|
||||
.list_objects_v2()
|
||||
.bucket(bucket)
|
||||
.prefix(prefix)
|
||||
.max_keys(1000)
|
||||
.set_continuation_token(token.take())
|
||||
.send()
|
||||
.await?;
|
||||
count += page.contents().len();
|
||||
match page.next_continuation_token() {
|
||||
Some(next) if page.is_truncated().unwrap_or(false) => token = Some(next.to_string()),
|
||||
_ => return Ok(count),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async fn admin(
|
||||
&self,
|
||||
method: http::Method,
|
||||
|
||||
@@ -17,13 +17,15 @@
|
||||
//! `common` is the shared environment: one RustFS under test, one programmable
|
||||
//! fake S3 source, admin-API wrappers, seeding and local-state assertions.
|
||||
//! `harness_self_test` proves the harness itself; `get_basic_test` covers the
|
||||
//! GET read-through (rustfs/backlog#2156). The fault, concurrency,
|
||||
//! GET read-through (rustfs/backlog#2156) and `backfill_test` the background
|
||||
//! backfill job (ODM-12, rustfs/backlog#2159). The fault, concurrency,
|
||||
//! interaction and real-source matrix is rustfs/backlog#2158; its lane split
|
||||
//! lives in `.config/nextest.toml` (fault / concurrency / real source run
|
||||
//! nightly, the rest in the merge lane).
|
||||
|
||||
pub mod common;
|
||||
|
||||
mod backfill_test;
|
||||
mod concurrency_test;
|
||||
mod fault_test;
|
||||
mod get_basic_test;
|
||||
|
||||
Reference in New Issue
Block a user