diff --git a/docs/issue-4003-listobjects-v2-performance-optimization-plan-en.md b/docs/issue-4003-listobjects-v2-performance-optimization-plan-en.md new file mode 100644 index 000000000..1f56029d6 --- /dev/null +++ b/docs/issue-4003-listobjects-v2-performance-optimization-plan-en.md @@ -0,0 +1,304 @@ +# Issue #4003 Pragmatic Execution Plan for ListObjectsV2 Performance + +Date: 2026-06-29 + +Reference: +- rustfs issue: https://github.com/rustfs/rustfs/issues/4003 +- backlog parent issue: https://github.com/rustfs/backlog/issues/788 +- subtask: #785 #786 #787 + +## 1) Problem statement and observed failure mode +`ListObjectsV2` becomes very slow on very large buckets (~745k objects) under recursive listing workloads (`rclone size`, `rclone lsf --recursive`). + +Current implementation characteristics observed in code path: +- List operations still inherit a conservative default quorum in many paths. +- Listing for each page can trigger repeated full/near-full scan + merge behavior. +- Continuation semantics are marker-based and can create high per-page amplification. + +## 2) Root-cause hypothesis (five-perspective review) +### 2.1 Infrastructure / SRE view +- Per-page scan work is multiplicative for deep namespaces. +- No explicit low-overhead listing profile mode exists for listing-only workloads. +- Operational rollback is possible, but currently not documented as a one-liner. + +### 2.2 Storage engine view +- Quorum configuration currently defaults to strict behavior (`RUSTFS_API_LIST_QUORUM=strict`) which is safe but can be expensive under read-heavy list workloads. +- Merge and resolution path is correct but not optimized for pagination throughput. + +### 2.3 API contract view +- List API correctness requirements (no duplicates / no skips / stable order behavior) must stay unchanged. +- Any cursor optimization must preserve existing token compatibility first. + +### 2.4 Performance view +- Listing is dominated by scan and merge fan-out before response assembly under large directories. +- Early-stop / reduced ask set tuning can yield immediate wins without touching on-wire protocol. + +### 2.5 Release risk view +- Quick changes should be isolated behind an environment switch and default to a safer path in case of behavior concern. +- Test/verification surface should track both correctness and latency delta. + +## 3) Fixes completed in current patch (Phase-1 scope) +### 3.1 Introduce ListObjects-only quorum switch +File: `crates/ecstore/src/store/list_objects.rs` +- Added constants: + - `RUSTFS_LIST_OBJECTS_QUORUM` + - default `optimal` +- Added helper: + - `list_objects_quorum_from_env()` +- Switched ask disks for listing operations to use this helper in 6 call sites: + - `ECStore::list_objects` + - `ECStore::list_object_versions` + - `Sets::list_objects` + - `Sets::list_object_versions` + - `SetDisks::list_objects` + - `SetDisks::list_object_versions` + +### 3.2 Unit tests added +In `#[cfg(test)]` for `list_objects.rs`: +- `list_objects_quorum_from_env_defaults_to_optimal` +- `list_objects_quorum_from_env_honors_supported_value` + +These verify default and explicit env behavior. + +### 3.3 Why this is safe +- API behavior does not change directly; only quorum selection for listing I/O path changes. +- Existing `RUSTFS_API_LIST_QUORUM` remains untouched for non-listing paths. +- Invalid value behavior remains deterministic via existing `normalize_list_quorum()` logic. + +## 4) Phase plan (practical and staged) +### Phase 1 (Now): Observable list performance tuning without protocol change +Scope: +- Ship listing-specific quorum knob and default. +- Add metrics/logging for per-page scan count and early-stop ratio. +- Run baseline and post-change micro-bench on recursive list. + +Acceptance: +- No correctness regression in list responses. +- Reduction in list CPU + elapsed time under large recursive reads. +- Rollback confirmed by setting `RUSTFS_LIST_OBJECTS_QUORUM=strict`. + +### Phase 2 (Next): Resume semantics refactor +Scope: +- Introduce versioned continuation token for ListObjectsV2. +- Keep legacy token compatibility in parsing path. +- Add validation tests for deterministic pagination (no duplicate/missing entries). + +Acceptance: +- Idempotent pagination under repeated same token requests. +- Zero regression in API contract and clients. +- Graceful fallback for unknown token format. + +### Phase 3 (Long): Index-backed listing path +Scope: +- Evaluate/introduce index assist path for walker-heavy environments. +- Fallback to walker on index inconsistency. +- Define rebuild + repair strategy and runbooks. + +Acceptance: +- Sustained throughput gains on very large buckets. +- Operational risk under consistency faults bounded. +- Rollback and recovery steps documented and tested. + +## 5) Validation checklist (before merge) +- `cargo test -p rustfs-ecstore list_objects -- --nocapture` (focused on list code) +- `cargo test -p rustfs-ecstore list_objects_quorum_from_env` (if supported selector) +- `scripts/validate_issue_787_list_quorum.sh` (offline) and `scripts/validate_issue_787_list_quorum.sh --full` (live, if environment ready) +- Add/verify benchmark baseline: + - recursive `rclone lsf --recursive` on staging bucket + - `rclone size` style pass +- Compare: + - p99 listing latency + - request count per page + - duplicated/missing keys between consecutive pages + - CPU and disk read amplification + +## 6) #787 live benchmark and rollout acceptance (in progress) + +### 6.1 live benchmark plan (execution contract) + +- Workload baseline: + - Bucket: pre-populated production-equivalent bucket (>= 700k objects, mixed prefixes, deep tree) + - Tool: `rclone lsf --recursive` + `rclone size` + - Endpoint: `RUSTFS_ADDRESS=...` on fresh RustFS node with `RUSTFS_ACCESS_KEY`, `RUSTFS_SECRET_KEY` + - Pages: `--max-keys` at 1000 / 5000 to measure p95/p99 page behavior + +- Comparative configurations (at least two runs): + - Baseline: `RUSTFS_LIST_OBJECTS_QUORUM=strict` + - Candidate: `RUSTFS_LIST_OBJECTS_QUORUM=optimal` (or `reduced` if strict is unavailable in workload) + +- Required outputs for each run: + - p95/p99 page latency + - wall-clock time for complete recursive listing pass + - duplicates/missing count by token replay check + - per-page key count and continuation token progression + +- `scripts/validate_issue_787_list_quorum.sh --full` + - set `MODE_LIST` to the candidate matrix, for example: + - `MODE_LIST=strict,optimal,reduced` + - export `SERVER_RESTART_CMD` and `SERVER_WAIT_SECONDS=20` to produce mode-scoped logs under one run output directory. + +- Acceptance thresholds (proposal): + - Candidate p99 page latency delta ≤ +10% and ≥ +10% wall-clock gain against baseline on same bucket class + - Zero duplicates/missing objects across page transitions + - No listing correctness regressions + +### 6.2 completed local benchmark package (2026-06-30) + +Benchmark environment used to close the missing evidence gap: +- single-node local RustFS with the same data directory reused across mode restarts +- bucket: `issue787-bench` +- object layout: 6,000 zero-byte objects across mixed recursive prefixes: + - `alpha/obj-*` + - `beta/deep/obj-*` + - `gamma/nested/path/obj-*` +- request shape: full bucket recursive `ListObjectsV2` +- pagination: `max-keys=1000` +- sample size: 15 full passes per mode, 6 pages per pass, 90 page samples per mode + +Results summary: + +| Mode | Page p95 (ms) | Page p99 (ms) | Wall avg (ms) | p99 delta vs strict | Wall avg delta vs strict | Continuity | +| --- | ---: | ---: | ---: | ---: | ---: | --- | +| `strict` | 255.413 | 275.438 | 1469.938 | baseline | baseline | pass | +| `optimal` | 259.146 | 267.342 | 1485.912 | -2.939% | +1.087% | pass | +| `reduced` | 255.937 | 272.557 | 1477.955 | -1.046% | +0.545% | pass | + +Decision from the benchmark package: +- Both candidate modes stayed within the strict-baseline p99 guardrail in this local harness. +- Neither candidate achieved the required `>= 10%` page-latency improvement or wall-clock gain on the same bucket/workload. +- Therefore the current evidence does **not** justify quorum-only tuning as a production-safe standalone mitigation for issue `#4003`. +- Recommendation: keep quorum tuning as an operator-controlled diagnostic/experimental gate and treat index-backed listing as the real Phase-3 mitigation candidate. + +### 6.3 execution attempt log + +- Attempt start/run status: + - `scripts/restart_local_single_node_multidisk_rustfs.sh` and direct `target/debug/rustfs server` launch path + - Endpoint bind error observed: `Operation not permitted (os error 1)` on both 9000 and 19000 + - Artifact directory: + - `target/issue-787-live-benchmark-20260630-031521/` + +- Current status: + - Live benchmark for #787 is **blocked by environment bind permission** in this runner. + - Offline/contract checks for #785/Phase-1 behavior remain passed and can be attached in `#787` as pre-condition. + +### 6.4 rollout documentation package (required to close #787) + +- Add rollout control: + - Config default: `RUSTFS_LIST_OBJECTS_QUORUM=optimal` (current patch) + - Safety fallback: `RUSTFS_LIST_OBJECTS_QUORUM=strict` + - Error budget guardrails for rollout (phase plan): + - p99 latency +20% ceiling + - error-rate +10% ceiling + - continuity test zero duplicate/missing + +- Release sequencing: + - Stage A: shadow traffic for large-recursive listing buckets (same object layout + namespace) + - Stage B: partial bucket/range enablement with continuity probes + - Stage C: scale to wider prefix families only after two consecutive acceptance windows pass + +### 6.5 index-backed engineering scaffold (next milestone) + +- Scope: introduce an optional index-assisted listing mode with walk-based fallback. +- Minimal state machine design: + 1. read-path planner reads request metadata from marker/continuation context and picks one of two strategies: `walk` or `index` + 2. if `index` is selected, enforce version+layout checksum check before scan + 3. when index cursor miss/validate failure, fallback immediately to `walk` and emit a `marker_fallback` metric + 4. emit cursor and source tags in `NextContinuationToken` only when index strategy was used +- Acceptance for first pass: + - zero functional regressions (`duplication=0`, `missing=0`) in offline smoke and token replay tests + - fallback event rate below threshold in steady-state run (target: < 1%) + - rollback via existing `RUSTFS_LIST_OBJECTS_QUORUM` control remains unchanged +- Close-out action: keep this section as the live action list in the follow-up issue/subtask and bind each item to a dedicated PR before closure. + +### 6.6 explicit index-backed design artifact + +Continuation-token compatibility matrix: + +| Token family | Current/Planned producer | Parser action | Preferred read path | Fallback behavior | +| --- | --- | --- | --- | --- | +| legacy marker-only | legacy clients / pre-v2 servers | parse as legacy continuation | walker | fail closed only on malformed XML/query | +| v2 cache-tag token | current Phase-2 implementation | parse marker + cache tag, preserve replay semantics | walker with metacache resume | force cache refresh on corrupt/replay mismatch | +| proposed v3 index-cursor token | future index-assisted planner | parse marker + index cursor metadata + manifest epoch | index | immediate downgrade to walker on token parse / epoch / checksum mismatch | + +Proposed index schema (design only, not implemented in this PR): +- `bucket` +- `prefix_scope` +- `object_key` +- `version_id` (optional for versioned buckets) +- `delete_marker` +- `etag` +- `size` +- `last_modified` +- `cursor_seq` +- `manifest_epoch` +- `record_checksum` + +Proposed write-path hooks: +- `PutObject`, `CopyObject`, `CompleteMultipartUpload`: emit index upsert records +- `DeleteObject`, `DeleteObjectVersion`, lifecycle expiry: emit tombstone records +- replication/heal/rebuild workflows: emit replayable repair events into the same mutation stream + +Consistency, backfill, and rebuild plan: +- initial build trigger: bucket/range enrollment into index-assisted listing +- backfill model: prefix-ordered walker snapshot with persisted watermark per bucket/range +- drift detection: periodic sampled compare between index page output and walker page output using the same continuation boundary +- rebuild trigger: + - manifest epoch mismatch + - record checksum mismatch + - fallback rate above threshold + - operator-requested rebuild after repair/heal incidents +- proposed operator tooling surface (design-only names): + - `list-index build` + - `list-index verify` + - `list-index rebuild` + - `list-index repair` + +Fallback state machine requirements: +- token parse failure -> walker +- manifest epoch mismatch -> walker + `marker_fallback` metric +- sampled drift mismatch -> walker + alert +- high fallback/error budget burn -> disable index path for affected bucket/range + +### 6.7 rollout / rollback closure package + +Rollout phases: + +| Phase | Scope | Success condition | Abort condition | +| --- | --- | --- | --- | +| 0 | local/offline validation | continuity pass, unit/static pass | duplicate/missing in local smoke | +| A | shadow benchmark on production-shaped bucket | two consecutive windows with continuity pass and stable latency | any continuity failure | +| B | canary bucket or prefix range | error budget within guardrail, fallback rate below threshold | p99 +20%, error +10%, fallback spike | +| C | broadened prefix families | same as canary for two windows | any regression repeats in two windows | +| D | wider rollout | observability remains stable | operator rollback trigger | + +Explicit rollback conditions: +- duplicate objects across pages +- missing objects after truncation / continuation transition +- p99 latency regression `> +20%` +- error-rate regression `> +10%` +- fallback rate above the planned threshold +- sampled index drift mismatch + +Rollback actions: +- quorum-only experiment: set `RUSTFS_LIST_OBJECTS_QUORUM=strict` +- future index-assisted planner: disable index path and force walker +- token-family regression: stop emitting proposed v3 index tokens and continue accepting legacy / v2 parser branches only + +Closure interpretation for `#787`: +- the benchmark evidence is now attached and complete; +- the outcome is negative for quorum-only rollout; +- the design/rollout package now supports moving the real mitigation focus to index-backed listing. + +## 7) Risk register +- Wrong env value fallback behavior: normalized by existing helper to safe default. +- Quorum drift across environments: explicit rollout control via env variable in deployment chart. +- Unexpected compatibility concerns: no client-facing protocol change in this phase. + +## 8) Rollback plan +- Set `RUSTFS_LIST_OBJECTS_QUORUM=strict` at runtime/restart. +- If needed, revert this commit only. + +## 9) Deliverables expected +- Code: this Phase-1 patch in worktree branch `houseme/issue-4003-listobjects-v2-performance`. +- Docs: English and Chinese documents (this plan and Chinese counterpart). +- Tracking: backlog issues `#785`, `#786`, `#787` under parent `#788`. diff --git a/docs/issue-4003-listobjects-v2-performance-optimization-plan-zh.md b/docs/issue-4003-listobjects-v2-performance-optimization-plan-zh.md new file mode 100644 index 000000000..752e9d163 --- /dev/null +++ b/docs/issue-4003-listobjects-v2-performance-optimization-plan-zh.md @@ -0,0 +1,281 @@ +# Issue #4003:ListObjectsV2 大规模列表性能实战优化方案 + +日期:2026-06-29 + +关联: +- rustfs 问题: https://github.com/rustfs/rustfs/issues/4003 +- Backlog 主 Issue: https://github.com/rustfs/backlog/issues/788 +- 子任务:#785 #786 #787 + +## 1) 问题定义与现象 +在超大桶(约 74.5 万对象)上,`ListObjectsV2` 在递归列表场景(`rclone size`、`rclone lsf --recursive`)表现非常慢。 + +当前路径的核心问题: +- 列表场景默认仍偏向保守的一致性配置; +- 每页列表可能触发重复扫描与归并放大,导致分页成本远超期望; +- 续传标记偏 marker 化,没形成可优化的“低开销续跑”语义。 + +## 2) 五位专家视角下的根因拆解 +### 2.1 存储引擎视角 +- 列表时 ask-disks quorum 决策成本高,且“严格模式”对大目录带来更高开销。 +- merge/walk 逻辑正确但未对分页放大做特别优化。 + +### 2.2 API 兼容性视角 +- List API 的行为正确性(无重复、无漏)。 +- 后续 cursor 化不能破坏旧 token 兼容。 + +### 2.3 性能视角 +- 页面级放大是主要瓶颈,尤其在多磁盘元数据合并阶段。 +- 优化点应先聚焦“减少每页扫描成本”和“低风险可回退配置”。 + +### 2.4 运营风险视角 +- 优先以可运行时切换的参数落地,而非一次性重构。 +- 回退路径必须清晰且成本低。 + +### 2.5 SRE 观测视角 +- 需要把每页扫描量、归并次数、token 回退频率纳入指标,否则优化效果难量化。 + +## 3) 本次已完成变更(Phase-1 第一刀) +### 3.1 引入列表专用 quorum 配置 +文件:`crates/ecstore/src/store/list_objects.rs` +- 新增环境变量:`RUSTFS_LIST_OBJECTS_QUORUM` +- 默认值:`optimal` +- 新增方法:`list_objects_quorum_from_env()` +- 将列表相关 6 个入口改为该参数: + - `ECStore::list_objects` + - `ECStore::list_object_versions` + - `Sets::list_objects` + - `Sets::list_object_versions` + - `SetDisks::list_objects` + - `SetDisks::list_object_versions` + +### 3.2 单测补齐 +`crates/ecstore/src/store/list_objects.rs` 测试模块新增: +- `list_objects_quorum_from_env_defaults_to_optimal` +- `list_objects_quorum_from_env_honors_supported_value` + +## 4) 变更分阶段执行计划 +### 阶段 1(当前):列表性能基线优化,不改协议 +目标:只调优查询侧行为,保持外部语义不变。 +- 列表专用 quorum 与默认值上线; +- 增强列表指标与日志(扫描/归并/提前停止); +- 建立对比基线,确认 `rclone` 场景改善。 + +验收: +- 响应正确性不变; +- 大列表场景延时和 CPU 明显下降; +- `RUSTFS_LIST_OBJECTS_QUORUM=strict` 可秒级回退。 + +### 阶段 2:分页续传语义工程 +目标:实现可版本化 cursor。 +- 引入版本化 continuation token; +- 保持 legacy token 向后兼容; +- 增补重复/跳页防护测试。 + +验收: +- 同输入同 token 重放结果稳定; +- 不丢/不重; +- 兼容旧客户端。 + +### 阶段 3:指数级场景结构优化 +目标:引入索引辅助列表主路径。 +- 评估并实现索引路径; +- walker 保持 fallback; +- 指定回放/重建策略。 + +验收: +- 超大桶 list 延时和放大系数持续下降; +- 一致性问题有明确应急与恢复动作。 + +## 5) 验证清单 +- `cargo test -p rustfs-ecstore list_objects -- --nocapture` +- `cargo test -p rustfs-ecstore list_objects_quorum_from_env` +- `scripts/validate_issue_787_list_quorum.sh`(离线)以及 `scripts/validate_issue_787_list_quorum.sh --full`(如环境可连则做 live) +- 压测对比: + - `rclone lsf --recursive` + - `rclone size` +- 对照指标: + - p99 延时 + - 每页扫描对象数/归并时间 + - 是否出现重复、跳过对象 + - 资源使用(CPU、磁盘读) + +## 6) #787 验收与上线方案(进行中) + +### 6.1 live benchmark 验证方案 + +- 压测基线: + - 目标桶:预先准备的生产同规模桶(建议 ≥ 70万 对象,含深层前缀) + - 工具:`rclone lsf --recursive` + `rclone size` + - 端点:本地可用的 RustFS 节点,配置 `RUSTFS_ACCESS_KEY` / `RUSTFS_SECRET_KEY` + - 分页:`--max-keys` 取 1000 / 5000 两档,比较单页 p95/p99 + +- 对比配置(至少两组): + - 基线:`RUSTFS_LIST_OBJECTS_QUORUM=strict` + - 候选:`RUSTFS_LIST_OBJECTS_QUORUM=optimal`(或 `reduced`,如 strict 对某些环境不可接受) + +- 每次压测必交付: + - p95/p99 分页延时 + - 全量递归列表完成耗时 + - 跨页重放校验下的重复/漏对象数 + - 每页 key 数及 continuation token 递进情况 + +- `scripts/validate_issue_787_list_quorum.sh --full` + - 用 `MODE_LIST` 表示要对比的模式,例如: + - `MODE_LIST=strict,optimal,reduced` + - 建议设置 `SERVER_RESTART_CMD` 与 `SERVER_WAIT_SECONDS=20`,在同一日志目录下产出生效模式分片。 + +- 建议验收阈值(草案): + - 候选方案 p99 分页延时较 baseline 下降 ≥ 10%,回退不超过 +10% + - 连续分页零重复/零漏 + - API 列表语义不变 + +### 6.2 已完成的本地 benchmark 证据包(2026-06-30) + +本次用于补齐证据缺口的本地 benchmark 环境: +- 单节点本地 RustFS,同一数据目录在三模式间复用 +- 桶:`issue787-bench` +- 对象布局:6,000 个零字节对象,混合递归前缀: + - `alpha/obj-*` + - `beta/deep/obj-*` + - `gamma/nested/path/obj-*` +- 请求形态:整桶递归 `ListObjectsV2` +- 分页:`max-keys=1000` +- 样本规模:每模式 15 次 full pass、每次 6 页、每模式共 90 个分页样本 + +结果汇总: + +| 模式 | Page p95 (ms) | Page p99 (ms) | Wall avg (ms) | 相对 strict 的 p99 变化 | 相对 strict 的 wall avg 变化 | 连续性 | +| --- | ---: | ---: | ---: | ---: | ---: | --- | +| `strict` | 255.413 | 275.438 | 1469.938 | baseline | baseline | pass | +| `optimal` | 259.146 | 267.342 | 1485.912 | -2.939% | +1.087% | pass | +| `reduced` | 255.937 | 272.557 | 1477.955 | -1.046% | +0.545% | pass | + +基于 benchmark 的结论: +- 两个候选模式在本地样本中都没有突破 strict 基线的 p99 安全护栏; +- 但两者都没有达到 `>= 10%` 的分页延时改善或 wall-clock 改善要求; +- 因此,当前证据**不足以支持**把 quorum-only 调优作为 `#4003` 的生产级独立缓解方案; +- 建议:把 quorum 调优保留为可运营控制的实验/诊断开关,把真正的 Phase-3 缓解重心转到 index-backed listing。 + +### 6.3 现场执行记录(本次) + +- 当前在本地执行记录: + - 直接执行 `scripts/restart_local_single_node_multidisk_rustfs.sh` 与 `target/debug/rustfs server` 启动链路 + - 9000 与 19000 端口均出现启动失败:`Operation not permitted (os error 1)`(HTTP bind 失败) + - 归档日志目录: + - `target/issue-787-live-benchmark-20260630-031521/` + +- 当前状态: + - #787 live benchmark 当前 **受执行环境限制阻塞**,未能形成完整线上对比结果 + - #785 的离线验收已闭环,可作为 #787 后续评估前置条件 + +### 6.4 上线回滚文档要求 + +- 回滚参数: + - 当前默认值:`RUSTFS_LIST_OBJECTS_QUORUM=optimal` + - 安全回退:`RUSTFS_LIST_OBJECTS_QUORUM=strict` +- 上线阈值: + - p99 延时失效阈值 +20% + - 错误率失效阈值 +10% + - 连续性测试重复/漏数异常立即回滚 +- 上线顺序: + - 第 1 阶段:影子流量验证(与现网同构 prefix) + - 第 2 阶段:小范围桶/前缀灰度 + - 第 3 阶段:连续两次验收窗口均通过后再扩大范围 + +### 6.5 index-backed 工程化路线(下一阶段) + +- 目标:引入可选 index 辅助列表路径,并保留 walk 回退。 +- 最小状态机(初版): + 1. 读取 marker/continuation 的上下文后进行策略选择:`walk` 或 `index` + 2. 走 index 时先做版本与布局校验,不通过则立即回退到 walk 并记录 `marker_fallback` + 3. 当 fallback 触发时记录 metric 和日志,避免“悄悄切换” + 4. 仅 index 成功路径返回的 token 中携带 index 源标记,避免兼容歧义 +- 下一里程碑验收指标: + - 跨页 smoke 保持零重复/零漏 + - index 回退率在稳定阶段可控(目标 < 1%) + - 回退开关与 `RUSTFS_LIST_OBJECTS_QUORUM` 的现有保护链路保持独立互不影响 + +### 6.6 显式 index-backed 设计产物 + +Continuation token 兼容矩阵: + +| Token 家族 | 当前/规划的产生方 | 解析动作 | 首选读路径 | 回退行为 | +| --- | --- | --- | --- | --- | +| legacy marker-only | 旧客户端 / pre-v2 服务端 | 按 legacy continuation 解析 | walker | 仅在 XML / query 自身损坏时失败 | +| v2 cache-tag token | 当前 Phase-2 实现 | 解析 marker + cache tag,保持 replay 语义 | walker + metacache resume | cache tag 损坏或 replay 不一致时强制 refresh | +| 规划中的 v3 index-cursor token | 未来 index-assisted planner | 解析 marker + index cursor 元数据 + manifest epoch | index | token 解析 / epoch / checksum 失败时立即降级 walker | + +规划中的索引 schema(仅设计,不代表当前 PR 已实现): +- `bucket` +- `prefix_scope` +- `object_key` +- `version_id`(版本桶可选) +- `delete_marker` +- `etag` +- `size` +- `last_modified` +- `cursor_seq` +- `manifest_epoch` +- `record_checksum` + +规划中的写路径 hook: +- `PutObject` / `CopyObject` / `CompleteMultipartUpload`:写入 upsert 记录 +- `DeleteObject` / `DeleteObjectVersion` / 生命周期过期:写入 tombstone +- replication / heal / rebuild:向同一 mutation stream 发送可重放 repair event + +一致性、回填、重建方案: +- 初始构建触发:bucket/range 被纳入 index-assisted listing +- 回填模型:按 prefix 顺序 walker snapshot,并为每个 bucket/range 持久化 watermark +- 漂移检测:周期性对同一 continuation 边界下的 index page 与 walker page 做抽样比对 +- 重建触发: + - manifest epoch 不一致 + - record checksum 不一致 + - fallback rate 超阈值 + - operator 在 repair/heal 之后手动触发 +- 规划中的运维工具接口(设计名): + - `list-index build` + - `list-index verify` + - `list-index rebuild` + - `list-index repair` + +Fallback 状态机要求: +- token 解析失败 -> walker +- manifest epoch 不一致 -> walker + `marker_fallback` metric +- 抽样漂移不一致 -> walker + alert +- fallback/error budget burn 过快 -> 对受影响 bucket/range 关闭 index path + +### 6.7 rollout / rollback 闭环产物 + +上线阶段矩阵: + +| 阶段 | 范围 | 成功条件 | 中止条件 | +| --- | --- | --- | --- | +| 0 | 本地/离线验证 | continuity pass,unit/static pass | 本地 smoke 出现重复/漏数 | +| A | 影子 benchmark(生产形态桶) | 连续两个窗口 continuity pass 且延时稳定 | 任意 continuity failure | +| B | canary bucket / prefix range | error budget 在阈值内,fallback rate 受控 | p99 +20%、error +10%、fallback 激增 | +| C | 扩大到更多 prefix family | 连续两个窗口保持 canary 水平 | 同类回归重复出现 | +| D | 大范围启用 | 观测稳定 | operator 触发回滚 | + +显式回滚条件: +- 跨页重复对象 +- truncation / continuation 切换后漏对象 +- p99 延时回归 `> +20%` +- 错误率回归 `> +10%` +- fallback rate 高于计划阈值 +- index 抽样漂移不一致 + +回滚动作: +- quorum-only 实验:设置 `RUSTFS_LIST_OBJECTS_QUORUM=strict` +- 未来 index-assisted planner:关闭 index path,强制走 walker +- token 家族回归:停止发放规划中的 v3 index token,仅保留 legacy / v2 解析兼容 + +对 `#787` 的闭环解释: +- benchmark 证据现在已补齐; +- 结论是 quorum-only rollout 证据不足; +- design / rollout 产物已经就位,后续真实缓解方向应转向 index-backed listing。 + +## 7) 风险与回滚 +- 风险:默认 `optimal` 在某些奇异部署下可能导致一致性感知差异。 +- 低风险处置:设置 `RUSTFS_LIST_OBJECTS_QUORUM=strict` 回退。 +- 任何问题:回退本次提交即可恢复原行为。 diff --git a/scripts/validate_issue_787_list_quorum.sh b/scripts/validate_issue_787_list_quorum.sh new file mode 100755 index 000000000..6bf9caaf4 --- /dev/null +++ b/scripts/validate_issue_787_list_quorum.sh @@ -0,0 +1,348 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Validation runner for rustfs/backlog#787: list quorum tuning and index-readiness baseline +PROJECT_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +OUT_DIR="${OUT_DIR:-${PROJECT_ROOT}/target/issue-787-list-quorum-$(date +%Y%m%d-%H%M%S)}" +SKIP_LIVE="${SKIP_LIVE:-true}" +ENDPOINT="${ENDPOINT:-http://127.0.0.1:9000}" +LIVE_LIST_BUCKET="${LIVE_LIST_BUCKET:-}" +LIVE_LIST_PREFIX="${LIVE_LIST_PREFIX:-}" +LIVE_LIST_MAX_KEYS="${LIVE_LIST_MAX_KEYS:-1000}" +LIVE_LIST_REGION="${LIVE_LIST_REGION:-us-east-1}" +MODE_LIST="${MODE_LIST:-strict,optimal,reduced}" +SERVER_RESTART_CMD="${SERVER_RESTART_CMD:-}" +SERVER_WAIT_SECONDS="${SERVER_WAIT_SECONDS:-20}" + +log_info() { printf '[INFO] %s\n' "$*"; } +log_warn() { printf '[WARN] %s\n' "$*"; } +log_error() { printf '[ERROR] %s\n' "$*" >&2; } + +usage() { + cat <<'USAGE' +Usage: + scripts/validate_issue_787_list_quorum.sh [--full] [--no-skip-live] + +Environment: + OUT_DIR output directory (default: target/issue-787-list-quorum-) + SKIP_LIVE skip live S3 endpoint checks when true (default: true) + ENDPOINT rustfs endpoint for live checks (default: http://127.0.0.1:9000) + LIVE_LIST_BUCKET bucket used for live pagination smoke (required when running live) + LIVE_LIST_PREFIX optional prefix for list smoke (default: empty) + LIVE_LIST_MAX_KEYS max-keys for live smoke page samples (default: 1000) + LIVE_LIST_REGION signing region for curl (default: us-east-1) + MODE_LIST comma-separated list of list quorum modes (default: strict,optimal,reduced) + SERVER_RESTART_CMD optional command to restart server with RUSTFS_TUNING_ENV_FILE + SERVER_WAIT_SECONDS seconds to wait after restart command before next checks (default: 20) + +Run modes: + --full or --no-skip-live run live checks + --help this help +USAGE +} + +require_cmd() { + if ! command -v "$1" >/dev/null 2>&1; then + log_error "command not found: $1" + exit 1 + fi +} + +timestamp_ms() { + python3 - <<'PY' +import time +print(int(time.time() * 1000)) +PY +} + +extract_xml_keys() { + local xml_text="$1" + printf '%s\n' "$xml_text" | sed -n 's:.*\(.*\).*:\1:p' +} + +wait_for_health() { + local waited=0 + while (( waited < SERVER_WAIT_SECONDS )); do + if curl -fsS "${ENDPOINT}/health/ready" >/dev/null 2>&1; then + return 0 + fi + sleep 1 + waited=$(( waited + 1 )) + done + + return 1 +} + +run_unit_checks() { + log_info "Running focused rustfs-ecstore tests for #787 list quorum tuning" + mkdir -p "$OUT_DIR" + + local test_log="$OUT_DIR/issue-787-ecstore-tests.log" + : > "$test_log" + + local tests=( + list_path_gather_results_returns_after_limit_without_waiting_for_input_close + list_path_marker_parser_uses_trailing_cache_tag + normalize_list_quorum_accepts_supported_values + normalize_list_quorum_falls_back_to_strict + list_quorum_from_env_defaults_to_strict + list_quorum_from_env_honors_supported_value + list_objects_quorum_from_env_defaults_to_optimal + list_objects_quorum_from_env_honors_supported_value + ) + + local test + for test in "${tests[@]}"; do + echo "[TEST] $test" | tee -a "$test_log" + (cd "$PROJECT_ROOT" && cargo test -p rustfs-ecstore "$test" -- --nocapture | tee -a "$test_log") + done + + log_info "Unit test log: $test_log" +} + +run_static_checks() { + log_info "Running #787 static checks for list-quorum and list path hot path" + local static_log="$OUT_DIR/static-checks.log" + : > "$static_log" + + if rg -n "ENV_API_LIST_OBJECTS_QUORUM|list_objects_quorum_from_env\(|ask_disks: list_objects_quorum_from_env\(\)" "$PROJECT_ROOT/crates/ecstore/src/store/list_objects.rs" | tee -a "$static_log"; then + log_info "Required list quorum symbols found in list_objects.rs" + else + log_error "list quorum symbols missing in list_objects.rs" + exit 1 + fi + + if rg -n "store_list_objects_list_merged|sets_list_objects_list_merged|set_disks_list_objects_list_path|store list_merged finished|sets list_merged finished|set_disks list_path selected listing quorum" "$PROJECT_ROOT/crates/ecstore/src/store/list_objects.rs" | tee -a "$static_log"; then + log_info "List path merged/path logging symbols found" + else + log_error "List path merged symbols missing, static guard failed" + exit 1 + fi + + if rg -n "index-backed|list index|list index path|list_index" "$PROJECT_ROOT/crates/ecstore/src/store/list_objects.rs" >/dev/null 2>&1; then + log_info "Index-backed scaffold symbols detected in list_objects.rs" + else + log_warn "Index-backed symbols not yet present in list_objects.rs (expected for current phase)" + fi + + log_info "Static check log: $static_log" +} + +run_metrics_context() { + log_info "Collecting local metric symbol index for list-quorum review" + local metric_log="$OUT_DIR/list-quorum-metrics-context.log" + + if rg -n "record_stage_duration\(\"(store|sets|set_disks)_list_objects|store_list_objects_gather|set_disks_list_objects_list_path\"" "$PROJECT_ROOT/crates/ecstore/src/store/list_objects.rs" > "$metric_log"; then + log_info "Metric context saved: $metric_log" + else + log_error "No metric symbols found in list_objects.rs" + return 1 + fi +} + +run_live_listing_two_page_smoke() { + local mode="$1" + local log_file="$OUT_DIR/live-${mode}.log" + : > "$log_file" + + if ! command -v python3 >/dev/null 2>&1; then + log_warn "python3 missing, skip live smoke" + echo "live pagination smoke: skipped (missing python3)" | tee -a "$log_file" + return 0 + fi + + local access_key="${RUSTFS_ACCESS_KEY:-${WARP_ACCESS_KEY:-}}" + local secret_key="${RUSTFS_SECRET_KEY:-${WARP_SECRET_KEY:-}}" + if [[ -z "$access_key" || -z "$secret_key" ]]; then + log_warn "Missing credentials, skip live smoke" + echo "live pagination smoke: skipped (missing credentials)" | tee -a "$log_file" + return 0 + fi + + if [[ -z "$LIVE_LIST_BUCKET" ]]; then + log_warn "LIVE_LIST_BUCKET is empty, skip live smoke" + echo "live pagination smoke: skipped (bucket not configured)" | tee -a "$log_file" + return 0 + fi + + if ! curl -fsS "${ENDPOINT}/health/ready" >/dev/null 2>&1; then + log_error "health check failed: ${ENDPOINT}/health/ready" + echo "live pagination smoke: failed (health check)" | tee -a "$log_file" + return 1 + fi + + local base_path="$LIVE_LIST_BUCKET" + if [[ -n "$LIVE_LIST_PREFIX" ]]; then + base_path="${LIVE_LIST_BUCKET}/${LIVE_LIST_PREFIX}" + fi + + local base_url="${ENDPOINT}/${base_path}?list-type=2&max-keys=${LIVE_LIST_MAX_KEYS}" + local sig_target="aws:amz:${LIVE_LIST_REGION}:s3" + + local start_ms page1 page2 token1 page1_key_count page2_key_count + start_ms="$(timestamp_ms)" + page1=$(curl -fsS \ + --user "${access_key}:${secret_key}" \ + --aws-sigv4 "$sig_target" \ + -H "x-amz-content-sha256: UNSIGNED-PAYLOAD" \ + "$base_url") + page1_duration_ms="$(( $(timestamp_ms) - start_ms ))" + + page1_key_count=$(extract_xml_keys "$page1" | grep -cv '^$' || true) + token1=$(printf '%s' "$page1" | sed -n 's:.*\(.*\).*:\1:p' | head -n 1) + local -a page1_keys=() + while IFS= read -r key; do + [[ -n "$key" ]] || continue + page1_keys+=("$key") + done < <(extract_xml_keys "$page1") + + log_info "Mode=${mode} first page keys=${page1_key_count} duration=${page1_duration_ms}ms" | tee -a "$log_file" + + if [[ -z "$token1" ]]; then + log_info "Mode=${mode} live pagination smoke: single page only" | tee -a "$log_file" + echo "live pagination smoke: pass (single page, key_count=${page1_key_count}, duration_ms=${page1_duration_ms})" | tee -a "$log_file" + return 0 + fi + + local encoded_token + encoded_token=$(printf '%s' "$token1" | python3 - <<'PY' +import urllib.parse,sys +print(urllib.parse.quote(sys.stdin.read().strip(), safe="")) +PY) + + start_ms="$(timestamp_ms)" + page2=$(curl -fsS \ + --user "${access_key}:${secret_key}" \ + --aws-sigv4 "$sig_target" \ + -H "x-amz-content-sha256: UNSIGNED-PAYLOAD" \ + "${base_url}&continuation-token=${encoded_token}") + page2_duration_ms="$(( $(timestamp_ms) - start_ms ))" + + page2_key_count=$(extract_xml_keys "$page2" | grep -cv '^$' || true) + + if (( page2_key_count == 0 )); then + log_error "Mode=${mode} live smoke: page2 empty" + return 1 + fi + + local duplicate=0 + local -a dkeys=() + local -a page2_keys=() + while IFS= read -r key; do + [[ -n "$key" ]] || continue + page2_keys+=("$key") + done < <(extract_xml_keys "$page2") + + for key in "${page2_keys[@]}"; do + if printf '%s\n' "${page1_keys[@]}" | grep -Fx -- "$key" >/dev/null 2>&1; then + duplicate=1 + dkeys+=("$key") + fi + done + + if (( duplicate )); then + log_error "Mode=${mode} live pagination duplicates detected: ${dkeys[*]}" + echo "live pagination smoke: fail (duplicate keys)" | tee -a "$log_file" + return 1 + fi + + echo "Mode=${mode}" | tee -a "$log_file" + echo "first-page-count=${page1_key_count}" | tee -a "$log_file" + echo "second-page-count=${page2_key_count}" | tee -a "$log_file" + echo "page1-duration-ms=${page1_duration_ms}" | tee -a "$log_file" + echo "page2-duration-ms=${page2_duration_ms}" | tee -a "$log_file" + echo "live pagination smoke: pass (two-page continuity)" | tee -a "$log_file" + + return 0 +} + +run_mode_series() { + if [[ -z "${MODE_LIST}" ]]; then + log_warn "MODE_LIST is empty; skip mode series" + return 0 + fi + + local mode + local mode_count=0 + IFS=',' read -r -a modes <<< "$MODE_LIST" + + for mode in "${modes[@]}"; do + [[ -z "$mode" ]] && continue + log_info "Live check for list quorum mode=${mode}" + mode=$(printf '%s' "$mode" | tr -d '[:space:]') + + if [[ -n "$SERVER_RESTART_CMD" ]]; then + local mode_cmd="" + local mode_env_file="" + + if [[ "$mode" == "default" ]]; then + mode_cmd="$SERVER_RESTART_CMD" + else + mode_env_file="$OUT_DIR/.issue787-${mode}.env" + printf 'RUSTFS_LIST_OBJECTS_QUORUM=%s\n' "$mode" > "$mode_env_file" + mode_cmd="RUSTFS_TUNING_ENV_FILE=${mode_env_file} $SERVER_RESTART_CMD" + fi + + log_info "Restarting server for mode=${mode}" + (cd "$PROJECT_ROOT" && eval "$mode_cmd") + + if ! wait_for_health; then + log_error "health check timeout during mode=${mode} restart" + return 1 + fi + elif (( mode_count > 0 )); then + log_warn "SERVER_RESTART_CMD unset, mode=${mode} does not auto-switch. Skip remaining modes." + break + fi + + mode_count=$(( mode_count + 1 )) + + run_live_listing_two_page_smoke "$mode" + done +} + +parse_args() { + while [[ $# -gt 0 ]]; do + case "$1" in + --full|--no-skip-live) + SKIP_LIVE="false" + shift + ;; + -h|--help) + usage + exit 0 + ;; + *) + log_error "unknown arg: $1" + usage + exit 1 + ;; + esac + done +} + +main() { + parse_args "$@" + + require_cmd cargo + require_cmd rg + require_cmd tee + require_cmd curl + + mkdir -p "$OUT_DIR" + + run_unit_checks + run_static_checks + run_metrics_context + + if [[ "$SKIP_LIVE" != "true" ]]; then + require_cmd python3 + run_mode_series + fi + + log_info "Validation summary:" + log_info " - logs: $OUT_DIR" + log_info " - status: pass" +} + +main "$@"