mirror of
https://github.com/rustfs/rustfs.git
synced 2026-07-26 08:18:18 +00:00
60ad15a7f9
* fix(obs): remove the dial9 task-dump switch that could never do anything Measured on a bench host (Linux x86_64) against the code merged in #4663: with `RUSTFS_RUNTIME_DIAL9_TASK_DUMP_ENABLED=true`, a `dial9-taskdump` build, and `--cfg tokio_taskdump`, dial9 recorded **zero** TaskDump events. dial9 captures a task dump only for futures it wrapped itself — those spawned through `dial9_tokio_telemetry::spawn`, which is where `TaskDumped<F>` gets applied. `tokio::spawn` gets no wrapper, and RustFS spawns with `tokio::spawn` throughout. Same workload, same binary, only the spawner changed: tokio::spawn -> 0 dumps dial9::spawn -> 14709 dumps, all with callchains Upstream documents this (README line 151) and tracks the doc gap at dial9-rs/dial9#477. I did not read it before wiring `with_task_dumps` in #4663, and so shipped exactly the kind of lying configuration knob that PR set out to delete. Remove it: the two environment variables, the config fields, the `with_task_dumps` call, and the `dial9-taskdump` feature — whose only effect was to constrain the build to Linux while recording nothing. Re-adding it only makes sense together with migrating the paths under investigation to dial9's spawner. Tracked as D9-16 in rustfs/backlog#1157. Also drop the `--cfg tokio_taskdump` requirement from the Makefile. Measured: dumps are captured with and without it (14709 vs 14674, within noise), and upstream never asked for it. That requirement was mine, invented and untested. Cargo.lock loses tokio's `backtrace` dependency, which `tokio/taskdump` pulled in. Co-Authored-By: heihutu <heihutu@gmail.com> * docs(obs): replace guessed dial9 retention numbers with measured ones Three corrections, all to claims I wrote in #4663 without measuring them. "Under a high poll rate that budget can wrap in minutes" was a guess. Measured on a single-node 4-drive cluster under warp mixed (66 MiB/s, 110 obj/s, 32 concurrent): 13023 events/s, 0.16 MiB/s, so the default 1 GiB budget wraps after roughly 108 minutes. Even at ten times the throughput that is ~11 minutes. State the measured rate and how to scale it instead. dial9 was described as the tool for drive stalls. It is not. RustFS does disk I/O on the blocking pool and through io_uring, never on an async worker, so a slow drive never lengthens a poll. Injecting 200 ms of latency on one of four drives cut throughput by 64% and left the poll distribution unchanged (polls >= 5 ms: 49 -> 56; p999: 2.67 ms -> 2.75 ms). Enabling dial9's CPU and sched profilers does not help: sched events are per-worker only, and the CPU profiler samples on-CPU while a stalled drive is an off-CPU wait. Say so plainly, and point at the `rustfs_io_*` metrics instead. What dial9 *is* good for, on the same traces: single polls of 418-625 ms with no fault injected at all — real worker stalls nothing else in the obs stack surfaces. Lead with that. Also link the two upstream issues filed for the gaps we documented: dial9-rs/dial9#658 (writer death unobservable) and #659 (worker-s3 CVEs). Measurements: rustfs/backlog#1157 (D9-11, D9-13, D9-18). Co-Authored-By: heihutu <heihutu@gmail.com> --------- Co-authored-by: heihutu <heihutu@gmail.com>
117 lines
5.1 KiB
TOML
117 lines
5.1 KiB
TOML
# Copyright 2024 RustFS Team
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
[package]
|
|
name = "rustfs-obs"
|
|
edition.workspace = true
|
|
license.workspace = true
|
|
repository.workspace = true
|
|
rust-version.workspace = true
|
|
version.workspace = true
|
|
homepage.workspace = true
|
|
description = "Observability and monitoring tools for RustFS, providing metrics, logging, and tracing capabilities."
|
|
keywords = ["observability", "metrics", "logging", "tracing", "RustFS"]
|
|
categories = ["web-programming", "development-tools::profiling", "asynchronous", "api-bindings", "development-tools::debugging"]
|
|
documentation = "https://docs.rs/rustfs-obs/latest/rustfs_obs/"
|
|
|
|
[features]
|
|
default = []
|
|
# Tokio runtime-level telemetry. Requires a `--cfg tokio_unstable` build; the
|
|
# build script fails the compile when that flag is missing. Off by default so
|
|
# ordinary builds neither pay for nor depend on Tokio's unstable API.
|
|
dial9 = ["dep:dial9-tokio-telemetry"]
|
|
#
|
|
# NOTE: there is deliberately no `dial9-taskdump` feature. dial9 only captures a
|
|
# task dump for futures it wrapped itself, i.e. those spawned via
|
|
# `dial9_tokio_telemetry::spawn`. RustFS spawns with `tokio::spawn` throughout,
|
|
# so enabling `tokio/taskdump` would cost a Linux-only build constraint and
|
|
# record nothing. Measured on an identical workload: 0 dumps via `tokio::spawn`,
|
|
# 14709 via `dial9::spawn`. Re-adding this feature only makes sense together
|
|
# with migrating the paths under investigation to dial9's spawner.
|
|
# See rustfs/backlog#1157 (D9-16) and dial9-rs/dial9#477.
|
|
#
|
|
# NOTE: there is deliberately no `dial9-s3` feature. dial9's `worker-s3` feature
|
|
# pulls aws-sdk-s3-transfer-manager 0.1.3 (its latest), which pins
|
|
# aws-smithy-http-client onto hyper-rustls 0.24 / rustls-webpki 0.101.7. That
|
|
# webpki carries RUSTSEC-2026-0098, -0099 and -0104, and `cargo deny` rejects it.
|
|
# Feature unification cannot drop a transitive dependency, so this can only be
|
|
# fixed upstream. Tracked in rustfs/backlog#1157 (D9-14).
|
|
gpu = ["dep:nvml-wrapper"]
|
|
pyroscope = ["dep:pyroscope"]
|
|
|
|
[[example]]
|
|
name = "dial9_smoke"
|
|
required-features = ["dial9"]
|
|
|
|
[lints]
|
|
workspace = true
|
|
|
|
[dependencies]
|
|
rustfs-audit = { workspace = true }
|
|
rustfs-common = { workspace = true }
|
|
rustfs-config = { workspace = true, features = ["constants", "observability"] }
|
|
# NOTE: This dependency on rustfs-ecstore is a known architectural limitation.
|
|
# The obs crate imports types from ecstore for metrics collection.
|
|
# Breaking this dependency would require defining traits in obs and
|
|
# implementing them in ecstore, which is a significant refactoring.
|
|
# See docs/architecture/obs-ecstore-dependency-inventory.md and
|
|
# https://github.com/rustfs/backlog/issues/735 for discussion.
|
|
rustfs-ecstore = { workspace = true }
|
|
rustfs-iam = { workspace = true }
|
|
rustfs-io-metrics = { workspace = true }
|
|
rustfs-notify = { workspace = true }
|
|
rustfs-security-governance = { workspace = true }
|
|
rustfs-storage-api = { workspace = true }
|
|
rustfs-utils = { workspace = true, features = ["ip"] }
|
|
chrono = { workspace = true }
|
|
flate2 = { workspace = true }
|
|
glob = { workspace = true }
|
|
jiff = { workspace = true }
|
|
metrics = { workspace = true }
|
|
crossbeam-channel = { workspace = true }
|
|
crossbeam-deque = { workspace = true }
|
|
crossbeam-utils = { workspace = true }
|
|
futures-util = { workspace = true }
|
|
num_cpus = { workspace = true }
|
|
opentelemetry = { workspace = true }
|
|
opentelemetry-appender-tracing = { workspace = true }
|
|
opentelemetry_sdk = { workspace = true }
|
|
opentelemetry-stdout = { workspace = true }
|
|
opentelemetry-otlp = { workspace = true }
|
|
opentelemetry-semantic-conventions = { workspace = true }
|
|
percent-encoding = { workspace = true }
|
|
serde = { workspace = true }
|
|
serde_json = { workspace = true }
|
|
tracing = { workspace = true, features = ["std", "attributes"] }
|
|
tracing-appender = { workspace = true }
|
|
tracing-error = { workspace = true }
|
|
tracing-opentelemetry = { workspace = true }
|
|
tracing-subscriber = { workspace = true, features = ["registry", "std", "fmt", "env-filter", "tracing-log", "time", "local-time", "json"] }
|
|
tokio = { workspace = true, features = ["sync", "fs", "rt-multi-thread", "rt", "time", "macros"] }
|
|
tokio-util = { workspace = true }
|
|
dial9-tokio-telemetry = { workspace = true, optional = true }
|
|
thiserror = { workspace = true }
|
|
zstd = { workspace = true, features = ["zstdmt"] }
|
|
sysinfo = { workspace = true }
|
|
nvml-wrapper = { workspace = true, optional = true }
|
|
|
|
[target.'cfg(any(target_os = "macos", all(target_os = "linux", target_env = "gnu", target_arch = "x86_64")))'.dependencies]
|
|
pyroscope = { workspace = true, features = ["backend-pprof-rs"], optional = true }
|
|
|
|
|
|
[dev-dependencies]
|
|
tokio = { workspace = true, features = ["full"] }
|
|
tempfile = { workspace = true }
|
|
temp-env = { workspace = true }
|