Files
pulse-triage[bot] ee6bb64d72 Honor cgroup v1 service memory limits
Resolve the process memory-controller hierarchy before selecting the tightest v1 hard limit, so systemd and nested appliance limits actually inform the existing Go runtime headroom.

Contract-Neutral: runtime memory-limit detection only; no API or persistence contract change
2026-08-30 02:15:36 +01:00

222 lines
6.3 KiB
Go

package server
import (
"math"
"os"
"path/filepath"
"runtime/debug"
"strconv"
"strings"
"github.com/rs/zerolog/log"
)
const (
cgroupV2Root = "/sys/fs/cgroup"
// cgroupV1NoLimit is the value cgroup v1 reports when no memory limit is
// set (PAGE_SIZE-rounded max int64).
cgroupV1NoLimit = int64(9223372036854771712)
// memoryLimitHeadroomPercent is how much of the cgroup limit the Go
// runtime may use. The remainder absorbs non-heap memory the GC cannot
// control: goroutine stacks, mmap'd files, and CGO allocations.
memoryLimitHeadroomPercent = 90
// minimumUsableMemoryLimit guards against nonsense cgroup values; below
// this a soft limit would just make the GC thrash.
minimumUsableMemoryLimit = int64(64 << 20)
)
// applyRuntimeMemoryLimit aligns the Go GC with the enclosing cgroup memory
// limit so a capped service (systemd MemoryMax, docker --memory, Kubernetes
// limits) tightens garbage collection as it approaches the cap instead of
// growing until the kernel OOM-kills it. Without this the GC paces itself
// purely off GOGC and is blind to the limit. Best effort: any failure leaves
// the runtime untouched.
func applyRuntimeMemoryLimit() {
if v := strings.TrimSpace(os.Getenv("GOMEMLIMIT")); v != "" {
// The runtime already honors the explicit operator setting.
log.Debug().Str("GOMEMLIMIT", v).Msg("runtime memory limit set from environment")
return
}
limit, ok := readCgroupMemoryLimit()
if !ok {
log.Debug().Msg("no cgroup memory limit detected; leaving GC defaults")
return
}
soft := limit / 100 * memoryLimitHeadroomPercent
if soft < minimumUsableMemoryLimit {
log.Warn().Int64("cgroupLimitBytes", limit).Msg("cgroup memory limit too small for a GC soft limit; leaving GC defaults")
return
}
debug.SetMemoryLimit(soft)
log.Info().
Int64("cgroupLimitBytes", limit).
Int64("goMemLimitBytes", soft).
Msg("aligned Go memory limit with cgroup memory limit")
}
func readCgroupMemoryLimit() (int64, bool) {
const procSelfCgroup = "/proc/self/cgroup"
if limit, ok := readCgroupV2MemoryLimit(cgroupV2Root, procSelfCgroup); ok {
return limit, true
}
if limit, ok := readCgroupV1HierarchyMemoryLimit("/sys/fs/cgroup/memory", procSelfCgroup); ok {
return limit, true
}
// Preserve the original best-effort fallback for unusual v1 mounts that
// expose the controller limit but not a readable /proc/self/cgroup.
return readCgroupV1MemoryLimit("/sys/fs/cgroup/memory/memory.limit_in_bytes")
}
// readCgroupV2MemoryLimit resolves the process's own cgroup from
// procSelfCgroup and walks from that directory up to the cgroup root, taking
// the smallest memory.max on the path. The limit may sit on any ancestor
// (systemd applies MemoryMax to the service cgroup; container runtimes to the
// container root).
func readCgroupV2MemoryLimit(root, procSelfCgroup string) (int64, bool) {
data, err := os.ReadFile(procSelfCgroup)
if err != nil {
return 0, false
}
rel := cgroupV2PathFrom(string(data))
if rel == "" {
return 0, false
}
lowest := int64(math.MaxInt64)
dir := filepath.Join(root, rel)
for {
if raw, err := os.ReadFile(filepath.Join(dir, "memory.max")); err == nil {
if v, ok := parseCgroupMemoryValue(string(raw)); ok && v < lowest {
lowest = v
}
}
if dir == root {
break
}
parent := filepath.Dir(dir)
if parent == dir {
break
}
dir = parent
}
if lowest == math.MaxInt64 {
return 0, false
}
return lowest, true
}
// cgroupV2PathFrom extracts the unified-hierarchy path from /proc/self/cgroup
// content ("0::/system.slice/pulse.service" -> "system.slice/pulse.service").
// A process in a cgroup namespace commonly sees its own cgroup as "0::/";
// filepath's "." preserves that valid root path without conflating it with a
// missing unified-hierarchy entry.
func cgroupV2PathFrom(content string) string {
for _, line := range strings.Split(content, "\n") {
if !strings.HasPrefix(line, "0::") {
continue
}
path := strings.TrimSpace(strings.TrimPrefix(line, "0::"))
if path == "/" {
return "."
}
return strings.TrimPrefix(path, "/")
}
return ""
}
// readCgroupV1HierarchyMemoryLimit resolves the process's memory-controller
// path and walks its ancestors just as the v2 reader does. Reading only the
// controller root misses a tighter systemd MemoryLimit/MemoryMax applied to a
// service cgroup on v1 hosts.
func readCgroupV1HierarchyMemoryLimit(root, procSelfCgroup string) (int64, bool) {
data, err := os.ReadFile(procSelfCgroup)
if err != nil {
return 0, false
}
rel := cgroupV1MemoryPathFrom(string(data))
if rel == "" {
return 0, false
}
lowest := int64(math.MaxInt64)
dir := filepath.Join(root, rel)
for {
if raw, err := os.ReadFile(filepath.Join(dir, "memory.limit_in_bytes")); err == nil {
if v, ok := parseCgroupMemoryValue(string(raw)); ok && v < cgroupV1NoLimit && v < lowest {
lowest = v
}
}
if dir == root {
break
}
parent := filepath.Dir(dir)
if parent == dir {
break
}
dir = parent
}
if lowest == math.MaxInt64 {
return 0, false
}
return lowest, true
}
// cgroupV1MemoryPathFrom extracts the path for the v1 memory controller from
// /proc/self/cgroup. Controllers may be mounted together, so "memory" can be
// one item in a comma-separated controller field.
func cgroupV1MemoryPathFrom(content string) string {
for _, line := range strings.Split(content, "\n") {
fields := strings.SplitN(line, ":", 3)
if len(fields) != 3 {
continue
}
hasMemory := false
for _, controller := range strings.Split(fields[1], ",") {
if strings.TrimSpace(controller) == "memory" {
hasMemory = true
break
}
}
if !hasMemory {
continue
}
path := strings.TrimSpace(fields[2])
if path == "/" {
return "."
}
return strings.TrimPrefix(path, "/")
}
return ""
}
func readCgroupV1MemoryLimit(limitFile string) (int64, bool) {
raw, err := os.ReadFile(limitFile)
if err != nil {
return 0, false
}
v, ok := parseCgroupMemoryValue(string(raw))
if !ok || v >= cgroupV1NoLimit {
return 0, false
}
return v, true
}
// parseCgroupMemoryValue parses a cgroup memory file value. "max" (v2's
// explicit no-limit marker) and non-numeric content report no limit.
func parseCgroupMemoryValue(raw string) (int64, bool) {
s := strings.TrimSpace(raw)
if s == "" || s == "max" {
return 0, false
}
v, err := strconv.ParseInt(s, 10, 64)
if err != nil || v <= 0 {
return 0, false
}
return v, true
}