Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion go.mod
Original file line number Diff line number Diff line change
Expand Up @@ -71,7 +71,7 @@ require (
golang.org/x/net v0.54.0 // indirect
golang.org/x/sys v0.45.0 // indirect
golang.org/x/term v0.43.0 // indirect
golang.org/x/text v0.37.0 // indirect
golang.org/x/text v0.39.0 // indirect
modernc.org/libc v1.72.0 // indirect
modernc.org/mathutil v1.7.1 // indirect
modernc.org/memory v1.11.0 // indirect
Expand Down
16 changes: 8 additions & 8 deletions go.sum
Original file line number Diff line number Diff line change
Expand Up @@ -152,22 +152,22 @@ golang.org/x/crypto v0.52.0 h1:RMs7fP2rXdep0CftQlK8Uf+kibLm7qkCcradZWYz988=
golang.org/x/crypto v0.52.0/go.mod h1:1QgfPxDqh0T2M/elOJtp9RvuR95kVjir0e6/BvEmGbc=
golang.org/x/exp v0.0.0-20251023183803-a4bb9ffd2546 h1:mgKeJMpvi0yx/sU5GsxQ7p6s2wtOnGAHZWCHUM4KGzY=
golang.org/x/exp v0.0.0-20251023183803-a4bb9ffd2546/go.mod h1:j/pmGrbnkbPtQfxEe5D0VQhZC6qKbfKifgD0oM7sR70=
golang.org/x/mod v0.35.0 h1:Ww1D637e6Pg+Zb2KrWfHQUnH2dQRLBQyAtpr/haaJeM=
golang.org/x/mod v0.35.0/go.mod h1:+GwiRhIInF8wPm+4AoT6L0FA1QWAad3OMdTRx4tFYlU=
golang.org/x/mod v0.37.0 h1:vF1DjpVEshcIqoEaauuHebaLk1O1forxjxBaVn884JQ=
golang.org/x/mod v0.37.0/go.mod h1:m8S8VeM9r4dzDwjrKO0a1sZP3YjeMamRRlD+fmR2Q/0=
golang.org/x/net v0.54.0 h1:2zJIZAxAHV/OHCDTCOHAYehQzLfSXuf/5SoL/Dv6w/w=
golang.org/x/net v0.54.0/go.mod h1:Sj4oj8jK6XmHpBZU/zWHw3BV3abl4Kvi+Ut7cQcY+cQ=
golang.org/x/sync v0.20.0 h1:e0PTpb7pjO8GAtTs2dQ6jYa5BWYlMuX047Dco/pItO4=
golang.org/x/sync v0.20.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
golang.org/x/sync v0.21.0 h1:HLII4xRRTtCRkxYp4HNFF0Js/Og6q2i++KXbg0gHCwM=
golang.org/x/sync v0.21.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
golang.org/x/sys v0.0.0-20210809222454-d867a43fc93e/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY=
golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/term v0.43.0 h1:S4RLU2sB31O/NCl+zFN9Aru9A/Cq2aqKpTZJ6B+DwT4=
golang.org/x/term v0.43.0/go.mod h1:lrhlHNdQJHO+1qVYiHfFKVuVioJIheAc3fBSMFYEIsk=
golang.org/x/text v0.37.0 h1:Cqjiwd9eSg8e0QAkyCaQTNHFIIzWtidPahFWR83rTrc=
golang.org/x/text v0.37.0/go.mod h1:a5sjxXGs9hsn/AJVwuElvCAo9v8QYLzvavO5z2PiM38=
golang.org/x/tools v0.44.0 h1:UP4ajHPIcuMjT1GqzDWRlalUEoY+uzoZKnhOjbIPD2c=
golang.org/x/tools v0.44.0/go.mod h1:KA0AfVErSdxRZIsOVipbv3rQhVXTnlU6UhKxHd1seDI=
golang.org/x/text v0.39.0 h1:UbZz4pLOvn600D6Oh6GGEI6VAmndrEBLv8/6BEXzyus=
golang.org/x/text v0.39.0/go.mod h1:3UwRclnC2g0TU9x8PZiyfOajCd1zaUNHF9cvqcQZ+ZM=
golang.org/x/tools v0.47.0 h1:7Kn5x/d1svx/PzryTsqeoZN4TZwqeH5pGWjefhLi/1Q=
golang.org/x/tools v0.47.0/go.mod h1:dFHnyTvFWY212G+h7ZY4Vsp/K3U4/7W9TyVaAul8uCA=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=
Expand Down
8 changes: 8 additions & 0 deletions internal/executor/executor.go
Original file line number Diff line number Diff line change
Expand Up @@ -2899,6 +2899,14 @@ func createTmuxWindow(daemonSession, windowName, workDir, script, allowedProject
return "", fmt.Errorf("security: refusing to create tmux window with invalid workDir: %s", workDir)
}

// Consult system memory pressure before adding another agent session. Warns by
// default and only defers when TY_MEMORY_GUARD=block (see memoryguard.go).
if note, gerr := guardMemoryForSpawn(taskID); gerr != nil {
return "", gerr
} else if note != "" {
log.Warn("memory guard: "+note, "task", taskID)
}

// Serialize check-then-create with the TUI/API spawn path (EnsureTaskWindow).
// Best-effort on timeout so a wedged holder can't block the daemon forever.
if release, lerr := executorlock.AcquireSpawn(executorSpawnLockDir(), taskID, spawnLockTimeout); lerr == nil {
Expand Down
116 changes: 116 additions & 0 deletions internal/executor/memoryguard.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,116 @@
package executor

// Memory admission guard for executor spawns.
//
// Why this exists: nothing in ty bounded how many agent sessions could be live at
// once. Each session is a claude process plus its MCP servers (~100-150MB resident,
// considerably more in footprint once macOS compresses the cold pages), so a queue
// that fans out freely can walk a workstation into swap thrashing — at which point
// every session, dev server and test run on the box gets slower, including the ones
// already doing useful work.
//
// Deliberately NOT a concurrency cap. A fixed "max N sessions" is a hidden wall:
// the right number depends entirely on how much RAM the machine has and what else
// is running, and a user who legitimately wants dozens of concurrent agents should
// not be told "no" by an arbitrary constant. So this gates on the machine's ACTUAL
// memory pressure instead. With headroom, spawn as many as you like; the guard only
// has an opinion once the kernel says memory is genuinely short.
//
// Default mode is "warn": log loudly, never block. That keeps behaviour unchanged
// for everyone else while making the condition visible. Set TY_MEMORY_GUARD=block
// to have spawns deferred under pressure instead.
//
// TY_MEMORY_GUARD=off|warn|block (default: warn)
// TY_MEMORY_GUARD_MIN_FREE_PCT=<0-100> (default: 20)
//
// The signal is "percent of memory still available", read per-platform by
// systemFreeMemoryPct (see memoryguard_darwin.go / memoryguard_linux.go). On any
// platform where it can't be read the guard is inert.
//
// Threshold semantics are the same everywhere: the fraction of memory still
// available, 0-100. On macOS that maps to Activity Monitor's pressure zones (green
// 100-50, yellow 50-30, red 30-0), so the default of 20 fires only well into the
// red — when the machine is already hurting, not merely busy.

import (
"errors"
"fmt"
"os"
"strconv"
"strings"
)

type memoryGuardMode int

const (
memoryGuardOff memoryGuardMode = iota
memoryGuardWarn
memoryGuardBlock
)

// defaultMemoryGuardMinFreePct is the free-memory percentage below which the guard
// considers the machine to be under real pressure. 20 sits inside Activity Monitor's
// red zone (30-0).
const defaultMemoryGuardMinFreePct = 20

// ErrMemoryPressure is returned by guardMemoryForSpawn in block mode when the
// machine is below the free-memory threshold. Callers should treat it as "try this
// task again shortly", not as a task failure.
var ErrMemoryPressure = errors.New("executor: deferring spawn, system memory pressure")

func memoryGuardModeFromEnv() memoryGuardMode {
switch strings.ToLower(strings.TrimSpace(os.Getenv("TY_MEMORY_GUARD"))) {
case "off", "0", "false", "disabled":
return memoryGuardOff
case "block", "defer", "enforce":
return memoryGuardBlock
default:
// Unset or unrecognised: warn. Never silently block.
return memoryGuardWarn
}
}

func memoryGuardMinFreePct() int {
raw := strings.TrimSpace(os.Getenv("TY_MEMORY_GUARD_MIN_FREE_PCT"))
if raw == "" {
return defaultMemoryGuardMinFreePct
}
n, err := strconv.Atoi(raw)
if err != nil || n < 0 || n > 100 {
return defaultMemoryGuardMinFreePct
}
return n
}

// systemFreeMemoryPct is implemented per platform; see memoryguard_darwin.go,
// memoryguard_linux.go and memoryguard_unsupported.go. It returns ok=false whenever
// the signal can't be read, which makes the guard inert rather than guessing.

// guardMemoryForSpawn consults system memory pressure before an executor spawn.
//
// It returns a human-readable note whenever the machine is under pressure (empty
// string otherwise) so the caller can surface it with whatever logger it has, and a
// non-nil error ONLY in block mode. Callers must always spawn when err == nil, even
// if note is non-empty — that is the whole point of the default warn mode.
func guardMemoryForSpawn(taskID int64) (note string, err error) {
mode := memoryGuardModeFromEnv()
if mode == memoryGuardOff {
return "", nil
}

freePct, ok := systemFreeMemoryPct()
if !ok {
return "", nil // no signal, no opinion
}
minFree := memoryGuardMinFreePct()
if freePct >= minFree {
return "", nil
}

if mode == memoryGuardBlock {
return "", fmt.Errorf("%w: %d%% memory free (threshold %d%%), task %d — set TY_MEMORY_GUARD=off or lower TY_MEMORY_GUARD_MIN_FREE_PCT to spawn anyway",
ErrMemoryPressure, freePct, minFree, taskID)
}
return fmt.Sprintf("system memory is low (%d%% free, threshold %d%%): spawning anyway; set TY_MEMORY_GUARD=block to defer spawns under pressure",
freePct, minFree), nil
}
39 changes: 39 additions & 0 deletions internal/executor/memoryguard_darwin.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,39 @@
//go:build darwin

package executor

import (
"context"
"os/exec"
"strconv"
"strings"
"time"
)

// memoryGuardSysctlTimeout bounds the sysctl call so a wedged exec can never delay
// a spawn. On timeout the guard reports "unknown" and stays out of the way.
const memoryGuardSysctlTimeout = 2 * time.Second

// systemFreeMemoryPct reads kern.memorystatus_level, the kernel's own
// percent-of-memory-still-free figure. It is what Activity Monitor's memory
// pressure display is derived from, so the number a user sees there and the number
// this guard acts on are the same one.
//
// Note this is deliberately NOT "free RAM" in the naive sense: macOS aggressively
// fills RAM with cache and compressed pages, so free-page counts read alarmingly low
// on a perfectly healthy machine. memorystatus_level already accounts for what the
// kernel can reclaim, which is why it's the right signal here.
func systemFreeMemoryPct() (pct int, ok bool) {
ctx, cancel := context.WithTimeout(context.Background(), memoryGuardSysctlTimeout)
defer cancel()

out, err := exec.CommandContext(ctx, "sysctl", "-n", "kern.memorystatus_level").Output()
if err != nil {
return 0, false
}
n, err := strconv.Atoi(strings.TrimSpace(string(out)))
if err != nil || n < 0 || n > 100 {
return 0, false
}
return n, true
}
45 changes: 45 additions & 0 deletions internal/executor/memoryguard_linux.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
//go:build linux

package executor

import "os"

// cgroup v2 paths. Inside a container the container's own cgroup is mounted at the
// hierarchy root, so these read that container's limit. On a bare host the root
// cgroup has no memory.max and the read simply fails, falling through to
// /proc/meminfo — which is what we want there anyway.
const (
cgroupV2MemoryMax = "/sys/fs/cgroup/memory.max"
cgroupV2MemoryCurrent = "/sys/fs/cgroup/memory.current"
cgroupV2MemoryStat = "/sys/fs/cgroup/memory.stat"
procMeminfo = "/proc/meminfo"
)

// systemFreeMemoryPct returns percent-of-memory-available on Linux.
//
// cgroup v2 is consulted first so a containerised agent server is measured against
// its own memory limit rather than the host's total RAM — otherwise a 2GB container
// on a 64GB host would look like it had endless headroom right up until the OOM
// killer fired. Falls back to /proc/meminfo when there is no cgroup limit.
//
// Deliberately not using PSI (/proc/pressure/memory): it is the better *pressure*
// signal, but it reports stall time rather than a fraction of memory, so it can't
// share TY_MEMORY_GUARD_MIN_FREE_PCT's meaning with the macOS path. Keeping one
// threshold that means the same thing on every platform is worth more here than a
// slightly sharper Linux signal.
func systemFreeMemoryPct() (pct int, ok bool) {
maxRaw, errMax := os.ReadFile(cgroupV2MemoryMax)
currentRaw, errCur := os.ReadFile(cgroupV2MemoryCurrent)
if errMax == nil && errCur == nil {
statRaw, _ := os.ReadFile(cgroupV2MemoryStat) // optional; absent just means no cache adjustment
if pct, ok := parseCgroupFreePct(string(maxRaw), string(currentRaw), string(statRaw)); ok {
return pct, true
}
}

data, err := os.ReadFile(procMeminfo)
if err != nil {
return 0, false
}
return parseMeminfoFreePct(string(data))
}
115 changes: 115 additions & 0 deletions internal/executor/memoryguard_procparse.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,115 @@
package executor

// Pure parsers for the Linux memory signals, deliberately kept free of build tags
// and of any file I/O so they can be unit-tested on any platform (including the
// macOS laptops where this is usually developed). The Linux-only file reading that
// feeds them lives in memoryguard_linux.go.

import (
"strconv"
"strings"
)

// parseMeminfoFreePct computes percent-of-memory-available from /proc/meminfo.
//
// MemAvailable is the right numerator: it is the kernel's own estimate of memory
// obtainable without swapping, already accounting for reclaimable page cache and
// slab. Using MemFree instead would report single-digit percentages on a healthy
// Linux box that is simply using its RAM for cache — and in block mode that would
// wedge an agent fleet for no reason.
//
// MemAvailable has been present since Linux 3.14 (2014). Older kernels fall back to
// MemFree + Buffers + Cached, the approximation MemAvailable itself replaced.
func parseMeminfoFreePct(data string) (pct int, ok bool) {
var total, available, free, buffers, cached int64
var haveAvailable bool

for _, line := range strings.Split(data, "\n") {
key, valueField, found := strings.Cut(line, ":")
if !found {
continue
}
fields := strings.Fields(valueField)
if len(fields) == 0 {
continue
}
v, err := strconv.ParseInt(fields[0], 10, 64)
if err != nil {
continue
}
switch key {
case "MemTotal":
total = v
case "MemAvailable":
available, haveAvailable = v, true
case "MemFree":
free = v
case "Buffers":
buffers = v
case "Cached": // must not match "SwapCached": strings.Cut keys are exact
cached = v
}
}

if total <= 0 {
return 0, false
}
if !haveAvailable {
available = free + buffers + cached
}
return clampPct(available * 100 / total), true
}

// parseCgroupFreePct computes percent-of-memory-available from cgroup v2 files, so
// a containerised agent server is measured against its own limit rather than the
// host's RAM. Returns ok=false when the cgroup is unlimited ("max"), which is the
// normal case on a bare VM — the caller then falls back to /proc/meminfo.
//
// memory.current counts page cache, which is reclaimable, so subtracting
// inactive_file from memory.stat is what keeps a cache-heavy container from looking
// permanently starved. Without that subtraction a long-running container would sit
// at ~0% "free" forever and, in block mode, stop spawning entirely.
func parseCgroupFreePct(maxRaw, currentRaw, statRaw string) (pct int, ok bool) {
maxRaw = strings.TrimSpace(maxRaw)
if maxRaw == "" || maxRaw == "max" {
return 0, false // no limit set; not a meaningful denominator
}
limit, err := strconv.ParseInt(maxRaw, 10, 64)
if err != nil || limit <= 0 {
return 0, false
}
current, err := strconv.ParseInt(strings.TrimSpace(currentRaw), 10, 64)
if err != nil || current < 0 {
return 0, false
}

// Treat reclaimable file cache as available.
var inactiveFile int64
for _, line := range strings.Split(statRaw, "\n") {
if name, value, found := strings.Cut(strings.TrimSpace(line), " "); found && name == "inactive_file" {
if v, err := strconv.ParseInt(value, 10, 64); err == nil && v >= 0 {
inactiveFile = v
}
break
}
}

used := current - inactiveFile
if used < 0 {
used = 0
}
if used > limit {
used = limit
}
return clampPct((limit - used) * 100 / limit), true
}

func clampPct(v int64) int {
if v < 0 {
return 0
}
if v > 100 {
return 100
}
return int(v)
}
Loading