From e5f407238ae79e9f296659a0419d9f0c304a9ec3 Mon Sep 17 00:00:00 2001 From: Benoit Sigoure Date: Thu, 6 Aug 2026 23:30:34 +0000 Subject: [PATCH 1/2] scaleset: reap runners stranded by GitHub Actions outages During the 2026-08-06 GitHub Actions outage, hundreds of scale set runners ended up stranded in states GARM never recovers from, pinning every scale set at max_runners and starving the fleet: 1. running/offline: the agent registered, then died or lost its connection to the actions service. GARM only tracked GitHub's offline status as metadata and never recycled such runners. 2. Phantom active: a job was assigned to a runner (the listener marks it active in GARM) but the agent never acquired it. The broker keeps the assignment while the runners list reports the runner as idle. GARM rejects the active to idle transition, so the runner stays active forever: never reaped, never given new jobs. 3. GitHub refusing deregistration with TaskAgentJobStillRunningException for both classes above, which aborted any deletion attempt. Changes: - Track when each runner is first observed in running/offline (in memory on the scale set worker; there is no status-changed timestamp in the DB and UpdatedAt is bumped by every write) and recycle it after DefaultRunnerOfflineTimeout (10 minutes) plus a deterministic per-runner jitter (fnv hash of the name, up to 5 minutes). The jitter prevents runners that went offline together from being reaped, respawned and going offline together in ever more synchronized batches. Offline status is only refreshed by the 5 minute consolidation pass, so effective reap latency is 10-15+ minutes. Runners actively executing a job can never be selected (the active to offline transition is rejected). - Cap offline reaps at max(5, max_runners/4) per consolidation pass. Provider delete/create operations fan out one goroutine per instance, and unbounded churn has overwhelmed the provider API. - Reap phantom active runners: if GARM considers a runner active but GitHub reports it idle on two consecutive consolidation passes, the job assignment was lost and will never arrive (a genuinely running job is reported busy, so a single stale snapshot cannot trigger this). - When GitHub refuses to deregister a runner, log the refusal and proceed with pending_delete instead of aborting. Destroying the instance is exactly what un-sticks GitHub's state: the agent disappears, GitHub fails the stuck job and releases the runner. With this, scale sets keep cycling stranded runners during an outage so fresh runners can pick up whatever jobs do go through, and the fleet recovers on its own once GitHub stabilizes. --- metrics/lifecycle.go | 1 + util/appdefaults/appdefaults.go | 7 + workers/scaleset/scaleset.go | 226 +++++++++++++++++++++++++++----- 3 files changed, 203 insertions(+), 31 deletions(-) diff --git a/metrics/lifecycle.go b/metrics/lifecycle.go index b516373cd..9d50fb98a 100644 --- a/metrics/lifecycle.go +++ b/metrics/lifecycle.go @@ -28,6 +28,7 @@ const ( OutcomeOrphaned = "orphaned" OutcomeManualDelete = "manual_delete" OutcomeStartupRecovery = "startup_recovery" + OutcomeOfflineTimeout = "offline_timeout" ) var ( diff --git a/util/appdefaults/appdefaults.go b/util/appdefaults/appdefaults.go index faf93ad85..8697e2d95 100644 --- a/util/appdefaults/appdefaults.go +++ b/util/appdefaults/appdefaults.go @@ -25,6 +25,13 @@ const ( // of time and no new updates have been made to it's state, it will be removed. DefaultRunnerBootstrapTimeout = 20 + // DefaultRunnerOfflineTimeout is the amount of time a scale set runner may + // remain in the running/offline state before it is recycled. Runners end up + // in this state when the github agent dies or fails to (re)connect to the + // actions service (e.g. during a github outage). Recycling them frees up + // capacity so fresh runners can pick up jobs. + DefaultRunnerOfflineTimeout = 10 * time.Minute + // DefaultGithubURL is the default URL where Github or Github Enterprise can be accessed. DefaultGithubURL = "https://github.com" diff --git a/workers/scaleset/scaleset.go b/workers/scaleset/scaleset.go index 2d5f9b3bb..495758227 100644 --- a/workers/scaleset/scaleset.go +++ b/workers/scaleset/scaleset.go @@ -17,6 +17,7 @@ import ( "context" "errors" "fmt" + "hash/fnv" "log/slog" "strconv" "strings" @@ -35,6 +36,7 @@ import ( "github.com/cloudbase/garm/params" "github.com/cloudbase/garm/runner/common" garmUtil "github.com/cloudbase/garm/util" + "github.com/cloudbase/garm/util/appdefaults" workersCommon "github.com/cloudbase/garm/workers/common" ) @@ -57,15 +59,17 @@ func NewWorker(ctx context.Context, store dbCommon.Store, scaleSet params.ScaleS return nil, fmt.Errorf("failed to get entity from the db: %w", err) } return &Worker{ - ctx: ctx, - controllerInfo: controllerInfo, - consumerID: consumerID, - store: store, - provider: provider, - scaleSet: scaleSet, - entity: entity, - runners: make(map[string]params.Instance), - pseudoPoolID: garmUtil.ScaleSetPseudoPoolID(scalesetEntity.ID, scaleSet.ID), + ctx: ctx, + controllerInfo: controllerInfo, + consumerID: consumerID, + store: store, + provider: provider, + scaleSet: scaleSet, + entity: entity, + runners: make(map[string]params.Instance), + pseudoPoolID: garmUtil.ScaleSetPseudoPoolID(scalesetEntity.ID, scaleSet.ID), + offlineSince: make(map[string]time.Time), + activeIdleMisses: make(map[string]int), }, nil } @@ -79,6 +83,23 @@ type Worker struct { scaleSet params.ScaleSet entity params.ForgeEntity runners map[string]params.Instance + // offlineSince tracks when we first observed a runner in the + // running/offline state, keyed by runner name. Instances have no + // "runner status changed at" timestamp in the DB and UpdatedAt is + // bumped by every write, so we track this in memory. Entries are + // cleared when the runner leaves the offline state or is removed; + // a GARM restart resets the clock, which at worst delays reaping + // by one timeout period. Protected by mux. + offlineSince map[string]time.Time + // activeIdleMisses counts consecutive consolidation passes in which a + // runner GARM considers "active" is reported "idle" by github. A runner + // genuinely executing a job is reported busy, so two consecutive idle + // observations (>5 minutes apart) mean the job assignment was lost on + // github's side (broker outage) and the runner will never receive it: + // GARM keeps it active forever (active->idle transitions are rejected) + // and it occupies a slot doing nothing. Such phantoms are reaped. + // Protected by mux. + activeIdleMisses map[string]int // pseudoPoolID is the stable pool ID reported to providers for this scale // set. It is derived from immutable IDs, so it is computed once. @@ -405,7 +426,13 @@ func (w *Worker) removeRunnerFromGithubAndSetPendingDelete(runnerName string, ag } if err := scaleSetCli.RemoveRunner(w.ctx, agentID); err != nil { if !errors.Is(err, runnerErrors.ErrNotFound) { - return fmt.Errorf("removing runner %s: %w", runnerName, err) + // Github may refuse to deregister a runner it believes is running a + // job (TaskAgentJobStillRunningException) even when the agent never + // actually acquired it (seen during github outages). Destroying the + // instance is what un-sticks that state on github's side: the agent + // disappears, github fails the stuck job and releases the runner. + // So log and proceed with pending_delete instead of aborting. + slog.WarnContext(w.ctx, "github refused to remove runner; deleting instance anyway", "runner_name", runnerName, "error", err) } } instance, err := w.setRunnerDBStatus(runnerName, commonParams.InstancePendingDelete) @@ -456,8 +483,34 @@ func (w *Worker) recordLifecycleEvent(outcome string) { ).Inc() } +// offlineTimeoutJitter returns a deterministic per-runner offset in +// [0, DefaultRunnerOfflineTimeout/2) added to the offline timeout. Runners +// that go offline together (e.g. during a github outage) would otherwise be +// reaped together, respawned together and go offline together again, with the +// batches growing more synchronized every cycle and hammering the provider +// API. Hashing the name spreads each batch over a jitter window without +// keeping extra state. +func offlineTimeoutJitter(runnerName string) time.Duration { + h := fnv.New64a() + h.Write([]byte(runnerName)) + return time.Duration(h.Sum64() % uint64(appdefaults.DefaultRunnerOfflineTimeout/2)) +} + +// maxOfflineReapsPerPass bounds how many offline runners we recycle in one +// consolidation pass, so a mass-offline event (github outage) trickles +// delete/create calls to the provider over several passes instead of issuing +// them all at once. +func (w *Worker) maxOfflineReapsPerPass() int { + limit := int(w.scaleSet.MaxRunners) / 4 + if limit < 5 { + limit = 5 + } + return limit +} + func (w *Worker) reapTimedOutRunners(runners map[string]params.RunnerReference) (func(), error) { lockNames := []string{} + offlineReaps := 0 unlockFn := func() { for _, name := range lockNames { @@ -466,10 +519,9 @@ func (w *Worker) reapTimedOutRunners(runners map[string]params.RunnerReference) } } + currentNames := make(map[string]struct{}, len(w.runners)) for _, runner := range w.runners { - if time.Since(runner.CreatedAt).Minutes() < float64(w.scaleSet.RunnerTimeout()) { - continue - } + currentNames[runner.Name] = struct{}{} switch runner.Status { case commonParams.InstancePendingDelete, commonParams.InstancePendingForceDelete, commonParams.InstanceDeleting, commonParams.InstanceDeleted: @@ -487,31 +539,80 @@ func (w *Worker) reapTimedOutRunners(runners map[string]params.RunnerReference) continue } - if runner.RunnerStatus != params.RunnerPending && runner.RunnerStatus != params.RunnerInstalling && runner.RunnerStatus != params.RunnerFailed { - slog.DebugContext(w.ctx, "runner is not pending, installing or failed; skipping", "runner_name", runner.Name) - continue + // A runner in running/offline is one whose agent registered with github + // but then died or lost its connection to the actions service (this + // happens en masse during github outages). It will never receive jobs + // again, but it occupies a max_runners slot. Track when we first saw it + // offline and recycle it once it exceeds the offline timeout. + offlineTimedOut := false + if runner.Status == commonParams.InstanceRunning && runner.RunnerStatus == params.RunnerOffline { + since, seen := w.offlineSince[runner.Name] + if !seen { + w.offlineSince[runner.Name] = time.Now() + } else { + timeout := appdefaults.DefaultRunnerOfflineTimeout + offlineTimeoutJitter(runner.Name) + offlineTimedOut = time.Since(since) >= timeout && offlineReaps < w.maxOfflineReapsPerPass() + } + } else { + delete(w.offlineSince, runner.Name) } - if ghRunner, ok := runners[runner.Name]; !ok || ghRunner.GetStatus() == params.RunnerOffline { - if ok := locking.TryLock(runner.Name, w.consumerID); !ok { - slog.DebugContext(w.ctx, "runner is locked; skipping", "runner_name", runner.Name) - continue + + bootstrapTimedOut := false + if time.Since(runner.CreatedAt).Minutes() >= float64(w.scaleSet.RunnerTimeout()) { + if runner.RunnerStatus == params.RunnerPending || runner.RunnerStatus == params.RunnerInstalling || runner.RunnerStatus == params.RunnerFailed { + ghRunner, ok := runners[runner.Name] + bootstrapTimedOut = !ok || ghRunner.GetStatus() == params.RunnerOffline } + } + if !offlineTimedOut && !bootstrapTimedOut { + continue + } + + if ok := locking.TryLock(runner.Name, w.consumerID); !ok { + slog.DebugContext(w.ctx, "runner is locked; skipping", "runner_name", runner.Name) + continue + } + + if offlineTimedOut { + slog.InfoContext( + w.ctx, "reaping runner offline for too long", + "runner_name", runner.Name, + "offline_for", time.Since(w.offlineSince[runner.Name]).String()) + } else { slog.InfoContext( w.ctx, "reaping timed-out/failed runner", "runner_name", runner.Name) + } - if err := w.removeRunnerFromGithubAndSetPendingDelete(runner.Name, runner.AgentID); err != nil { - // Don't let a single poisoned runner (e.g. one that raced into a - // status that can no longer transition to pending_delete through - // some other codepath) abort reaping for the rest of the batch. - // Log it, release just this runner's lock, and keep going. - slog.ErrorContext(w.ctx, "error removing runner", "runner_name", runner.Name, "error", err) - locking.Unlock(runner.Name, false) - continue - } + if err := w.removeRunnerFromGithubAndSetPendingDelete(runner.Name, runner.AgentID); err != nil { + // Don't let a single poisoned runner (e.g. one that raced into a + // status that can no longer transition to pending_delete through + // some other codepath) abort reaping for the rest of the batch. + // Log it, release just this runner's lock, and keep going. + slog.ErrorContext(w.ctx, "error removing runner", "runner_name", runner.Name, "error", err) + locking.Unlock(runner.Name, false) + continue + } + if offlineTimedOut { + offlineReaps++ + w.recordLifecycleEvent(metrics.OutcomeOfflineTimeout) + } else { w.recordLifecycleEvent(metrics.OutcomeBootstrapTimeout) - lockNames = append(lockNames, runner.Name) + } + delete(w.offlineSince, runner.Name) + lockNames = append(lockNames, runner.Name) + } + + // Prune tracking entries for runners that no longer exist. + for name := range w.offlineSince { + if _, ok := currentNames[name]; !ok { + delete(w.offlineSince, name) + } + } + for name := range w.activeIdleMisses { + if _, ok := currentNames[name]; !ok { + delete(w.activeIdleMisses, name) } } return unlockFn, nil @@ -534,7 +635,8 @@ func (w *Worker) consolidateRunnerState(listedAt time.Time, runners []params.Run // Cross check what exists in github with what we have in the database. for name, runner := range ghRunnersByName { status := runner.GetStatus() - if _, ok := dbRunnersByName[name]; !ok { + dbRunner, ok := dbRunnersByName[name] + if !ok { // runner appears to be active. Is it not managed by GARM? if status != params.RunnerIdle && status != params.RunnerActive { slog.InfoContext(w.ctx, "runner does not exist in GARM; removing from github", "runner_name", name) @@ -547,6 +649,68 @@ func (w *Worker) consolidateRunnerState(listedAt time.Time, runners []params.Run } continue } + + // Sync github's view of the runner (online idle/busy or offline) onto + // the instance. Without this, a runner whose agent died after setup + // stays "idle" in GARM forever, while github considers it offline and + // never assigns it jobs. Only touch runners that finished installing; + // runners in earlier lifecycle states are expected to be offline. + switch dbRunner.RunnerStatus { + case params.RunnerIdle, params.RunnerActive, params.RunnerOffline: + default: + continue + } + switch status { + case params.RunnerIdle, params.RunnerActive, params.RunnerOffline: + default: + continue + } + if dbRunner.RunnerStatus == status { + delete(w.activeIdleMisses, name) + continue + } + if dbRunner.RunnerStatus == params.RunnerActive && status == params.RunnerIdle { + // GARM thinks this runner is executing a job, github reports it + // idle. Either the job just started (our runner list snapshot is + // stale) or the job assignment was lost on github's side and the + // runner will never receive it. A real job shows up as busy on the + // next pass, so only act after two consecutive idle observations. + w.activeIdleMisses[name]++ + if w.activeIdleMisses[name] < 2 { + continue + } + if ok := locking.TryLock(name, w.consumerID); !ok { + slog.DebugContext(w.ctx, "runner is locked; skipping phantom active reap", "runner_name", name) + continue + } + slog.InfoContext(w.ctx, "reaping phantom active runner; github reports it idle", "runner_name", name, "misses", w.activeIdleMisses[name]) + if err := w.removeRunnerFromGithubAndSetPendingDelete(name, dbRunner.AgentID); err != nil { + slog.ErrorContext(w.ctx, "error removing phantom active runner", "runner_name", name, "error", err) + locking.Unlock(name, false) + continue + } + delete(w.activeIdleMisses, name) + // Hold the lock until consolidation finishes, like the other + // pending_delete paths below, so the provider worker doesn't act + // on the status change while we still cross-check state. + defer locking.Unlock(name, false) + continue + } + delete(w.activeIdleMisses, name) + if ok := locking.TryLock(name, w.consumerID); !ok { + slog.DebugContext(w.ctx, "runner is locked; skipping runner status sync", "runner_name", name) + continue + } + slog.InfoContext(w.ctx, "syncing runner status from github", "runner_name", name, "old_status", dbRunner.RunnerStatus, "new_status", status) + updatedRunner, err := w.store.UpdateInstance(w.ctx, name, params.UpdateInstanceParams{RunnerStatus: status}) + locking.Unlock(name, false) + if err != nil { + if !errors.Is(err, runnerErrors.ErrNotFound) { + slog.ErrorContext(w.ctx, "error updating runner status", "runner_name", name, "error", err) + } + continue + } + w.runners[updatedRunner.ID] = updatedRunner } unlockFn, err := w.reapTimedOutRunners(ghRunnersByName) From 83ce5a2dfab3bb1340b2cee42c3ec4cce1ed1bd7 Mon Sep 17 00:00:00 2001 From: Benoit Sigoure Date: Fri, 7 Aug 2026 06:05:33 +0000 Subject: [PATCH 2/2] scaleset: do not reap active runners the github list reports idle The github runners list is not a reliable busy indicator for scale set runners: it reports idle for runners that are actively executing a job. The phantom-active reaper introduced earlier trusted that signal and cancelled ~150 running jobs over the course of an hour once real load returned after the outage. Never act on the list for active runners. Job completion and failure are handled by the listener, and the offline reaper covers dead agents (a runner whose agent dies mid job goes offline and the active to offline transition being rejected keeps GARM from syncing it, but github eventually fails the job and the JobCompleted message terminates the instance). Co-Authored-By: Claude Fable 5 --- workers/scaleset/scaleset.go | 67 +++++++++--------------------------- 1 file changed, 16 insertions(+), 51 deletions(-) diff --git a/workers/scaleset/scaleset.go b/workers/scaleset/scaleset.go index 495758227..45406461d 100644 --- a/workers/scaleset/scaleset.go +++ b/workers/scaleset/scaleset.go @@ -59,17 +59,16 @@ func NewWorker(ctx context.Context, store dbCommon.Store, scaleSet params.ScaleS return nil, fmt.Errorf("failed to get entity from the db: %w", err) } return &Worker{ - ctx: ctx, - controllerInfo: controllerInfo, - consumerID: consumerID, - store: store, - provider: provider, - scaleSet: scaleSet, - entity: entity, - runners: make(map[string]params.Instance), - pseudoPoolID: garmUtil.ScaleSetPseudoPoolID(scalesetEntity.ID, scaleSet.ID), - offlineSince: make(map[string]time.Time), - activeIdleMisses: make(map[string]int), + ctx: ctx, + controllerInfo: controllerInfo, + consumerID: consumerID, + store: store, + provider: provider, + scaleSet: scaleSet, + entity: entity, + runners: make(map[string]params.Instance), + pseudoPoolID: garmUtil.ScaleSetPseudoPoolID(scalesetEntity.ID, scaleSet.ID), + offlineSince: make(map[string]time.Time), }, nil } @@ -91,15 +90,6 @@ type Worker struct { // a GARM restart resets the clock, which at worst delays reaping // by one timeout period. Protected by mux. offlineSince map[string]time.Time - // activeIdleMisses counts consecutive consolidation passes in which a - // runner GARM considers "active" is reported "idle" by github. A runner - // genuinely executing a job is reported busy, so two consecutive idle - // observations (>5 minutes apart) mean the job assignment was lost on - // github's side (broker outage) and the runner will never receive it: - // GARM keeps it active forever (active->idle transitions are rejected) - // and it occupies a slot doing nothing. Such phantoms are reaped. - // Protected by mux. - activeIdleMisses map[string]int // pseudoPoolID is the stable pool ID reported to providers for this scale // set. It is derived from immutable IDs, so it is computed once. @@ -610,11 +600,6 @@ func (w *Worker) reapTimedOutRunners(runners map[string]params.RunnerReference) delete(w.offlineSince, name) } } - for name := range w.activeIdleMisses { - if _, ok := currentNames[name]; !ok { - delete(w.activeIdleMisses, name) - } - } return unlockFn, nil } @@ -666,37 +651,17 @@ func (w *Worker) consolidateRunnerState(listedAt time.Time, runners []params.Run continue } if dbRunner.RunnerStatus == status { - delete(w.activeIdleMisses, name) continue } if dbRunner.RunnerStatus == params.RunnerActive && status == params.RunnerIdle { - // GARM thinks this runner is executing a job, github reports it - // idle. Either the job just started (our runner list snapshot is - // stale) or the job assignment was lost on github's side and the - // runner will never receive it. A real job shows up as busy on the - // next pass, so only act after two consecutive idle observations. - w.activeIdleMisses[name]++ - if w.activeIdleMisses[name] < 2 { - continue - } - if ok := locking.TryLock(name, w.consumerID); !ok { - slog.DebugContext(w.ctx, "runner is locked; skipping phantom active reap", "runner_name", name) - continue - } - slog.InfoContext(w.ctx, "reaping phantom active runner; github reports it idle", "runner_name", name, "misses", w.activeIdleMisses[name]) - if err := w.removeRunnerFromGithubAndSetPendingDelete(name, dbRunner.AgentID); err != nil { - slog.ErrorContext(w.ctx, "error removing phantom active runner", "runner_name", name, "error", err) - locking.Unlock(name, false) - continue - } - delete(w.activeIdleMisses, name) - // Hold the lock until consolidation finishes, like the other - // pending_delete paths below, so the provider worker doesn't act - // on the status change while we still cross-check state. - defer locking.Unlock(name, false) + // The github runners list is NOT a reliable busy indicator for + // scale set runners: it reports "idle" for runners that are + // actively executing a job (observed 2026-08-07: reaping on this + // signal cancelled ~150 running jobs). Never touch active runners + // based on the list; job completion/failure is handled by the + // listener and the offline reaper covers dead agents. continue } - delete(w.activeIdleMisses, name) if ok := locking.TryLock(name, w.consumerID); !ok { slog.DebugContext(w.ctx, "runner is locked; skipping runner status sync", "runner_name", name) continue