From 7ecf63169b71c3485bc59113df00a6127ad83ddd Mon Sep 17 00:00:00 2001 From: Aaron Echols <649815+dasunsrule32@users.noreply.github.com> Date: Tue, 6 Oct 2026 23:30:09 -0400 Subject: [PATCH] feat(daemon): debounce container change notifications to avoid reconcile storms (#438) * fix(store): debounce container change notifications to avoid reconcile storms SubscribeContainers forwarded every raw row-level change event from Corrosion as its own signal, with no coalescing. Each signal triggers a full reconcile (ListContainers + regenerate) in every independent subscriber (caddy-controller, DNS resolver) on every machine in the cluster. With enough container/health-check churn across services, this produces a sustained stream of events (observed ~1.6-1.7/sec cluster-wide on a 45-service, 7-machine cluster) and burns meaningful CPU on every machine continuously, regardless of whether that machine runs Caddy locally. On a resource-constrained node this escalated into kernel RCU stall warnings and a full freeze. Coalesce bursts of events into a single signal per debounce window instead of forwarding one per event. * bound debounced changes by the window --------- Co-authored-by: Pasha Sviderski --- internal/machine/store/container.go | 28 ++++++++++++++++++++++++++-- 1 file changed, 26 insertions(+), 2 deletions(-) diff --git a/internal/machine/store/container.go b/internal/machine/store/container.go index 7a789584..0fb6d18a 100644 --- a/internal/machine/store/container.go +++ b/internal/machine/store/container.go @@ -23,6 +23,10 @@ const ( // SyncStatusOutdated indicates that a container record may be outdated, for example, due to being unable // to retrieve the container's state from the Docker daemon or when the machine is being stopped or restarted. SyncStatusOutdated = "outdated" + + // containerChangesDebounceInterval defines how long to wait before notifying subscribers about container changes. + // Multiple changes within this window are grouped into a single notification to prevent system overload. + containerChangesDebounceInterval = 250 * time.Millisecond ) type ContainerRecord struct { @@ -267,6 +271,13 @@ func (s *Store) SubscribeContainers(ctx context.Context) ([]ContainerRecord, <-c changes := make(chan struct{}) go func() { defer close(changes) + // Coalesce bursts of rapid-fire row-level change events (e.g. from frequent health-check updates + // across many services) into a single signal per debounce window, instead of forwarding one signal + // per event. Every subscriber of this channel (e.g. the Caddy and DNS reconcilers) otherwise reruns + // its full reconciliation on every single event, which can burn significant CPU across the cluster + // when there's a lot of container churn. + var debouncer *time.Timer + var debouncerCh <-chan time.Time for { select { case <-ctx.Done(): @@ -279,8 +290,21 @@ func (s *Store) SubscribeContainers(ctx context.Context) ([]ContainerRecord, <-c } return } - // Just signal that there is a change in the containers list. - changes <- struct{}{} + if debouncerCh == nil { + if debouncer == nil { + debouncer = time.NewTimer(containerChangesDebounceInterval) + } else { + debouncer.Reset(containerChangesDebounceInterval) + } + debouncerCh = debouncer.C + } + case <-debouncerCh: + select { + case changes <- struct{}{}: + case <-ctx.Done(): + return + } + debouncerCh = nil } } }()