plugin/shed: add UDP overload protection plugin (#8312)

* plugin/shed: add UDP overload protection plugin

UDP responses written back through one listener socket serialize on the
Go runtime's internal fdMutex, which allows at most 2^20-1 concurrent
operations per file descriptor and panics the process when exceeded.
CoreDNS serves UDP with one goroutine per query, all writing through the
shared packet connection, so a sustained overload parks every excess
in-flight query in that wait queue until the process dies with
"too many concurrent operations on a single file or socket". Observed
in production: ~2.8M goroutines and 60GiB RSS before the panic.

The shed plugin makes the panic structurally unreachable. It installs,
via Config.UDPDecorateWriterFunc, a per-socket bounded evict-oldest
stack drained newest-first by a single writer goroutine, so the fd
never sees more than one writer and residual capacity under overload
always goes to the freshest response. While a socket's stack is full,
arriving queries are dropped before any plugin runs. Drops are silent
(the client's resolver retries elsewhere) and counted in
coredns_shed_dropped_total{server, reason}.

plugin/shed/fdmutex_test.go demonstrates the failure and the fix with
one shared flood harness. Two subprocess tests reproduce the exact
runtime panic without the plugin's write discipline - one deterministic
(a held write plus >2^20 queued writers), one with nothing held or
mocked; both exercise the Go runtime rather than the plugin, so they
are gated behind SHED_FLOOD_TEST=1. The counterfactual - the same load
through the plugin's stack, completing with every response accounted
for as written or dropped - runs in every test invocation, including
-race, at 50k responders, and at the full 1.5M with SHED_FLOOD_TEST=1:

    SHED_FLOOD_TEST=1 go test ./plugin/shed/

Signed-off-by: Ryan Brewster <rpb@anthropic.com>

* test: add shed e2e test

Query a shed-enabled server over UDP (the plugin's deferred
single-writer path) and TCP (which shed passes through), and check
that coredns_shed_dropped_total is exported with its reason label.

No-Verification-Needed: test-only change
Signed-off-by: Ryan Brewster <rpb@anthropic.com>

---------

Signed-off-by: Ryan Brewster <rpb@anthropic.com>
This commit is contained in:
rpb-ant
2026-07-27 05:13:25 -04:00
committed by GitHub
parent 989bf4a9fd
commit 76056dd2e5
13 changed files with 1166 additions and 0 deletions

176
plugin/shed/stack.go Normal file
View File

@@ -0,0 +1,176 @@
package shed
import (
"sync"
"sync/atomic"
"github.com/coredns/coredns/core/dnsserver"
"github.com/miekg/dns"
"github.com/prometheus/client_golang/prometheus"
)
// decorateWriterFactory is installed as Config.UDPDecorateWriterFunc by
// setup. dnsserver's ServePacket calls it once per UDP listener socket,
// before the socket serves its first packet, so the socket's state and
// writer goroutine exist before ServeDNS ever looks them up. miekg/dns then
// applies the returned dns.DecorateWriter once per packet, wrapping the
// response writer.
func (s *Shed) decorateWriterFactory(srv *dnsserver.Server) dns.DecorateWriter {
st := s.mintState(srv)
return func(w dns.Writer) dns.Writer {
return &stackWriter{stack: st.stack, inner: w}
}
}
// stackWriter is the per-packet transport wrapper. miekg/dns runs every
// message transform (including TSIG) before handing Write the packed bytes,
// so the deferred operation is precisely the serialized syscall.
type stackWriter struct {
stack *respStack
inner dns.Writer // the raw response writer; its Write is the terminal syscall
}
// Write pushes the packed bytes and reports success: from here on "written"
// means "queued for the socket's writer goroutine". The pushed slice is
// exclusively owned — miekg/dns packs each response into a fresh allocation.
func (sw *stackWriter) Write(data []byte) (int, error) {
if sw.stack.push(pendingResp{w: sw.inner, data: data}) {
sw.stack.dropped.Inc()
}
return len(data), nil
}
// pendingResp is one captured response awaiting the socket's writer
// goroutine: the packed bytes and the raw writer that puts them on the wire.
type pendingResp struct {
w dns.Writer
data []byte
}
// respStack is a per-socket bounded ring of pending responses with a single
// insertion cursor and no head index: entries occupy the size slots before
// next (mod depth), so when the ring is full the slot at next holds the
// oldest entry and pushing over it is the eviction.
type respStack struct {
dropped prometheus.Counter // responses evicted, write-failed, or pushed after close
mu sync.Mutex
buf []pendingResp // ring; len(buf) is the fixed depth
next int // index of the next push
size int // occupied slots
closed bool // set by close; pushes are rejected from then on
n atomic.Int64 // size mirror for the lock-free full() check
notify chan struct{} // cap 1: writer wake-up
stop chan struct{} // closed on shutdown
}
func newRespStack(depth int, dropped prometheus.Counter) *respStack {
return &respStack{
dropped: dropped,
buf: make([]pendingResp, depth),
notify: make(chan struct{}, 1),
stop: make(chan struct{}),
}
}
// push adds p as the newest entry, evicting the oldest when full. Never
// blocks. Reports whether a response was dropped as a result: the evicted
// oldest, or — on a closed stack — p itself.
func (rs *respStack) push(p pendingResp) (dropped bool) {
rs.mu.Lock()
switch {
case rs.closed:
rs.mu.Unlock()
return true
case rs.size == len(rs.buf):
dropped = true // the slot at next holds the oldest entry
default:
rs.size++
}
rs.buf[rs.next] = p
rs.next = (rs.next + 1) % len(rs.buf)
rs.n.Store(int64(rs.size))
rs.mu.Unlock()
select {
case rs.notify <- struct{}{}:
default:
}
return dropped
}
// pop removes and returns the newest entry.
func (rs *respStack) pop() (pendingResp, bool) {
rs.mu.Lock()
if rs.size == 0 {
rs.mu.Unlock()
return pendingResp{}, false
}
rs.next = (rs.next - 1 + len(rs.buf)) % len(rs.buf)
p := rs.buf[rs.next]
rs.buf[rs.next] = pendingResp{} // release the response bytes
rs.size--
rs.n.Store(int64(rs.size))
rs.mu.Unlock()
return p, true
}
// full is the lock-free view used by the coupled-shed predicate.
func (rs *respStack) full() bool { return rs.n.Load() >= int64(len(rs.buf)) }
// close stops the writer goroutine — it drains whatever is stacked, then
// exits — and rejects any straggler pushes.
func (rs *respStack) close() {
rs.mu.Lock()
rs.closed = true
rs.mu.Unlock()
close(rs.stop)
}
// waitNonempty blocks until the stack has work, or reports false once the
// stack is closed and empty. The re-check on the stop arm matters: the
// select may pick stop over a pending notify, but entries accepted before
// the close must still be served — closed guarantees no new pushes, so the
// drain terminates.
func (rs *respStack) waitNonempty() bool {
if rs.n.Load() > 0 {
return true
}
select {
case <-rs.notify:
return true
case <-rs.stop:
return rs.n.Load() > 0
}
}
// writerLoop is the socket's single writer: it waits for pending responses,
// pops the one that is newest at write time, and writes it to the wire.
func (rs *respStack) writerLoop() {
for {
if !rs.waitNonempty() {
return
}
if p, ok := rs.pop(); ok {
rs.write(p)
}
}
}
// write performs the deferred raw write; a response that fails to reach the
// wire is a counted drop. A writer panic must not kill the process — that is
// the failure class this plugin removes — so it is recovered, like
// dnsserver does for synchronous writes.
func (rs *respStack) write(p pendingResp) {
defer func() {
if rec := recover(); rec != nil {
rs.dropped.Inc()
log.Errorf("Recovered panic in shed writer: %v", rec)
}
}()
if _, err := p.w.Write(p.data); err != nil {
rs.dropped.Inc()
log.Debugf("Deferred response write failed: %s", err)
}
}