mirror of
https://github.com/coredns/coredns.git
synced 2026-08-20 23:08:28 -04:00
plugin/shed: add UDP overload protection plugin (#8312)
* plugin/shed: add UDP overload protection plugin
UDP responses written back through one listener socket serialize on the
Go runtime's internal fdMutex, which allows at most 2^20-1 concurrent
operations per file descriptor and panics the process when exceeded.
CoreDNS serves UDP with one goroutine per query, all writing through the
shared packet connection, so a sustained overload parks every excess
in-flight query in that wait queue until the process dies with
"too many concurrent operations on a single file or socket". Observed
in production: ~2.8M goroutines and 60GiB RSS before the panic.
The shed plugin makes the panic structurally unreachable. It installs,
via Config.UDPDecorateWriterFunc, a per-socket bounded evict-oldest
stack drained newest-first by a single writer goroutine, so the fd
never sees more than one writer and residual capacity under overload
always goes to the freshest response. While a socket's stack is full,
arriving queries are dropped before any plugin runs. Drops are silent
(the client's resolver retries elsewhere) and counted in
coredns_shed_dropped_total{server, reason}.
plugin/shed/fdmutex_test.go demonstrates the failure and the fix with
one shared flood harness. Two subprocess tests reproduce the exact
runtime panic without the plugin's write discipline - one deterministic
(a held write plus >2^20 queued writers), one with nothing held or
mocked; both exercise the Go runtime rather than the plugin, so they
are gated behind SHED_FLOOD_TEST=1. The counterfactual - the same load
through the plugin's stack, completing with every response accounted
for as written or dropped - runs in every test invocation, including
-race, at 50k responders, and at the full 1.5M with SHED_FLOOD_TEST=1:
SHED_FLOOD_TEST=1 go test ./plugin/shed/
Signed-off-by: Ryan Brewster <rpb@anthropic.com>
* test: add shed e2e test
Query a shed-enabled server over UDP (the plugin's deferred
single-writer path) and TCP (which shed passes through), and check
that coredns_shed_dropped_total is exported with its reason label.
No-Verification-Needed: test-only change
Signed-off-by: Ryan Brewster <rpb@anthropic.com>
---------
Signed-off-by: Ryan Brewster <rpb@anthropic.com>
This commit is contained in:
164
plugin/shed/stack_test.go
Normal file
164
plugin/shed/stack_test.go
Normal file
@@ -0,0 +1,164 @@
|
||||
package shed
|
||||
|
||||
import (
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/coredns/coredns/core/dnsserver"
|
||||
|
||||
"github.com/miekg/dns"
|
||||
)
|
||||
|
||||
// chanWriter hands each written payload to a channel — the race-safe way to
|
||||
// observe the writer goroutine's deferred writes.
|
||||
type chanWriter struct {
|
||||
got chan []byte
|
||||
}
|
||||
|
||||
func (w *chanWriter) Write(p []byte) (int, error) {
|
||||
w.got <- p
|
||||
return len(p), nil
|
||||
}
|
||||
|
||||
func TestStackEvictsOldestPopsNewest(t *testing.T) {
|
||||
rs := newRespStack(3, droppedTotal.WithLabelValues(t.Name(), "response")) // no writer goroutine: pure data structure test
|
||||
for i := 1; i <= 5; i++ {
|
||||
dropped := rs.push(pendingResp{data: []byte{byte(i)}})
|
||||
if want := i > 3; dropped != want {
|
||||
t.Errorf("push %d: dropped = %v, want %v", i, dropped, want)
|
||||
}
|
||||
}
|
||||
if !rs.full() {
|
||||
t.Error("expected full stack after overfilling")
|
||||
}
|
||||
// 1 and 2 were evicted; the survivors pop newest-first.
|
||||
for _, want := range []byte{5, 4, 3} {
|
||||
p, ok := rs.pop()
|
||||
if !ok || p.data[0] != want {
|
||||
t.Fatalf("pop = %v, %v; want entry %d", p.data, ok, want)
|
||||
}
|
||||
}
|
||||
if _, ok := rs.pop(); ok {
|
||||
t.Error("expected empty stack")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStackCloseRejectsPushDrainsRest(t *testing.T) {
|
||||
rs := newRespStack(4, droppedTotal.WithLabelValues(t.Name(), "response"))
|
||||
|
||||
// A stale notify token on an empty open stack wakes the writer, which
|
||||
// must tolerate the failed pop (writerLoop's pop-ok check).
|
||||
rs.push(pendingResp{data: []byte{9}})
|
||||
rs.pop() // pop directly, leaving the push's token buffered
|
||||
if !rs.waitNonempty() {
|
||||
t.Error("a stale token should report as work")
|
||||
}
|
||||
if _, ok := rs.pop(); ok {
|
||||
t.Error("pop should find nothing behind a stale token")
|
||||
}
|
||||
|
||||
rs.push(pendingResp{data: []byte{1}})
|
||||
rs.close()
|
||||
if !rs.push(pendingResp{data: []byte{2}}) {
|
||||
t.Error("push on closed stack should report a drop")
|
||||
}
|
||||
// Entries accepted before the close must still be served.
|
||||
if !rs.waitNonempty() {
|
||||
t.Fatal("waitNonempty should report the pre-close entry")
|
||||
}
|
||||
if p, ok := rs.pop(); !ok || p.data[0] != 1 {
|
||||
t.Fatalf("pop = %v, %v; want pre-close entry", p, ok)
|
||||
}
|
||||
// The pre-close push's token may still be buffered; drain it so the
|
||||
// final wait deterministically takes the stop arm.
|
||||
select {
|
||||
case <-rs.notify:
|
||||
default:
|
||||
}
|
||||
if rs.waitNonempty() {
|
||||
t.Error("waitNonempty should report false once closed and drained")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecoratorCapturesWriteAndWriterWrites(t *testing.T) {
|
||||
s := newShed(t, nil)
|
||||
srv := &dnsserver.Server{}
|
||||
dec := s.decorateWriterFactory(srv)
|
||||
if _, ok := registry.Load(srv); !ok {
|
||||
t.Fatal("factory should pre-register the socket's state")
|
||||
}
|
||||
|
||||
cw := &chanWriter{got: make(chan []byte, 1)}
|
||||
data := packedReply(t)
|
||||
w := dec(cw)
|
||||
if _, ok := w.(*stackWriter); !ok {
|
||||
t.Fatalf("decorator returned %T, want *stackWriter", w)
|
||||
}
|
||||
if _, err := w.Write(data); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
select {
|
||||
case got := <-cw.got:
|
||||
m := new(dns.Msg)
|
||||
if err := m.Unpack(got); err != nil {
|
||||
t.Fatalf("writer goroutine wrote unparseable bytes: %s", err)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("writer goroutine never performed the deferred write")
|
||||
}
|
||||
}
|
||||
|
||||
// panicWriter panics on its first Write, then counts.
|
||||
type panicWriter struct {
|
||||
writes atomic.Int64
|
||||
}
|
||||
|
||||
func (w *panicWriter) Write(p []byte) (int, error) {
|
||||
if w.writes.Add(1) == 1 {
|
||||
panic("writer exploded")
|
||||
}
|
||||
return len(p), nil
|
||||
}
|
||||
|
||||
func TestWriterPanicRecovered(t *testing.T) {
|
||||
s := newShed(t, nil)
|
||||
srv := &dnsserver.Server{}
|
||||
dec := s.decorateWriterFactory(srv)
|
||||
|
||||
pw := &panicWriter{}
|
||||
data := packedReply(t)
|
||||
if _, err := dec(pw).Write(data); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := dec(pw).Write(data); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// The writer goroutine must survive the first write's panic and still
|
||||
// perform the second.
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for pw.writes.Load() < 2 {
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("writer performed %d writes, want 2 (goroutine died on panic?)", pw.writes.Load())
|
||||
}
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
}
|
||||
|
||||
func TestShutdownIsInstanceScoped(t *testing.T) {
|
||||
old := newShed(t, nil)
|
||||
cur := newShed(t, nil)
|
||||
oldSrv, newSrv := &dnsserver.Server{}, &dnsserver.Server{}
|
||||
old.decorateWriterFactory(oldSrv)
|
||||
cur.decorateWriterFactory(newSrv)
|
||||
|
||||
if err := old.shutdown(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, ok := registry.Load(oldSrv); ok {
|
||||
t.Error("old instance's entry should be swept")
|
||||
}
|
||||
if _, ok := registry.Load(newSrv); !ok {
|
||||
t.Error("new instance's entry must survive the old instance's shutdown")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user