From d6f2a6e8b374ee619f7a6888d542bf83ca4dce55 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:39:28 +0200 Subject: [PATCH 01/34] test(iouring): failing-first tests for the io_uring hand-off's fd lifetime (celeris#657) PR 2 of 3 for #657 (face 2). These tests fail on main (5b2e83b) and pin what the fix must do; the fix follows in separate commits. Unit tests drive one connection through the real request, send and hand-off code with the kernel taken out of the loop: every SQE the engine places is read back and removed unsubmitted, and every completion is written by the test, so each ordering a cancel can resolve in is a deterministic input. - TestTransplantNeverHandsOffArmedRecv: no hand-off while the recv is armed; a reported cancel of its own tag (0x09) instead, hand-off at the recv's -ECANCELED, a request that beats the cancel served and HELD, and the same against the real kernel. - TestTransplantReapMissIsRetried: a missed reap is retried, never followed by a hand-off; a stale miss cannot clear a newer reap. - TestHoldReleasedWhenDrainStops: a held conn gets its recv armed at the SEND completion when it is not handed off (processCQE, the worker loop, a refused hand-off). - TestHoldRescuedByCheckTimeouts: the timeout sweep's belt, counted. - TestOneOwnerPerHandoff: a conn claimed by its dispatch goroutine is not also moved by tryTransplant; finishAsyncTransplant leaves a slot it no longer owns alone. - TestNoDrainSQESequenceIsUnchanged: with no drain set the per-request SQEs (opcode, flags incl. IO_LINK, tag) of the sync, WRITEV and direct-body tails. Passes on main; it is the witness that the fix changes nothing on the steady-state path. Engine-level (real worker loop, keep-alive clients across a StartTransplant): TestHandoffHasNothingInFlight, TestStaleRecvDataCounted (client losses == stale data CQEs, and both 0), TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen and TestWorkerParksWithNothingPending. Refs #657 --- engine/iouring/fd_lifetime_engine_test.go | 519 +++++++++++++++++++ engine/iouring/fd_lifetime_fixture_test.go | 271 ++++++++++ engine/iouring/fd_lifetime_test.go | 557 +++++++++++++++++++++ 3 files changed, 1347 insertions(+) create mode 100644 engine/iouring/fd_lifetime_engine_test.go create mode 100644 engine/iouring/fd_lifetime_fixture_test.go create mode 100644 engine/iouring/fd_lifetime_test.go diff --git a/engine/iouring/fd_lifetime_engine_test.go b/engine/iouring/fd_lifetime_engine_test.go new file mode 100644 index 00000000..5095824b --- /dev/null +++ b/engine/iouring/fd_lifetime_engine_test.go @@ -0,0 +1,519 @@ +//go:build linux + +package iouring + +import ( + "bufio" + "context" + "errors" + "io" + "log/slog" + "net" + "net/http" + "os" + "sort" + "strconv" + "strings" + "sync" + "sync/atomic" + "syscall" + "testing" + "time" + + "github.com/goceleris/celeris/engine" + "github.com/goceleris/celeris/protocol/h2/stream" + "github.com/goceleris/celeris/resource" +) + +// celeris#657 face 2 end to end: a real io_uring engine, keep-alive clients +// that never stop sending, and a hand-off across StartTransplant. On the base +// every hand-off leaves the connection's next recv armed, and some of those +// recvs take a request the client then waits on forever. + +// startFDLEngine runs an io_uring engine on a free loopback port. Workers is +// 2 (the minimum Resources accepts); RLIMIT_MEMLOCK decides how many actually +// start (one at 8 MiB), and the count is logged as workers=N so a run can be +// checked against the shape it was registered for. +func startFDLEngine(t *testing.T, h stream.Handler, mut func(*resource.Config)) (*Engine, string) { + t.Helper() + ln, err := net.Listen("tcp", "127.0.0.1:0") + if err != nil { + t.Fatalf("pick port: %v", err) + } + addr := ln.Addr().String() + _ = ln.Close() + cfg := resource.Config{ + Addr: addr, + Protocol: engine.HTTP1, + Resources: resource.Resources{Workers: 2}, + Logger: slog.New(slog.NewTextHandler(io.Discard, nil)), + } + if mut != nil { + mut(&cfg) + } + e, err := New(cfg, h) + if err != nil { + skipOrFail656(t, "iouring engine unavailable: %v", err) + } + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error, 1) + go func() { done <- e.Listen(ctx) }() + t.Cleanup(func() { + cancel() + select { + case <-done: + case <-time.After(5 * time.Second): + t.Error("engine did not stop within 5s") + } + }) + for deadline := time.Now().Add(8 * time.Second); ; { + if c, derr := net.DialTimeout("tcp", addr, 200*time.Millisecond); derr == nil { + _ = c.Close() + if e.NumWorkers() > 0 { + break + } + } + select { + case err := <-done: + skipOrFail656(t, "iouring engine failed to start: %v", err) + default: + } + if time.Now().After(deadline) { + t.Fatal("engine did not start listening within 8s") + } + time.Sleep(20 * time.Millisecond) + } + t.Logf("celeris657 engine workers=%d", e.NumWorkers()) + return e, addr +} + +// servingTarget is a stand-in epoll engine: every adopted descriptor is served +// as HTTP/1 keep-alive by a goroutine, so a client whose conn was handed off +// keeps getting answers — unless a recv left on the source took its request. +type servingTarget struct { + adopted atomic.Int64 + onAdopt func() // runs first, on the source worker's thread + wg sync.WaitGroup + mu sync.Mutex + conns []net.Conn +} + +func (s *servingTarget) AdoptConn(fd int, _ engine.Carryover) error { + if s.onAdopt != nil { + s.onAdopt() + } + f := os.NewFile(uintptr(fd), "adopted") + c, err := net.FileConn(f) + _ = f.Close() + if err != nil { + return nil // fd closed above; nothing for the source to reclaim + } + s.adopted.Add(1) + s.mu.Lock() + s.conns = append(s.conns, c) + s.mu.Unlock() + s.wg.Add(1) + go func() { + defer s.wg.Done() + br := bufio.NewReader(c) + for { + req, err := http.ReadRequest(br) + if err != nil { + return + } + _, _ = io.Copy(io.Discard, req.Body) + _ = req.Body.Close() + if _, err := c.Write([]byte("HTTP/1.1 200 OK\r\ncontent-type: text/plain\r\ncontent-length: 2\r\n\r\nok")); err != nil { + return + } + } + }() + return nil +} + +func (s *servingTarget) close() { + s.mu.Lock() + for _, c := range s.conns { + _ = c.Close() + } + s.mu.Unlock() + s.wg.Wait() +} + +// refusingTarget refuses every hand-off, so the source must keep (reclaim) +// each connection it offers. +type refusingTarget struct{ refused atomic.Int64 } + +func (r *refusingTarget) AdoptConn(int, engine.Carryover) error { + r.refused.Add(1) + return errors.New("refusingTarget: no") +} + +// loadResult is what the clients saw. +type loadResult struct { + ok int64 + errs int64 + byClass map[string]int64 +} + +// runKeepAliveLoad drives n keep-alive clients against addr, each sending one +// request at a time with a 1 s read deadline, until stop is closed. A request +// the server never answers surfaces as that client's read timeout; the client +// then stops, so each lost request is exactly one error. +func runKeepAliveLoad(t *testing.T, addr string, n int, stop <-chan struct{}) func() loadResult { + t.Helper() + var ok, errsN atomic.Int64 + var mu sync.Mutex + byClass := map[string]int64{} + var wg sync.WaitGroup + for i := 0; i < n; i++ { + c, err := net.DialTimeout("tcp", addr, 2*time.Second) + if err != nil { + t.Fatalf("dial %d: %v", i, err) + } + wg.Add(1) + go func() { + defer wg.Done() + defer func() { _ = c.Close() }() + br := bufio.NewReader(c) + fail := func(err error) { + errsN.Add(1) + cls := "other" + switch { + case errors.Is(err, os.ErrDeadlineExceeded): + cls = "read_timeout" + case errors.Is(err, io.EOF), errors.Is(err, io.ErrUnexpectedEOF): + cls = "eof" + case errors.Is(err, syscall.ECONNRESET): + cls = "reset" + } + mu.Lock() + byClass[cls]++ + mu.Unlock() + } + for { + select { + case <-stop: + return + default: + } + if _, err := c.Write([]byte("GET / HTTP/1.1\r\nHost: x\r\n\r\n")); err != nil { + fail(err) + return + } + _ = c.SetReadDeadline(time.Now().Add(time.Second)) + resp, err := http.ReadResponse(br, nil) + if err != nil { + fail(err) + return + } + _, err = io.Copy(io.Discard, resp.Body) + _ = resp.Body.Close() + if err != nil { + fail(err) + return + } + ok.Add(1) + } + }() + } + return func() loadResult { + wg.Wait() + mu.Lock() + defer mu.Unlock() + cp := map[string]int64{} + for k, v := range byClass { + cp[k] = v + } + return loadResult{ok: ok.Load(), errs: errsN.Load(), byClass: cp} + } +} + +func classes(m map[string]int64) string { + var ks []string + for k := range m { + ks = append(ks, k) + } + sort.Strings(ks) + var b strings.Builder + for _, k := range ks { + b.WriteString(k + "=" + strconv.FormatInt(m[k], 10) + " ") + } + return strings.TrimSpace(b.String()) +} + +// transplantUnderLoad starts n clients, calls StartTransplant(tgt) after +// warm, keeps the load up for after, stops it and returns what the clients +// saw. The stale CQEs of any stolen request arrive within the load window; +// the final wait lets the last of them land before the counters are read. +func transplantUnderLoad(t *testing.T, e *Engine, addr string, n int, tgt engine.TransplantTarget, + warm, after time.Duration, +) loadResult { + t.Helper() + stop := make(chan struct{}) + wait := runKeepAliveLoad(t, addr, n, stop) + time.Sleep(warm) + e.StartTransplant(tgt) + time.Sleep(after) + close(stop) + res := wait() + e.StopTransplant() + time.Sleep(100 * time.Millisecond) + t.Logf("celeris657 load conns=%d ok=%d errs=%d classes=[%s] W1T=%d W1U=%d W1C=%d W2=%d", + n, res.ok, res.errs, classes(res.byClass), + e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), + e.metrics.handoffLoss.staleRecvDataUnattributed.Load(), + e.metrics.handoffLoss.staleRecvDataClosed.Load(), + e.metrics.handoffLoss.handoffInFlight.Load()) + return res +} + +// TestHandoffHasNothingInFlight: at every hand-off, nothing may still be in +// flight on the connection (TransplantHandoffInFlight does not move), and no +// stale recv data may appear afterwards, with 128 busy keep-alives across a +// StartTransplant. Base: every hand-off is made with the linked RECV armed. +func TestHandoffHasNothingInFlight(t *testing.T) { + e, addr := startFDLEngine(t, transplantTestHandler{}, nil) + var maxSeen atomic.Uint64 + tgt := &servingTarget{} + tgt.onAdopt = func() { + if v := e.metrics.handoffLoss.handoffInFlight.Load(); v > maxSeen.Load() { + maxSeen.Store(v) + } + } + defer tgt.close() + const conns = 128 + res := transplantUnderLoad(t, e, addr, conns, tgt, 300*time.Millisecond, 1500*time.Millisecond) + if v := maxSeen.Load(); v != 0 { + t.Errorf("TransplantHandoffInFlight read %d at a hand-off, want 0 at every one: a conn "+ + "left with an op still able to resolve its fd", v) + } + if tr, un := e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), + e.metrics.handoffLoss.staleRecvDataUnattributed.Load(); tr != 0 || un != 0 { + t.Errorf("stale recv data after the hand-offs: Transplanted=%d Unattributed=%d, want 0/0", tr, un) + } + if res.errs != 0 { + t.Errorf("clients saw %d errors (%s), want 0", res.errs, classes(res.byClass)) + } + if got := tgt.adopted.Load(); got != conns { + t.Errorf("%d of %d busy conns were handed off, want all", got, conns) + } +} + +// TestStaleRecvDataCounted is the celeris#657 join as a test: every request a +// client lost is a stale recv CQE with data that the witness counted as +// Transplanted or Unattributed — and with the fix there are none of either. +func TestStaleRecvDataCounted(t *testing.T) { + e, addr := startFDLEngine(t, transplantTestHandler{}, nil) + tgt := &servingTarget{} + defer tgt.close() + res := transplantUnderLoad(t, e, addr, 128, tgt, 300*time.Millisecond, 1500*time.Millisecond) + w1 := int64(e.metrics.handoffLoss.staleRecvDataTransplanted.Load() + + e.metrics.handoffLoss.staleRecvDataUnattributed.Load()) + if res.errs != w1 { + t.Errorf("join broken: clients lost %d requests (%s) but the witness counted %d stale "+ + "data CQEs (Transplanted+Unattributed)", res.errs, classes(res.byClass), w1) + } + if res.errs != 0 || w1 != 0 { + t.Errorf("lost requests = %d, stale data = %d, want 0 and 0", res.errs, w1) + } +} + +// TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen: a conn held for a +// hand-off that does not happen must be served on as if nothing had been +// attempted — (a) the target refuses every conn, (b) the drain stops while +// held responses are in flight, five times over. Clients see no error, no +// stale data appears, and no hold is left for the timeout sweep to rescue. +func TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen(t *testing.T) { + check := func(t *testing.T, e *Engine, res loadResult) { + t.Helper() + if res.errs != 0 { + t.Errorf("clients saw %d errors (%s), want 0", res.errs, classes(res.byClass)) + } + if tr, un := e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), + e.metrics.handoffLoss.staleRecvDataUnattributed.Load(); tr != 0 || un != 0 { + t.Errorf("stale recv data: Transplanted=%d Unattributed=%d, want 0/0", tr, un) + } + if n := e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Errorf("TransplantHandoffInFlight = %d, want 0", n) + } + if n := metric(t, e, "TransplantHoldRescued"); n != 0 { + t.Errorf("TransplantHoldRescued = %d, want 0: a held conn was stranded until the "+ + "timeout sweep found it", n) + } + if res.ok == 0 { + t.Error("clients completed no requests") + } + } + + t.Run("target_refuses", func(t *testing.T) { + e, addr := startFDLEngine(t, transplantTestHandler{}, nil) + tgt := &refusingTarget{} + res := transplantUnderLoad(t, e, addr, 64, tgt, 200*time.Millisecond, 1200*time.Millisecond) + if tgt.refused.Load() == 0 { + t.Fatal("no hand-off was attempted; the test exercised nothing") + } + check(t, e, res) + }) + + t.Run("drain_stops_while_held", func(t *testing.T) { + e, addr := startFDLEngine(t, transplantTestHandler{}, nil) + tgt := &refusingTarget{} + stop := make(chan struct{}) + wait := runKeepAliveLoad(t, addr, 64, stop) + time.Sleep(200 * time.Millisecond) + for i := 0; i < 5; i++ { + e.StartTransplant(tgt) + time.Sleep(100 * time.Millisecond) + e.StopTransplant() + time.Sleep(100 * time.Millisecond) + } + time.Sleep(1200 * time.Millisecond) // a stranded conn's 1 s read deadline expires in here + close(stop) + res := wait() + t.Logf("celeris657 flaps=5 ok=%d errs=%d classes=[%s] refused=%d", res.ok, res.errs, + classes(res.byClass), tgt.refused.Load()) + if tgt.refused.Load() == 0 { + t.Fatal("no hand-off was attempted; the test exercised nothing") + } + check(t, e, res) + }) +} + +// testHoldReleasedInTheWorkerLoop is TestHoldReleasedWhenDrainStops through +// the worker's own event loop (the inlined udSend dispatch site), made +// deterministic by a response large enough to keep its SEND in flight until +// the client reads it: the request arrives with a drain set (so the response +// is held), the drain stops, and only then does the SEND complete. +func testHoldReleasedInTheWorkerLoop(t *testing.T) { + big := make([]byte, 3<<20) + e, addr := startFDLEngine(t, bigBodyHandler{big: big}, nil) + d := net.Dialer{Control: func(_, _ string, rc syscall.RawConn) error { + var serr error + _ = rc.Control(func(fd uintptr) { + serr = syscall.SetsockoptInt(int(fd), syscall.SOL_SOCKET, syscall.SO_RCVBUF, 4096) + }) + return serr + }} + c, err := d.Dial("tcp", addr) + if err != nil { + t.Fatalf("dial: %v", err) + } + defer func() { _ = c.Close() }() + br := bufio.NewReader(c) + get := func(path string, timeout time.Duration) (int, error) { + if _, err := c.Write([]byte("GET " + path + " HTTP/1.1\r\nHost: x\r\n\r\n")); err != nil { + return 0, err + } + _ = c.SetReadDeadline(time.Now().Add(timeout)) + resp, err := http.ReadResponse(br, nil) + if err != nil { + return 0, err + } + n, err := io.Copy(io.Discard, resp.Body) + _ = resp.Body.Close() + return int(n), err + } + if _, err := get("/", 2*time.Second); err != nil { + t.Fatalf("warm-up request: %v", err) + } + tgt := &fdlTarget{} + e.StartTransplant(tgt) + if _, err := c.Write([]byte("GET /big HTTP/1.1\r\nHost: x\r\n\r\n")); err != nil { + t.Fatalf("write /big: %v", err) + } + time.Sleep(300 * time.Millisecond) // served and held; its SEND waits on this client + e.StopTransplant() + _ = c.SetReadDeadline(time.Now().Add(10 * time.Second)) + resp, err := http.ReadResponse(br, nil) + if err != nil { + t.Fatalf("read /big: %v", err) + } + n, err := io.Copy(io.Discard, resp.Body) + _ = resp.Body.Close() + if err != nil || int(n) != len(big) { + t.Fatalf("read /big body: %d of %d bytes, err %v", n, len(big), err) + } + if _, err := get("/", 2*time.Second); err != nil { + t.Fatalf("the request after a held response whose drain stopped got %v: the conn was "+ + "left with no recv armed", err) + } + if tgt.adopted.Load() != 0 { + t.Fatalf("the conn was handed off (%d) although its SEND was in flight until the drain stopped", + tgt.adopted.Load()) + } + if n := metric(t, e, "TransplantHoldRescued"); n != 0 { + t.Fatalf("TransplantHoldRescued = %d, want 0: the release at the SEND completion did not run", n) + } +} + +type bigBodyHandler struct{ big []byte } + +func (h bigBodyHandler) HandleStream(_ context.Context, s *stream.Stream) error { + if s.ResponseWriter == nil { + return nil + } + body := []byte("ok") + if s.Path == "/big" { + body = h.big + } + return s.ResponseWriter.WriteResponse(s, 200, + [][2]string{{"content-type", "application/octet-stream"}, {"content-length", strconv.Itoa(len(body))}}, body) +} + +// TestWorkerParksWithNothingPending (A5): a worker that parks must first +// submit what its last iteration queued. The last connection's close queues +// the cancel of its header timer in the same iteration that finds the worker +// idle; on the base it parks with that SQE unsubmitted, and it stays so until +// something wakes the worker. +func TestWorkerParksWithNothingPending(t *testing.T) { + e, addr := startFDLEngine(t, transplantTestHandler{}, func(c *resource.Config) { + c.ReadHeaderTimeout = 10 * time.Second // a kernel timer in flight per conn + }) + c, err := net.DialTimeout("tcp", addr, 2*time.Second) + if err != nil { + t.Fatalf("dial: %v", err) + } + br := bufio.NewReader(c) + if _, err := c.Write([]byte("GET / HTTP/1.1\r\nHost: x\r\n\r\n")); err != nil { + t.Fatalf("write: %v", err) + } + _ = c.SetReadDeadline(time.Now().Add(2 * time.Second)) + resp, err := http.ReadResponse(br, nil) + if err != nil { + t.Fatalf("read: %v", err) + } + _, _ = io.Copy(io.Discard, resp.Body) + _ = resp.Body.Close() + if err := e.PauseAccept(); err != nil { + t.Fatalf("PauseAccept: %v", err) + } + _ = c.Close() + + e.mu.Lock() + workers := append([]*Worker(nil), e.workers...) + e.mu.Unlock() + for deadline := time.Now().Add(5 * time.Second); ; { + parked := 0 + for _, w := range workers { + if w.suspended.Load() { + parked++ + } + } + if parked == len(workers) { + break + } + if time.Now().After(deadline) { + t.Fatalf("%d of %d workers parked within 5s", parked, len(workers)) + } + time.Sleep(10 * time.Millisecond) + } + // suspended is stored after the park decision, so everything the worker + // wrote before it — including the SQ ring's pending count — is visible + // here, and a parked worker writes nothing. + for i, w := range workers { + if p := w.ring.Pending(); p != 0 { + t.Errorf("worker %d parked with %d SQE(s) queued and never submitted", i, p) + } + } +} diff --git a/engine/iouring/fd_lifetime_fixture_test.go b/engine/iouring/fd_lifetime_fixture_test.go new file mode 100644 index 00000000..3d2923f7 --- /dev/null +++ b/engine/iouring/fd_lifetime_fixture_test.go @@ -0,0 +1,271 @@ +//go:build linux + +package iouring + +import ( + "context" + "errors" + "reflect" + "strconv" + "sync/atomic" + "testing" + "unsafe" + + "golang.org/x/sys/unix" + + "github.com/goceleris/celeris/engine" + "github.com/goceleris/celeris/protocol/h2/stream" + "github.com/goceleris/celeris/resource" +) + +// celeris#657, face 2: an io_uring hand-off must never leave a read that can +// still resolve the descriptor. These helpers drive one connection through +// the real request, send and hand-off code with the kernel taken out of the +// loop: every SQE the engine places is read back and removed from the SQ ring +// unsubmitted, and every completion is written by the test. That makes each +// ordering the kernel can produce (a cancel that hits, one that misses, data +// that beats the cancel) a deterministic, repeatable input. + +// wantReapTag is the op tag of the hand-off's own recv cancel (DECISION P2: +// udTransplantReap = 0x09<<56, a tag of its own so its completion can be told +// apart from the WebSocket pause's cancel, celeris#596). Spelled as a literal +// so this file compiles on a tree that does not have the constant yet. +const wantReapTag uint64 = 0x09 << 56 + +// sqeRec is one SQE as the kernel would read it. +type sqeRec struct { + op uint8 + flags uint8 + fd int32 + addr uint64 + ud uint64 +} + +func (r sqeRec) tag() uint64 { return r.ud & udMask } + +func (r sqeRec) String() string { + name := map[uint8]string{opSEND: "SEND", opRECV: "RECV", opASYNCCANCEL: "ASYNC_CANCEL", opWRITEV: "WRITEV", + opTIMEOUT: "TIMEOUT"}[r.op] + if name == "" { + name = "op" + strconv.Itoa(int(r.op)) + } + return name + "{flags:" + strconv.Itoa(int(r.flags)) + " tag:0x" + strconv.FormatUint(r.tag()>>56, 16) + + " fd:" + strconv.Itoa(int(r.fd)) + "}" +} + +// takeSQEs returns every SQE placed on r since the last call, in placement +// order, and removes them from the SQ ring without submitting them: the kernel +// never sees them, so every completion in these tests is the test's own. +func takeSQEs(r *Ring) []sqeRec { + head := atomic.LoadUint32((*uint32)(r.sqHead)) + tail := atomic.LoadUint32((*uint32)(r.sqTail)) + var out []sqeRec + for i := head; i != tail; i++ { + b := r.sqes[uintptr(i&r.sqMask)*sqeSize:] + out = append(out, sqeRec{ + op: b[0], + flags: b[1], + fd: *(*int32)(unsafe.Pointer(&b[4])), + addr: *(*uint64)(unsafe.Pointer(&b[16])), + ud: *(*uint64)(unsafe.Pointer(&b[32])), + }) + } + atomic.StoreUint32((*uint32)(r.sqTail), head) + r.pending = 0 + return out +} + +// fdlHandler answers "ok", or a body large enough for the scatter-gather +// (WRITEV) path on /large. +type fdlHandler struct{} + +var fdlLargeBody = make([]byte, 16<<10) + +func (fdlHandler) HandleStream(_ context.Context, s *stream.Stream) error { + if s.ResponseWriter == nil { + return nil + } + body := []byte("ok") + if s.Path == "/large" { + body = fdlLargeBody + } + return s.ResponseWriter.WriteResponse(s, 200, + [][2]string{{"content-type", "text/plain"}, {"content-length", strconv.Itoa(len(body))}}, body) +} + +const fdlGET = "GET / HTTP/1.1\r\nHost: x\r\n\r\n" + +// fdlTarget stands in for the epoll engine on the receiving end of a hand-off. +// It closes what it adopts, or, with refuse set, refuses it the way a target +// that cannot take the fd does (the source then reclaims it). +type fdlTarget struct { + adopted atomic.Int64 + refuse bool +} + +func (r *fdlTarget) AdoptConn(fd int, _ engine.Carryover) error { + if r.refuse { + return errors.New("fdlTarget: refused") + } + r.adopted.Add(1) + _ = unix.Close(fd) + return nil +} + +// fdlFixture is one HTTP/1 keep-alive connection on a ring-backed Worker, set +// up the way onAcceptedFD sets up a real one. e is the engine the worker's +// celeris#657 counters belong to, so tests read them through Metrics() — +// the path the validation artifact reads. +type fdlFixture struct { + t *testing.T + e *Engine + w *Worker + cs *connState + fd, peer int + gen uint32 + tgt *fdlTarget + now int64 +} + +func newFDLFixture(t *testing.T, async bool) *fdlFixture { + t.Helper() + fd, peer := socketPairFDs(t) + _ = unix.SetNonblock(fd, true) + e := &Engine{} + w := newLedgerWorker(fd) + // Room for the descriptor a refused hand-off is reclaimed on. + if len(w.conns) < 1024 { + w.conns = make([]*connState, 1024) + } + w.ring = newTestRing(t) + w.handler = fdlHandler{} + w.cfg = resource.Config{Protocol: engine.HTTP1, MaxRequestBodySize: 1 << 20} + w.resolved.BufferSize = 4096 + w.reqCount = new(atomic.Uint64) + w.liveConns = make([]int, 0, 4) + w.async = async + // The engine-wide counters createWorkers wires, taken from e so a + // Metrics() read sees them. + w.recvArm = &e.metrics.recvArm + w.handoffLoss = &e.metrics.handoffLoss + w.transplantDetached = &e.metrics.transplantDetached + w.transplantHandoffRefused = &e.metrics.transplantHandoffRefused + w.transplantAdoptRefused = &e.metrics.transplantAdoptRefused + + cs := acquireConnState(context.Background(), fd, 4096, async) + cs.writeFn = w.makeWriteFn(cs) + cs.protocol.Store(int32(engine.HTTP1)) + cs.detected = true + w.initProtocol(cs) + w.conns[fd] = cs + w.connCount = 1 + w.addLiveConn(cs) + w.maxFD = fd + w.activeConns.Add(1) + f := &fdlFixture{t: t, e: e, w: w, cs: cs, fd: fd, peer: peer, gen: cs.generation, tgt: &fdlTarget{}} + t.Cleanup(func() { + _ = unix.Close(peer) + // Every descriptor the worker still owns (the fixture's own, or the + // one a refused hand-off was reclaimed on): nothing else closes them. + for _, c := range w.conns { + if c != nil { + _ = unix.Close(c.fd) + } + } + }) + return f +} + +// startDrain / stopDrain are Engine.StartTransplant / StopTransplant for this +// one worker. A new holder each time, as StartTransplant makes. +func (f *fdlFixture) startDrain() { + f.w.transplant.Store(&transplantTargetHolder{target: f.tgt}) +} + +func (f *fdlFixture) stopDrain() { f.w.transplant.Store(nil) } + +// armFirstRecv is onAcceptedFD's first recv arm. +func (f *fdlFixture) armFirstRecv() { + f.t.Helper() + if !f.w.prepareRecv(f.cs, f.cs.buf) { + f.t.Fatal("first recv arm refused on an empty ring") + } + if got := takeSQEs(f.w.ring); len(got) != 1 || got[0].op != opRECV { + f.t.Fatalf("first arm placed %v, want one RECV", got) + } +} + +func (f *fdlFixture) process(c *completionEntry) { + f.now++ + f.w.processCQE(context.Background(), c, f.now) +} + +func (f *fdlFixture) recvCQE(res int32) *completionEntry { + return &completionEntry{UserData: encodeUserDataGen(udRecv, f.fd, f.gen), Res: res} +} + +// reapCQE is the completion of the hand-off's own recv cancel. +func (f *fdlFixture) reapCQE(res int32) *completionEntry { + return &completionEntry{UserData: encodeUserDataGen(wantReapTag, f.fd, f.gen), Res: res} +} + +// sendCQE completes the SEND in flight in full. +func (f *fdlFixture) sendCQE() *completionEntry { + f.t.Helper() + if !f.cs.sending { + f.t.Fatal("sendCQE: no SEND in flight") + } + n := len(f.cs.sendBuf) + len(f.cs.sendBody) + return &completionEntry{UserData: encodeUserDataGen(udSend, f.fd, f.gen), Res: int32(n)} +} + +// deliver completes the armed recv with req's bytes in cs.buf. +func (f *fdlFixture) deliver(req string) { + n := copy(f.cs.buf, req) + f.process(f.recvCQE(int32(n))) +} + +// serveOne runs one request to completion with no drain set, leaving the +// state every keep-alive conn is in between requests on the base: the +// response flushed and its linked RECV armed. +func (f *fdlFixture) serveOne() { + f.t.Helper() + f.deliver(fdlGET) + if got := takeSQEs(f.w.ring); len(got) != 2 || got[0].op != opSEND || got[1].op != opRECV { + f.t.Fatalf("serving a request with no drain placed %v, want SEND then its linked RECV", got) + } + f.process(f.sendCQE()) + if got := takeSQEs(f.w.ring); len(got) != 0 { + f.t.Fatalf("the SEND's completion placed %v, want nothing", got) + } + if !f.cs.recvArmed || f.cs.sending { + f.t.Fatalf("after one request: recvArmed=%v sending=%v, want true/false", f.cs.recvArmed, f.cs.sending) + } +} + +// metric reads one EngineMetrics field by name, so a test can pin a counter +// that a tree may not have yet: a missing field fails the test by name +// instead of failing the package build. +func metric(t *testing.T, e *Engine, name string) uint64 { + t.Helper() + v := reflect.ValueOf(e.Metrics()).FieldByName(name) + if !v.IsValid() { + t.Fatalf("engine.EngineMetrics has no field %s (celeris#657 PR-2 counter)", name) + } + return v.Uint() +} + +// isReap reports whether s is the hand-off's reported cancel of cs's recv: +// matched on the recv's own user_data (op, fd AND generation, so it can never +// cancel the next owner of the fd number), tagged with the reap tag, and +// REPORTED (no CQE_SKIP_SUCCESS), so a miss produces a completion. +func (f *fdlFixture) isReap(s sqeRec) bool { + return s.op == opASYNCCANCEL && s.flags&sqeCQESkipSuccess == 0 && + s.addr == encodeUserDataGen(udRecv, f.fd, f.gen) && + s.ud == encodeUserDataGen(wantReapTag, f.fd, f.gen) +} + +func fdIsOpen(fd int) bool { + _, err := unix.FcntlInt(uintptr(fd), unix.F_GETFD, 0) + return err == nil +} diff --git a/engine/iouring/fd_lifetime_test.go b/engine/iouring/fd_lifetime_test.go new file mode 100644 index 00000000..e2e26fb0 --- /dev/null +++ b/engine/iouring/fd_lifetime_test.go @@ -0,0 +1,557 @@ +//go:build linux + +package iouring + +import ( + "context" + "testing" + "time" + + "golang.org/x/sys/unix" +) + +// celeris#657 face 2 (the fd-lifetime rule, R0): a connection leaves io_uring +// only when no read can still resolve its descriptor. The hand-off used to dup +// the fd, close the original and cancel the armed recv with a SKIP_SUCCESS +// cancel, while that recv could still complete with the client's next request +// or, through a recycled fd number, another connection's. Measured: client +// errors == joined stale-data CQEs, 1,707 == 1,707 in 48 of 48 runs. +// +// The fix has two halves. HOLD: while a drain is set, an eligible response is +// flushed with no recv behind it, so its SEND completion finds nothing in +// flight and hands the conn off. REAP: a conn whose recv is already armed is +// not handed off; the worker cancels that recv with a reported cancel of its +// own tag and hands off at the recv's -ECANCELED. A cancel that misses is +// retried; it is never followed by a hand-off. + +// TestTransplantNeverHandsOffArmedRecv pins R0 at the sync hand-off site. +func TestTransplantNeverHandsOffArmedRecv(t *testing.T) { + // The udSend dispatch site exactly as run()'s inlined switch runs it: + // the send is handled, then, with a drain set, tryTransplant. The base + // hands the conn off here with its linked RECV still armed. + t.Run("dispatch_site_refuses_and_reaps", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + f.deliver(fdlGET) + if got := takeSQEs(f.w.ring); len(got) != 2 || got[1].op != opRECV || got[0].flags&sqeIOLink == 0 { + t.Fatalf("setup: request placed %v, want SEND|LINK then RECV", got) + } + f.startDrain() + c := f.sendCQE() + if !f.w.staleConnCQE(c, f.fd, c.UserData) { + f.w.handleSend(c, f.fd, 1) + if f.w.transplant.Load() != nil { + f.w.tryTransplant(f.fd) + } + } + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("handed off %d time(s) with the linked RECV armed: that recv reads the "+ + "client's next request after the fd moved (TransplantHandoffInFlight=%d)", + n, f.e.metrics.handoffLoss.handoffInFlight.Load()) + } + if f.w.conns[f.fd] != f.cs || !f.cs.recvArmed { + t.Fatal("the conn left the table, or its recv was forgotten, without a hand-off") + } + sqes := takeSQEs(f.w.ring) + if len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("placed %v, want exactly one REPORTED cancel of this recv's user_data "+ + "tagged 0x09 (the reap)", sqes) + } + }) + + // A request served while a drain is set is HELD: its response goes out + // with no recv behind it, so its SEND completion finds nothing in flight + // and hands the conn off there. Through processCQE, which the + // listener-close harvest uses. + t.Run("held_response_hands_off_with_nothing_in_flight", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + f.startDrain() + f.deliver(fdlGET) // the linked recv takes request 2 while the drain is set + sq := takeSQEs(f.w.ring) + if len(sq) != 1 || sq[0].op != opSEND || sq[0].flags&sqeIOLink != 0 { + t.Fatalf("a request served while a drain is set placed %v, want one UNLINKED SEND "+ + "and no recv behind it (HOLD)", sq) + } + if f.cs.recvArmed { + t.Fatal("a held response left a recv armed") + } + f.process(f.sendCQE()) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the held conn's SEND completion made %d hand-offs, want 1", n) + } + if got := takeSQEs(f.w.ring); len(got) != 0 { + t.Fatalf("the hand-off placed %v, want nothing: no op was in flight to cancel", got) + } + if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) + } + if n := metric(t, f.e, "TransplantHeld"); n != 1 { + t.Errorf("TransplantHeld = %d, want 1", n) + } + }) + + t.Run("cancelled_recv_hands_off", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + // Request 2 is served before the drain starts: its response goes out + // with a linked RECV, and the drain is set while that SEND is in flight. + f.deliver(fdlGET) + _ = takeSQEs(f.w.ring) + f.startDrain() + f.process(f.sendCQE()) + sqes := takeSQEs(f.w.ring) + if n := f.tgt.adopted.Load(); n != 0 || len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("SEND completion with the linked RECV armed: %d hand-off(s), placed %v; "+ + "want 0 and exactly one reap", n, sqes) + } + closes := f.w.closeCount.Load() + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the reaped recv's -ECANCELED made %d hand-offs, want 1 (closeCount %d -> %d: "+ + "a -ECANCELED routed to the generic error branch closes a healthy conn)", + n, closes, f.w.closeCount.Load()) + } + if f.w.closeCount.Load() != closes { + t.Fatal("the reaped conn was closed") + } + if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) + } + // The cancel's own completion (a hit) arrives for a conn that left. + c := f.reapCQE(1) + if !f.w.staleConnCQE(c, f.fd, c.UserData) { + t.Fatal("the reap's completion was not stale after the hand-off") + } + }) + + t.Run("data_first_is_served_and_held", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + f.deliver(fdlGET) + _ = takeSQEs(f.w.ring) + f.startDrain() + f.process(f.sendCQE()) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("placed %v, want one reap", sqes) + } + reqs := f.w.reqBatch + // The client's next request beats the cancel: the recv completes + // with data. It must be served here, not dropped, and its response + // held. + f.deliver(fdlGET) + if f.w.reqBatch != reqs+1 { + t.Fatalf("the request that beat the reap was not served (reqBatch %d -> %d)", reqs, f.w.reqBatch) + } + sqes := takeSQEs(f.w.ring) + if len(sqes) != 1 || sqes[0].op != opSEND || sqes[0].flags&sqeIOLink != 0 { + t.Fatalf("its response placed %v, want one UNLINKED SEND and no recv (HOLD)", sqes) + } + if f.tgt.adopted.Load() != 0 { + t.Fatal("handed off with the response still in flight") + } + // The reap's cancel then reports its miss: no hand-off follows it. + f.process(f.reapCQE(0)) + if f.tgt.adopted.Load() != 0 { + t.Fatal("a missed reap was followed by a hand-off") + } + f.process(f.sendCQE()) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the held response's SEND completion made %d hand-offs, want 1", n) + } + if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) + } + }) + + // The kernel, not the test, completes the ops: the reap's SQE really + // cancels the armed recv (so its user_data target is right), and the + // hand-off follows the -ECANCELED in whichever order the two CQEs come. + t.Run("kernel_cancels_the_armed_recv", func(t *testing.T) { + f := newFDLFixture(t, false) + if !f.w.prepareRecv(f.cs, f.cs.buf) { + t.Fatal("arm refused") + } + if _, err := f.w.ring.Submit(); err != nil { + t.Fatalf("submit recv: %v", err) + } + f.startDrain() + f.w.tryTransplant(f.fd) + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("tryTransplant handed off an idle conn with its recv armed (%d)", n) + } + if f.w.ring.Pending() != 1 { + t.Fatalf("tryTransplant placed %d SQEs, want the reap", f.w.ring.Pending()) + } + if _, err := f.w.ring.Submit(); err != nil { + t.Fatalf("submit reap: %v", err) + } + got := 0 + for deadline := time.Now().Add(2 * time.Second); got < 2 && time.Now().Before(deadline); { + _ = f.w.ring.WaitCQETimeout(100 * time.Millisecond) + head, tail := f.w.ring.BeginCQ() + for ; head != tail; head++ { + c := *f.w.ring.cqeAt(head) + if decodeOp(c.UserData) == udRecv && c.Res != -int32(unix.ECANCELED) { + t.Fatalf("the recv completed with %d, want -ECANCELED", c.Res) + } + f.w.processCQE(context.Background(), &c, time.Now().UnixNano()) + got++ + } + f.w.ring.EndCQ(head) + } + if got != 2 { + t.Fatalf("kernel produced %d completions, want the recv's -ECANCELED and the reap's own", got) + } + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("%d hand-offs after the kernel cancelled the recv, want 1", n) + } + }) +} + +// TestTransplantReapMissIsRetried: a reap that finds nothing to cancel (the +// recv completed first, or is still linked behind its SEND and not issued yet) +// is retried and never followed by a hand-off; and its count is a COUNT, so a +// stale miss cannot clear a newer reap (the celeris#484/#596 lesson). +func TestTransplantReapMissIsRetried(t *testing.T) { + armedAndReaped := func(t *testing.T) *fdlFixture { + t.Helper() + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + f.deliver(fdlGET) + _ = takeSQEs(f.w.ring) + f.startDrain() + f.process(f.sendCQE()) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("SEND completion with the linked RECV armed placed %v, want one reap", sqes) + } + return f + } + + t.Run("miss_is_retried_never_handed_off", func(t *testing.T) { + f := armedAndReaped(t) + f.process(f.reapCQE(-int32(unix.ENOENT))) + if f.tgt.adopted.Load() != 0 || !f.cs.recvArmed { + t.Fatal("a missed reap was followed by a hand-off with the recv still armed") + } + // The next loop iteration retries it. + f.w.drainDetachQueue() + sqes := takeSQEs(f.w.ring) + if len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("the next iteration placed %v, want the reap again", sqes) + } + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the retried reap's -ECANCELED made %d hand-offs, want 1", n) + } + if n := metric(t, f.e, "TransplantReapMisses"); n != 1 { + t.Errorf("TransplantReapMisses = %d, want 1", n) + } + if n := metric(t, f.e, "TransplantReaps"); n != 2 { + t.Errorf("TransplantReaps = %d, want 2", n) + } + }) + + t.Run("stale_miss_cannot_clear_a_newer_reap", func(t *testing.T) { + f := armedAndReaped(t) + // Reap 1 is in flight. The client's request beats it; the drain + // stops before the response, so the response goes out with a new + // linked RECV (recv 2). + f.stopDrain() + f.deliver(fdlGET) + if sqes := takeSQEs(f.w.ring); len(sqes) != 2 || sqes[1].op != opRECV { + t.Fatalf("with the drain stopped the response placed %v, want SEND then RECV", sqes) + } + // A new drain (the next flap): recv 2 gets a reap of its own while + // reap 1's completion is still owed. + f.startDrain() + f.process(f.sendCQE()) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("recv 2 got %v, want its own reap while reap 1 is still owed", sqes) + } + // Reap 1 reports its miss (its recv had already completed). This must + // not forget reap 2, which is about to hit. + f.process(f.reapCQE(0)) + f.w.drainDetachQueue() + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 { + t.Fatalf("a stale miss re-reaped a recv that already has a reap in flight: %v", sqes) + } + closes := f.w.closeCount.Load() + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if f.w.closeCount.Load() != closes { + t.Fatal("reap 2's -ECANCELED closed the conn: the stale miss cleared the reap's state " + + "and the cancel fell through to the generic error branch") + } + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("reap 2's -ECANCELED made %d hand-offs, want 1", n) + } + }) +} + +// TestHoldReleasedWhenDrainStops: a held conn (response flushed with no recv +// behind it) must get its recv armed at that SEND's completion when it is not +// handed off there — here because the drain stopped in between. Otherwise it +// waits for a request it can never read. +func TestHoldReleasedWhenDrainStops(t *testing.T) { + held := func(t *testing.T) *fdlFixture { + t.Helper() + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + f.startDrain() + f.deliver(fdlGET) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || sqes[0].op != opSEND || sqes[0].flags&sqeIOLink != 0 { + t.Fatalf("a response while a drain is set placed %v, want one UNLINKED SEND (HOLD)", sqes) + } + // serveOne left the linked RECV armed; it took request 2 above, so + // the conn now has no recv at all. + if f.cs.recvArmed { + t.Fatal("held conn has a recv armed") + } + return f + } + + t.Run("processCQE_send_rearms_after_drain_stops", func(t *testing.T) { + f := held(t) + f.stopDrain() + f.process(f.sendCQE()) + sqes := takeSQEs(f.w.ring) + if len(sqes) != 1 || sqes[0].op != opRECV || sqes[0].tag() != udRecv { + t.Fatalf("the held conn's SEND completion placed %v, want its recv armed", sqes) + } + if !f.cs.recvArmed || f.tgt.adopted.Load() != 0 { + t.Fatalf("recvArmed=%v adopted=%d, want true/0", f.cs.recvArmed, f.tgt.adopted.Load()) + } + if n := metric(t, f.e, "TransplantHoldRescued"); n != 0 { + t.Errorf("TransplantHoldRescued = %d, want 0", n) + } + }) + + t.Run("processCQE_send_hands_off_while_drain_set", func(t *testing.T) { + f := held(t) + f.process(f.sendCQE()) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("processCQE handled the held conn's SEND completion with a drain set and "+ + "made %d hand-offs, want 1", n) + } + }) + + // The worker's own loop: the inlined udSend dispatch site. + t.Run("worker_loop_rearms_after_drain_stops", testHoldReleasedInTheWorkerLoop) + + t.Run("refused_hand_off_rearms", func(t *testing.T) { + f := held(t) + f.tgt.refuse = true + f.process(f.sendCQE()) + // The target refused: reclaimTransplant re-attached the dup'd fd as + // a new conn with its own recv; the old slot is empty. + if f.w.transplantHandoffRefused.Load() != 1 { + t.Fatalf("hand-off refusals = %d, want 1", f.w.transplantHandoffRefused.Load()) + } + armed := 0 + for _, s := range takeSQEs(f.w.ring) { + if s.op == opRECV { + armed++ + } + } + if armed != 1 { + t.Fatalf("after a refused hand-off %d recv(s) were armed, want exactly 1", armed) + } + }) +} + +// TestHoldRescuedByCheckTimeouts is the belt under the release: a held conn +// whose send is done and that was neither handed off nor re-armed (a path that +// skipped the release) gets its recv armed by the timeout sweep, and the +// rescue is counted. The counter must stay 0 in every real run. +func TestHoldRescuedByCheckTimeouts(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + f.startDrain() + f.deliver(fdlGET) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || sqes[0].op != opSEND { + t.Fatalf("placed %v, want one held SEND", sqes) + } + // The SEND completes through handleSend alone: no dispatch site ran, so + // nothing released the hold. + f.stopDrain() + c := f.sendCQE() + if !f.w.staleConnCQE(c, f.fd, c.UserData) { + f.w.handleSend(c, f.fd, 1) + } + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 || f.cs.recvArmed { + t.Fatalf("handleSend alone placed %v (recvArmed=%v); the release belongs to the dispatch site", sqes, f.cs.recvArmed) + } + f.w.checkTimeouts() + sqes := takeSQEs(f.w.ring) + if len(sqes) != 1 || sqes[0].op != opRECV { + t.Fatalf("checkTimeouts placed %v, want the stranded conn's recv", sqes) + } + if n := metric(t, f.e, "TransplantHoldRescued"); n != 1 { + t.Fatalf("TransplantHoldRescued = %d, want 1", n) + } + // A second sweep finds nothing more to do. + f.w.checkTimeouts() + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 { + t.Fatalf("a second sweep placed %v", sqes) + } +} + +// TestOneOwnerPerHandoff (A6): a promoted async conn whose dispatch goroutine +// has claimed its own hand-off (transplantPending) belongs to that claim, and +// a finishAsyncTransplant for a connState that no longer owns its slot must +// not touch the fd number. Measured on the base: one identity moved twice in +// 167 runs, 20 us apart, the only run with a negative gauge. +func TestOneOwnerPerHandoff(t *testing.T) { + // claimedAsync is a promoted async conn parked at a boundary whose + // goroutine has marked it and exited. Nothing is in flight, so the only + // thing between it and a second hand-off is the ownership check. + claimedAsync := func(t *testing.T) *fdlFixture { + t.Helper() + f := newFDLFixture(t, true) + f.cs.asyncPromoted.Store(true) + f.cs.transplantPending.Store(true) + f.startDrain() + return f + } + + t.Run("tryTransplant_refuses_a_claimed_conn", func(t *testing.T) { + f := claimedAsync(t) + f.w.tryTransplant(f.fd) + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("tryTransplant moved a conn its dispatch goroutine had claimed (%d hand-offs); "+ + "finishAsyncTransplant would then move it again", n) + } + if n := metric(t, f.e, "TransplantDoubleClaim"); n != 1 { + t.Errorf("TransplantDoubleClaim = %d, want 1", n) + } + }) + + t.Run("finishAsyncTransplant_refuses_a_stale_owner", func(t *testing.T) { + f := claimedAsync(t) + old := f.cs + // The slot now belongs to a different connState on the same fd + // number (the old one left and the number was reused). + next := acquireConnState(context.Background(), f.fd, 4096, true) + f.w.conns[f.fd] = next + f.w.finishAsyncTransplant(old) + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("finishAsyncTransplant handed off fd %d for a connState that no longer owns "+ + "it (%d hand-offs): that is the next owner's socket", f.fd, n) + } + if f.w.conns[f.fd] != next || !fdIsOpen(f.fd) { + t.Fatal("the next owner of the fd number lost its slot or its descriptor") + } + if n := metric(t, f.e, "TransplantDoubleClaim"); n != 1 { + t.Errorf("TransplantDoubleClaim = %d, want 1", n) + } + f.w.conns[f.fd] = old // let the fixture's cleanup close the fd + }) + + // The measured interleaving end to end: the goroutine claims the + // hand-off and enqueues itself; before the worker drains the queue, a + // SEND completion for the same conn reaches the udSend dispatch site. + t.Run("send_completion_between_claim_and_drain", func(t *testing.T) { + f := claimedAsync(t) + f.w.enqueueDetach(f.cs) + f.cs.writeBuf = append(f.cs.writeBuf, "HTTP/1.1 200 OK\r\ncontent-length: 2\r\n\r\nok"...) + if f.w.flushSend(f.cs) { + t.Fatal("setup: flushSend found the ring full") + } + _ = takeSQEs(f.w.ring) + c := f.sendCQE() + if !f.w.staleConnCQE(c, f.fd, c.UserData) { + f.w.handleSend(c, f.fd, 1) + if f.w.transplant.Load() != nil { + f.w.tryTransplant(f.fd) + } + } + if f.w.conns[f.fd] == nil { + // The sync path took the conn and closed its fd. In the measured + // run the number had already been reused by the time the queued + // claim ran; reuse it here the same way (a decoy socket on the + // same number), so a second hand-off is visible as a hand-off. + a, b := socketPairFDs(t) + t.Cleanup(func() { _ = unix.Close(b) }) + if a != f.fd { // the kernel may already have handed out the freed number + if err := unix.Dup3(a, f.fd, unix.O_CLOEXEC); err != nil { + t.Fatalf("dup3: %v", err) + } + _ = unix.Close(a) + } + } + f.w.drainDetachQueue() + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the conn was handed off %d times, want exactly once", n) + } + if n := metric(t, f.e, "TransplantDoubleClaim"); n != 1 { + t.Errorf("TransplantDoubleClaim = %d, want 1 (the sync claim refused)", n) + } + }) +} + +// TestNoDrainSQESequenceIsUnchanged is the witness that routing every +// response tail through one hold-aware helper costs nothing when no drain is +// set: per request, the SQEs placed (opcode, flags including IO_LINK, op tag, +// fd) are exactly the base's. It passes on the base by construction; a change +// to the steady-state sequence fails it. +func TestNoDrainSQESequenceIsUnchanged(t *testing.T) { + type want struct { + op uint8 + flags uint8 + tag uint64 + } + check := func(t *testing.T, f *fdlFixture, got []sqeRec, w []want) { + t.Helper() + ok := len(got) == len(w) + for i := 0; ok && i < len(w); i++ { + ok = got[i].op == w[i].op && got[i].flags == w[i].flags && got[i].tag() == w[i].tag && + int(got[i].fd) == f.fd + } + if !ok { + t.Fatalf("per-request SQEs %v, want %+v", got, w) + } + } + + t.Run("sync_tail", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + for i := 0; i < 3; i++ { + f.deliver(fdlGET) + check(t, f, takeSQEs(f.w.ring), []want{{opSEND, sqeIOLink, udSend}, {opRECV, 0, udRecv}}) + f.process(f.sendCQE()) + check(t, f, takeSQEs(f.w.ring), nil) + } + }) + + t.Run("sync_tail_writev", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + f.deliver("GET /large HTTP/1.1\r\nHost: x\r\n\r\n") + check(t, f, takeSQEs(f.w.ring), []want{{opWRITEV, 0, udSend}, {opRECV, 0, udRecv}}) + f.process(f.sendCQE()) + check(t, f, takeSQEs(f.w.ring), nil) + }) + + t.Run("direct_body_tail", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + // Half the body: no response yet, and the next recv goes straight + // into the body buffer. + f.deliver("POST / HTTP/1.1\r\nHost: x\r\nContent-Length: 20\r\n\r\n0123456789") + check(t, f, takeSQEs(f.w.ring), []want{{opRECV, 0, udRecv}}) + if !f.cs.recvIntoBody || len(f.cs.bodyRecvPin) < 10 { + t.Fatalf("setup: the body recv was not armed into the body buffer") + } + n := copy(f.cs.bodyRecvPin, "abcdefghij") + f.process(f.recvCQE(int32(n))) + check(t, f, takeSQEs(f.w.ring), []want{{opSEND, 0, udSend}, {opRECV, 0, udRecv}}) + f.process(f.sendCQE()) + check(t, f, takeSQEs(f.w.ring), nil) + }) +} From bbae2ba99128903fce52ad7051621cf3cb89a44c Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:43:14 +0200 Subject: [PATCH 02/34] refactor(iouring): one hand-off primitive for both io_uring hand-off sites (celeris#657) tryTransplant and finishAsyncTransplant each carried their own copy of the commit sequence (dup, detach, witnesses, cancel, release, close, adopt). Both now call handOff once their gates have passed, so the fd-lifetime gate that follows is written once and the celeris#657 witnesses stay where a gate regression would move them. No behaviour change: the order of every step and the release kind per site are as before. Refs #657 --- engine/iouring/transplant_source.go | 78 +++++++++++------------------ 1 file changed, 30 insertions(+), 48 deletions(-) diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 6798d99e..22faa889 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -100,24 +100,39 @@ func (w *Worker) tryTransplant(fd int) { return } - // Dup the fd so epoll gets a clean, ops-free descriptor immediately while - // io_uring cancels + drains its armed recv on the original. + w.handOff(cs, fd, h, false) +} + +// handOff is the commit point both hand-off sites share (tryTransplant for a +// sync conn, finishAsyncTransplant for a promoted async one), reached only +// after the caller's gates have passed. It dups the fd for epoll, detaches the +// conn from io_uring the way hijackConn does (drop it from the live set and the +// conn table, cancel what is still armed, defer the connState release to the +// terminal CQEs), closes the ORIGINAL fd (the dup keeps the socket alive for +// epoll) and hands the dup over. A failed dup leaves the conn untouched and +// reports false; the hand-off is retried at the conn's next boundary. +// +// detachedRelease picks the release: the async site's dispatch goroutine may +// still be in its deferred recover()/Done() block referencing cs, so that site +// holds cs alive with no pool recycle until the kernel ops drain (mirrors +// finishCloseDetached). +func (w *Worker) handOff(cs *connState, fd int, h *transplantTargetHolder, detachedRelease bool) bool { newFD, err := unix.Dup(fd) if err != nil { - return + return false } if serr := unix.SetNonblock(newFD, true); serr != nil { _ = unix.Close(newFD) - return + return false } carry := engine.Carryover{RemoteAddr: cs.remoteAddr} - // Detach from io_uring (mirror hijackConn): drop from live set/conn table, - // cancel the armed recv, defer the connState release to its terminal CQE, and - // close the ORIGINAL fd (the dup keeps the socket alive for epoll). // Unlink from the dirty list (celeris#527): the original fd is closed // below and its number may be re-accepted, so a dirty-loop retry would - // call prepareRecv on a stranger's socket. + // call prepareRecv on a stranger's socket. For the async site this is the + // one teardown path with no cs.sending guard at all, so it is the only + // way a detached connState reaches the list with a SEND in flight -- the + // immortal-entry case that pins the worker at 100% CPU (celeris#529). w.removeDirty(cs) w.removeLiveConn(cs) w.conns[fd] = nil @@ -141,12 +156,17 @@ func (w *Worker) tryTransplant(fd int) { w.noteHandoffInFlight(cs) w.cancelConnOps(fd, cs) w.noteHandedOffInflight(cs) - w.queuePendingRelease(cs) + if detachedRelease { + w.queuePendingReleaseDetached(cs) + } else { + w.queuePendingRelease(cs) + } _ = unix.Close(fd) if err := h.target.AdoptConn(newFD, carry); err != nil { w.reclaimTransplant(newFD, carry, err) } + return true } // reclaimTransplant takes back a connection this worker had already @@ -244,43 +264,5 @@ func (w *Worker) finishAsyncTransplant(cs *connState) { if cs.sending || cs.zcNotifPending || len(cs.sendBuf) != 0 || len(cs.writeBuf) != 0 { return } - fd := cs.fd - newFD, err := unix.Dup(fd) - if err != nil { - return // can't dup — leave the conn; transplant retried at its next park - } - if serr := unix.SetNonblock(newFD, true); serr != nil { - _ = unix.Close(newFD) - return - } - carry := engine.Carryover{RemoteAddr: cs.remoteAddr} - - // Unlink from the dirty list (celeris#527). This is the one teardown - // path with no cs.sending guard at all, so it is the only way a detached - // connState reaches the list with a SEND in flight — the immortal-entry - // case that pins the worker at 100% CPU. See also celeris#529. - w.removeDirty(cs) - w.removeLiveConn(cs) - w.conns[fd] = nil - w.connCount-- - w.activeConns.Add(-1) - // Same detach-for-transplant ledger entry as tryTransplant's, on the - // self-initiated async path — and, for the same reason, no closeCount - // bump: a detach is not a close (celeris#624). - if w.transplantDetached != nil { - w.transplantDetached.Add(1) - } - // The same celeris#657 witnesses as tryTransplant's. - w.noteHandoffInFlight(cs) - w.cancelConnOps(fd, cs) - w.noteHandedOffInflight(cs) - // Detached release: the dispatch goroutine may still be in its deferred - // recover()/Done() block referencing cs; hold cs alive (no pool recycle) until - // the kernel recv drains. Mirrors finishCloseDetached. - w.queuePendingReleaseDetached(cs) - _ = unix.Close(fd) - - if err := h.target.AdoptConn(newFD, carry); err != nil { - w.reclaimTransplant(newFD, carry, err) - } + w.handOff(cs, cs.fd, h, true) } From f06310b82287d00cad123ee3990a437952bf8174 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:44:05 +0200 Subject: [PATCH 03/34] feat(engine): counters for the io_uring hand-off's fd-lifetime rule (celeris#657) EngineMetrics gains five io_uring-only, cumulative fields, exported the way the PR-1 witnesses are (the engine's handoffLoss set, Metrics(), and the adaptive sum over both sub-engines): - TransplantHeld: responses flushed with the next recv held for a hand-off (HOLD). - TransplantReaps / TransplantReapMisses: reported cancels of an armed recv submitted so the conn can be handed off (REAP), and those that matched nothing. - TransplantHoldRescued: held conns the timeout sweep had to rescue. Must stay 0. - TransplantDoubleClaim: hand-offs refused because another path owned the conn's hand-off (a double hand-off prevented). Nothing increments them yet; the rule that does follows. Refs #657 --- adaptive/engine.go | 10 ++++ adaptive/handoff_loss_metrics_test.go | 9 +++ engine/engine.go | 40 +++++++++++++ engine/iouring/engine.go | 5 ++ engine/iouring/handoff_loss.go | 66 +++++++++++++++++++-- engine/iouring/handoff_loss_metrics_test.go | 10 ++++ 6 files changed, 136 insertions(+), 4 deletions(-) diff --git a/adaptive/engine.go b/adaptive/engine.go index 761da4c7..2360890f 100644 --- a/adaptive/engine.go +++ b/adaptive/engine.go @@ -1078,6 +1078,16 @@ func (e *Engine) Metrics() engine.EngineMetrics { StaleRecvDataTransplanted: pm.StaleRecvDataTransplanted + sm.StaleRecvDataTransplanted, StaleRecvDataUnattributed: pm.StaleRecvDataUnattributed + sm.StaleRecvDataUnattributed, TransplantHandoffInFlight: pm.TransplantHandoffInFlight + sm.TransplantHandoffInFlight, + // The fd-lifetime rule's counters (celeris#657 PR-2), io_uring-only + // and cumulative. A hold, a reap and a refused double claim each + // happen on the one sub-engine making the hand-off, so the sum + // counts each once; like the witnesses above, the standby's half + // is where a revert's hand-offs are made. + TransplantHeld: pm.TransplantHeld + sm.TransplantHeld, + TransplantReaps: pm.TransplantReaps + sm.TransplantReaps, + TransplantReapMisses: pm.TransplantReapMisses + sm.TransplantReapMisses, + TransplantHoldRescued: pm.TransplantHoldRescued + sm.TransplantHoldRescued, + TransplantDoubleClaim: pm.TransplantDoubleClaim + sm.TransplantDoubleClaim, // The celeris#607 recv-stall and linked-recv ledger. io_uring-only, // so the epoll half contributes zero and a switch simply moves which diff --git a/adaptive/handoff_loss_metrics_test.go b/adaptive/handoff_loss_metrics_test.go index 89be9fea..57a87c40 100644 --- a/adaptive/handoff_loss_metrics_test.go +++ b/adaptive/handoff_loss_metrics_test.go @@ -22,10 +22,14 @@ func TestMetricsSumsTheHandoffLossWitnesses(t *testing.T) { e.primary.(*mockEngine).SetMetrics(engine.EngineMetrics{ StaleRecvDataClosed: 1, StaleRecvDataTransplanted: 2, StaleRecvDataUnattributed: 3, TransplantHandoffInFlight: 4, + TransplantHeld: 5, TransplantReaps: 6, TransplantReapMisses: 7, + TransplantHoldRescued: 8, TransplantDoubleClaim: 9, }) e.secondary.(*mockEngine).SetMetrics(engine.EngineMetrics{ StaleRecvDataClosed: 10, StaleRecvDataTransplanted: 20, StaleRecvDataUnattributed: 30, TransplantHandoffInFlight: 40, + TransplantHeld: 50, TransplantReaps: 60, TransplantReapMisses: 70, + TransplantHoldRescued: 80, TransplantDoubleClaim: 90, }) m := e.Metrics() @@ -37,6 +41,11 @@ func TestMetricsSumsTheHandoffLossWitnesses(t *testing.T) { {"StaleRecvDataTransplanted", m.StaleRecvDataTransplanted, 22}, {"StaleRecvDataUnattributed", m.StaleRecvDataUnattributed, 33}, {"TransplantHandoffInFlight", m.TransplantHandoffInFlight, 44}, + {"TransplantHeld", m.TransplantHeld, 55}, + {"TransplantReaps", m.TransplantReaps, 66}, + {"TransplantReapMisses", m.TransplantReapMisses, 77}, + {"TransplantHoldRescued", m.TransplantHoldRescued, 88}, + {"TransplantDoubleClaim", m.TransplantDoubleClaim, 99}, } { if c.got != c.want { t.Errorf("Metrics().%s = %d, want %d (sum of both sub-engines)", diff --git a/engine/engine.go b/engine/engine.go index 4024664f..93a56f59 100644 --- a/engine/engine.go +++ b/engine/engine.go @@ -484,6 +484,46 @@ type EngineMetrics struct { //nolint:revive // user-approved name // io_uring-only; zero on other engines. On the adaptive engine it is // the sum over both sub-engines. TransplantHandoffInFlight uint64 + // TransplantHeld, TransplantReaps and TransplantReapMisses count how the + // io_uring hand-off keeps its fd-lifetime rule (celeris#657): a + // connection leaves io_uring only when no read can still resolve its + // descriptor, which is what takes TransplantHandoffInFlight and + // StaleRecvDataTransplanted to 0. + // + // - TransplantHeld: responses flushed with the connection's next recv + // held back because a drain was set, so the hand-off at that send's + // completion finds nothing in flight. + // - TransplantReaps: reported cancels of an already-armed recv, submitted + // so the connection can be handed off at that recv's cancellation. + // - TransplantReapMisses: those cancels that matched nothing (the recv + // had completed, or was not issued yet). A miss is retried and is + // never followed by a hand-off. + // + // Rates, not invariants: all three are zero while no drain runs. + // io_uring-only and cumulative; zero on other engines. On the adaptive + // engine each is the sum over both sub-engines. + TransplantHeld uint64 + TransplantReaps uint64 + TransplantReapMisses uint64 + // TransplantHoldRescued counts connections whose recv was held for a + // hand-off that did not happen and that no completion released: the + // timeout sweep found them with their response sent and no recv armed, + // and armed it. It is a belt under the release at every send + // completion, and must stay 0; a non-zero value is a connection that sat + // unable to read until the sweep came by (celeris#657). io_uring-only; + // on the adaptive engine the sum over both sub-engines. + TransplantHoldRescued uint64 + // TransplantDoubleClaim counts io_uring hand-offs refused because another + // path already owned that connection's hand-off: the sync path finding a + // connection whose async dispatch goroutine had claimed its own hand-off, + // or the async completion finding its connection no longer owns its + // descriptor slot. Each is a connection that would otherwise have been + // handed off twice, the second time as whatever socket then held the + // descriptor number (celeris#657). The first case needs a send + // completion to land between the claim and the worker's drain of it, so + // a low non-zero rate is the check working, not a fault. io_uring-only; + // on the adaptive engine the sum over both sub-engines. + TransplantDoubleClaim uint64 } // FillErrorClasses copies one engine's per-cause error tally into m and diff --git a/engine/iouring/engine.go b/engine/iouring/engine.go index 3b249dc8..8f913b7e 100644 --- a/engine/iouring/engine.go +++ b/engine/iouring/engine.go @@ -446,6 +446,11 @@ func (e *Engine) Metrics() engine.EngineMetrics { StaleRecvDataTransplanted: e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), StaleRecvDataUnattributed: e.metrics.handoffLoss.staleRecvDataUnattributed.Load(), TransplantHandoffInFlight: e.metrics.handoffLoss.handoffInFlight.Load(), + TransplantHeld: e.metrics.handoffLoss.held.Load(), + TransplantReaps: e.metrics.handoffLoss.reaps.Load(), + TransplantReapMisses: e.metrics.handoffLoss.reapMisses.Load(), + TransplantHoldRescued: e.metrics.handoffLoss.holdRescued.Load(), + TransplantDoubleClaim: e.metrics.handoffLoss.doubleClaim.Load(), } // ErrorCount and its eleven buckets, together, from one snapshot // (celeris#645). diff --git a/engine/iouring/handoff_loss.go b/engine/iouring/handoff_loss.go index 00209127..15c40064 100644 --- a/engine/iouring/handoff_loss.go +++ b/engine/iouring/handoff_loss.go @@ -48,15 +48,73 @@ import "sync/atomic" // (kernelInflight != 0) or a SEND_ZC notification pending — the // precondition of every loss above. // -// Both are direct atomic adds: they fire on the stale-CQE and hand-off -// paths only, never on the per-request path, and like the celeris#586 -// witnesses they are per-event invariants that a per-iteration batch -// could lose at loop exit. +// The fd-lifetime rule that removes that loss (celeris#657 PR-2: no hand-off +// while a read can still resolve the fd) keeps its own counters here too: +// +// - held: responses flushed with the next recv HELD, because a drain was +// set and the conn was one the hand-off at that SEND's completion +// accepts (HOLD). A rate. +// - reaps / reapMisses: reported cancels of an armed recv submitted so a +// conn could be handed off (REAP), and those whose own completion said +// they matched nothing (the recv had completed, or was not issued yet). +// A miss is retried, never followed by a hand-off. Rates. +// - holdRescued: held conns the timeout sweep found with their response +// sent, not handed off and no recv armed — a path that skipped the +// release. The belt under releaseHold. Must stay 0. +// - doubleClaim: hand-offs refused because another path owned the conn's +// hand-off: tryTransplant on a conn whose dispatch goroutine had already +// claimed it (transplantPending), or finishAsyncTransplant for a +// connState that no longer owns its fd slot. Each is a double hand-off +// prevented; before the checks existed one identity was measured moving +// twice in 167 runs. Rare, not impossible: the first half fires whenever +// a SEND completion lands between a goroutine's claim and the drain. +// +// All are direct atomic adds: they fire on the stale-CQE, drain and hand-off +// paths only, never on the per-request path while no drain is set, and like +// the celeris#586 witnesses they are per-event invariants that a +// per-iteration batch could lose at loop exit. type handoffLossStats struct { staleRecvDataClosed atomic.Uint64 staleRecvDataTransplanted atomic.Uint64 staleRecvDataUnattributed atomic.Uint64 handoffInFlight atomic.Uint64 + held atomic.Uint64 + reaps atomic.Uint64 + reapMisses atomic.Uint64 + holdRescued atomic.Uint64 + doubleClaim atomic.Uint64 +} + +// The fd-lifetime counters are nil-safe: a hand-built test Worker has none. + +func (s *handoffLossStats) noteHeld() { + if s != nil { + s.held.Add(1) + } +} + +func (s *handoffLossStats) noteReap() { + if s != nil { + s.reaps.Add(1) + } +} + +func (s *handoffLossStats) noteReapMiss() { + if s != nil { + s.reapMisses.Add(1) + } +} + +func (s *handoffLossStats) noteHoldRescued() { + if s != nil { + s.holdRescued.Add(1) + } +} + +func (s *handoffLossStats) noteDoubleClaim() { + if s != nil { + s.doubleClaim.Add(1) + } } // noteStaleRecvData counts one stale recv CQE that carried data, under the diff --git a/engine/iouring/handoff_loss_metrics_test.go b/engine/iouring/handoff_loss_metrics_test.go index b5ffdfa2..2a1ea4ad 100644 --- a/engine/iouring/handoff_loss_metrics_test.go +++ b/engine/iouring/handoff_loss_metrics_test.go @@ -18,6 +18,11 @@ func TestMetricsCarriesTheHandoffLossWitnesses(t *testing.T) { e.metrics.handoffLoss.staleRecvDataTransplanted.Store(5) e.metrics.handoffLoss.staleRecvDataUnattributed.Store(7) e.metrics.handoffLoss.handoffInFlight.Store(11) + e.metrics.handoffLoss.held.Store(13) + e.metrics.handoffLoss.reaps.Store(17) + e.metrics.handoffLoss.reapMisses.Store(19) + e.metrics.handoffLoss.holdRescued.Store(23) + e.metrics.handoffLoss.doubleClaim.Store(29) m := e.Metrics() for _, c := range []struct { @@ -28,6 +33,11 @@ func TestMetricsCarriesTheHandoffLossWitnesses(t *testing.T) { {"StaleRecvDataTransplanted", m.StaleRecvDataTransplanted, 5}, {"StaleRecvDataUnattributed", m.StaleRecvDataUnattributed, 7}, {"TransplantHandoffInFlight", m.TransplantHandoffInFlight, 11}, + {"TransplantHeld", m.TransplantHeld, 13}, + {"TransplantReaps", m.TransplantReaps, 17}, + {"TransplantReapMisses", m.TransplantReapMisses, 19}, + {"TransplantHoldRescued", m.TransplantHoldRescued, 23}, + {"TransplantDoubleClaim", m.TransplantDoubleClaim, 29}, } { if c.got != c.want { t.Errorf("Metrics().%s = %d, want %d — the witness exists but cannot "+ From 4692b6e50583fd5d1ff20db34d84d7c822336952 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:46:27 +0200 Subject: [PATCH 04/34] fix(iouring): no hand-off while the connection's recv can still resolve its fd (celeris#657) P1 (R0 gate) and P2 (REAP) of the fd-lifetime rule. Both hand-off sites, tryTransplant and finishAsyncTransplant, now refuse a connection with recvArmed, kernelInflight != 0 or zcNotifPending. When the only op in flight is the recv, they reap it: a REPORTED cancel matched on the recv's own user_data (generation included) under a new tag, udTransplantReap = 0x09<<56. The hand-off runs again at the recv's -ECANCELED, which handleRecv now routes first, before the generic negative-result branch that closes the conn. finishAsyncTransplant reaps instead of cancelling, dup'ing and closing with the recv armed. - If the request arrives first, the recv completes with it and it is served here as usual. - A cancel that reports a miss is retried on the next loop iteration while the recv is still armed (drainDetachQueue runs the retries). A miss is never followed by a hand-off. - The per-connection state is a COUNT of outstanding reaps plus a flag saying they all target recvs that have since completed, so a newer recv can get its own reap and a stale miss cannot clear it (the celeris#484/#596 lesson). - A pause's cancel keeps precedence in the -ECANCELED routing; detached WS/SSE conns are never reaped. The PR-1 witness tests drove the sites with a recv armed, which the gate now refuses. They check the refusal and then drive handOff, the commit point the gate guards, so the witnesses stay proven. Refs #657 --- engine/iouring/conn.go | 28 +++++ engine/iouring/cqe.go | 10 ++ engine/iouring/fd_lifetime.go | 173 ++++++++++++++++++++++++++++ engine/iouring/handoff_loss_test.go | 63 +++++++--- engine/iouring/transplant_source.go | 47 ++++++-- engine/iouring/worker.go | 29 +++++ 6 files changed, 322 insertions(+), 28 deletions(-) create mode 100644 engine/iouring/fd_lifetime.go diff --git a/engine/iouring/conn.go b/engine/iouring/conn.go index 5df0cfa5..c938c618 100644 --- a/engine/iouring/conn.go +++ b/engine/iouring/conn.go @@ -195,6 +195,31 @@ type connState struct { // Worker-thread only, like recvPaused. recvCancelPending uint16 + // transplantReap counts the hand-off's reported recv cancels (REAP, + // celeris#657) whose effect has not been observed yet: incremented when + // startReap submits one, decremented by the -ECANCELED of the recv one + // of them cancelled or by a cancel's own completion reporting that it + // matched nothing. A COUNT, for the reason recvCancelPending is one: a + // reap can be outstanding for a recv that has since completed while a + // newer reap targets the recv armed after it, and the older one's miss + // must not clear the state the newer one's -ECANCELED needs — as a bool + // it did, and that -ECANCELED fell through handleRecv's generic error + // branch and closed a healthy connection (celeris#484/#596). + // + // reapStale: every reap counted in transplantReap was aimed at a recv + // that has since completed, so the recv armed now has none aimed at it + // and may get its own. + // + // transplantHold: the conn's last response was flushed with NO recv + // behind it, because a drain was set and the hand-off at that SEND's + // completion was expected to take the conn (HOLD). releaseHold arms the + // recv there if the hand-off does not happen. + // + // All three worker-thread only, like recvCancelPending. + transplantReap uint16 + reapStale bool + transplantHold bool + // Async handler dispatch (Worker.async=true, HTTP1 only): // Incoming recv bytes are appended under asyncInMu by the worker. // A single dispatch goroutine per conn drains asyncInBuf via a @@ -440,6 +465,9 @@ func releaseConnState(cs *connState) { cs.recvPaused = false cs.recvPauseDesired.Store(false) cs.recvCancelPending = 0 + cs.transplantReap = 0 + cs.reapStale = false + cs.transplantHold = false cs.headerTimerSpec = kernelTimespec{} cs.headerTimerArmed = false cs.forceRSTClose = false diff --git a/engine/iouring/cqe.go b/engine/iouring/cqe.go index 3fd27e4e..98a8a46d 100644 --- a/engine/iouring/cqe.go +++ b/engine/iouring/cqe.go @@ -65,6 +65,16 @@ const ( // left to correct. udRecvCancel uint64 = 0x05 << 56 udH2Wakeup uint64 = 0x07 << 56 + // udTransplantReap tags the reported ASYNC_CANCEL a hand-off submits for + // a connection's armed recv (celeris#657, the REAP half of the + // fd-lifetime rule): the connection is handed to epoll at that recv's + // own -ECANCELED, never while it can still complete. A tag of its own, + // not udRecvCancel's, so its miss can be told from the WebSocket + // pause's (celeris#596) and routed to handleTransplantReap. Conn-bound + // (stamped with the generation, so it passes the stale-CQE gate) and, + // like udRecvCancel, NOT a terminalOp: it stays outside the + // kernelInflight accounting. + udTransplantReap uint64 = 0x09 << 56 // udHeaderTimer tags IORING_OP_TIMEOUT SQEs submitted by initProtocol / // ProcessH1's arm-callback to enforce ReadHeaderTimeout per-conn. The // timer fires absolutely at the deadline; CQE handler closes the conn diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go new file mode 100644 index 00000000..720f482d --- /dev/null +++ b/engine/iouring/fd_lifetime.go @@ -0,0 +1,173 @@ +//go:build linux + +package iouring + +import "golang.org/x/sys/unix" + +// The fd-lifetime rule of the io_uring→epoll hand-off (celeris#657, face 2): +// a connection leaves io_uring only when no read can still resolve its +// descriptor. +// +// The hand-off dups the fd for epoll and closes the original. Closing does not +// end an io_uring recv (the op holds its own file reference), and a recv SQE +// that has not reached the kernel yet resolves the fd NUMBER when it does. So +// a recv left armed across the hand-off could take the client's next request +// off the socket epoll now owns, or, once a later dup or accept reused the +// number, another connection's request: measured, client errors == stale data +// CQEs, 1,707 == 1,707 in 48 of 48 runs, while the #624 ledger balanced. +// +// Two mechanisms keep the rule; both hand-off sites refuse otherwise. +// +// - REAP. A connection whose recv is armed is not handed off. The worker +// submits a REPORTED cancel of exactly that recv (matched on its +// user_data, generation included, so it cannot touch the fd number's next +// owner) under a tag of its own, and hands off at the recv's -ECANCELED — +// the recv's own terminal completion, not the cancel's result. If the +// request arrives first, the recv completes with it: it is served here +// and its response is HELD. If the cancel reports that it matched +// nothing (the recv completed first, or is still linked behind its SEND +// and not issued), the reap is retried on the next loop iteration while +// the recv is still armed. A miss is never followed by a hand-off. +// - HOLD. While a drain is set, the response of a connection the hand-off +// would accept is flushed with no recv behind it (not linked, not +// standalone). Its SEND completion then finds nothing in flight and hands +// the conn off; the client's next request waits in the socket buffer for +// epoll. releaseHold arms the recv at that completion whenever the +// hand-off does not happen, and checkTimeouts rescues (and counts) any +// held conn a path left unreleased. + +// onlyRecvInFlight reports whether the one op the R0 gate found in flight is +// the recv — the case REAP can clear. Anything else (a send, a SEND_ZC +// notification, an accounting the recv alone does not explain) refuses the +// hand-off until the conn's next completion. +func onlyRecvInFlight(cs *connState) bool { + return cs.recvArmed && cs.kernelInflight == 1 && !cs.zcNotifPending +} + +// startReap submits a reported cancel of cs's armed recv, unless a reap is +// already aimed at that recv. Worker thread only. A full SQ ring defers it to +// the next iteration's retry. +func (w *Worker) startReap(cs *connState) { + if cs.transplantReap > 0 && !cs.reapStale { + return // the armed recv already has a reap on its way + } + sqe := w.getCancelSQE() + if sqe == nil { + w.queueReapRetry(cs) + return + } + prepCancelUserDataReported(sqe, encodeUserDataGen(udRecv, cs.fd, cs.generation)) + setSQEUserData(sqe, encodeUserDataGen(udTransplantReap, cs.fd, cs.generation)) + cs.transplantReap++ + cs.reapStale = false + w.handoffLoss.noteReap() +} + +// reapOutcome sees every recv completion of a conn with a reap outstanding +// (transplantReap > 0), first thing in handleRecv. It consumes the recv's +// -ECANCELED, which is the reap landing: the recv is gone and read nothing, +// so the hand-off runs again (the generic negative-result branch would close +// a healthy conn). A pause's cancel has precedence: a detached conn is never +// reaped, so both outstanding at once is only defensive. Any other terminal +// completion means the recv the reaps were aimed at completed by itself — a +// request, a FIN, an error — and is handled as usual; the reaps still counted +// now target a recv that is gone. +func (w *Worker) reapOutcome(c *completionEntry, fd int, cs *connState) bool { + if c.Res == -int32(unix.ECANCELED) && !cs.recvPaused && cs.recvCancelPending == 0 { + cs.transplantReap-- + cs.reapStale = true + w.rerunHandOff(fd, cs) + // Not handed off (the drain stopped, a gate refused, the dup failed): + // the conn stays and is served here, so it needs its recv back. + if w.conns[fd] == cs && !cs.closing && !cs.recvArmed && !cs.recvPaused && !cs.transplantHold { + if !w.prepareRecv(cs, w.pickRecvTarget(cs)) { + cs.needsRecv = true + w.markDirty(cs) + } + } + return true + } + if !cqeHasMore(c.Flags) { + cs.reapStale = true + } + return false +} + +// handleTransplantReap processes a reap cancel's own completion. With +// IORING_ASYNC_CANCEL_ALL, res > 0 is a hit whose -ECANCELED reapOutcome +// retires, and -EALREADY means the recv was found executing and completes on +// its own, which is treated the same way (as handleRecvCancel treats it). +// res == 0 or -ENOENT is the miss: the only event that can report it. +func (w *Worker) handleTransplantReap(c *completionEntry, fd int) { + if fd < 0 || fd >= len(w.conns) { + return + } + cs := w.conns[fd] + if cs == nil { + return + } + if c.Res > 0 || c.Res == -int32(unix.EALREADY) { + return + } + w.handoffLoss.noteReapMiss() + if cs.transplantReap > 0 { + cs.transplantReap-- + } + if cs.transplantReap > 0 { + return // another reap is still out; its own outcome decides + } + cs.reapStale = false + if cs.recvArmed && !cs.closing { + w.queueReapRetry(cs) + } +} + +// rerunHandOff re-runs the hand-off for cs at the site that owns it: a +// promoted async conn (its dispatch goroutine has exited) through +// finishAsyncTransplant, anything else through tryTransplant. A promoted conn +// whose goroutine is running again owns itself and claims its own hand-off at +// its next park. +func (w *Worker) rerunHandOff(fd int, cs *connState) { + if w.async && cs.asyncPromoted.Load() { + cs.asyncInMu.Lock() + running := cs.asyncRun + cs.asyncInMu.Unlock() + if !running { + w.finishAsyncTransplant(cs) + } + return + } + w.tryTransplant(fd) +} + +// queueReapRetry defers a reap to the next loop iteration. Keyed by (fd, +// generation), so a conn that closes in between, or whose number is reused, +// is skipped rather than touched. +func (w *Worker) queueReapRetry(cs *connState) { + w.reapRetry = append(w.reapRetry, encodeConnOpKey(cs.fd, cs.generation)) +} + +// retryReaps re-runs the hand-off for every conn whose reap missed, or could +// not be placed, while its recv was still armed. Runs once per loop +// iteration, at the head of drainDetachQueue: after the io_uring_enter that +// processed the miss, so a linked recv that was not issued yet has been by +// now, and the new cancel finds it. +func (w *Worker) retryReaps() { + keys := w.reapRetry + w.reapRetry = w.reapRetrySpare[:0] + for _, k := range keys { + fd := int(k & fdMask) + if fd >= len(w.conns) { + continue + } + cs := w.conns[fd] + if cs == nil || cs.generation != uint32((k&genMask)>>genShift) || cs.closing || !cs.recvArmed { + continue + } + if w.transplant.Load() == nil { + continue // the drain stopped: the recv simply stays armed + } + w.rerunHandOff(fd, cs) + } + w.reapRetrySpare = keys[:0] +} diff --git a/engine/iouring/handoff_loss_test.go b/engine/iouring/handoff_loss_test.go index c68587f5..4af5deae 100644 --- a/engine/iouring/handoff_loss_test.go +++ b/engine/iouring/handoff_loss_test.go @@ -16,6 +16,13 @@ import ( // is waiting on. Its CQE is stale, so staleConnCQE dropped it with no trace // while the #624 ledger balanced. These tests pin the witnesses one counter // at a time; each fails if its increment is removed. +// +// Since the fd-lifetime rule (celeris#657 PR-2) both hand-off sites refuse a +// conn with anything in flight and reap its recv instead, so the loss these +// witnesses count can only come back through a regression of that gate. The +// tests below therefore check that each site refuses the state, then drive +// handOff — the commit point both sites share, which the gate guards — with +// it, to prove the witnesses still see what such a regression would do. // handoffLossCounts reads the four witnesses in one go, for messages. type handoffLossCounts struct { @@ -59,26 +66,30 @@ func staleRecv(fd int, gen uint32, res int32) *completionEntry { return &completionEntry{UserData: encodeUserDataGen(udRecv, fd, gen), Res: res} } -// TestStaleRecvDataCountsATransplantedConn drives the whole sync path: -// tryTransplant hands off a conn with its recv armed, and that recv then -// completes with a request's bytes. The CQE is stale (the slot is empty), -// its identity was registered by the hand-off, so it must count as -// Transplanted — and as nothing else. The CQE is terminal and the identity -// owes exactly one op, so noteStaleTerminalOp retires the identity on this -// very CQE: the count must be taken before that, or it reads Unattributed. +// TestStaleRecvDataCountsATransplantedConn drives the sync hand-off of a +// conn with its recv armed, and that recv then completes with a request's +// bytes. The CQE is stale (the slot is empty), its identity was registered by +// the hand-off, so it must count as Transplanted — and as nothing else. The +// CQE is terminal and the identity owes exactly one op, so +// noteStaleTerminalOp retires the identity on this very CQE: the count must +// be taken before that, or it reads Unattributed. func TestStaleRecvDataCountsATransplantedConn(t *testing.T) { fd, other := socketPairFDs(t) defer func() { _ = unix.Close(other) }() w, tgt := newHandoffLossWorker(t, fd) const gen = 5 - w.conns[fd] = idleSyncConn(fd, gen) + cs := idleSyncConn(fd, gen) + w.conns[fd] = cs w.connCount = 1 w.activeConns.Add(1) w.tryTransplant(fd) - if w.conns[fd] != nil || tgt.adopted.Load() != 1 { - t.Fatal("the conn was not handed off — the eligibility gates rejected " + - "the setup, so this test proves nothing about the counter") + if w.conns[fd] != cs || tgt.adopted.Load() != 0 { + t.Fatal("tryTransplant handed off a conn with its recv armed: the R0 gate is gone") + } + if !w.handOff(cs, fd, w.transplant.Load(), false) || w.conns[fd] != nil || tgt.adopted.Load() != 1 { + t.Fatal("the conn was not handed off — the setup was rejected, so this test " + + "proves nothing about the counter") } c := staleRecv(fd, gen, 27) // the next request, read by the old recv @@ -98,7 +109,7 @@ func TestStaleRecvDataCountsATransplantedConn(t *testing.T) { // TestStaleRecvDataCountsAnAsyncTransplantedConn is the same loss on the // self-initiated path a promoted async conn takes (finishAsyncTransplant), -// which registers its identity through its own call site. +// which commits with the detached release. func TestStaleRecvDataCountsAnAsyncTransplantedConn(t *testing.T) { fd, other := socketPairFDs(t) defer func() { _ = unix.Close(other) }() @@ -110,7 +121,10 @@ func TestStaleRecvDataCountsAnAsyncTransplantedConn(t *testing.T) { w.activeConns.Add(1) w.finishAsyncTransplant(cs) - if w.conns[fd] != nil || tgt.adopted.Load() != 1 { + if w.conns[fd] != cs || tgt.adopted.Load() != 0 { + t.Fatal("finishAsyncTransplant handed off a conn with its recv armed: the R0 gate is gone") + } + if !w.handOff(cs, fd, w.transplant.Load(), true) || w.conns[fd] != nil || tgt.adopted.Load() != 1 { t.Fatal("the async conn was not handed off — setup rejected, the counter is unproven") } @@ -241,7 +255,10 @@ func TestStaleRecvDataCountsEachMultishotCompletion(t *testing.T) { // TestTransplantHandoffInFlightCountsTryTransplant pins the precondition // witness on the sync hand-off site: it counts a hand-off made with a recv // armed, a kernel op outstanding, or a SEND_ZC notification pending, and -// does not count one made with nothing in flight. +// does not count one made with nothing in flight. tryTransplant must refuse +// each of the three states (celeris#657 PR-2) and hand off the empty one; the +// three are then committed through handOff, the point a gate regression +// would reach. // // Each of the three single-term states differs from "nothing in flight" in // exactly one term, so removing any one term from the predicate fails this @@ -280,6 +297,13 @@ func TestTransplantHandoffInFlightCountsTryTransplant(t *testing.T) { w.activeConns.Add(1) w.tryTransplant(fd) + inFlight := st.recvArmed || st.kernelInflight != 0 || st.zcNotifPending + if inFlight { + if w.conns[fd] != cs || tgt.adopted.Load() != int64(i) { + t.Fatalf("%s: tryTransplant handed off with an op in flight: the R0 gate is gone", st.name) + } + w.handOff(cs, fd, w.transplant.Load(), false) + } if w.conns[fd] != nil || tgt.adopted.Load() != int64(i+1) { t.Fatalf("%s: the conn was not handed off — the counter is unproven", st.name) } @@ -291,9 +315,10 @@ func TestTransplantHandoffInFlightCountsTryTransplant(t *testing.T) { } // TestTransplantHandoffInFlightCountsFinishAsyncTransplant is the same -// witness on the async hand-off site. finishAsyncTransplant already refuses a -// conn with a SEND or a SEND_ZC notification outstanding, so what it can hand -// off with an op in flight is the armed recv. +// witness on the async hand-off site. finishAsyncTransplant already refused a +// conn with a SEND or a SEND_ZC notification outstanding; since celeris#657 +// PR-2 it refuses the armed recv too, so the armed state is committed through +// handOff with the detached release, as a gate regression would. func TestTransplantHandoffInFlightCountsFinishAsyncTransplant(t *testing.T) { fdIdle, otherIdle := socketPairFDs(t) defer func() { _ = unix.Close(otherIdle) }() @@ -319,6 +344,10 @@ func TestTransplantHandoffInFlightCountsFinishAsyncTransplant(t *testing.T) { w.connCount++ w.activeConns.Add(1) w.finishAsyncTransplant(armed) + if w.conns[fdArmed] != armed || tgt.adopted.Load() != 1 { + t.Fatal("finishAsyncTransplant handed off with the recv armed: the R0 gate is gone") + } + w.handOff(armed, fdArmed, w.transplant.Load(), true) if w.conns[fdArmed] != nil || tgt.adopted.Load() != 2 { t.Fatal("the armed async conn was not handed off — setup rejected") } diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 22faa889..a4fd0052 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -43,11 +43,13 @@ func (e *Engine) StopTransplant() { } // tryTransplant detaches an idle H1 conn from this worker and hands it to the -// epoll target when it is at a clean, fully-flushed request boundary. Runs on the -// worker thread (after handleRecv/handleSend). Mirrors hijackConn's detach (cancel -// the armed recv, defer the connState release to its terminal CQE) but hands a -// freshly dup'd non-blocking fd to epoll instead of wrapping it in a net.Conn. -// No-op for ineligible conns; retried at their next boundary. +// epoll target when it is at a clean, fully-flushed request boundary with +// nothing in flight on its descriptor (celeris#657). Runs on the worker thread +// (after handleRecv/handleSend, and when a reap lands). A conn whose recv is +// armed is not moved: its recv is reaped first (fd_lifetime.go). Otherwise it +// mirrors hijackConn's detach but hands a freshly dup'd non-blocking fd to +// epoll instead of wrapping it in a net.Conn (handOff). No-op for ineligible +// conns; retried at their next boundary. // // This worker-side path handles SYNC conns (and async conns that never promoted, // asyncRun==false — they run inline on the worker, so the worker owns their state). @@ -99,6 +101,16 @@ func (w *Worker) tryTransplant(fd int) { if !cs.h1State.AtRequestBoundary() || cs.h1State.HasPendingData() { return } + // The fd-lifetime rule (celeris#657, R0): nothing that can still resolve + // this descriptor may be in flight when it moves. The usual case is the + // recv armed after the last response (its linked RECV, or the idle + // conn's standalone one): reap it and hand off at its -ECANCELED. + if cs.recvArmed || cs.kernelInflight != 0 || cs.zcNotifPending { + if onlyRecvInFlight(cs) { + w.startReap(cs) + } + return + } w.handOff(cs, fd, h, false) } @@ -227,12 +239,13 @@ func (w *Worker) enqueueDetach(cs *connState) { } // finishAsyncTransplant completes a self-initiated async transplant on the WORKER -// thread (from drainDetachQueue), after the dispatch goroutine has marked the conn -// (transplantPending) and exited. It dups the fd for epoll, runs the worker-owned -// detach (cancel the armed recv, defer the connState release to its terminal CQE), -// closes the original, and hands the dup to epoll. If the drain was stopped in the -// meantime, or the dup fails, it leaves the conn in place — its next recv respawns -// the dispatch goroutine (asyncRun is already false), so nothing is lost. +// thread (from drainDetachQueue, or when its reap lands), after the dispatch +// goroutine has marked the conn (transplantPending) and exited. The conn's recv +// is reaped first (celeris#657); with nothing in flight, handOff dups the fd for +// epoll, runs the worker-owned detach, closes the original, and hands the dup +// over. If the drain was stopped in the meantime, or the dup fails, it leaves +// the conn in place — its next recv respawns the dispatch goroutine (asyncRun is +// already false), so nothing is lost. func (w *Worker) finishAsyncTransplant(cs *connState) { h := w.transplant.Load() if h == nil { @@ -264,5 +277,17 @@ func (w *Worker) finishAsyncTransplant(cs *connState) { if cs.sending || cs.zcNotifPending || len(cs.sendBuf) != 0 || len(cs.writeBuf) != 0 { return } + // The fd-lifetime rule (celeris#657, R0), as in tryTransplant. This used + // to cancel, dup and close with the recv still armed; a request that + // recv took was lost (measured on main at an idle revert). The goroutine + // has exited, so the recv the feed path armed for it is the one op the + // conn can have: reap it, and come back here at its -ECANCELED. A + // request that beats the cancel respawns the goroutine as usual. + if cs.recvArmed || cs.kernelInflight != 0 || cs.zcNotifPending { + if onlyRecvInFlight(cs) { + w.startReap(cs) + } + return + } w.handOff(cs, cs.fd, h, true) } diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index 1ef6bbcb..de408ed3 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -482,6 +482,13 @@ type Worker struct { // this only has to outlive that. shutdownDriverHold []*driverConn + // reapRetry holds the (fd, generation) identities whose hand-off reap + // missed, or found the SQ ring full, while their recv was still armed; + // retryReaps re-runs them once per loop iteration (celeris#657). + // reapRetrySpare is the second buffer of the swap. Worker thread only. + reapRetry []uint64 + reapRetrySpare []uint64 + // transplant (#383 reverse) is non-nil while a drain-to-epoll is in progress. // Set by Engine.StartTransplant (controller goroutine), read on this worker's // own thread after each handleRecv; when set, an idle H1 conn at a clean @@ -1222,6 +1229,12 @@ func (w *Worker) run(ctx context.Context) { if !w.staleConnCQE(entry, fd, ud) { w.handleRecvCancel(entry, fd) } + case udTransplantReap: + // And for the hand-off's reap: its miss is the only + // event that says the recv is still armed (celeris#657). + if !w.staleConnCQE(entry, fd, ud) { + w.handleTransplantReap(entry, fd) + } case udDriverRecv: w.handleDriverRecv(entry, fd) case udDriverSend: @@ -1590,6 +1603,11 @@ func (w *Worker) processCQE(ctx context.Context, c *completionEntry, now int64) return } w.handleRecvCancel(c, fd) + case udTransplantReap: + if w.staleConnCQE(c, fd, ud) { + return + } + w.handleTransplantReap(c, fd) case udDriverRecv: w.handleDriverRecv(c, fd) case udDriverSend: @@ -2413,6 +2431,12 @@ func (w *Worker) handleRecv(c *completionEntry, fd int, now int64) { } return } + // A hand-off reap is outstanding (celeris#657): its -ECANCELED is the + // hand-off point and must never reach the generic negative-result + // branch below, which would close a healthy conn. + if cs.transplantReap > 0 && w.reapOutcome(c, fd, cs) { + return + } if c.Res <= 0 { if cqeHasBuffer(c.Flags) && w.bufRing != nil { @@ -4434,6 +4458,11 @@ func (w *Worker) pickRecvTarget(cs *connState) []byte { } func (w *Worker) drainDetachQueue() { + // The other deferred hand-off work first: reaps that missed, or found + // the SQ ring full, while their recv was still armed (celeris#657). + if len(w.reapRetry) > 0 { + w.retryReaps() + } if w.detachQPending.Load() == 0 { return } From c9b4a66adddb631ca6b3682dbca8d5c4443eabce Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:47:05 +0200 Subject: [PATCH 05/34] fix(iouring): hold the next recv of a response served while a drain is set (celeris#657) P3 (HOLD). While a drain is set and there is no provided-buffer ring, the response of a connection the hand-off would accept is flushed unlinked and its next recv is not armed at all. That SEND's completion then finds nothing in flight and hands the connection off; the client's next request waits in the socket buffer for epoll. Every recv-arming tail after a response now goes through one helper, respondAndArm: the sync tail (flushSendLink, or flushSend on a buffer ring) and the direct-body tail (flushSend, then a standalone recv). With no drain set it places exactly the SQEs the two tails placed before (TestNoDrainSQESequenceIsUnchanged); the added cost per request is one atomic pointer load. Held connections the hand-off does not take are released by the next commit. Refs #657 --- engine/iouring/fd_lifetime.go | 31 ++++++++++++++++++- engine/iouring/worker.go | 57 ++++++++++++++++++++++------------- 2 files changed, 66 insertions(+), 22 deletions(-) diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go index 720f482d..7904a021 100644 --- a/engine/iouring/fd_lifetime.go +++ b/engine/iouring/fd_lifetime.go @@ -2,7 +2,11 @@ package iouring -import "golang.org/x/sys/unix" +import ( + "golang.org/x/sys/unix" + + "github.com/goceleris/celeris/engine" +) // The fd-lifetime rule of the io_uring→epoll hand-off (celeris#657, face 2): // a connection leaves io_uring only when no read can still resolve its @@ -171,3 +175,28 @@ func (w *Worker) retryReaps() { } w.reapRetrySpare = keys[:0] } + +// holdEligible reports whether cs's response, about to be flushed, can be +// sent with its next recv HELD: exactly the conns tryTransplant would hand off +// at that SEND's completion (plain HTTP/1 at a clean boundary, not detached, +// not H2, not a promoted async conn, nothing pending) and only when a SEND is +// coming, since that completion is what releases the hold. Worker thread, +// under cs.detachMu when the conn has one. +func (w *Worker) holdEligible(cs *connState) bool { + if cs.fixedFile || cs.recvPaused || cs.closing { + return false + } + if !cs.detected || cs.h1State == nil || engine.Protocol(cs.protocol.Load()) != engine.HTTP1 { + return false + } + if cs.h1State.Detached.Load() || cs.h2State != nil || cs.asyncH2Promoted.Load() { + return false + } + if w.async && cs.asyncPromoted.Load() { + return false + } + if !cs.h1State.AtRequestBoundary() || cs.h1State.HasPendingData() { + return false + } + return cs.sending || len(cs.writeBuf) > 0 || len(cs.sendBuf) > 0 || len(cs.bodyBuf) > 0 +} diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index de408ed3..8235f93c 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -2566,21 +2566,10 @@ func (w *Worker) handleRecv(c *completionEntry, fd int, now int64) { return } } - if mu := cs.detachMu; mu != nil { - mu.Lock() - } - if w.flushSend(cs) { - w.markDirty(cs) - } - if mu := cs.detachMu; mu != nil { - mu.Unlock() - } - if !cqeHasMore(c.Flags) && !cs.recvLinked && !cs.recvPaused { - if !w.prepareRecv(cs, w.pickRecvTarget(cs)) { - cs.needsRecv = true - w.markDirty(cs) - } - } + // The direct-body tail: flushed unlinked (the next recv may target + // the body buffer again), then a standalone recv — or none, held for + // a hand-off (celeris#657). + w.respondAndArm(cs, fd, c, false, false) return } @@ -2974,28 +2963,51 @@ func (w *Worker) handleRecv(c *completionEntry, fd int, now int64) { // buffers. The linked SEND→RECV lets the kernel start RECV immediately // after SEND completes, eliminating one loop iteration per request. cs.recvLinked = false + w.respondAndArm(cs, fd, c, true, true) +} + +// respondAndArm is the one tail every H1 request path in handleRecv takes +// once the request has produced its response (celeris#657): flush the +// response, then arm the connection's next recv, chained behind the SEND +// (link, single-shot recv only; flushSendLink falls back to an unlinked send +// itself when it must) or standalone. capCheck applies the back-pressure +// close of the sync tail. +// +// With a drain set and a conn the hand-off would take, the response is +// flushed UNLINKED and no recv is armed at all (HOLD): the conn is handed off +// at that SEND's completion with nothing in the kernel, or releaseHold arms +// the recv there if it is not. With no drain set it places exactly the SQEs +// the two tails placed before they shared it +// (TestNoDrainSQESequenceIsUnchanged). +func (w *Worker) respondAndArm(cs *connState, fd int, c *completionEntry, link, capCheck bool) { // Hoist the detachMu load: the same mu is checked Lock and Unlock on - // the success path, plus once on the early-return overflow path. One - // pointer load instead of three. + // the success path, plus once on the early-return overflow path. mu := cs.detachMu if mu != nil { mu.Lock() } // Back-pressure: capture pending size inside the lock so concurrent // goroutine writes via the guarded writeFn don't race the read. - pending := len(cs.writeBuf) + len(cs.sendBuf) - if pending > cs.sendCap() { + if capCheck && len(cs.writeBuf)+len(cs.sendBuf) > cs.sendCap() { if mu != nil { mu.Unlock() } w.closeConn(fd) return } - if w.bufRing == nil { + hold := w.bufRing == nil && w.transplant.Load() != nil && w.holdEligible(cs) + switch { + case hold: + cs.transplantHold = true + w.handoffLoss.noteHeld() + if w.flushSend(cs) { + w.markDirty(cs) + } + case link && w.bufRing == nil: if w.flushSendLink(cs) { w.markDirty(cs) } - } else { + default: if w.flushSend(cs) { w.markDirty(cs) } @@ -3003,6 +3015,9 @@ func (w *Worker) handleRecv(c *completionEntry, fd int, now int64) { if mu != nil { mu.Unlock() } + if hold { + return + } // For multishot recv, CQE_F_MORE means the kernel will produce more CQEs // without needing a new SQE. Only re-arm if multishot ended. From e4376d2622bfcb5cc082b8b108c865308d96cdd6 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:47:43 +0200 Subject: [PATCH 06/34] fix(iouring): release every held connection the hand-off does not take (celeris#657) P4, the release guarantee for HOLD. A held connection has no recv armed, so if the hand-off at its SEND completion does not happen (the drain stopped, a gate refused, the dup failed) nothing would ever read its next request. - releaseHold runs after the transplant attempt at both inlined dispatch sites, unconditionally: once nothing is left to send, the hold is cleared and the recv armed. - processCQE's udRecv and udSend cases gain the transplant attempt they lacked, and the release. The listener-close harvest handles completions there. - checkTimeouts rescues a held connection found with its send done and arms its recv, counted in TransplantHoldRescued, which must stay 0. Refs #657 --- engine/iouring/fd_lifetime.go | 52 +++++++++++++++++++++++++++++++++++ engine/iouring/worker.go | 24 ++++++++++++++++ 2 files changed, 76 insertions(+) diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go index 7904a021..7c82fea7 100644 --- a/engine/iouring/fd_lifetime.go +++ b/engine/iouring/fd_lifetime.go @@ -200,3 +200,55 @@ func (w *Worker) holdEligible(cs *connState) bool { } return cs.sending || len(cs.writeBuf) > 0 || len(cs.sendBuf) > 0 || len(cs.bodyBuf) > 0 } + +// releaseHold runs after the hand-off attempt at every recv/send dispatch +// site, unconditionally: a conn held for a hand-off (transplantHold) whose +// response has been sent but that is still here — the drain stopped, a gate +// refused, the dup failed — gets its recv armed. No counter gates the call; a +// drifting one could skip a re-arm and leave a conn that cannot read. Small +// enough to inline: one slot load and one flag per recv/send completion. +func (w *Worker) releaseHold(fd int) { + if cs := w.conns[fd]; cs != nil && cs.transplantHold { + w.releaseHoldSlow(cs, false) + } +} + +// releaseHoldSlow clears cs's hold and arms its recv once nothing is left to +// send; while something is, the next SEND completion decides. rescued marks +// the timeout sweep's belt, which is counted. Worker thread. +func (w *Worker) releaseHoldSlow(cs *connState, rescued bool) { + if cs.closing { + cs.transplantHold = false + return + } + if mu := cs.detachMu; mu != nil { + // A held conn is never a promoted async one, so nothing contends + // this; TryLock only because checkTimeouts never blocks on it. + if !mu.TryLock() { + return + } + defer mu.Unlock() + } + if cs.sending || cs.zcNotifPending || len(cs.sendBuf) > 0 || len(cs.writeBuf) > 0 || len(cs.bodyBuf) > 0 { + return + } + cs.transplantHold = false + if rescued { + w.handoffLoss.noteHoldRescued() + } + if cs.recvArmed || cs.recvPaused { + return + } + if !w.prepareRecv(cs, w.pickRecvTarget(cs)) { + cs.needsRecv = true + w.markDirty(cs) + } +} + +// rescueHold is checkTimeouts' belt under releaseHold: every path that +// completes a held conn's send releases it, so finding one here with its send +// done means a path skipped the release, and the conn could not have read +// until now. Arms the recv and counts it (TransplantHoldRescued, must stay 0). +func (w *Worker) rescueHold(cs *connState) { + w.releaseHoldSlow(cs, true) +} diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index 8235f93c..11511939 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -1193,6 +1193,10 @@ func (w *Worker) run(ctx context.Context) { if w.transplant.Load() != nil { w.tryTransplant(fd) } + // Unconditionally, after the attempt: a conn held for + // a hand-off that did not happen gets its recv back + // (celeris#657). + w.releaseHold(fd) } case udSend: if !w.staleConnCQE(entry, fd, ud) { @@ -1204,6 +1208,9 @@ func (w *Worker) run(ctx context.Context) { if w.transplant.Load() != nil { w.tryTransplant(fd) } + // A held conn's SEND completion is where it either + // left (above) or gets its recv back (celeris#657). + w.releaseHold(fd) } case udAccept: w.handleAccept(ctx, entry, fd, now) @@ -1579,11 +1586,23 @@ func (w *Worker) processCQE(ctx context.Context, c *completionEntry, now int64) return } w.handleRecv(c, fd, now) + // The same hand-off attempt and hold release as the inlined + // dispatch (celeris#657). The listener-close harvest processes + // completions here, and a held conn whose SEND completion landed in + // it would otherwise be neither handed off nor re-armed. + if w.transplant.Load() != nil { + w.tryTransplant(fd) + } + w.releaseHold(fd) case udSend: if w.staleConnCQE(c, fd, ud) { return } w.handleSend(c, fd, now) + if w.transplant.Load() != nil { + w.tryTransplant(fd) + } + w.releaseHold(fd) case udClose: if w.staleConnCQE(c, fd, ud) { return @@ -5111,6 +5130,11 @@ func (w *Worker) checkTimeouts() { // reading never completes the SEND at all (celeris#498). The remaining // timeouts below are meaningless here: the handler is gone, so there is // no read to time out and no new bytes can join the queue. + // A conn held for a hand-off that neither happened nor was released + // (celeris#657): arm its recv. Counted; must stay 0. + if cs.transplantHold && !cs.closing { + w.rescueHold(cs) + } if cs.closing { if now-cs.lastActivity > closingDrainTimeoutNanos { // Everything closeConn does before deferring (detach From a3d18e190093fca639f68c9e8add2ae73a2f39ca Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:47:59 +0200 Subject: [PATCH 07/34] fix(iouring): one owner per io_uring hand-off (celeris#657) P5 (A6). A promoted async conn whose dispatch goroutine had parked and claimed its own hand-off (transplantPending, asyncRun cleared) could be moved a second time: a SEND completion landing before the worker drained the claim reached tryTransplant, which only skipped a RUNNING goroutine and moved the conn as if it were sync; drainDetachQueue's finishAsyncTransplant then dup'd cs.fd, a number the first move had closed and the process had reused. Measured on main: one identity moved twice, 20 us apart, in 1 of 167 runs, joined to the only negative gauge. - tryTransplant reads the claim with asyncRun under asyncInMu (the lock the goroutine sets both under) and refuses a claimed conn. - finishAsyncTransplant returns unless w.conns[cs.fd] == cs and the conn is not closing. Both refusals count in TransplantDoubleClaim. Refs #657 --- engine/iouring/transplant_source.go | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index a4fd0052..2670f1c9 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -76,13 +76,28 @@ func (w *Worker) tryTransplant(fd int) { } // Promoted async conn — owned by its dispatch goroutine, which self-transplants // at its own park boundary. Skip here to avoid racing it. + // + // One owner per hand-off (celeris#657, A6): a goroutine that has parked + // and claimed its own hand-off (transplantPending) has exited too, so + // asyncRun alone reads as "not running" and this path used to move the + // conn as if it were sync — then drainDetachQueue's finishAsyncTransplant + // moved it again, dup'ing a descriptor number the first move had closed + // and the process had reused (measured: one identity moved twice, 20 us + // apart, in 1 of 167 runs, the run with the only negative gauge). The + // goroutine sets the claim and clears asyncRun under asyncInMu, so both + // are read under it here. if w.async { cs.asyncInMu.Lock() running := cs.asyncRun + claimed := cs.transplantPending.Load() cs.asyncInMu.Unlock() if running { return } + if claimed { + w.handoffLoss.noteDoubleClaim() + return + } } // Eligibility. Only a plain HTTP/1 keep-alive at a clean, fully-flushed boundary @@ -251,6 +266,14 @@ func (w *Worker) finishAsyncTransplant(cs *connState) { if h == nil { return // drain stopped — leave the conn; next recv respawns its goroutine } + // One owner per hand-off (celeris#657, A6): act only for the connState + // that still owns its slot. If anything else moved or closed it since the + // goroutine's claim, cs.fd is a number some other connection may hold by + // now, and dup'ing it would hand that connection off. + if cs.fd < 0 || cs.fd >= len(w.conns) || w.conns[cs.fd] != cs || cs.closing { + w.handoffLoss.noteDoubleClaim() + return + } // Re-validate egress here, on the worker thread (celeris#529). // // asyncTransplantEligible runs on the DISPATCH goroutine and deliberately From 17ab243b57b5af121fd2c8fa25ed6625572558c4 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:48:10 +0200 Subject: [PATCH 08/34] fix(iouring): submit pending SQEs before a worker parks (celeris#657) P6 (A5). The iteration that closes or hands off a worker's last connection queues that connection's close-path cancels in the same pass that finds the worker idle, and the DRAINING->SUSPENDED park is indefinite, so the worker parked with them unsubmitted (measured 1-24 pending SQEs at parks). It now submits before taking wakeMu and setting suspended. Refs #657 --- engine/iouring/worker.go | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index 11511939..cdeb32a1 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -1446,6 +1446,16 @@ func (w *Worker) run(ctx context.Context) { if w.listenFD < 0 && w.connCount == 0 && !w.hasDriverConns.Load() && w.driverActionPending.Load() == 0 && w.detachQPending.Load() == 0 && w.acceptPaused.Load() { + // Submit what this iteration queued before parking (celeris#657, + // A5). The iteration that closes or hands off the last conn + // queues its close-path cancels (the header timer's, a send's) + // in the same pass that finds the worker idle, and the park is + // indefinite: those SQEs used to sit unsubmitted until something + // woke the worker, measured 1-24 pending at parks. Outside + // wakeMu, which is a leaf; SQPOLL submits by itself. + if !w.sqpoll && w.ring.Pending() > 0 { + _, _ = w.ring.Submit() + } w.wakeMu.Lock() if !w.acceptPaused.Load() || w.driverActionPending.Load() != 0 || w.detachQPending.Load() != 0 { From 248351831417dbdd0942af370d9577c5e9bb74ee Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:52:05 +0200 Subject: [PATCH 09/34] test(iouring): keep the worker-loop hold-release subtest off an accepting target (celeris#657) The warm-up response's SEND completion can reach the worker after the client has read the response, and so after StartTransplant. The idle conn is then reaped and handed to the target before /big is sent; the test's accepting target closed it, and the client read a reset (8 of 15 runs on 17ab243, reported with adopted=1 held=0). The target now refuses, so such a conn is reclaimed onto io_uring and /big is served and held there as the test intends; it also asserts the response really was held, and reports the hand-off counters when the read fails. Refs #657 --- engine/iouring/fd_lifetime_engine_test.go | 26 ++++++++++++++++++----- 1 file changed, 21 insertions(+), 5 deletions(-) diff --git a/engine/iouring/fd_lifetime_engine_test.go b/engine/iouring/fd_lifetime_engine_test.go index 5095824b..41f25357 100644 --- a/engine/iouring/fd_lifetime_engine_test.go +++ b/engine/iouring/fd_lifetime_engine_test.go @@ -385,6 +385,12 @@ func TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen(t *testing.T) { // deterministic by a response large enough to keep its SEND in flight until // the client reads it: the request arrives with a drain set (so the response // is held), the drain stops, and only then does the SEND complete. +// +// The target refuses every conn. The warm-up response's SEND completion can +// reach the worker after the client has read it, and so after StartTransplant: +// the idle conn is then reaped and offered to the target before /big is sent. +// Refused, it is reclaimed onto io_uring and /big is served there as intended; +// an accepting target would own (and here close) the conn instead. func testHoldReleasedInTheWorkerLoop(t *testing.T) { big := make([]byte, 3<<20) e, addr := startFDLEngine(t, bigBodyHandler{big: big}, nil) @@ -417,30 +423,40 @@ func testHoldReleasedInTheWorkerLoop(t *testing.T) { if _, err := get("/", 2*time.Second); err != nil { t.Fatalf("warm-up request: %v", err) } - tgt := &fdlTarget{} + tgt := &fdlTarget{refuse: true} e.StartTransplant(tgt) if _, err := c.Write([]byte("GET /big HTTP/1.1\r\nHost: x\r\n\r\n")); err != nil { t.Fatalf("write /big: %v", err) } time.Sleep(300 * time.Millisecond) // served and held; its SEND waits on this client + if n := metric(t, e, "TransplantHeld"); n == 0 { + t.Fatal("the /big response was not held: the test exercised nothing") + } e.StopTransplant() + state := func() string { + m := e.Metrics() + return "adopted=" + strconv.FormatInt(tgt.adopted.Load(), 10) + + " detached=" + strconv.FormatUint(m.TransplantDetached, 10) + + " refused=" + strconv.FormatUint(m.TransplantHandoffRefused, 10) + + " held=" + strconv.FormatUint(metric(t, e, "TransplantHeld"), 10) + + " closes=" + strconv.FormatUint(m.CloseCount, 10) + } _ = c.SetReadDeadline(time.Now().Add(10 * time.Second)) resp, err := http.ReadResponse(br, nil) if err != nil { - t.Fatalf("read /big: %v", err) + t.Fatalf("read /big: %v (%s)", err, state()) } n, err := io.Copy(io.Discard, resp.Body) _ = resp.Body.Close() if err != nil || int(n) != len(big) { - t.Fatalf("read /big body: %d of %d bytes, err %v", n, len(big), err) + t.Fatalf("read /big body: %d of %d bytes, err %v (%s)", n, len(big), err, state()) } if _, err := get("/", 2*time.Second); err != nil { t.Fatalf("the request after a held response whose drain stopped got %v: the conn was "+ "left with no recv armed", err) } if tgt.adopted.Load() != 0 { - t.Fatalf("the conn was handed off (%d) although its SEND was in flight until the drain stopped", - tgt.adopted.Load()) + t.Fatalf("the refusing target adopted %d conns", tgt.adopted.Load()) } if n := metric(t, e, "TransplantHoldRescued"); n != 0 { t.Fatalf("TransplantHoldRescued = %d, want 0: the release at the SEND completion did not run", n) From 4089b91829d5965de0026679ec3fb3d74b1efb99 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:52:38 +0200 Subject: [PATCH 10/34] ci: run the io_uring hand-off fd-lifetime tests with skipping forbidden, on one worker (celeris#657) The unit job runs ./engine/iouring without -v, so a skip there is invisible. The ten fd-lifetime tests run again by name with -v and CELERIS_REQUIRE_IOURING_WORKERS=1, with the witness step's exact tally (RUN and PASS counts, no SKIP line). It is also the one-worker leg: the runner's 8 MiB memlock funds a single io_uring worker, the shape the loss was measured in, and the step fails unless every engine the tests start logs workers=1. Refs #657 --- .github/workflows/ci.yml | 31 +++++++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index be32204e..6c260350 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -165,6 +165,37 @@ jobs: echo "was one renamed or removed, did one skip, or did one fail?" exit 1 fi + # celeris#657 PR-2: the fd-lifetime tests of the io_uring hand-off, run + # the same way: by name, -v, skipping forbidden, exact tally. This is + # the one-worker leg: the runner's own 8 MiB memlock funds a single + # io_uring worker, the shape in which every hand-off loss was measured + # (all connections on one ring). The engine-level tests log + # "celeris657 engine workers=N"; the tally below requires every such + # line to read workers=1, so a runner that funds more workers fails + # this step instead of silently testing another shape. + - name: celeris#657 fd-lifetime tests, one io_uring worker (skipping forbidden) + shell: bash + env: + CELERIS_REQUIRE_IOURING_WORKERS: "1" + run: | + set -o pipefail + echo "memlock (KiB): $(ulimit -l)" + names='TestTransplantNeverHandsOffArmedRecv|TestTransplantReapMissIsRetried|TestHoldReleasedWhenDrainStops|TestHoldRescuedByCheckTimeouts|TestOneOwnerPerHandoff|TestNoDrainSQESequenceIsUnchanged|TestHandoffHasNothingInFlight|TestStaleRecvDataCounted|TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen|TestWorkerParksWithNothingPending' + want=$(( $(tr '|' '\n' <<<"$names" | wc -l) )) + go test -race -count=1 -timeout=300s -v -run "^(${names})\$" \ + ./engine/iouring/ 2>&1 | tee /tmp/fdlife657.log + ran=$(grep -cE "^=== RUN (${names})\$" /tmp/fdlife657.log || true) + passed=$(grep -cE "^--- PASS: (${names}) \(" /tmp/fdlife657.log || true) + skipped=$(grep -cE '^[[:space:]]*--- SKIP' /tmp/fdlife657.log || true) + engines=$(grep -cE 'celeris657 engine workers=[0-9]+' /tmp/fdlife657.log || true) + one=$(grep -cE 'celeris657 engine workers=1$' /tmp/fdlife657.log || true) + echo "celeris#657 fd-lifetime tests: want $want, ran $ran, passed $passed, SKIP lines $skipped, engines $engines at workers=1: $one" + if [ "$ran" -ne "$want" ] || [ "$passed" -ne "$want" ] || [ "$skipped" -ne 0 ] || [ "$engines" -eq 0 ] || [ "$one" -ne "$engines" ]; then + echo "expected exactly $want top-level fd-lifetime tests to run and PASS with no SKIP line," + echo "every engine at one io_uring worker -- was one renamed or removed, did one skip or fail," + echo "or did this runner fund a different worker count?" + exit 1 + fi - name: middleware/compress run: go test -race -count=1 -timeout=120s ./... working-directory: middleware/compress From 308ffca69ee852c28cc7d0ec72d1f9b2170b413f Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 08:56:17 +0200 Subject: [PATCH 11/34] test(iouring): cover the async hand-off site under load, and log the fd-lifetime counters (celeris#657) TestHandoffHasNothingInFlight gains an async subtest: every route async, so each conn is promoted to a dispatch goroutine, claims its own hand-off at its park and leaves through finishAsyncTransplant, which must reap the recv the feed path armed. The load helper logs the new counters (held, reaps, misses, rescued, double claims) next to the witnesses. Refs #657 --- engine/iouring/fd_lifetime_engine_test.go | 97 ++++++++++++++++------- 1 file changed, 70 insertions(+), 27 deletions(-) diff --git a/engine/iouring/fd_lifetime_engine_test.go b/engine/iouring/fd_lifetime_engine_test.go index 41f25357..9f45adfb 100644 --- a/engine/iouring/fd_lifetime_engine_test.go +++ b/engine/iouring/fd_lifetime_engine_test.go @@ -11,6 +11,7 @@ import ( "net" "net/http" "os" + "reflect" "sort" "strconv" "strings" @@ -259,47 +260,89 @@ func transplantUnderLoad(t *testing.T, e *Engine, addr string, n int, tgt engine res := wait() e.StopTransplant() time.Sleep(100 * time.Millisecond) - t.Logf("celeris657 load conns=%d ok=%d errs=%d classes=[%s] W1T=%d W1U=%d W1C=%d W2=%d", + t.Logf("celeris657 load conns=%d ok=%d errs=%d classes=[%s] W1T=%d W1U=%d W1C=%d W2=%d "+ + "held=%d reaps=%d misses=%d rescued=%d doubleclaim=%d detached=%d", n, res.ok, res.errs, classes(res.byClass), e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), e.metrics.handoffLoss.staleRecvDataUnattributed.Load(), e.metrics.handoffLoss.staleRecvDataClosed.Load(), - e.metrics.handoffLoss.handoffInFlight.Load()) + e.metrics.handoffLoss.handoffInFlight.Load(), + metricOr(e, "TransplantHeld"), metricOr(e, "TransplantReaps"), metricOr(e, "TransplantReapMisses"), + metricOr(e, "TransplantHoldRescued"), metricOr(e, "TransplantDoubleClaim"), + e.metrics.transplantDetached.Load()) return res } +// metricOr is metric for a log line: -1 when the tree has no such field. +func metricOr(e *Engine, name string) int64 { + v := reflect.ValueOf(e.Metrics()).FieldByName(name) + if !v.IsValid() { + return -1 + } + return int64(v.Uint()) +} + // TestHandoffHasNothingInFlight: at every hand-off, nothing may still be in // flight on the connection (TransplantHandoffInFlight does not move), and no // stale recv data may appear afterwards, with 128 busy keep-alives across a -// StartTransplant. Base: every hand-off is made with the linked RECV armed. +// StartTransplant. Base: every hand-off is made with the conn's recv armed. +// The sync conns leave through tryTransplant (a held response's SEND +// completion, or a reap); the async ones are promoted to a dispatch goroutine +// that claims its own hand-off at its park, and leave through +// finishAsyncTransplant, which reaps the recv the feed path armed. func TestHandoffHasNothingInFlight(t *testing.T) { - e, addr := startFDLEngine(t, transplantTestHandler{}, nil) - var maxSeen atomic.Uint64 - tgt := &servingTarget{} - tgt.onAdopt = func() { - if v := e.metrics.handoffLoss.handoffInFlight.Load(); v > maxSeen.Load() { - maxSeen.Store(v) - } - } - defer tgt.close() - const conns = 128 - res := transplantUnderLoad(t, e, addr, conns, tgt, 300*time.Millisecond, 1500*time.Millisecond) - if v := maxSeen.Load(); v != 0 { - t.Errorf("TransplantHandoffInFlight read %d at a hand-off, want 0 at every one: a conn "+ - "left with an op still able to resolve its fd", v) - } - if tr, un := e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), - e.metrics.handoffLoss.staleRecvDataUnattributed.Load(); tr != 0 || un != 0 { - t.Errorf("stale recv data after the hand-offs: Transplanted=%d Unattributed=%d, want 0/0", tr, un) - } - if res.errs != 0 { - t.Errorf("clients saw %d errors (%s), want 0", res.errs, classes(res.byClass)) - } - if got := tgt.adopted.Load(); got != conns { - t.Errorf("%d of %d busy conns were handed off, want all", got, conns) + for _, tc := range []struct { + name string + async bool + }{{"sync", false}, {"async", true}} { + t.Run(tc.name, func(t *testing.T) { + var h stream.Handler = transplantTestHandler{} + var mut func(*resource.Config) + if tc.async { + h = asyncRouteHandler{} + mut = func(c *resource.Config) { c.AsyncHandlers = true } + } + e, addr := startFDLEngine(t, h, mut) + var maxSeen atomic.Uint64 + tgt := &servingTarget{} + tgt.onAdopt = func() { + if v := e.metrics.handoffLoss.handoffInFlight.Load(); v > maxSeen.Load() { + maxSeen.Store(v) + } + } + defer tgt.close() + const conns = 128 + res := transplantUnderLoad(t, e, addr, conns, tgt, 300*time.Millisecond, 1500*time.Millisecond) + if tc.async { + if n := e.Metrics().AsyncPromotedConns; n == 0 { + t.Fatal("no conn was promoted to async dispatch: the async path was not exercised") + } + } + if v := maxSeen.Load(); v != 0 { + t.Errorf("TransplantHandoffInFlight read %d at a hand-off, want 0 at every one: a conn "+ + "left with an op still able to resolve its fd", v) + } + if tr, un := e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), + e.metrics.handoffLoss.staleRecvDataUnattributed.Load(); tr != 0 || un != 0 { + t.Errorf("stale recv data after the hand-offs: Transplanted=%d Unattributed=%d, want 0/0", tr, un) + } + if res.errs != 0 { + t.Errorf("clients saw %d errors (%s), want 0", res.errs, classes(res.byClass)) + } + if got := tgt.adopted.Load(); got != conns { + t.Errorf("%d of %d busy conns were handed off, want all", got, conns) + } + }) } } +// asyncRouteHandler is transplantTestHandler with every route async, so each +// conn is promoted to its own dispatch goroutine on its first request. +type asyncRouteHandler struct{ transplantTestHandler } + +func (asyncRouteHandler) RouteAsync(_, _ string) bool { return true } +func (asyncRouteHandler) HasAsyncRoutes() bool { return true } + // TestStaleRecvDataCounted is the celeris#657 join as a test: every request a // client lost is a stale recv CQE with data that the witness counted as // Transplanted or Unattributed — and with the fix there are none of either. From 14dcffb841a1d853d25d69559777191f864e5292 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 13:56:56 +0200 Subject: [PATCH 12/34] ci: guard every celeris#657 io_uring test in one witness step, and run the revert tests on one worker (celeris#657) The unit job's celeris#657 witness step now carries all twenty-one tests: the eleven hand-off loss witnesses from PR-1 and the ten fd-lifetime tests this PR adds. It runs them by name with -v, CELERIS_REQUIRE_IOURING_WORKERS=1 and an exact tally (twenty-one RUN, twenty-one PASS, no SKIP line). Each engine-level test logs its io_uring worker count, and the step requires every such line to read workers=1 at the runner's own 8 MiB memlock, the one-ring shape in which the hand-off losses were measured. The adaptive job gains a second step for TestReverseTransplant and TestBidirectionalFlap at the runner's memlock. The job's existing step raises memlock, so until now those two tests never ran with io_uring on a single worker. The new step forbids skipping (neither test reads a REQUIRE variable, so the tally fails on any SKIP line), requires exactly two RUN and two PASS, and requires every "io_uring engine listening" line to read workers=1. It runs even when the step before it failed. --- .github/workflows/ci.yml | 103 ++++++++++++++++++++++++--------------- 1 file changed, 63 insertions(+), 40 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 6c260350..2947d440 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -127,12 +127,14 @@ jobs: go test -race -count=1 -timeout=300s $pkgs # celeris#657: the step above runs ./engine/iouring WITHOUT -v, so a skip # there prints nothing and the package still reports `ok`. That is how the - # celeris#656 leak shipped with no cover. Five of the celeris#657 hand-off - # loss witness tests can skip: four build a ring through newTestRing, and - # one builds io_uring workers. So all eleven witness tests run again here - # by name, with the PASS-count interlock the `iouring` job uses - # (celeris#664): -v, CELERIS_REQUIRE_IOURING_WORKERS=1 to turn every - # environment skip into a failure, and an exact tally. + # celeris#656 leak shipped with no cover. Many celeris#657 tests can skip: + # of the eleven hand-off loss witness tests (PR-1), four build a ring + # through newTestRing and one builds io_uring workers; of the ten + # fd-lifetime tests (PR-2), six build a ring through newTestRing and + # four (plus one subtest of the six) start a whole io_uring engine. So + # all twenty-one run again here by name, with the PASS-count interlock + # the `iouring` job uses (celeris#664): -v, CELERIS_REQUIRE_IOURING_WORKERS=1 + # to turn every environment skip into a failure, and an exact tally. # # The tally counts TOP-LEVEL results: `=== RUN Name` with nothing after # the name, and `--- PASS: Name (`. go test prints the elapsed time after @@ -145,53 +147,35 @@ jobs: # measured (CI and local runner-shape containers), not guaranteed. If a # later change makes them not fit, the require-workers env turns the # ENOMEM into a red job instead of a skip. That is intended. - - name: celeris#657 hand-off loss witness tests (skipping forbidden) + # + # That 8 MiB is also what makes this the one-worker leg of the + # fd-lifetime tests: through the Listen path it funds a single io_uring + # worker, the shape in which every hand-off loss was measured (all + # connections on one ring). Each engine-level fd-lifetime test logs + # "celeris657 engine workers=N"; the step requires at least one such + # line and every one to read workers=1, so a runner that funds more + # workers fails here instead of silently testing another shape. + - name: celeris#657 hand-off witness and fd-lifetime tests, one io_uring worker (skipping forbidden) shell: bash env: CELERIS_REQUIRE_IOURING_WORKERS: "1" run: | set -o pipefail echo "memlock (KiB): $(ulimit -l)" - names='TestStaleRecvDataCountsATransplantedConn|TestStaleRecvDataCountsAnAsyncTransplantedConn|TestStaleRecvDataCountsAClosedConn|TestStaleRecvDataCountsAnUnattributedIdentity|TestStaleRecvDataIgnoresCompletionsWithoutData|TestStaleRecvDataCountsEachMultishotCompletion|TestTransplantHandoffInFlightCountsTryTransplant|TestTransplantHandoffInFlightCountsFinishAsyncTransplant|TestMetricsCarriesTheHandoffLossWitnesses|TestWorkersShareTheHandoffLossWitnesses|TestClosedOpsEntryStaysThirtyTwoBytes' + pr1='TestStaleRecvDataCountsATransplantedConn|TestStaleRecvDataCountsAnAsyncTransplantedConn|TestStaleRecvDataCountsAClosedConn|TestStaleRecvDataCountsAnUnattributedIdentity|TestStaleRecvDataIgnoresCompletionsWithoutData|TestStaleRecvDataCountsEachMultishotCompletion|TestTransplantHandoffInFlightCountsTryTransplant|TestTransplantHandoffInFlightCountsFinishAsyncTransplant|TestMetricsCarriesTheHandoffLossWitnesses|TestWorkersShareTheHandoffLossWitnesses|TestClosedOpsEntryStaysThirtyTwoBytes' + pr2='TestTransplantNeverHandsOffArmedRecv|TestTransplantReapMissIsRetried|TestHoldReleasedWhenDrainStops|TestHoldRescuedByCheckTimeouts|TestOneOwnerPerHandoff|TestNoDrainSQESequenceIsUnchanged|TestHandoffHasNothingInFlight|TestStaleRecvDataCounted|TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen|TestWorkerParksWithNothingPending' + names="${pr1}|${pr2}" want=$(( $(tr '|' '\n' <<<"$names" | wc -l) )) go test -race -count=1 -timeout=300s -v -run "^(${names})\$" \ ./engine/iouring/ 2>&1 | tee /tmp/witness657.log ran=$(grep -cE "^=== RUN (${names})\$" /tmp/witness657.log || true) passed=$(grep -cE "^--- PASS: (${names}) \(" /tmp/witness657.log || true) skipped=$(grep -cE '^[[:space:]]*--- SKIP' /tmp/witness657.log || true) - echo "celeris#657 witness tests: want $want, ran $ran, passed $passed, SKIP lines $skipped" - if [ "$ran" -ne "$want" ] || [ "$passed" -ne "$want" ] || [ "$skipped" -ne 0 ]; then - echo "expected exactly $want top-level celeris#657 tests to run and PASS with no SKIP line --" - echo "was one renamed or removed, did one skip, or did one fail?" - exit 1 - fi - # celeris#657 PR-2: the fd-lifetime tests of the io_uring hand-off, run - # the same way: by name, -v, skipping forbidden, exact tally. This is - # the one-worker leg: the runner's own 8 MiB memlock funds a single - # io_uring worker, the shape in which every hand-off loss was measured - # (all connections on one ring). The engine-level tests log - # "celeris657 engine workers=N"; the tally below requires every such - # line to read workers=1, so a runner that funds more workers fails - # this step instead of silently testing another shape. - - name: celeris#657 fd-lifetime tests, one io_uring worker (skipping forbidden) - shell: bash - env: - CELERIS_REQUIRE_IOURING_WORKERS: "1" - run: | - set -o pipefail - echo "memlock (KiB): $(ulimit -l)" - names='TestTransplantNeverHandsOffArmedRecv|TestTransplantReapMissIsRetried|TestHoldReleasedWhenDrainStops|TestHoldRescuedByCheckTimeouts|TestOneOwnerPerHandoff|TestNoDrainSQESequenceIsUnchanged|TestHandoffHasNothingInFlight|TestStaleRecvDataCounted|TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen|TestWorkerParksWithNothingPending' - want=$(( $(tr '|' '\n' <<<"$names" | wc -l) )) - go test -race -count=1 -timeout=300s -v -run "^(${names})\$" \ - ./engine/iouring/ 2>&1 | tee /tmp/fdlife657.log - ran=$(grep -cE "^=== RUN (${names})\$" /tmp/fdlife657.log || true) - passed=$(grep -cE "^--- PASS: (${names}) \(" /tmp/fdlife657.log || true) - skipped=$(grep -cE '^[[:space:]]*--- SKIP' /tmp/fdlife657.log || true) - engines=$(grep -cE 'celeris657 engine workers=[0-9]+' /tmp/fdlife657.log || true) - one=$(grep -cE 'celeris657 engine workers=1$' /tmp/fdlife657.log || true) - echo "celeris#657 fd-lifetime tests: want $want, ran $ran, passed $passed, SKIP lines $skipped, engines $engines at workers=1: $one" + engines=$(grep -cE 'celeris657 engine workers=[0-9]+$' /tmp/witness657.log || true) + one=$(grep -cE 'celeris657 engine workers=1$' /tmp/witness657.log || true) + echo "celeris#657 witness and fd-lifetime tests: want $want, ran $ran, passed $passed, SKIP lines $skipped, engines $engines at workers=1: $one" if [ "$ran" -ne "$want" ] || [ "$passed" -ne "$want" ] || [ "$skipped" -ne 0 ] || [ "$engines" -eq 0 ] || [ "$one" -ne "$engines" ]; then - echo "expected exactly $want top-level fd-lifetime tests to run and PASS with no SKIP line," + echo "expected exactly $want top-level celeris#657 tests to run and PASS with no SKIP line, and" echo "every engine at one io_uring worker -- was one renamed or removed, did one skip or fail," echo "or did this runner fund a different worker count?" exit 1 @@ -253,6 +237,45 @@ jobs: go test -race -count=1 -timeout=600s -v \ -skip '^(TestBidirectionalFlapAsync|TestRampH1Sync|TestRampH1Async)$' \ ./adaptive/... + # celeris#657 PR-2: the one-worker leg of the two revert tests. The step + # above raises memlock, so io_uring there runs several workers and + # TestReverseTransplant / TestBidirectionalFlap never meet the shape in + # which the io_uring -> epoll hand-off lost requests: every connection + # on ONE io_uring ring. This step does not raise memlock. At the + # runner's own 8 MiB io_uring is capped to one worker (12 MiB per + # worker, engine/iouring/ring.go minMemlockPerWorker), and the step + # requires at least one "io_uring engine listening" line and every one + # to read workers=1, so a runner that funds more workers fails here + # instead of silently testing another shape. + # + # Neither test reads a CELERIS_REQUIRE_* variable: they t.Skip when + # io_uring is unavailable. The tally is what forbids that: it fails on + # ANY `--- SKIP` line, and on any top-level count other than two RUN and + # two PASS (same patterns as the unit job's celeris#657 step). It runs + # even when the step above failed, so a known flake there cannot hide + # this leg's result. + - name: Adaptive — one-worker revert tests (runner memlock, skipping forbidden) + if: ${{ !cancelled() }} + shell: bash + run: | + set -o pipefail + echo "memlock (KiB): $(ulimit -l)" + names='TestReverseTransplant|TestBidirectionalFlap' + want=$(( $(tr '|' '\n' <<<"$names" | wc -l) )) + go test -race -count=1 -timeout=300s -v -run "^(${names})\$" \ + ./adaptive/ 2>&1 | tee /tmp/revert657.log + ran=$(grep -cE "^=== RUN (${names})\$" /tmp/revert657.log || true) + passed=$(grep -cE "^--- PASS: (${names}) \(" /tmp/revert657.log || true) + skipped=$(grep -cE '^[[:space:]]*--- SKIP' /tmp/revert657.log || true) + engines=$(grep -cE 'io_uring engine listening .*workers=[0-9]+' /tmp/revert657.log || true) + one=$(grep -cE 'io_uring engine listening .*workers=1( |$)' /tmp/revert657.log || true) + echo "celeris#657 one-worker revert tests: want $want, ran $ran, passed $passed, SKIP lines $skipped, io_uring engines $engines at workers=1: $one" + if [ "$ran" -ne "$want" ] || [ "$passed" -ne "$want" ] || [ "$skipped" -ne 0 ] || [ "$engines" -eq 0 ] || [ "$one" -ne "$engines" ]; then + echo "expected exactly $want top-level revert tests to run and PASS with no SKIP line, and" + echo "every io_uring engine at one worker -- was one renamed or removed, did one skip or fail," + echo "or did this runner fund a different worker count?" + exit 1 + fi iouring: name: io_uring init-failure regression (./engine/iouring) From 667d4d39d4b3870ca01c3a87ca97093009e5c2cb Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 15:40:54 +0200 Subject: [PATCH 13/34] fix(iouring): make the hand-off reap safe without cancel flags, and split the double-claim counter (celeris#657) Round 2 of #681 (PR 2 of 3 for #657). Review findings C1-C4, C6 and the m15 survivor. C1. The reap cancels an armed recv with IORING_ASYNC_CANCEL_ALL, and cancel flags exist only from Linux 5.19. Through 5.18 the kernel fails the cancel with -EINVAL and leaves the recv armed (measured on 5.15.0-191), and handleTransplantReap read -EINVAL as a miss and requeued it: one failing cancel per loop iteration, per connection, for as long as the drain lasted. - New probeAsyncCancelFlags (cached, like the other runtime probes) submits exactly the reap's SQE form against a user_data nothing carries: res >= 0 or -ENOENT means accepted, -EINVAL rejected. New stores the answer, logs it (async_cancel_flags), and createWorkers copies it to every worker. - startReap places no reap where the flags are rejected and queues no retry. The connection keeps its recv until the recv completes on its own; a sync connection is then HELD at its next response and leaves at that SEND's completion with nothing in flight. Counted in the new TransplantReapUnsupported (a rate, 0 from 5.19). - handleTransplantReap retries only a miss (0, -ENOENT). Any other non-hit result is counted in the new TransplantReapFailed (must stay 0) and is neither retried nor followed by a hand-off. The other cancel sites have the same pre-5.19 problem on main; they are not changed here. C2. A hand-off that fails at its dup (EMFILE) left the conn in place, and the recv re-armed for it was reaped again at once: a RECV, a cancel and two completions per iteration while the failure lasted. handOff now sets reapSuppressed on a dup or SetNonblock failure; startReap places nothing while it is set; handleRecv clears it when data arrives. The dup goes through a Worker.dupFD test seam (nil = unix.Dup). C3. TransplantDoubleClaim counted two different things. tryTransplant finding a conn whose dispatch goroutine had claimed its own hand-off is ordering (a completion landed between the park and the drain of the claim); it is now TransplantClaimDeferred, a rate. finishAsyncTransplant counts TransplantDoubleClaim only when the connState no longer owns its slot, which in async mode means another hand-off of the same conn; a closing conn and the range check are plain refusals. So a release gate of TransplantDoubleClaim == 0 now means what it says. C4. reapOutcome consumes the reaped recv's last completion, so it now does what handleRecv does at any recv's last completion: it returns a flagged provided buffer to the ring and clears recvLinked. C6. The A5 comment said SQPOLL "submits by itself"; an idle SQ thread needs the NEED_WAKEUP kick. Comment corrected; no tier enables SQPOLL. m15 (removing releaseHold after the inlined udRecv dispatch) survived round 1's mutation run. It is an equivalent mutant, so the call is removed, together with the same call in processCQE's udRecv case, which has the same status (round 1's m12 was killed only through the send site). Proof, from code at 86d46ac: - transplantHold is set in one place, respondAndArm, only with no provided-buffer ring (single-shot: the recv that carried the request has just completed, nothing is armed), and only when holdEligible's last clause holds (cs.sending or bytes pending). flushSend never completes a send synchronously. - Every recv-arming path clears the hold first (releaseHoldSlow, rescueHold) or refuses a held conn (reapOutcome), and detached, paused, H2 and promoted async conns are never held. - So a held conn has no recv armed, and the only recv completion that reaches a recv-site release with the hold set is the one whose own handleRecv just set it, with its SEND pending: releaseHoldSlow returns at its egress check without changing anything. - A closing conn only has the flag cleared, which nothing observes (checkTimeouts skips closing conns; releaseConnState resets it). Measured (86d46ac plus a counting overlay, -race, 8 MiB, one worker): the recv-inline site was entered 60,559 times with the hold set and changed nothing, processCQE's recv site 6 times, nothing; in the same runs the send-inline site released 86 of 90 entries (the positive control). The send-site releases and the checkTimeouts rescue stay. Tests (engine/iouring): TestTransplantReapFailureIsNotRetried, TestNoReapWithoutAsyncCancelFlags, TestReapSuppressedAfterFailedHandOff, TestReapedRecvLeavesNoLinkOrBuffer, TestAsyncCancelProbeClassifies, TestAsyncCancelProbeOnThisKernel, TestWorkersCarryTheAsyncCancelProbe and TestReapOnTheRunningKernel; TestOneOwnerPerHandoff gains the split and a closing-conn case; kernel_cancels_the_armed_recv also checks a kernel that rejects the flags. The new EngineMetrics fields are summed by the adaptive engine and pinned by both metrics tests. --- adaptive/engine.go | 5 + adaptive/handoff_loss_metrics_test.go | 5 + engine/engine.go | 43 ++- engine/iouring/async_cancel_probe_test.go | 215 +++++++++++++ engine/iouring/conn.go | 11 +- engine/iouring/engine.go | 30 +- engine/iouring/fd_lifetime.go | 79 ++++- engine/iouring/fd_lifetime_fixture_test.go | 17 + engine/iouring/fd_lifetime_test.go | 330 +++++++++++++++++++- engine/iouring/handoff_loss.go | 49 ++- engine/iouring/handoff_loss_metrics_test.go | 6 + engine/iouring/probe.go | 94 ++++++ engine/iouring/transplant_source.go | 33 +- engine/iouring/worker.go | 37 ++- 14 files changed, 896 insertions(+), 58 deletions(-) create mode 100644 engine/iouring/async_cancel_probe_test.go diff --git a/adaptive/engine.go b/adaptive/engine.go index 2360890f..8455997a 100644 --- a/adaptive/engine.go +++ b/adaptive/engine.go @@ -1088,6 +1088,11 @@ func (e *Engine) Metrics() engine.EngineMetrics { TransplantReapMisses: pm.TransplantReapMisses + sm.TransplantReapMisses, TransplantHoldRescued: pm.TransplantHoldRescued + sm.TransplantHoldRescued, TransplantDoubleClaim: pm.TransplantDoubleClaim + sm.TransplantDoubleClaim, + // The same rule for the refusals and fallbacks around them: each + // is an event on the one sub-engine attempting the hand-off. + TransplantClaimDeferred: pm.TransplantClaimDeferred + sm.TransplantClaimDeferred, + TransplantReapFailed: pm.TransplantReapFailed + sm.TransplantReapFailed, + TransplantReapUnsupported: pm.TransplantReapUnsupported + sm.TransplantReapUnsupported, // The celeris#607 recv-stall and linked-recv ledger. io_uring-only, // so the epoll half contributes zero and a switch simply moves which diff --git a/adaptive/handoff_loss_metrics_test.go b/adaptive/handoff_loss_metrics_test.go index 57a87c40..d5f2dc7a 100644 --- a/adaptive/handoff_loss_metrics_test.go +++ b/adaptive/handoff_loss_metrics_test.go @@ -24,12 +24,14 @@ func TestMetricsSumsTheHandoffLossWitnesses(t *testing.T) { StaleRecvDataUnattributed: 3, TransplantHandoffInFlight: 4, TransplantHeld: 5, TransplantReaps: 6, TransplantReapMisses: 7, TransplantHoldRescued: 8, TransplantDoubleClaim: 9, + TransplantClaimDeferred: 12, TransplantReapFailed: 13, TransplantReapUnsupported: 14, }) e.secondary.(*mockEngine).SetMetrics(engine.EngineMetrics{ StaleRecvDataClosed: 10, StaleRecvDataTransplanted: 20, StaleRecvDataUnattributed: 30, TransplantHandoffInFlight: 40, TransplantHeld: 50, TransplantReaps: 60, TransplantReapMisses: 70, TransplantHoldRescued: 80, TransplantDoubleClaim: 90, + TransplantClaimDeferred: 120, TransplantReapFailed: 130, TransplantReapUnsupported: 140, }) m := e.Metrics() @@ -46,6 +48,9 @@ func TestMetricsSumsTheHandoffLossWitnesses(t *testing.T) { {"TransplantReapMisses", m.TransplantReapMisses, 77}, {"TransplantHoldRescued", m.TransplantHoldRescued, 88}, {"TransplantDoubleClaim", m.TransplantDoubleClaim, 99}, + {"TransplantClaimDeferred", m.TransplantClaimDeferred, 132}, + {"TransplantReapFailed", m.TransplantReapFailed, 143}, + {"TransplantReapUnsupported", m.TransplantReapUnsupported, 154}, } { if c.got != c.want { t.Errorf("Metrics().%s = %d, want %d (sum of both sub-engines)", diff --git a/engine/engine.go b/engine/engine.go index 93a56f59..b662f2bc 100644 --- a/engine/engine.go +++ b/engine/engine.go @@ -513,17 +513,40 @@ type EngineMetrics struct { //nolint:revive // user-approved name // unable to read until the sweep came by (celeris#657). io_uring-only; // on the adaptive engine the sum over both sub-engines. TransplantHoldRescued uint64 - // TransplantDoubleClaim counts io_uring hand-offs refused because another - // path already owned that connection's hand-off: the sync path finding a - // connection whose async dispatch goroutine had claimed its own hand-off, - // or the async completion finding its connection no longer owns its - // descriptor slot. Each is a connection that would otherwise have been - // handed off twice, the second time as whatever socket then held the - // descriptor number (celeris#657). The first case needs a send - // completion to land between the claim and the worker's drain of it, so - // a low non-zero rate is the check working, not a fault. io_uring-only; - // on the adaptive engine the sum over both sub-engines. + // TransplantDoubleClaim counts io_uring hand-offs refused because the + // connection had already left its descriptor slot when its async + // dispatch goroutine's claim to hand it off was acted on. In async mode + // only another hand-off of the same connection vacates the slot that way + // (a close marks the queued claim first, and a hijack is refused), so + // each count is a connection that would otherwise have been handed off + // twice, the second time as whatever socket then held the descriptor + // number (celeris#657). Must stay 0: a release gate can require it. + // io_uring-only; on the adaptive engine the sum over both sub-engines. TransplantDoubleClaim uint64 + // TransplantClaimDeferred counts io_uring hand-off attempts on the + // worker's own path that found the connection's async dispatch + // goroutine had already claimed the hand-off, and left it to that claim. + // It is ordering, not a fault: it fires whenever a completion of the + // connection lands between the goroutine's park and the worker's drain + // of the claim. A rate. io_uring-only; on the adaptive engine the sum + // over both sub-engines. + TransplantClaimDeferred uint64 + // TransplantReapFailed counts hand-off recv cancels (TransplantReaps) + // whose completion was neither a hit nor a miss, for example -EINVAL + // from a kernel that rejects the cancel flags the startup probe found + // accepted. Such a cancel is not retried and is never followed by a + // hand-off; the connection stays until its recv completes on its own. + // Must stay 0. io_uring-only; on the adaptive engine the sum over both + // sub-engines. + TransplantReapFailed uint64 + // TransplantReapUnsupported counts hand-off recv cancels not placed + // because the kernel rejects the IORING_ASYNC_CANCEL flags they need + // (Linux 5.19 added them; the io_uring engine probes for them at + // startup). The connection stays on io_uring until its armed recv + // completes on its own, and is handed off after its next response. + // A rate, 0 on every kernel from 5.19. io_uring-only; on the adaptive + // engine the sum over both sub-engines. + TransplantReapUnsupported uint64 } // FillErrorClasses copies one engine's per-cause error tally into m and diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go new file mode 100644 index 00000000..ec8b4dd2 --- /dev/null +++ b/engine/iouring/async_cancel_probe_test.go @@ -0,0 +1,215 @@ +//go:build linux + +package iouring + +import ( + "context" + "strings" + "testing" + "time" + + "golang.org/x/sys/unix" + + "github.com/goceleris/celeris/engine" + "github.com/goceleris/celeris/probe" + "github.com/goceleris/celeris/resource" +) + +// celeris#681 C1: the hand-off's REAP cancels an armed recv with +// IORING_ASYNC_CANCEL_ALL, and cancel flags exist only from Linux 5.19. +// Through 5.18 the kernel fails every such cancel with -EINVAL and leaves the +// recv armed (measured on 5.15.0-191), so the engine probes for the flags at +// startup and places no reap where they are rejected. + +// TestAsyncCancelProbeClassifies pins how the probe reads its cancel's +// completion: a cancel with CANCEL_ALL of a user_data nothing carries. +func TestAsyncCancelProbeClassifies(t *testing.T) { + for _, tc := range []struct { + res int32 + ok bool + reason string + }{ + {0, true, ""}, // CANCEL_ALL: res is the number cancelled, and nothing was + {1, true, ""}, + {-int32(unix.ENOENT), true, ""}, // the flag-free form's miss + {-int32(unix.EINVAL), false, "5.19"}, + {-int32(unix.EBADF), false, "cqe.res=-9"}, + {-int32(unix.ECANCELED), false, "cqe.res=-125"}, + } { + ok, reason := classifyAsyncCancelProbe(tc.res) + if ok != tc.ok || (tc.reason == "") != (reason == "") || !strings.Contains(reason, tc.reason) { + t.Errorf("classifyAsyncCancelProbe(%d) = (%v, %q), want (%v, containing %q)", + tc.res, ok, reason, tc.ok, tc.reason) + } + } +} + +// kernelHasCancelFlags reports whether the running kernel's version is one +// that has IORING_ASYNC_CANCEL flags (5.19 or later). +func kernelHasCancelFlags(t *testing.T) (bool, string) { + t.Helper() + var u unix.Utsname + if err := unix.Uname(&u); err != nil { + t.Fatalf("uname: %v", err) + } + rel := unix.ByteSliceToString(u.Release[:]) + kv, err := probe.ParseKernelVersion(rel) + if err != nil { + t.Fatalf("parse kernel release %q: %v", rel, err) + } + return kv.AtLeast(5, 19), rel +} + +// TestAsyncCancelProbeOnThisKernel runs the probe itself against the running +// kernel and checks its answer against the kernel's version. +func TestAsyncCancelProbeOnThisKernel(t *testing.T) { + r, err := NewRing(8, 0, 0) + if err != nil { + skipOrFail656(t, "io_uring unavailable: %v", err) + } + _ = r.Close() + ok, reason := probeAsyncCancelFlags() + want, rel := kernelHasCancelFlags(t) + t.Logf("celeris681 async cancel flags probe: kernel=%s ok=%v reason=%q", rel, ok, reason) + if ok != want { + t.Fatalf("probeAsyncCancelFlags() = (%v, %q) on kernel %s, want %v", ok, reason, rel, want) + } + if cok, _ := probeAsyncCancelFlagsCached(); cok != ok { + t.Fatalf("the cached probe says %v, the probe %v", cok, ok) + } +} + +// TestWorkersCarryTheAsyncCancelProbe: New stores the probe's answer and +// createWorkers gives every worker that answer, so a worker on a kernel +// without cancel flags never places a reap, and one on a kernel with them +// does. +func TestWorkersCarryTheAsyncCancelProbe(t *testing.T) { + e, err := New(resource.Config{ + Addr: "127.0.0.1:0", + Protocol: engine.HTTP1, + Resources: resource.Resources{Workers: 2}, + }, transplantTestHandler{}) + if err != nil { + skipOrFail656(t, "iouring engine unavailable: %v", err) + } + if want, _ := probeAsyncCancelFlagsCached(); e.asyncCancelFlags != want { + t.Fatalf("New stored asyncCancelFlags=%v, the probe says %v", e.asyncCancelFlags, want) + } + resolved := e.cfg.Resources.Resolve() + for _, answer := range []bool{false, true} { + e.asyncCancelFlags = answer + workers, err := e.createWorkers(SelectTier(e.profile, 0), make([]int, resolved.Workers), resolved) + if err != nil { + skipOrFail656(t, "cannot create io_uring workers here: %v", err) + } + for i, w := range workers { + if w.asyncCancelFlags != answer { + t.Errorf("engine answer %v: worker %d has asyncCancelFlags=%v", answer, i, w.asyncCancelFlags) + } + } + for _, w := range workers { + w.shutdown() + } + } +} + +// TestReapOnTheRunningKernel drives the hand-off of an idle conn whose recv +// is armed through the running kernel, which completes every op: with the +// probe's own answer, and with the flags taken as rejected. With the flags a +// reap cancels the recv and the conn is handed off at its -ECANCELED. Without +// them no reap is placed, on this iteration or any later one; the client's +// next request completes the recv, its response is held, and the conn is +// handed off at that SEND's completion with nothing in flight. +func TestReapOnTheRunningKernel(t *testing.T) { + // run submits and lets the kernel complete ops until pred holds, + // processing every completion the way the worker loop does. + run := func(t *testing.T, f *fdlFixture, what string, pred func() bool) { + t.Helper() + if _, err := f.w.ring.Submit(); err != nil { + t.Fatalf("%s: submit: %v", what, err) + } + for deadline := time.Now().Add(2 * time.Second); !pred(); { + if time.Now().After(deadline) { + t.Fatalf("%s: not done within 2s (adopted %d, recvArmed %v, sending %v)", + what, f.tgt.adopted.Load(), f.cs.recvArmed, f.cs.sending) + } + _ = f.w.ring.WaitCQETimeout(50 * time.Millisecond) + head, tail := f.w.ring.BeginCQ() + for ; head != tail; head++ { + c := *f.w.ring.cqeAt(head) + f.w.processCQE(context.Background(), &c, time.Now().UnixNano()) + } + f.w.ring.EndCQ(head) + if _, err := f.w.ring.Submit(); err != nil { + t.Fatalf("%s: submit: %v", what, err) + } + } + } + idleArmed := func(t *testing.T, flags bool) *fdlFixture { + t.Helper() + f := newFDLFixture(t, false) + f.w.asyncCancelFlags = flags + if !f.w.prepareRecv(f.cs, f.cs.buf) { + t.Fatal("arm refused") + } + if _, err := f.w.ring.Submit(); err != nil { + t.Fatalf("submit recv: %v", err) + } + f.startDrain() + f.w.tryTransplant(f.fd) + return f + } + withoutFlags := func(t *testing.T, f *fdlFixture) { + t.Helper() + if n := f.w.ring.Pending(); n != 0 { + t.Fatalf("tryTransplant placed %d SQE(s) without cancel flags, want none", n) + } + for i := 1; i <= 3; i++ { + f.w.drainDetachQueue() + if n := f.w.ring.Pending(); n != 0 { + t.Fatalf("loop iteration %d placed %d SQE(s) without cancel flags, want none", i, n) + } + } + if f.tgt.adopted.Load() != 0 || !f.cs.recvArmed { + t.Fatalf("adopted %d, recvArmed %v: want 0 and true", f.tgt.adopted.Load(), f.cs.recvArmed) + } + if _, err := unix.Write(f.peer, []byte(fdlGET)); err != nil { + t.Fatalf("client write: %v", err) + } + run(t, f, "the next request", func() bool { return f.tgt.adopted.Load() == 1 }) + if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) + } + if n := f.e.metrics.handoffLoss.held.Load(); n != 1 { + t.Fatalf("TransplantHeld = %d, want 1: the conn left after its held response", n) + } + if n := f.e.metrics.handoffLoss.reaps.Load(); n != 0 { + t.Fatalf("TransplantReaps = %d, want 0", n) + } + } + + t.Run("probe_answer", func(t *testing.T) { + ok, _ := probeAsyncCancelFlagsCached() + want, rel := kernelHasCancelFlags(t) + t.Logf("celeris681 kernel=%s probe=%v", rel, ok) + if ok != want { + t.Fatalf("the probe says %v on kernel %s, want %v", ok, rel, want) + } + f := idleArmed(t, ok) + if !ok { + withoutFlags(t, f) + return + } + if n := f.w.ring.Pending(); n != 1 { + t.Fatalf("tryTransplant placed %d SQE(s), want the reap", n) + } + run(t, f, "the reap", func() bool { return f.tgt.adopted.Load() == 1 }) + if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) + } + }) + + t.Run("flags_rejected", func(t *testing.T) { + withoutFlags(t, idleArmed(t, false)) + }) +} diff --git a/engine/iouring/conn.go b/engine/iouring/conn.go index c938c618..04cae5e6 100644 --- a/engine/iouring/conn.go +++ b/engine/iouring/conn.go @@ -215,10 +215,18 @@ type connState struct { // completion was expected to take the conn (HOLD). releaseHold arms the // recv there if the hand-off does not happen. // - // All three worker-thread only, like recvCancelPending. + // reapSuppressed: the conn's last hand-off failed at handOff (the dup or + // its non-blocking switch, e.g. EMFILE), so no reap is placed for it + // until it next receives data. Without it, the recv re-armed after the + // failure was reaped again at once and the hand-off failed again: a + // RECV, a cancel and two completions per loop iteration for as long as + // the failure and the drain both lasted. + // + // All four worker-thread only, like recvCancelPending. transplantReap uint16 reapStale bool transplantHold bool + reapSuppressed bool // Async handler dispatch (Worker.async=true, HTTP1 only): // Incoming recv bytes are appended under asyncInMu by the worker. @@ -468,6 +476,7 @@ func releaseConnState(cs *connState) { cs.transplantReap = 0 cs.reapStale = false cs.transplantHold = false + cs.reapSuppressed = false cs.headerTimerSpec = kernelTimespec{} cs.headerTimerArmed = false cs.forceRSTClose = false diff --git a/engine/iouring/engine.go b/engine/iouring/engine.go index 8f913b7e..d3ac50c1 100644 --- a/engine/iouring/engine.go +++ b/engine/iouring/engine.go @@ -91,6 +91,10 @@ type Engine struct { // at construction so Metrics() doesn't pay the type-assertion per // call. Snapshot-at-Listen — static post-Start. asyncRoutes int + // asyncCancelFlags is the probeAsyncCancelFlags answer: whether this + // kernel accepts IORING_ASYNC_CANCEL_* flags (5.19+). Every worker gets + // a copy; the hand-off's REAP needs them (celeris#657). + asyncCancelFlags bool } // New creates a new io_uring engine. @@ -162,6 +166,18 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { } } + // The io_uring→epoll hand-off cancels an armed recv before it moves a + // connection (REAP, celeris#657), and that cancel needs + // IORING_ASYNC_CANCEL flags, which exist from Linux 5.19. Where the + // kernel rejects them the hand-off never reaps: a connection whose recv + // is armed stays on io_uring until that recv completes on its own, and + // its next response is then HELD and handed off with nothing in flight. + asyncCancelFlags, acReason := probeAsyncCancelFlagsCached() + if !asyncCancelFlags { + cfg.Logger.Info("async cancel flags runtime probe failed: the io_uring→epoll hand-off will not cancel an armed recv (celeris#657)", + "reason", acReason) + } + tier := SelectTier(profile, 2*time.Second) if tier == nil { return nil, fmt.Errorf("no suitable io_uring tier available") @@ -177,13 +193,15 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { // was gated off (celeris#541). "fixed_files", fixedFilesEnabled(tier.SupportsFixedFiles()), "send_zc", tier.SupportsSendZC(), + "async_cancel_flags", asyncCancelFlags, ) e := &Engine{ - tier: tier, - profile: profile, - cfg: cfg, - handler: handler, + tier: tier, + profile: profile, + cfg: cfg, + handler: handler, + asyncCancelFlags: asyncCancelFlags, } // Snapshot the static AsyncRoutes count from the handler so // Metrics() doesn't pay a type-assertion per call (#300 G3). @@ -350,6 +368,7 @@ func (e *Engine) createWorkers(tier TierStrategy, cpus []int, w.transplantHandoffRefused = &e.metrics.transplantHandoffRefused w.transplantAdoptRefused = &e.metrics.transplantAdoptRefused w.handoffLoss = &e.metrics.handoffLoss // celeris#657 witnesses + w.asyncCancelFlags = e.asyncCancelFlags workers[i] = w } return workers, nil @@ -451,6 +470,9 @@ func (e *Engine) Metrics() engine.EngineMetrics { TransplantReapMisses: e.metrics.handoffLoss.reapMisses.Load(), TransplantHoldRescued: e.metrics.handoffLoss.holdRescued.Load(), TransplantDoubleClaim: e.metrics.handoffLoss.doubleClaim.Load(), + TransplantClaimDeferred: e.metrics.handoffLoss.claimDeferred.Load(), + TransplantReapFailed: e.metrics.handoffLoss.reapFailed.Load(), + TransplantReapUnsupported: e.metrics.handoffLoss.reapUnsupported.Load(), } // ErrorCount and its eleven buckets, together, from one snapshot // (celeris#645). diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go index 7c82fea7..191c43d4 100644 --- a/engine/iouring/fd_lifetime.go +++ b/engine/iouring/fd_lifetime.go @@ -31,7 +31,13 @@ import ( // and its response is HELD. If the cancel reports that it matched // nothing (the recv completed first, or is still linked behind its SEND // and not issued), the reap is retried on the next loop iteration while -// the recv is still armed. A miss is never followed by a hand-off. +// the recv is still armed. A miss is never followed by a hand-off, and +// neither is any other result: a cancel that fails outright is counted +// and not retried. A reap needs IORING_ASYNC_CANCEL flags (Linux 5.19); +// on a kernel that rejects them (probeAsyncCancelFlags) none is placed, +// and the conn stays until its recv completes on its own (startReap). +// None is placed either after a hand-off of the conn failed at its dup, +// until the conn next receives data. // - HOLD. While a drain is set, the response of a connection the hand-off // would accept is flushed with no recv behind it (not linked, not // standalone). Its SEND completion then finds nothing in flight and hands @@ -51,7 +57,26 @@ func onlyRecvInFlight(cs *connState) bool { // startReap submits a reported cancel of cs's armed recv, unless a reap is // already aimed at that recv. Worker thread only. A full SQ ring defers it to // the next iteration's retry. +// +// Two conditions place no reap and queue no retry. The conn keeps its recv +// armed and stays until that recv completes on its own: a sync conn then has +// the request it brings served here and its response HELD, and leaves at that +// SEND's completion with nothing in flight; a promoted async conn, which is +// never held, stays on io_uring (placement only: nothing is in flight when it +// moves, so no request can be lost). The conditions: +// - the kernel rejects IORING_ASYNC_CANCEL flags (before 5.19, found by +// probeAsyncCancelFlags): every reap would fail with -EINVAL and leave +// the recv armed. Counted (TransplantReapUnsupported). +// - the conn's last hand-off failed at its dup (reapSuppressed): reaping +// the recv re-armed after that failure would fail the same way at once. func (w *Worker) startReap(cs *connState) { + if !w.asyncCancelFlags { + w.handoffLoss.noteReapUnsupported() + return + } + if cs.reapSuppressed { + return + } if cs.transplantReap > 0 && !cs.reapStale { return // the armed recv already has a reap on its way } @@ -76,8 +101,19 @@ func (w *Worker) startReap(cs *connState) { // completion means the recv the reaps were aimed at completed by itself — a // request, a FIN, an error — and is handled as usual; the reaps still counted // now target a recv that is gone. +// +// Consuming the completion here means doing what handleRecv does for every +// recv completion it ends: a provided buffer the completion carries goes back +// to the ring (a cancelled recv consumed none, so this is defensive), and +// recvLinked, which only the chained recv's own completion clears, is cleared, +// since this completion is that recv's last. func (w *Worker) reapOutcome(c *completionEntry, fd int, cs *connState) bool { if c.Res == -int32(unix.ECANCELED) && !cs.recvPaused && cs.recvCancelPending == 0 { + if cqeHasBuffer(c.Flags) && w.bufRing != nil { + w.bufRing.PushBuffer(cqeBufferID(c.Flags)) + w.hasBufReturns = true + } + cs.recvLinked = false cs.transplantReap-- cs.reapStale = true w.rerunHandOff(fd, cs) @@ -101,7 +137,17 @@ func (w *Worker) reapOutcome(c *completionEntry, fd int, cs *connState) bool { // IORING_ASYNC_CANCEL_ALL, res > 0 is a hit whose -ECANCELED reapOutcome // retires, and -EALREADY means the recv was found executing and completes on // its own, which is treated the same way (as handleRecvCancel treats it). -// res == 0 or -ENOENT is the miss: the only event that can report it. +// res == 0 or -ENOENT is the miss: the only event that can report it, and +// the only one retried. +// +// Anything else is a cancel that failed, not an answer about the recv: the +// kernel matched nothing because it never looked. -EINVAL is what a kernel +// that rejects IORING_ASYNC_CANCEL flags returns (before 5.19; the startup +// probe keeps reaps off there). Retrying it placed the same failing cancel +// on every loop iteration for as long as the drain lasted. It is counted +// (TransplantReapFailed, which must stay 0), not retried, and not followed +// by a hand-off: the recv stays armed and the conn is served here until that +// recv completes on its own. func (w *Worker) handleTransplantReap(c *completionEntry, fd int) { if fd < 0 || fd >= len(w.conns) { return @@ -113,7 +159,12 @@ func (w *Worker) handleTransplantReap(c *completionEntry, fd int) { if c.Res > 0 || c.Res == -int32(unix.EALREADY) { return } - w.handoffLoss.noteReapMiss() + miss := c.Res == 0 || c.Res == -int32(unix.ENOENT) + if miss { + w.handoffLoss.noteReapMiss() + } else { + w.handoffLoss.noteReapFailed() + } if cs.transplantReap > 0 { cs.transplantReap-- } @@ -121,7 +172,7 @@ func (w *Worker) handleTransplantReap(c *completionEntry, fd int) { return // another reap is still out; its own outcome decides } cs.reapStale = false - if cs.recvArmed && !cs.closing { + if miss && cs.recvArmed && !cs.closing { w.queueReapRetry(cs) } } @@ -201,12 +252,20 @@ func (w *Worker) holdEligible(cs *connState) bool { return cs.sending || len(cs.writeBuf) > 0 || len(cs.sendBuf) > 0 || len(cs.bodyBuf) > 0 } -// releaseHold runs after the hand-off attempt at every recv/send dispatch -// site, unconditionally: a conn held for a hand-off (transplantHold) whose -// response has been sent but that is still here — the drain stopped, a gate -// refused, the dup failed — gets its recv armed. No counter gates the call; a -// drifting one could skip a re-arm and leave a conn that cannot read. Small -// enough to inline: one slot load and one flag per recv/send completion. +// releaseHold runs after the hand-off attempt at every send dispatch site +// (the inlined one and processCQE's), unconditionally: a conn held for a +// hand-off (transplantHold) whose response has been sent but that is still +// here — the drain stopped, a gate refused, the dup failed — gets its recv +// armed. No counter gates the call; a drifting one could skip a re-arm and +// leave a conn that cannot read. Small enough to inline: one slot load and +// one flag per send completion. +// +// The recv dispatch sites do not call it. A held conn has no recv armed: +// every path that arms one clears the hold first (releaseHoldSlow) or +// refuses a held conn (reapOutcome), so the only recv completion that finds +// a hold is the one whose own response just set it, with that SEND still in +// flight, and releaseHoldSlow would return at its egress check. Measured the +// same way: 60,559 recv-site entries with the hold set, none changed a thing. func (w *Worker) releaseHold(fd int) { if cs := w.conns[fd]; cs != nil && cs.transplantHold { w.releaseHoldSlow(cs, false) diff --git a/engine/iouring/fd_lifetime_fixture_test.go b/engine/iouring/fd_lifetime_fixture_test.go index 3d2923f7..9754ba4f 100644 --- a/engine/iouring/fd_lifetime_fixture_test.go +++ b/engine/iouring/fd_lifetime_fixture_test.go @@ -151,6 +151,11 @@ func newFDLFixture(t *testing.T, async bool) *fdlFixture { w.transplantDetached = &e.metrics.transplantDetached w.transplantHandoffRefused = &e.metrics.transplantHandoffRefused w.transplantAdoptRefused = &e.metrics.transplantAdoptRefused + // createWorkers copies the engine's async-cancel-flags probe answer; the + // kernel never sees this fixture's SQEs, so it reads "accepted", the + // answer of every kernel from 5.19. A tree without the probe reaps + // unconditionally, which is the same thing. + trySetWorkerField(w, "asyncCancelFlags", true) cs := acquireConnState(context.Background(), fd, 4096, async) cs.writeFn = w.makeWriteFn(cs) @@ -255,6 +260,18 @@ func metric(t *testing.T, e *Engine, name string) uint64 { return v.Uint() } +// trySetWorkerField sets w's field name to v and reports whether w has such a +// field. With metric, it lets a test pin a field a tree may not have yet: the +// test then fails on its assertions, instead of the package failing to build. +func trySetWorkerField(w *Worker, name string, v any) bool { + f := reflect.ValueOf(w).Elem().FieldByName(name) + if !f.IsValid() { + return false + } + reflect.NewAt(f.Type(), unsafe.Pointer(f.UnsafeAddr())).Elem().Set(reflect.ValueOf(v)) + return true +} + // isReap reports whether s is the hand-off's reported cancel of cs's recv: // matched on the recv's own user_data (op, fd AND generation, so it can never // cancel the next owner of the fd number), tagged with the reap tag, and diff --git a/engine/iouring/fd_lifetime_test.go b/engine/iouring/fd_lifetime_test.go index e2e26fb0..515a6fc4 100644 --- a/engine/iouring/fd_lifetime_test.go +++ b/engine/iouring/fd_lifetime_test.go @@ -170,6 +170,10 @@ func TestTransplantNeverHandsOffArmedRecv(t *testing.T) { // The kernel, not the test, completes the ops: the reap's SQE really // cancels the armed recv (so its user_data target is right), and the // hand-off follows the -ECANCELED in whichever order the two CQEs come. + // A kernel that rejects IORING_ASYNC_CANCEL flags (before 5.19) fails the + // reap with -EINVAL and leaves the recv armed; the fixture places the + // reap regardless of the startup probe, so on such a kernel this checks + // that the failure is neither retried nor followed by a hand-off. t.Run("kernel_cancels_the_armed_recv", func(t *testing.T) { f := newFDLFixture(t, false) if !f.w.prepareRecv(f.cs, f.cs.buf) { @@ -189,22 +193,50 @@ func TestTransplantNeverHandsOffArmedRecv(t *testing.T) { if _, err := f.w.ring.Submit(); err != nil { t.Fatalf("submit reap: %v", err) } - got := 0 - for deadline := time.Now().Add(2 * time.Second); got < 2 && time.Now().Before(deadline); { + var reapRes *int32 + recvCancelled := false + for deadline := time.Now().Add(2 * time.Second); time.Now().Before(deadline); { + if reapRes != nil && (recvCancelled || *reapRes == -int32(unix.EINVAL)) { + break + } _ = f.w.ring.WaitCQETimeout(100 * time.Millisecond) head, tail := f.w.ring.BeginCQ() for ; head != tail; head++ { c := *f.w.ring.cqeAt(head) - if decodeOp(c.UserData) == udRecv && c.Res != -int32(unix.ECANCELED) { - t.Fatalf("the recv completed with %d, want -ECANCELED", c.Res) + switch decodeOp(c.UserData) { + case udRecv: + if c.Res != -int32(unix.ECANCELED) { + t.Fatalf("the recv completed with %d, want -ECANCELED", c.Res) + } + recvCancelled = true + case wantReapTag: + r := c.Res + reapRes = &r } f.w.processCQE(context.Background(), &c, time.Now().UnixNano()) - got++ } f.w.ring.EndCQ(head) } - if got != 2 { - t.Fatalf("kernel produced %d completions, want the recv's -ECANCELED and the reap's own", got) + if reapRes == nil { + t.Fatal("the reap produced no completion of its own") + } + if *reapRes == -int32(unix.EINVAL) { + if recvCancelled || f.tgt.adopted.Load() != 0 { + t.Fatalf("the kernel rejected the reap (-EINVAL), yet recvCancelled=%v and %d hand-off(s)", + recvCancelled, f.tgt.adopted.Load()) + } + for i := 1; i <= 3; i++ { + f.w.drainDetachQueue() + if n := f.w.ring.Pending(); n != 0 { + t.Fatalf("loop iteration %d placed %d SQE(s) after the kernel rejected the reap: the "+ + "failed cancel is retried, one failing cancel per iteration for as long as the drain lasts", i, n) + } + } + t.Logf("celeris681 the kernel rejected the reap's cancel flags (-EINVAL): not retried, no hand-off") + return + } + if !recvCancelled { + t.Fatalf("the reap completed with %d but the recv's -ECANCELED never came", *reapRes) } if n := f.tgt.adopted.Load(); n != 1 { t.Fatalf("%d hand-offs after the kernel cancelled the recv, want 1", n) @@ -420,6 +452,9 @@ func TestOneOwnerPerHandoff(t *testing.T) { return f } + // Leaving a claimed conn to its claim is ordering, not a double claim: + // it is counted as TransplantClaimDeferred, a rate, so that + // TransplantDoubleClaim can be held at 0 (celeris#681 C3). t.Run("tryTransplant_refuses_a_claimed_conn", func(t *testing.T) { f := claimedAsync(t) f.w.tryTransplant(f.fd) @@ -427,8 +462,11 @@ func TestOneOwnerPerHandoff(t *testing.T) { t.Fatalf("tryTransplant moved a conn its dispatch goroutine had claimed (%d hand-offs); "+ "finishAsyncTransplant would then move it again", n) } - if n := metric(t, f.e, "TransplantDoubleClaim"); n != 1 { - t.Errorf("TransplantDoubleClaim = %d, want 1", n) + if n := metric(t, f.e, "TransplantDoubleClaim"); n != 0 { + t.Errorf("TransplantDoubleClaim = %d, want 0: deferring to a claim is not a double claim", n) + } + if n := metric(t, f.e, "TransplantClaimDeferred"); n != 1 { + t.Errorf("TransplantClaimDeferred = %d, want 1", n) } }) @@ -453,6 +491,22 @@ func TestOneOwnerPerHandoff(t *testing.T) { f.w.conns[f.fd] = old // let the fixture's cleanup close the fd }) + // A claimed conn that is closing (a close deferred behind a send) is + // refused like any closing conn. It still owns its slot, so this is no + // double claim, and counting it would make a gate of 0 fail on a close. + t.Run("closing_conn_is_refused_not_counted", func(t *testing.T) { + f := claimedAsync(t) + f.cs.closing = true + f.w.finishAsyncTransplant(f.cs) + f.cs.closing = false + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("finishAsyncTransplant handed off a closing conn (%d hand-offs)", n) + } + if n := metric(t, f.e, "TransplantDoubleClaim"); n != 0 { + t.Errorf("TransplantDoubleClaim = %d, want 0: a closing conn that owns its slot is no double claim", n) + } + }) + // The measured interleaving end to end: the goroutine claims the // hand-off and enqueues itself; before the worker drains the queue, a // SEND completion for the same conn reaches the udSend dispatch site. @@ -489,8 +543,11 @@ func TestOneOwnerPerHandoff(t *testing.T) { if n := f.tgt.adopted.Load(); n != 1 { t.Fatalf("the conn was handed off %d times, want exactly once", n) } - if n := metric(t, f.e, "TransplantDoubleClaim"); n != 1 { - t.Errorf("TransplantDoubleClaim = %d, want 1 (the sync claim refused)", n) + if n := metric(t, f.e, "TransplantClaimDeferred"); n != 1 { + t.Errorf("TransplantClaimDeferred = %d, want 1 (the sync attempt left the conn to its claim)", n) + } + if n := metric(t, f.e, "TransplantDoubleClaim"); n != 0 { + t.Errorf("TransplantDoubleClaim = %d, want 0: the conn was handed off once", n) } }) } @@ -555,3 +612,254 @@ func TestNoDrainSQESequenceIsUnchanged(t *testing.T) { check(t, f, takeSQEs(f.w.ring), nil) }) } + +// reapedFixture is a keep-alive conn whose response SEND completed while a +// drain was set and its linked RECV was armed: the hand-off refused it and +// placed one reap of that recv, which the test now completes as it likes. +func reapedFixture(t *testing.T) *fdlFixture { + t.Helper() + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + f.deliver(fdlGET) + _ = takeSQEs(f.w.ring) + f.startDrain() + f.process(f.sendCQE()) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("SEND completion with the linked RECV armed placed %v, want one reap", sqes) + } + return f +} + +// heldHandOff delivers the client's next request on f's armed recv and +// checks that its response is HELD (one unlinked SEND, no recv) and that the +// SEND's completion hands the conn off with nothing in flight: the way a conn +// the hand-off could not reap leaves. +func heldHandOff(t *testing.T, f *fdlFixture) { + t.Helper() + f.deliver(fdlGET) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || sqes[0].op != opSEND || sqes[0].flags&sqeIOLink != 0 { + t.Fatalf("the next request's response placed %v, want one UNLINKED SEND and no recv (HOLD)", sqes) + } + f.process(f.sendCQE()) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the held response's SEND completion made %d hand-offs, want 1", n) + } + if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) + } +} + +// TestTransplantReapFailureIsNotRetried (celeris#681 C1): a reap whose own +// completion is neither a hit nor a miss is a cancel that failed, and the +// kernel will fail the next one the same way. -EINVAL is what a kernel that +// rejects IORING_ASYNC_CANCEL flags returns (before 5.19). Retrying it placed +// a failing cancel on every loop iteration for as long as the drain lasted. +// It must be counted, not retried and not followed by a hand-off; the conn +// then leaves the way any unreaped conn does, after its next response. +func TestTransplantReapFailureIsNotRetried(t *testing.T) { + for _, tc := range []struct { + name string + res int32 + }{ + {"EINVAL", -int32(unix.EINVAL)}, + {"EBADF", -int32(unix.EBADF)}, + {"ECANCELED", -int32(unix.ECANCELED)}, + } { + t.Run(tc.name, func(t *testing.T) { + f := reapedFixture(t) + f.process(f.reapCQE(tc.res)) + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("a reap that failed with %d was followed by %d hand-off(s) with the recv armed", tc.res, n) + } + for i := 1; i <= 3; i++ { + f.w.drainDetachQueue() + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 { + t.Fatalf("loop iteration %d placed %v after the reap failed with %d: a failed cancel "+ + "is retried, one per iteration for as long as the drain lasts", i, sqes, tc.res) + } + } + if !f.cs.recvArmed || f.w.conns[f.fd] != f.cs { + t.Fatalf("recvArmed=%v, in table=%v: the conn must keep its recv and stay", + f.cs.recvArmed, f.w.conns[f.fd] == f.cs) + } + if n := metric(t, f.e, "TransplantReapFailed"); n != 1 { + t.Errorf("TransplantReapFailed = %d, want 1", n) + } + if n := metric(t, f.e, "TransplantReapMisses"); n != 0 { + t.Errorf("TransplantReapMisses = %d, want 0: a failed cancel is not a miss", n) + } + heldHandOff(t, f) + }) + } +} + +// TestNoReapWithoutAsyncCancelFlags (celeris#681 C1): on a kernel that +// rejects IORING_ASYNC_CANCEL flags, which the engine's startup probe finds, +// no reap is ever placed, at either hand-off site or on a later iteration: it +// could only fail with -EINVAL and leave the recv armed. The conn keeps its +// recv and stays until that recv completes; a sync conn then leaves after its +// next response, held, with nothing in flight. +func TestNoReapWithoutAsyncCancelFlags(t *testing.T) { + noFlags := func(t *testing.T, f *fdlFixture) { + t.Helper() + if !trySetWorkerField(f.w, "asyncCancelFlags", false) { + t.Error("Worker has no asyncCancelFlags field: nothing tells the hand-off the kernel rejects cancel flags") + } + } + checkNothingPlaced := func(t *testing.T, f *fdlFixture, where string) { + t.Helper() + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 { + t.Fatalf("%s placed %v on a kernel without cancel flags, want nothing: a reap fails with "+ + "-EINVAL there and leaves the recv armed", where, sqes) + } + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("%s handed the conn off %d time(s) with its recv armed", where, n) + } + for i := 1; i <= 3; i++ { + f.w.drainDetachQueue() + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 { + t.Fatalf("loop iteration %d after %s placed %v, want nothing", i, where, sqes) + } + } + if !f.cs.recvArmed { + t.Fatalf("after %s the conn has no recv armed", where) + } + } + + t.Run("sync_site", func(t *testing.T) { + f := newFDLFixture(t, false) + noFlags(t, f) + f.armFirstRecv() + f.serveOne() + f.deliver(fdlGET) + _ = takeSQEs(f.w.ring) + f.startDrain() + f.process(f.sendCQE()) // the SEND completes with its linked RECV armed + checkNothingPlaced(t, f, "the SEND completion") + if n := metric(t, f.e, "TransplantReaps"); n != 0 { + t.Errorf("TransplantReaps = %d, want 0", n) + } + if n := metric(t, f.e, "TransplantReapUnsupported"); n != 1 { + t.Errorf("TransplantReapUnsupported = %d, want 1", n) + } + heldHandOff(t, f) + }) + + // A promoted async conn whose goroutine claimed its own hand-off: the + // recv the feed path armed for it is the one op in flight. + t.Run("async_site", func(t *testing.T) { + f := newFDLFixture(t, true) + noFlags(t, f) + f.cs.asyncPromoted.Store(true) + f.armFirstRecv() + f.startDrain() + f.cs.transplantPending.Store(true) + f.w.enqueueDetach(f.cs) + f.w.drainDetachQueue() + checkNothingPlaced(t, f, "the drain of the goroutine's claim") + if n := metric(t, f.e, "TransplantReaps"); n != 0 { + t.Errorf("TransplantReaps = %d, want 0", n) + } + }) +} + +// TestReapSuppressedAfterFailedHandOff (celeris#681 C2): a hand-off that +// fails at its dup (a process out of descriptors) leaves the conn in place, +// and the recv re-armed for it used to be reaped again at once, failing the +// same way: a RECV, a cancel and two completions per loop iteration for as +// long as the failure and the drain lasted. No reap may follow a failed +// hand-off until the conn receives data again, and once the dup works the +// conn leaves normally. +func TestReapSuppressedAfterFailedHandOff(t *testing.T) { + f := reapedFixture(t) + dupCalls := 0 + failDup := func(int) (int, error) { + dupCalls++ + return -1, unix.EMFILE + } + if !trySetWorkerField(f.w, "dupFD", failDup) { + t.Error("Worker has no dupFD seam: the dup failure cannot be injected, the real dup runs") + } + // The reap lands; the hand-off it re-runs fails at the dup. + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if n := f.tgt.adopted.Load(); n != 0 { + t.Fatalf("the conn was handed off (%d) although its dup failed", n) + } + sqes := takeSQEs(f.w.ring) + if len(sqes) != 1 || sqes[0].op != opRECV || sqes[0].tag() != udRecv { + t.Fatalf("after the failed hand-off the completion placed %v, want only the conn's recv re-armed; "+ + "a reap of it fails the hand-off again at once", sqes) + } + for i := 1; i <= 3; i++ { + f.w.drainDetachQueue() + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 { + t.Fatalf("loop iteration %d placed %v after the failed hand-off, want nothing", i, sqes) + } + } + // The client's next request: served here and HELD (the drain is still + // set); its SEND completion tries the hand-off, whose dup fails again, + // so the recv is re-armed, and again not reaped. + f.deliver(fdlGET) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || sqes[0].op != opSEND { + t.Fatalf("the next request placed %v, want its held SEND", sqes) + } + f.process(f.sendCQE()) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || sqes[0].op != opRECV { + t.Fatalf("the held SEND's completion, with the dup failing, placed %v, want only the recv re-armed", sqes) + } + if dupCalls != 2 { + t.Errorf("dup tried %d times, want 2: once at the reap, once at the held response", dupCalls) + } + // The descriptors are back: the next response leaves. + trySetWorkerField(f.w, "dupFD", (func(int) (int, error))(nil)) + heldHandOff(t, f) +} + +// TestReapedRecvLeavesNoLinkOrBuffer (celeris#681 C4): reapOutcome consumes +// the reaped recv's -ECANCELED, the last completion that recv has, so it +// must leave what handleRecv leaves after any recv's last completion: no +// stale recvLinked (only the chained recv's own completion clears it), and a +// provided buffer the completion carries returned to the ring. +func TestReapedRecvLeavesNoLinkOrBuffer(t *testing.T) { + t.Run("linked_recv_is_unlinked", func(t *testing.T) { + f := newFDLFixture(t, false) + f.armFirstRecv() + f.serveOne() + if !f.cs.recvLinked { + t.Fatal("setup: serveOne left no linked RECV") + } + f.startDrain() + f.w.tryTransplant(f.fd) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("placed %v, want the reap of the linked RECV", sqes) + } + f.stopDrain() // the conn stays: its recv is re-armed standalone + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || sqes[0].op != opRECV || sqes[0].flags&sqeIOLink != 0 { + t.Fatalf("placed %v, want one standalone RECV", sqes) + } + if f.cs.recvLinked { + t.Fatal("recvLinked is still set after the linked recv's last completion: the recv armed now is standalone") + } + }) + + t.Run("provided_buffer_goes_back", func(t *testing.T) { + f := reapedFixture(t) + br, err := NewBufferRing(f.w.ring, 7, 4, 64) + if err != nil { + skipOrFail656(t, "provided buffer ring unavailable: %v", err) + } + t.Cleanup(func() { br.Close(f.w.ring) }) + f.w.bufRing = br + tail := br.tail + f.stopDrain() + c := f.recvCQE(-int32(unix.ECANCELED)) + c.Flags = cqeFBuffer | 2<<16 + f.process(c) + if br.tail != tail+1 || !f.w.hasBufReturns { + t.Fatalf("provided buffer 2 not returned (ring tail %d -> %d, hasBufReturns=%v)", + tail, br.tail, f.w.hasBufReturns) + } + }) +} diff --git a/engine/iouring/handoff_loss.go b/engine/iouring/handoff_loss.go index 15c40064..368b5d6f 100644 --- a/engine/iouring/handoff_loss.go +++ b/engine/iouring/handoff_loss.go @@ -58,16 +58,30 @@ import "sync/atomic" // conn could be handed off (REAP), and those whose own completion said // they matched nothing (the recv had completed, or was not issued yet). // A miss is retried, never followed by a hand-off. Rates. +// - reapFailed: reaps whose completion was neither a hit nor a miss (for +// example -EINVAL from a kernel that rejects the cancel flags the +// startup probe found accepted). Not retried and never followed by a +// hand-off. Must stay 0. +// - reapUnsupported: reaps not placed because this kernel rejects the +// IORING_ASYNC_CANCEL flags a reap needs (probeAsyncCancelFlags; they +// exist from 5.19). The conn stays until its recv completes on its own. +// A rate, 0 on every kernel from 5.19. // - holdRescued: held conns the timeout sweep found with their response // sent, not handed off and no recv armed — a path that skipped the // release. The belt under releaseHold. Must stay 0. -// - doubleClaim: hand-offs refused because another path owned the conn's -// hand-off: tryTransplant on a conn whose dispatch goroutine had already -// claimed it (transplantPending), or finishAsyncTransplant for a -// connState that no longer owns its fd slot. Each is a double hand-off -// prevented; before the checks existed one identity was measured moving -// twice in 167 runs. Rare, not impossible: the first half fires whenever -// a SEND completion lands between a goroutine's claim and the drain. +// - doubleClaim: hand-offs refused at finishAsyncTransplant because the +// connState no longer owns its fd slot: something moved it out of the +// table since its dispatch goroutine claimed the hand-off, and in async +// mode that something is another hand-off of the same conn (a close +// marks the queued claim detachClosed first, and hijack is refused). +// Before the checks existed one identity was measured moving twice in +// 167 runs. Must stay 0. +// - claimDeferred: tryTransplant finding a conn whose dispatch goroutine +// has claimed its own hand-off (transplantPending) and leaving it to that +// claim. Counted before tryTransplant's other gates, so it is ordering, +// not a fault: it fires whenever a completion of the conn (its own +// response SEND, typically) lands between the goroutine's park and the +// drain of its claim. A rate. // // All are direct atomic adds: they fire on the stale-CQE, drain and hand-off // paths only, never on the per-request path while no drain is set, and like @@ -83,6 +97,9 @@ type handoffLossStats struct { reapMisses atomic.Uint64 holdRescued atomic.Uint64 doubleClaim atomic.Uint64 + claimDeferred atomic.Uint64 + reapFailed atomic.Uint64 + reapUnsupported atomic.Uint64 } // The fd-lifetime counters are nil-safe: a hand-built test Worker has none. @@ -117,6 +134,24 @@ func (s *handoffLossStats) noteDoubleClaim() { } } +func (s *handoffLossStats) noteClaimDeferred() { + if s != nil { + s.claimDeferred.Add(1) + } +} + +func (s *handoffLossStats) noteReapFailed() { + if s != nil { + s.reapFailed.Add(1) + } +} + +func (s *handoffLossStats) noteReapUnsupported() { + if s != nil { + s.reapUnsupported.Add(1) + } +} + // noteStaleRecvData counts one stale recv CQE that carried data, under the // class of its (fd, generation) identity. Must run BEFORE // noteStaleTerminalOp, which deletes the closedOps entry once the kernel diff --git a/engine/iouring/handoff_loss_metrics_test.go b/engine/iouring/handoff_loss_metrics_test.go index 2a1ea4ad..773b0354 100644 --- a/engine/iouring/handoff_loss_metrics_test.go +++ b/engine/iouring/handoff_loss_metrics_test.go @@ -23,6 +23,9 @@ func TestMetricsCarriesTheHandoffLossWitnesses(t *testing.T) { e.metrics.handoffLoss.reapMisses.Store(19) e.metrics.handoffLoss.holdRescued.Store(23) e.metrics.handoffLoss.doubleClaim.Store(29) + e.metrics.handoffLoss.claimDeferred.Store(31) + e.metrics.handoffLoss.reapFailed.Store(37) + e.metrics.handoffLoss.reapUnsupported.Store(41) m := e.Metrics() for _, c := range []struct { @@ -38,6 +41,9 @@ func TestMetricsCarriesTheHandoffLossWitnesses(t *testing.T) { {"TransplantReapMisses", m.TransplantReapMisses, 19}, {"TransplantHoldRescued", m.TransplantHoldRescued, 23}, {"TransplantDoubleClaim", m.TransplantDoubleClaim, 29}, + {"TransplantClaimDeferred", m.TransplantClaimDeferred, 31}, + {"TransplantReapFailed", m.TransplantReapFailed, 37}, + {"TransplantReapUnsupported", m.TransplantReapUnsupported, 41}, } { if c.got != c.want { t.Errorf("Metrics().%s = %d, want %d — the witness exists but cannot "+ diff --git a/engine/iouring/probe.go b/engine/iouring/probe.go index 4b2f1ac4..7e47eba8 100644 --- a/engine/iouring/probe.go +++ b/engine/iouring/probe.go @@ -33,6 +33,9 @@ var ( cachedMultiAccept sync.Once cachedMultiAcceptOK bool cachedMultiAcceptReas string + cachedAsyncCancel sync.Once + cachedAsyncCancelOK bool + cachedAsyncCancelReas string ) // probeSendZCCached returns the cached SendZC probe result, running the @@ -68,6 +71,15 @@ func probeMultishotAcceptCached() (bool, string) { return cachedMultiAcceptOK, cachedMultiAcceptReas } +// probeAsyncCancelFlagsCached returns the cached async-cancel-flags probe +// result. +func probeAsyncCancelFlagsCached() (bool, string) { + cachedAsyncCancel.Do(func() { + cachedAsyncCancelOK, cachedAsyncCancelReas = probeAsyncCancelFlags() + }) + return cachedAsyncCancelOK, cachedAsyncCancelReas +} + // SEND_ZC ioprio flags and notification result values. const ( sendZCReportUsage = 1 << 3 // IORING_SEND_ZC_REPORT_USAGE: request ZC usage info in notification @@ -449,3 +461,85 @@ func probeMultishotAccept() (bool, string) { } return true, "" } + +// The async-cancel-flags probe's own user_data values: the op it tries to +// cancel (nothing in its private ring carries it) and the cancel itself. +const ( + asyncCancelProbeTarget uint64 = 0xCA_4CE1_7A26_E7 + asyncCancelProbeTag uint64 = 0xCA_4CE1_7A6 +) + +// probeAsyncCancelFlags tests whether the kernel accepts the +// IORING_ASYNC_CANCEL_* flags, which exist from Linux 5.19. The io_uring→epoll +// hand-off's REAP (celeris#657) cancels an armed recv with +// IORING_ASYNC_CANCEL_ALL, keyed on the recv's user_data +// (prepCancelUserDataReported). Through 5.18, io_async_cancel_prep rejects any +// non-zero cancel_flags with -EINVAL, and no IORING_FEAT bit reports the +// flags, so the kernel has to be asked. Version-based selection is no answer +// either: the Base tier covers every 5.10-5.18 kernel, and a vendor kernel can +// claim a version its feature surface does not match. +// +// The probe submits exactly the reap's SQE form against a user_data that +// nothing carries and reads the cancel's own completion; see +// classifyAsyncCancelProbe for how the result reads. The cancel runs inline +// at submit on every kernel measured, so its completion is normally there +// when Submit returns; a short wait covers any that is not. +func probeAsyncCancelFlags() (bool, string) { + ring, err := NewRing(8, 0, 0) + if err != nil { + return false, "NewRing failed: " + err.Error() + } + defer func() { _ = ring.Close() }() + + sqe := ring.GetSQE() + if sqe == nil { + return false, "GetSQE returned nil" + } + prepCancelUserDataReported(sqe, asyncCancelProbeTarget) + setSQEUserData(sqe, asyncCancelProbeTag) + if _, err := ring.Submit(); err != nil { + return false, "Submit failed: " + err.Error() + } + head, tail := ring.BeginCQ() + if head == tail { + if err := ring.SubmitAndWaitTimeout(500 * time.Millisecond); err != nil { + return false, "SubmitAndWaitTimeout failed: " + err.Error() + } + head, tail = ring.BeginCQ() + if head == tail { + return false, "no CQE produced for the cancel (waited 500ms)" + } + } + cqe := ring.cqeAt(head) + ud, res := cqe.UserData, cqe.Res + ring.EndCQ(head + 1) + if ud != asyncCancelProbeTag { + return false, fmt.Sprintf("unexpected CQE user_data %#x (want the cancel's %#x)", ud, asyncCancelProbeTag) + } + return classifyAsyncCancelProbe(res) +} + +// classifyAsyncCancelProbe reads the completion of the probe's cancel, a +// cancel with IORING_ASYNC_CANCEL_ALL of a user_data nothing carries: +// +// - res >= 0: the flags were accepted. With CANCEL_ALL, res is the number +// of ops cancelled, so the miss the probe makes is 0. +// - -ENOENT: accepted too. It is how a cancel without CANCEL_ALL reports a +// miss; no kernel measured answers the probe with it. +// - -EINVAL: rejected. Measured on 5.15.0-191: every cancel form celeris +// builds returns -EINVAL there and leaves its target running. +// - anything else: not an answer the probe understands, so the flags are +// treated as rejected and the value is reported. +// +// Split out of probeAsyncCancelFlags so every outcome can be checked against +// a synthetic result. +func classifyAsyncCancelProbe(res int32) (bool, string) { + switch { + case res >= 0, res == -int32(unix.ENOENT): + return true, "" + case res == -int32(unix.EINVAL): + return false, "IORING_ASYNC_CANCEL flags rejected: cqe.res=-22 (EINVAL); the kernel predates Linux 5.19" + default: + return false, fmt.Sprintf("the cancel completed with cqe.res=%d, which the probe does not recognise", res) + } +} diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 2670f1c9..02da465d 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -85,7 +85,10 @@ func (w *Worker) tryTransplant(fd int) { // and the process had reused (measured: one identity moved twice, 20 us // apart, in 1 of 167 runs, the run with the only negative gauge). The // goroutine sets the claim and clears asyncRun under asyncInMu, so both - // are read under it here. + // are read under it here. Leaving the conn to its claim is ordering, not + // a double claim (a completion of the conn landed between the park and + // the drain of the claim), and it is counted as such, before any of the + // gates below: TransplantClaimDeferred, a rate. if w.async { cs.asyncInMu.Lock() running := cs.asyncRun @@ -95,7 +98,7 @@ func (w *Worker) tryTransplant(fd int) { return } if claimed { - w.handoffLoss.noteDoubleClaim() + w.handoffLoss.noteClaimDeferred() return } } @@ -137,19 +140,26 @@ func (w *Worker) tryTransplant(fd int) { // conn table, cancel what is still armed, defer the connState release to the // terminal CQEs), closes the ORIGINAL fd (the dup keeps the socket alive for // epoll) and hands the dup over. A failed dup leaves the conn untouched and -// reports false; the hand-off is retried at the conn's next boundary. +// reports false; the hand-off is retried at the conn's next boundary, with no +// reap placed for it until it next receives data (reapSuppressed). // // detachedRelease picks the release: the async site's dispatch goroutine may // still be in its deferred recover()/Done() block referencing cs, so that site // holds cs alive with no pool recycle until the kernel ops drain (mirrors // finishCloseDetached). func (w *Worker) handOff(cs *connState, fd int, h *transplantTargetHolder, detachedRelease bool) bool { - newFD, err := unix.Dup(fd) + dup := unix.Dup + if w.dupFD != nil { + dup = w.dupFD + } + newFD, err := dup(fd) if err != nil { + cs.reapSuppressed = true return false } if serr := unix.SetNonblock(newFD, true); serr != nil { _ = unix.Close(newFD) + cs.reapSuppressed = true return false } carry := engine.Carryover{RemoteAddr: cs.remoteAddr} @@ -270,10 +280,23 @@ func (w *Worker) finishAsyncTransplant(cs *connState) { // that still owns its slot. If anything else moved or closed it since the // goroutine's claim, cs.fd is a number some other connection may hold by // now, and dup'ing it would hand that connection off. - if cs.fd < 0 || cs.fd >= len(w.conns) || w.conns[cs.fd] != cs || cs.closing { + // + // Only the slot check is a double claim, and only it is counted + // (TransplantDoubleClaim, which must stay 0). In async mode a close marks + // the queued claim detachClosed, so drainDetachQueue skips it before this + // point, and hijack is refused; what else vacates the slot is another + // hand-off of this conn. A closing conn (a close deferred behind a send) + // is an ordinary refusal, and the range check is defensive. + if cs.fd < 0 || cs.fd >= len(w.conns) { + return + } + if w.conns[cs.fd] != cs { w.handoffLoss.noteDoubleClaim() return } + if cs.closing { + return + } // Re-validate egress here, on the worker thread (celeris#529). // // asyncTransplantEligible runs on the DISPATCH goroutine and deliberately diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index cdeb32a1..b6a4b833 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -488,6 +488,14 @@ type Worker struct { // reapRetrySpare is the second buffer of the swap. Worker thread only. reapRetry []uint64 reapRetrySpare []uint64 + // asyncCancelFlags: this kernel accepts IORING_ASYNC_CANCEL_* flags + // (probeAsyncCancelFlags, 5.19+). A reap is only placed when it does; + // createWorkers copies the engine's answer. Read-only after init. + asyncCancelFlags bool + // dupFD duplicates the descriptor a hand-off moves: unix.Dup when nil. + // A test seam, so a test can make the dup fail the way a process out + // of descriptors does (EMFILE). + dupFD func(fd int) (int, error) // transplant (#383 reverse) is non-nil while a drain-to-epoll is in progress. // Set by Engine.StartTransplant (controller goroutine), read on this worker's @@ -1193,10 +1201,11 @@ func (w *Worker) run(ctx context.Context) { if w.transplant.Load() != nil { w.tryTransplant(fd) } - // Unconditionally, after the attempt: a conn held for - // a hand-off that did not happen gets its recv back - // (celeris#657). - w.releaseHold(fd) + // No releaseHold here (celeris#657): a held conn has + // no recv armed, so the only recv completion that can + // find a hold is the one whose own response set it, + // with that SEND still in flight. The SEND completion + // below is where a hold is released. } case udSend: if !w.staleConnCQE(entry, fd, ud) { @@ -1452,7 +1461,11 @@ func (w *Worker) run(ctx context.Context) { // in the same pass that finds the worker idle, and the park is // indefinite: those SQEs used to sit unsubmitted until something // woke the worker, measured 1-24 pending at parks. Outside - // wakeMu, which is a leaf; SQPOLL submits by itself. + // wakeMu, which is a leaf. Not under SQPOLL, where the kernel's + // SQ thread submits; an SQ thread that has gone idle would need + // the NEED_WAKEUP kick the submit branch of this loop gives it, + // but no tier enables SQPOLL today (SQPollIdle is 0 in all + // three), so that case is not handled here. if !w.sqpoll && w.ring.Pending() > 0 { _, _ = w.ring.Submit() } @@ -1596,14 +1609,13 @@ func (w *Worker) processCQE(ctx context.Context, c *completionEntry, now int64) return } w.handleRecv(c, fd, now) - // The same hand-off attempt and hold release as the inlined - // dispatch (celeris#657). The listener-close harvest processes - // completions here, and a held conn whose SEND completion landed in - // it would otherwise be neither handed off nor re-armed. + // The same hand-off attempt as the inlined dispatch (celeris#657). + // The listener-close harvest processes completions here. As there, + // no hold can be released at a recv completion; the udSend case + // below releases it. if w.transplant.Load() != nil { w.tryTransplant(fd) } - w.releaseHold(fd) case udSend: if w.staleConnCQE(c, fd, ud) { return @@ -1612,6 +1624,8 @@ func (w *Worker) processCQE(ctx context.Context, c *completionEntry, now int64) if w.transplant.Load() != nil { w.tryTransplant(fd) } + // A held conn whose SEND completion lands in the listener-close + // harvest would otherwise be neither handed off nor re-armed. w.releaseHold(fd) case udClose: if w.staleConnCQE(c, fd, ud) { @@ -2538,6 +2552,9 @@ func (w *Worker) handleRecv(c *completionEntry, fd int, now int64) { } cs.lastActivity = now + // Data: the conn is serving again, so a hand-off that failed at its dup + // may be tried (and its recv reaped) once more (celeris#657). + cs.reapSuppressed = false // c.Res > 0 here (the c.Res <= 0 cases returned above): bytes received // on this recv CQE, regardless of which buffer they landed in. w.bytesReadBatch += uint64(c.Res) From 1bad54f73cdf7c21972b33fac2e316c3f3157b90 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 15:41:08 +0200 Subject: [PATCH 14/34] test(adaptive): commit T4, the multi-ring gate for the io_uring hand-off loss (celeris#657) TestFlapConnsPerRing fixes both engines at two workers, so 256 keep-alive connections are 128 per io_uring ring whatever memlock funds, and runs three promote/revert cycles under continuous load, requiring zero client errors. Step 0 validated it as the CI gate for face 2 (main FAIL 6/6), and the #681 campaign measured it: base 5b2e83b FAIL 8/8 (1,713 lost requests, each one a stale recv the witness counted), fix PASS 8/8. Until now it existed only as an overlay. The file is the measured step-0 source with four changes: its headers no longer say "overlay only", the conditional six-cycle variant (never triggered) is dropped, the two files are one, and a premise guard skips it when RLIMIT_MEMLOCK funds fewer than two io_uring workers, failing instead under CELERIS_REQUIRE_UPSWITCH=1 as the adaptive CI job sets it. --- adaptive/flap_conns_per_ring_test.go | 221 +++++++++++++++++++++++++++ 1 file changed, 221 insertions(+) create mode 100644 adaptive/flap_conns_per_ring_test.go diff --git a/adaptive/flap_conns_per_ring_test.go b/adaptive/flap_conns_per_ring_test.go new file mode 100644 index 00000000..7c1b0d60 --- /dev/null +++ b/adaptive/flap_conns_per_ring_test.go @@ -0,0 +1,221 @@ +//go:build linux + +package adaptive + +// celeris#657 face 2, T4: the CI-shape test for the io_uring -> epoll hand-off loss (celeris#681 round 2 commits it +// from the step-0 overlay the campaign measured: base 5b2e83b FAIL 8/8, fix PASS 8/8). +// +// The sync loss at a revert depends on linked SEND->RECV chains PER RING, not on the io_uring worker count, and +// CI's adaptive job raises memlock to unlimited, which gives io_uring its full worker count and so few conns per +// ring that the loss does not show. T4 pins the variable instead of the environment: Resources.Workers=2 fixes +// both engines at 2 workers, so 256 keep-alive conns are 128 per ring after a promotion, whatever the memlock. +// +// Shape (DECISION.md step 0; red-team.md section 7, T4): controller frozen, THREE promote/revert cycles under +// continuous back-to-back load, 2.5 s after each switch (more than a 2 s client read deadline, so a request lost +// at switch k is counted before switch k+1). It requires ZERO client errors: a client error here is a request the +// server never answered. (The StaleRecvData half of T4's assertion needs PR-1's counter, absent on 985a386.) + +import ( + "bufio" + "context" + "errors" + "fmt" + "io" + "net" + "net/http" + "os" + "sort" + "strings" + "sync" + "sync/atomic" + "syscall" + "testing" + "time" + + "github.com/goceleris/celeris/engine" + "github.com/goceleris/celeris/probe" + "github.com/goceleris/celeris/protocol/h2/stream" + "github.com/goceleris/celeris/resource" +) + +func TestFlapConnsPerRing(t *testing.T) { s0FlapConnsPerRing(t, 3) } + +func s0FlapConnsPerRing(t *testing.T, cycles int) { + if testing.Short() { + t.Skip("integration") + } + const ( + workers = 2 + perRing = 128 + conns = workers * perRing + dwell = 2500 * time.Millisecond + ) + // Two io_uring workers need 24 MiB of RLIMIT_MEMLOCK (12 MiB each, engine/iouring minMemlockPerWorker). Below + // that io_uring caps itself to one ring and the PREMISE check below fails on the environment, not the engine. + // The adaptive CI job raises memlock and sets CELERIS_REQUIRE_UPSWITCH=1, which turns this skip into a failure. + if m := maxWorkersForMemlock(); m >= 0 && m < workers { + msg := fmt.Sprintf("RLIMIT_MEMLOCK funds %d io_uring worker(s), T4 needs %d", m, workers) + if os.Getenv("CELERIS_REQUIRE_UPSWITCH") == "1" { + t.Fatal(msg + " -- CELERIS_REQUIRE_UPSWITCH=1 forbids skipping") + } + t.Skip(msg) + } + e, addr, stop := s0Bind(t, resource.Config{Addr: "127.0.0.1:0", Protocol: engine.HTTP1, + Resources: resource.Resources{Workers: workers}}, respHandler{}) + defer stop() + + stopLoad := make(chan struct{}) + var okCount, errCount atomic.Int64 + var cen s0Census + wg := s0Drive(addr, conns, stopLoad, &okCount, &errCount, &cen) + time.Sleep(500 * time.Millisecond) + t.Logf("S0T4 START cycles=%d conns=%d epoll=%d ok=%d err=%d", cycles, conns, + e.primary.Metrics().ActiveConnections, okCount.Load(), errCount.Load()) + + for sw := 1; sw <= 2*cycles; sw++ { + dir := "promote" + if sw%2 == 0 { + dir = "revert" + } + errBefore := errCount.Load() + t0 := time.Now() + e.ForceSwitch() + took := time.Since(t0) + time.Sleep(dwell) + pm, sm := e.primary.Metrics(), e.secondary.Metrics() + t.Logf("S0T4 SWITCH n=%d dir=%s took_ms=%d epoll=%d io_uring=%d wE=%d wI=%d ok=%d err=%d err_this=%d census=%s", + sw, dir, took.Milliseconds(), pm.ActiveConnections, sm.ActiveConnections, pm.Workers, sm.Workers, + okCount.Load(), errCount.Load(), errCount.Load()-errBefore, cen.String()) + } + close(stopLoad) + wg.Wait() + ew, iw := e.primary.Metrics().Workers, e.secondary.Metrics().Workers + t.Logf("S0T4 RESULT cycles=%d conns=%d wE=%d wI=%d conns_per_ring=%d ok=%d err=%d census=%s", + cycles, conns, ew, iw, conns/max(iw, 1), okCount.Load(), errCount.Load(), cen.String()) + if ew != workers || iw != workers { + t.Errorf("S0T4 PREMISE: want %d workers on both engines, got epoll=%d io_uring=%d", workers, ew, iw) + } + if n := errCount.Load(); n != 0 { + t.Errorf("S0T4 LOSS: %d of %d keep-alive clients lost a request across %d promote/revert cycles at %d conns per ring (census %s)", + n, conns, cycles, conns/max(iw, 1), cen.String()) + } +} + +// The client driver and bind helper of TestFlapConnsPerRing (from the celeris#657 step-0 overlay). The driver keeps +// driveKeepAlive's shape (write a request, read the response with a 2 s deadline, stop at the first error, never +// redial), so an error count is the number of connections that lost a request. It adds a per-class census that +// never saturates. The identifiers keep their s0 prefix. + +type s0Census struct { + mu sync.Mutex + m map[string]int +} + +func (c *s0Census) add(kind string, err error) { + cls := "other" + switch { + case errors.Is(err, os.ErrDeadlineExceeded): + cls = "timeout" + case errors.Is(err, syscall.ECONNRESET): + cls = "reset" + case errors.Is(err, syscall.EPIPE): + cls = "epipe" + case errors.Is(err, io.EOF), errors.Is(err, io.ErrUnexpectedEOF): + cls = "eof" + } + c.mu.Lock() + if c.m == nil { + c.m = map[string]int{} + } + c.m[kind+":"+cls]++ + c.mu.Unlock() +} + +func (c *s0Census) String() string { + c.mu.Lock() + defer c.mu.Unlock() + if len(c.m) == 0 { + return "none" + } + keys := make([]string, 0, len(c.m)) + for k := range c.m { + keys = append(keys, k) + } + sort.Strings(keys) + var b strings.Builder + for _, k := range keys { + fmt.Fprintf(&b, "%s=%d,", k, c.m[k]) + } + return strings.TrimSuffix(b.String(), ",") +} + +// s0Drive opens conns keep-alive HTTP/1.1 connections that loop write-request / read-response back to back +// until stop is closed. +func s0Drive(addr string, conns int, stop <-chan struct{}, ok, errc *atomic.Int64, cen *s0Census) *sync.WaitGroup { + var wg sync.WaitGroup + for i := 0; i < conns; i++ { + wg.Add(1) + go func() { + defer wg.Done() + c, derr := net.DialTimeout("tcp", addr, 2*time.Second) + if derr != nil { + errc.Add(1) + cen.add("dial", derr) + return + } + defer func() { _ = c.Close() }() + br := bufio.NewReader(c) + for { + select { + case <-stop: + return + default: + } + if _, werr := c.Write([]byte("GET / HTTP/1.1\r\nHost: x\r\n\r\n")); werr != nil { + errc.Add(1) + cen.add("write", werr) + return + } + _ = c.SetReadDeadline(time.Now().Add(2 * time.Second)) + resp, rerr := http.ReadResponse(br, nil) + if rerr != nil { + errc.Add(1) + cen.add("read", rerr) + return + } + _, _ = io.Copy(io.Discard, resp.Body) + _ = resp.Body.Close() + ok.Add(1) + } + }() + } + return &wg +} + +// s0Bind builds an adaptive engine from cfg, disables the switch cooldown, FREEZES the controller (so the +// only switches are the test's ForceSwitch calls; performSwitch ignores the freeze), starts Listen and waits +// for the bind. +func s0Bind(t *testing.T, cfg resource.Config, h stream.Handler) (*Engine, string, func()) { + t.Helper() + if !probe.Probe().IOUringTier.Available() { + t.Skip("io_uring unavailable: needs both sub-engines") + } + e, err := New(cfg, h, nil) + if err != nil { + t.Skipf("adaptive.New unsupported here: %v", err) + } + e.ctrl.cooldown = 0 + e.FreezeSwitching() + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error, 1) + go func() { done <- e.Listen(ctx) }() + for dl := time.Now().Add(3 * time.Second); e.Addr() == nil && time.Now().Before(dl); { + time.Sleep(10 * time.Millisecond) + } + if e.Addr() == nil { + cancel() + <-done + t.Fatal("adaptive engine never bound") + } + return e, e.Addr().String(), func() { cancel(); <-done } +} From 174a0ab4191b3f6ee9de5fcfb71980e8220cc37a Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 15:41:08 +0200 Subject: [PATCH 15/34] ci: gate T4 in the adaptive race step, require err=0 in the one-worker leg, and run the new fd-lifetime tests (celeris#657) - adaptive race step (memlock unlimited): now shell: bash with pipefail and a tee, plus a tally for TestFlapConnsPerRing: exactly one top-level RUN and PASS, no SKIP line for it, and exactly one RESULT line showing both engines at two workers and err=0. At raised memlock the two revert tests did not lose on the base; T4 did in 8 of 8 runs. - one-worker revert leg: the verdicts of TestReverseTransplant and TestBidirectionalFlap tolerate up to 64 lost requests (a base flap run PASSED with 56), so the step also requires each test's own summary line (async=false) to read err=0. - unit witness step: the eight new fd-lifetime tests join the PR-2 list, 29 in all. --- .github/workflows/ci.yml | 62 +++++++++++++++++++++++++++++++--------- 1 file changed, 49 insertions(+), 13 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 2947d440..34b53f1f 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -127,12 +127,13 @@ jobs: go test -race -count=1 -timeout=300s $pkgs # celeris#657: the step above runs ./engine/iouring WITHOUT -v, so a skip # there prints nothing and the package still reports `ok`. That is how the - # celeris#656 leak shipped with no cover. Many celeris#657 tests can skip: - # of the eleven hand-off loss witness tests (PR-1), four build a ring - # through newTestRing and one builds io_uring workers; of the ten - # fd-lifetime tests (PR-2), six build a ring through newTestRing and - # four (plus one subtest of the six) start a whole io_uring engine. So - # all twenty-one run again here by name, with the PASS-count interlock + # celeris#656 leak shipped with no cover. Most celeris#657 tests can skip: + # they build a ring through newTestRing, build io_uring workers, or start + # a whole io_uring engine. That holds for the eleven hand-off loss witness + # tests (PR-1, the first list) and the eighteen fd-lifetime tests (PR-2, + # the second list; eight of them pin the async-cancel-flags probe, the + # reap-failure and dup-failure paths and the reaped recv's cleanup). So + # all twenty-nine run again here by name, with the PASS-count interlock # the `iouring` job uses (celeris#664): -v, CELERIS_REQUIRE_IOURING_WORKERS=1 # to turn every environment skip into a failure, and an exact tally. # @@ -163,7 +164,7 @@ jobs: set -o pipefail echo "memlock (KiB): $(ulimit -l)" pr1='TestStaleRecvDataCountsATransplantedConn|TestStaleRecvDataCountsAnAsyncTransplantedConn|TestStaleRecvDataCountsAClosedConn|TestStaleRecvDataCountsAnUnattributedIdentity|TestStaleRecvDataIgnoresCompletionsWithoutData|TestStaleRecvDataCountsEachMultishotCompletion|TestTransplantHandoffInFlightCountsTryTransplant|TestTransplantHandoffInFlightCountsFinishAsyncTransplant|TestMetricsCarriesTheHandoffLossWitnesses|TestWorkersShareTheHandoffLossWitnesses|TestClosedOpsEntryStaysThirtyTwoBytes' - pr2='TestTransplantNeverHandsOffArmedRecv|TestTransplantReapMissIsRetried|TestHoldReleasedWhenDrainStops|TestHoldRescuedByCheckTimeouts|TestOneOwnerPerHandoff|TestNoDrainSQESequenceIsUnchanged|TestHandoffHasNothingInFlight|TestStaleRecvDataCounted|TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen|TestWorkerParksWithNothingPending' + pr2='TestTransplantNeverHandsOffArmedRecv|TestTransplantReapMissIsRetried|TestHoldReleasedWhenDrainStops|TestHoldRescuedByCheckTimeouts|TestOneOwnerPerHandoff|TestNoDrainSQESequenceIsUnchanged|TestHandoffHasNothingInFlight|TestStaleRecvDataCounted|TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen|TestWorkerParksWithNothingPending|TestTransplantReapFailureIsNotRetried|TestNoReapWithoutAsyncCancelFlags|TestReapSuppressedAfterFailedHandOff|TestReapedRecvLeavesNoLinkOrBuffer|TestAsyncCancelProbeClassifies|TestAsyncCancelProbeOnThisKernel|TestWorkersCarryTheAsyncCancelProbe|TestReapOnTheRunningKernel' names="${pr1}|${pr2}" want=$(( $(tr '|' '\n' <<<"$names" | wc -l) )) go test -race -count=1 -timeout=300s -v -run "^(${names})\$" \ @@ -228,15 +229,40 @@ jobs: # connections queued on the paused engine, and on a GitHub runner a # 2048-conn promotion loses about a third of them). The other three # up-switch tests keep running. + # + # celeris#657 face 2 has its multi-ring gate here: TestFlapConnsPerRing + # (T4) fixes both engines at two workers, so its 256 keep-alive + # connections are 128 per io_uring ring whatever the runner funds, and + # runs three promote/revert cycles under load. At raised memlock the two + # revert tests below spread their 64 connections over several rings and + # did not lose on the base; T4 lost requests in 8 of 8 base runs. The + # tally makes T4 impossible to lose quietly: exactly one top-level RUN + # and PASS, no SKIP line for it, and its own RESULT line must show both + # engines at two workers and err=0 (the requests its clients lost). - name: Adaptive — race tests (memlock raised, up-switch required) + shell: bash env: CELERIS_REQUIRE_UPSWITCH: "1" run: | + set -o pipefail sudo prlimit --pid "$$" --memlock=unlimited:unlimited echo "memlock (KiB): $(ulimit -l)" go test -race -count=1 -timeout=600s -v \ -skip '^(TestBidirectionalFlapAsync|TestRampH1Sync|TestRampH1Async)$' \ - ./adaptive/... + ./adaptive/... 2>&1 | tee /tmp/adaptive-race.log + t4='TestFlapConnsPerRing' + ran=$(grep -cE "^=== RUN ${t4}\$" /tmp/adaptive-race.log || true) + passed=$(grep -cE "^--- PASS: ${t4} \(" /tmp/adaptive-race.log || true) + skipped=$(grep -cE "^[[:space:]]*--- SKIP: ${t4}[ /]" /tmp/adaptive-race.log || true) + results=$(grep -cE 'S0T4 RESULT cycles=[0-9]+ conns=[0-9]+ wE=[0-9]+ wI=[0-9]+ ' /tmp/adaptive-race.log || true) + clean=$(grep -cE 'S0T4 RESULT cycles=3 conns=256 wE=2 wI=2 conns_per_ring=128 ok=[0-9]+ err=0 ' /tmp/adaptive-race.log || true) + echo "celeris#657 T4: want 1, ran $ran, passed $passed, SKIP lines $skipped, RESULT lines $results, at two workers with err=0: $clean" + if [ "$ran" -ne 1 ] || [ "$passed" -ne 1 ] || [ "$skipped" -ne 0 ] || [ "$results" -ne 1 ] || [ "$clean" -ne 1 ]; then + echo "expected TestFlapConnsPerRing to run once and PASS with no SKIP line, and its RESULT line to show" + echo "both engines at two workers (128 connections per io_uring ring) and err=0 -- was it renamed or" + echo "removed, did it skip, or did this runner fund a different worker count?" + exit 1 + fi # celeris#657 PR-2: the one-worker leg of the two revert tests. The step # above raises memlock, so io_uring there runs several workers and # TestReverseTransplant / TestBidirectionalFlap never meet the shape in @@ -254,6 +280,13 @@ jobs: # two PASS (same patterns as the unit job's celeris#657 step). It runs # even when the step above failed, so a known flake there cannot hide # this leg's result. + # + # Their verdicts tolerate up to one lost request per connection (err <= + # 64), so a PASS alone does not mean nothing was lost: on the base, the + # flap test PASSED once with 56 lost requests. The step therefore also + # reads each test's own summary line (the revert test's "before revert" + # line and the flap test's "total" line, both for async=false) and + # requires err=0 on both. - name: Adaptive — one-worker revert tests (runner memlock, skipping forbidden) if: ${{ !cancelled() }} shell: bash @@ -269,11 +302,14 @@ jobs: skipped=$(grep -cE '^[[:space:]]*--- SKIP' /tmp/revert657.log || true) engines=$(grep -cE 'io_uring engine listening .*workers=[0-9]+' /tmp/revert657.log || true) one=$(grep -cE 'io_uring engine listening .*workers=1( |$)' /tmp/revert657.log || true) - echo "celeris#657 one-worker revert tests: want $want, ran $ran, passed $passed, SKIP lines $skipped, io_uring engines $engines at workers=1: $one" - if [ "$ran" -ne "$want" ] || [ "$passed" -ne "$want" ] || [ "$skipped" -ne 0 ] || [ "$engines" -eq 0 ] || [ "$one" -ne "$engines" ]; then - echo "expected exactly $want top-level revert tests to run and PASS with no SKIP line, and" - echo "every io_uring engine at one worker -- was one renamed or removed, did one skip or fail," - echo "or did this runner fund a different worker count?" + summaries=$(grep -cE '\[async=false\] (before revert: .*\| ok=[0-9]+ err=[0-9]+$|total ok=[0-9]+ err=[0-9]+ \|)' /tmp/revert657.log || true) + lossless=$(grep -cE '\[async=false\] (before revert: .*\| ok=[0-9]+ err=0$|total ok=[0-9]+ err=0 \|)' /tmp/revert657.log || true) + echo "celeris#657 one-worker revert tests: want $want, ran $ran, passed $passed, SKIP lines $skipped, io_uring engines $engines at workers=1: $one, summaries $summaries with err=0: $lossless" + if [ "$ran" -ne "$want" ] || [ "$passed" -ne "$want" ] || [ "$skipped" -ne 0 ] || [ "$engines" -eq 0 ] || [ "$one" -ne "$engines" ] || [ "$summaries" -ne "$want" ] || [ "$lossless" -ne "$summaries" ]; then + echo "expected exactly $want top-level revert tests to run and PASS with no SKIP line, every" + echo "io_uring engine at one worker, and each test's summary line to report err=0 -- was one renamed" + echo "or removed, did one skip or fail, did this runner fund a different worker count, or were" + echo "requests lost (a PASS verdict tolerates up to one per connection)?" exit 1 fi From c910ff0226a993bea7a4bb311479db487e88e4bd Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 15:44:32 +0200 Subject: [PATCH 16/34] test(iouring): log the round-2 hand-off counters in the engine-level load line (celeris#657) The celeris657 load line of the engine-level fd-lifetime tests now also prints TransplantClaimDeferred, TransplantReapFailed and TransplantReapUnsupported (read by name, -1 on a tree without them), so a run on a kernel without cancel flags shows which path its connections took. --- engine/iouring/fd_lifetime_engine_test.go | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/engine/iouring/fd_lifetime_engine_test.go b/engine/iouring/fd_lifetime_engine_test.go index 9f45adfb..ebc5bad6 100644 --- a/engine/iouring/fd_lifetime_engine_test.go +++ b/engine/iouring/fd_lifetime_engine_test.go @@ -261,7 +261,8 @@ func transplantUnderLoad(t *testing.T, e *Engine, addr string, n int, tgt engine e.StopTransplant() time.Sleep(100 * time.Millisecond) t.Logf("celeris657 load conns=%d ok=%d errs=%d classes=[%s] W1T=%d W1U=%d W1C=%d W2=%d "+ - "held=%d reaps=%d misses=%d rescued=%d doubleclaim=%d detached=%d", + "held=%d reaps=%d misses=%d rescued=%d doubleclaim=%d detached=%d "+ + "claimdeferred=%d reapfailed=%d reapunsupported=%d", n, res.ok, res.errs, classes(res.byClass), e.metrics.handoffLoss.staleRecvDataTransplanted.Load(), e.metrics.handoffLoss.staleRecvDataUnattributed.Load(), @@ -269,7 +270,9 @@ func transplantUnderLoad(t *testing.T, e *Engine, addr string, n int, tgt engine e.metrics.handoffLoss.handoffInFlight.Load(), metricOr(e, "TransplantHeld"), metricOr(e, "TransplantReaps"), metricOr(e, "TransplantReapMisses"), metricOr(e, "TransplantHoldRescued"), metricOr(e, "TransplantDoubleClaim"), - e.metrics.transplantDetached.Load()) + e.metrics.transplantDetached.Load(), + metricOr(e, "TransplantClaimDeferred"), metricOr(e, "TransplantReapFailed"), + metricOr(e, "TransplantReapUnsupported")) return res } From 48dfe1abca5367eee83d54575e4252133c9fa448 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 15:47:20 +0200 Subject: [PATCH 17/34] test(iouring): drive the cancel probe's rejection path on any kernel, and pin that data lifts the reap suppression (celeris#657) - probeAsyncCancelFlags is probeAsyncCancel(IORING_ASYNC_CANCEL_ALL). TestAsyncCancelProbeOnThisKernel also submits a cancel_flags bit no kernel defines, which every kernel rejects with -EINVAL, so a probe that stopped reading the kernel's answer fails on a 5.19+ kernel too. - TestReapSuppressedAfterFailedHandOff now ends with the conn receiving data again, being served with the drain stopped, and having its linked RECV reaped once the drain is back: the suppression lasts until data, not forever. --- engine/iouring/async_cancel_probe_test.go | 9 ++++++++- engine/iouring/fd_lifetime_test.go | 21 +++++++++++++++++++-- engine/iouring/probe.go | 9 +++++++++ 3 files changed, 36 insertions(+), 3 deletions(-) diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index ec8b4dd2..85399439 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -61,7 +61,10 @@ func kernelHasCancelFlags(t *testing.T) (bool, string) { } // TestAsyncCancelProbeOnThisKernel runs the probe itself against the running -// kernel and checks its answer against the kernel's version. +// kernel and checks its answer against the kernel's version, and checks that +// the probe reads the kernel's answer: a cancel flag no kernel defines +// (bit 31) is rejected with -EINVAL everywhere, so the same path must then +// report the flags unsupported. func TestAsyncCancelProbeOnThisKernel(t *testing.T) { r, err := NewRing(8, 0, 0) if err != nil { @@ -77,6 +80,10 @@ func TestAsyncCancelProbeOnThisKernel(t *testing.T) { if cok, _ := probeAsyncCancelFlagsCached(); cok != ok { t.Fatalf("the cached probe says %v, the probe %v", cok, ok) } + if bad, why := probeAsyncCancel(1 << 31); bad || !strings.Contains(why, "EINVAL") { + t.Fatalf("a cancel flag no kernel defines was read as (%v, %q), want rejected with EINVAL: "+ + "the probe does not read the kernel's answer", bad, why) + } } // TestWorkersCarryTheAsyncCancelProbe: New stores the probe's answer and diff --git a/engine/iouring/fd_lifetime_test.go b/engine/iouring/fd_lifetime_test.go index 515a6fc4..0068f597 100644 --- a/engine/iouring/fd_lifetime_test.go +++ b/engine/iouring/fd_lifetime_test.go @@ -811,9 +811,26 @@ func TestReapSuppressedAfterFailedHandOff(t *testing.T) { if dupCalls != 2 { t.Errorf("dup tried %d times, want 2: once at the reap, once at the held response", dupCalls) } - // The descriptors are back: the next response leaves. + // The descriptors are back, and the conn receives data again, which + // lifts the suppression: served with the drain stopped, its response + // goes out with a linked RECV, and once the drain is set again that recv + // is reaped and the conn leaves at its -ECANCELED. trySetWorkerField(f.w, "dupFD", (func(int) (int, error))(nil)) - heldHandOff(t, f) + f.stopDrain() + f.deliver(fdlGET) + if sqes := takeSQEs(f.w.ring); len(sqes) != 2 || sqes[1].op != opRECV { + t.Fatalf("a request with the drain stopped placed %v, want SEND then its linked RECV", sqes) + } + f.startDrain() + f.process(f.sendCQE()) + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("after the conn received data again its armed recv got %v, want a reap: the "+ + "suppression outlived the failure it was for", sqes) + } + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the reap's -ECANCELED made %d hand-offs, want 1", n) + } } // TestReapedRecvLeavesNoLinkOrBuffer (celeris#681 C4): reapOutcome consumes diff --git a/engine/iouring/probe.go b/engine/iouring/probe.go index 7e47eba8..293b22a2 100644 --- a/engine/iouring/probe.go +++ b/engine/iouring/probe.go @@ -485,6 +485,14 @@ const ( // at submit on every kernel measured, so its completion is normally there // when Submit returns; a short wait covers any that is not. func probeAsyncCancelFlags() (bool, string) { + return probeAsyncCancel(cancelAll) +} + +// probeAsyncCancel is probeAsyncCancelFlags with the cancel_flags word +// given: the probe passes the reap's own (IORING_ASYNC_CANCEL_ALL), and a +// test passes a bit no kernel defines, which every kernel rejects, to drive +// the rejection through the same submit and completion path. +func probeAsyncCancel(cancelFlags uint32) (bool, string) { ring, err := NewRing(8, 0, 0) if err != nil { return false, "NewRing failed: " + err.Error() @@ -496,6 +504,7 @@ func probeAsyncCancelFlags() (bool, string) { return false, "GetSQE returned nil" } prepCancelUserDataReported(sqe, asyncCancelProbeTarget) + *(*uint32)(unsafe.Pointer(&(*[sqeSize]byte)(sqe)[28])) = cancelFlags setSQEUserData(sqe, asyncCancelProbeTag) if _, err := ring.Submit(); err != nil { return false, "Submit failed: " + err.Error() From 16c9719df12794fc0135b8a783f8d85e7357ce4c Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 20:43:38 +0200 Subject: [PATCH 18/34] fix(iouring): claim no async hand-off at a park on a worker that cannot reap (celeris#657) Review R1 on #681. Where the kernel rejects IORING_ASYNC_CANCEL flags, a promoted async conn under a reverse drain never left io_uring, yet its dispatch goroutine claimed the hand-off at every park: the worker's drain of the claim found the recv the feed path had armed, which only a reap can clear, and refused; the next request respawned the goroutine. A spawn and a detach-queue round trip per request, with no end while the drain lasted. asyncTransplantEligible now refuses on a worker without cancel flags, so the goroutine parks as it does with no drain set and the conn stays on io_uring (placement only). asyncCancelFlags is set before the worker starts and never written again, so the dispatch goroutine reads it race-free. Tests: TestNoReapWithoutAsyncCancelFlags/async_park_claims_nothing runs the real dispatch goroutine through three requests with the drain set and requires no claim and one goroutine for all three; on 48dfe1a it fails with 3 of 3 parks claimed, 3 goroutines and TransplantReapUnsupported = 3. Its control (control_async_park_claims_with_flags) shows the same rig sees a claim, a reap and the hand-off where the worker can reap. TestReapSuppressedAfterFailedHandOff/async_conn_retries_at_its_next_park pins the async side of a failed dup: the data that respawns the goroutine clears reapSuppressed before any park, so the hand-off is retried at the next park, once per request, and the conn leaves once the dup works. --- engine/iouring/fd_lifetime.go | 5 +- engine/iouring/fd_lifetime_fixture_test.go | 121 +++++++++++++++++++++ engine/iouring/fd_lifetime_test.go | 112 +++++++++++++++++++ engine/iouring/transplant_source.go | 15 +++ engine/iouring/worker.go | 2 + 5 files changed, 254 insertions(+), 1 deletion(-) diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go index 191c43d4..a64ceaa4 100644 --- a/engine/iouring/fd_lifetime.go +++ b/engine/iouring/fd_lifetime.go @@ -66,7 +66,10 @@ func onlyRecvInFlight(cs *connState) bool { // moves, so no request can be lost). The conditions: // - the kernel rejects IORING_ASYNC_CANCEL flags (before 5.19, found by // probeAsyncCancelFlags): every reap would fail with -EINVAL and leave -// the recv armed. Counted (TransplantReapUnsupported). +// the recv armed. Counted (TransplantReapUnsupported). A promoted async +// conn never gets here on such a worker: its dispatch goroutine makes no +// claim (asyncTransplantEligible) and stays parked, still running, so +// tryTransplant leaves it alone too (celeris#681 R1). // - the conn's last hand-off failed at its dup (reapSuppressed): reaping // the recv re-armed after that failure would fail the same way at once. func (w *Worker) startReap(cs *connState) { diff --git a/engine/iouring/fd_lifetime_fixture_test.go b/engine/iouring/fd_lifetime_fixture_test.go index 9754ba4f..7d8b2cce 100644 --- a/engine/iouring/fd_lifetime_fixture_test.go +++ b/engine/iouring/fd_lifetime_fixture_test.go @@ -6,9 +6,14 @@ import ( "context" "errors" "reflect" + "regexp" + "runtime" "strconv" + "strings" + "sync" "sync/atomic" "testing" + "time" "unsafe" "golang.org/x/sys/unix" @@ -286,3 +291,119 @@ func fdIsOpen(fd int) bool { _, err := unix.FcntlInt(uintptr(fd), unix.F_GETFD, 0) return err == nil } + +// goidHandler is fdlHandler recording the goroutine each request ran on. A +// promoted async conn runs its requests on its dispatch goroutine, so a +// goroutine that exits at its park and is respawned by the next request shows +// up as a new id (celeris#681 R1). +type goidHandler struct { + mu sync.Mutex + ids []uint64 +} + +func (h *goidHandler) HandleStream(ctx context.Context, s *stream.Stream) error { + h.mu.Lock() + h.ids = append(h.ids, goroutineID()) + h.mu.Unlock() + return fdlHandler{}.HandleStream(ctx, s) +} + +func (h *goidHandler) served() []uint64 { + h.mu.Lock() + defer h.mu.Unlock() + return append([]uint64(nil), h.ids...) +} + +// goroutineID is the running goroutine's id, from its stack header +// ("goroutine 18 [running]:"). +func goroutineID() uint64 { + var b [64]byte + s := strings.TrimPrefix(string(b[:runtime.Stack(b[:], false)]), "goroutine ") + id, _ := strconv.ParseUint(s[:strings.IndexByte(s, ' ')], 10, 64) + return id +} + +// goroutineWaitsOnCond reports whether goroutine id is blocked in +// sync.Cond.Wait, which is where a dispatch goroutine parks. +func goroutineWaitsOnCond(id uint64) bool { + buf := make([]byte, 1<<16) + for { + n := runtime.Stack(buf, true) + if n < len(buf) { + buf = buf[:n] + break + } + buf = make([]byte, 2*len(buf)) + } + return regexp.MustCompile(`(?m)^goroutine ` + strconv.FormatUint(id, 10) + ` \[sync\.Cond\.Wait`).Match(buf) +} + +// asyncParkFixture is fdlFixture's conn promoted to async dispatch and served +// by h, on a worker whose asyncCancelFlags is flags, with its first recv armed +// and a drain set: every request delivered now runs on the conn's dispatch +// goroutine, which writes the response itself and then reaches its park. The +// cleanup ends that goroutine and waits for it before the fixture closes the +// descriptors. +func asyncParkFixture(t *testing.T, flags bool) (*fdlFixture, *goidHandler) { + t.Helper() + f := newFDLFixture(t, true) + h := &goidHandler{} + f.w.handler = h + trySetWorkerField(f.w, "asyncCancelFlags", flags) + f.cs.asyncPromoted.Store(true) + f.armFirstRecv() + f.startDrain() + t.Cleanup(func() { + f.cs.asyncInMu.Lock() + f.cs.asyncClosed.Store(true) + f.cs.asyncCond.Broadcast() + f.cs.asyncInMu.Unlock() + f.w.asyncWG.Wait() + }) + return f, h +} + +// asyncRequest delivers one request to f's promoted conn, reads its response +// at the client, and waits until the dispatch goroutine that answered it has +// reached its park: there it either claimed its own hand-off +// (transplantPending, enqueued for the worker, and exited) or waits on its +// cond for the next request. It reports which, and the SQEs the delivery +// placed. +func (f *fdlFixture) asyncRequest(h *goidHandler) (claimed bool, placed []sqeRec) { + f.t.Helper() + n := len(h.served()) + f.deliver(fdlGET) + placed = takeSQEs(f.w.ring) + fds := []unix.PollFd{{Fd: int32(f.peer), Events: unix.POLLIN}} + if k, err := unix.Poll(fds, 10000); err != nil || k != 1 { + f.t.Fatalf("no response at the client within 10s (poll %d, %v)", k, err) + } + var resp [512]byte + if k, err := unix.Read(f.peer, resp[:]); err != nil || !strings.HasPrefix(string(resp[:max(k, 0)]), "HTTP/1.1 200") { + f.t.Fatalf("client read %q, %v: want a 200 response", resp[:max(k, 0)], err) + } + ids := h.served() + if len(ids) != n+1 { + f.t.Fatalf("the handler ran %d times for one request", len(ids)-n) + } + for deadline := time.Now().Add(10 * time.Second); ; time.Sleep(time.Millisecond) { + if f.cs.transplantPending.Load() || f.w.detachQPending.Load() != 0 { + return true, placed + } + if goroutineWaitsOnCond(ids[n]) { + return false, placed + } + if time.Now().After(deadline) { + f.t.Fatal("the dispatch goroutine neither claimed its hand-off nor parked within 10s of its response") + } + } +} + +// distinct counts the distinct values in ids. +func distinct(ids []uint64) int { + seen := map[uint64]bool{} + for _, id := range ids { + seen[id] = true + } + return len(seen) +} diff --git a/engine/iouring/fd_lifetime_test.go b/engine/iouring/fd_lifetime_test.go index 0068f597..65951e25 100644 --- a/engine/iouring/fd_lifetime_test.go +++ b/engine/iouring/fd_lifetime_test.go @@ -762,6 +762,67 @@ func TestNoReapWithoutAsyncCancelFlags(t *testing.T) { t.Errorf("TransplantReaps = %d, want 0", n) } }) + + // celeris#681 R1: the dispatch goroutine itself, run for real through + // three requests with the drain set. On a worker that cannot reap, a + // claim can only be refused (the recv the feed path armed after the + // request cannot be reaped, and a promoted conn is never held), and the + // goroutine that exited to make it is respawned by the next request: a + // spawn and a detach-queue round trip per request for as long as the drain + // lasts, with the conn never leaving. So the goroutine must not claim: it + // parks, and one goroutine serves every request. + t.Run("async_park_claims_nothing", func(t *testing.T) { + f, h := asyncParkFixture(t, false) + claims := 0 + for i := 1; i <= 3; i++ { + claimed, placed := f.asyncRequest(h) + if len(placed) != 1 || placed[0].op != opRECV || placed[0].tag() != udRecv { + t.Fatalf("request %d placed %v, want only the feed path's RECV", i, placed) + } + if claimed { + claims++ + } + f.w.drainDetachQueue() // the worker's next loop iteration + if sqes := takeSQEs(f.w.ring); len(sqes) != 0 { + t.Fatalf("the loop iteration after request %d placed %v, want nothing", i, sqes) + } + } + if got := distinct(h.served()); claims != 0 || got != 1 { + t.Errorf("the dispatch goroutine claimed the hand-off at %d of 3 parks, and the 3 requests ran on %d "+ + "goroutines, want 0 claims and 1 goroutine: on a worker that cannot reap every claim is refused and "+ + "the next request respawns the goroutine", claims, got) + } + if n := metric(t, f.e, "TransplantReapUnsupported"); n != 0 { + t.Errorf("TransplantReapUnsupported = %d, want 0: no claim was made, so no reap was refused", n) + } + if n := metric(t, f.e, "TransplantClaimDeferred"); n != 0 { + t.Errorf("TransplantClaimDeferred = %d, want 0", n) + } + if f.tgt.adopted.Load() != 0 || f.w.conns[f.fd] != f.cs || !f.cs.recvArmed { + t.Fatalf("adopted %d, in table %v, recvArmed %v: want 0, true, true (the conn stays, reading)", + f.tgt.adopted.Load(), f.w.conns[f.fd] == f.cs, f.cs.recvArmed) + } + }) + + // The control for the subtest above: the same park on a worker that can + // reap claims the hand-off, the worker's drain of the claim reaps the + // feed path's recv, and the conn leaves at that recv's -ECANCELED. It + // shows that asyncRequest sees a claim when one is made. + t.Run("control_async_park_claims_with_flags", func(t *testing.T) { + f, h := asyncParkFixture(t, true) + claimed, _ := f.asyncRequest(h) + if !claimed { + t.Fatal("the dispatch goroutine parked without claiming the hand-off on a worker that can reap") + } + f.w.drainDetachQueue() + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("the drain of the claim placed %v, want the reap of the feed path's recv", sqes) + } + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the reap's -ECANCELED made %d hand-offs, want 1", n) + } + }) } // TestReapSuppressedAfterFailedHandOff (celeris#681 C2): a hand-off that @@ -831,6 +892,57 @@ func TestReapSuppressedAfterFailedHandOff(t *testing.T) { if n := f.tgt.adopted.Load(); n != 1 { t.Fatalf("the reap's -ECANCELED made %d hand-offs, want 1", n) } + + // A promoted async conn (celeris#681 R1). Its hand-off is reached only + // through its dispatch goroutine's claim at a park, and the goroutine runs + // only on data, which lifts the suppression first: at a park the flag is + // never set. So a hand-off that failed at its dup is retried at the + // conn's next park, one reap and one dup per request while the dup keeps + // failing (a sync conn tries the dup once per request too, at its held + // response), and the conn leaves at the first park after the dup works. + t.Run("async_conn_retries_at_its_next_park", func(t *testing.T) { + f, h := asyncParkFixture(t, true) + dupCalls := 0 + trySetWorkerField(f.w, "dupFD", func(int) (int, error) { + dupCalls++ + return -1, unix.EMFILE + }) + for i := 1; i <= 2; i++ { + if claimed, _ := f.asyncRequest(h); !claimed { + t.Fatalf("request %d: the dispatch goroutine parked without claiming the hand-off", i) + } + if f.cs.reapSuppressed { + t.Fatalf("request %d: reapSuppressed is set at the park", i) + } + f.w.drainDetachQueue() + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("request %d: the drain of the claim placed %v, want the reap of the feed path's recv", i, sqes) + } + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if f.tgt.adopted.Load() != 0 || !f.cs.reapSuppressed { + t.Fatalf("request %d: adopted %d, reapSuppressed %v after the dup failed: want 0 and true", + i, f.tgt.adopted.Load(), f.cs.reapSuppressed) + } + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || sqes[0].op != opRECV { + t.Fatalf("request %d: after the failed hand-off the completion placed %v, want only the recv re-armed", i, sqes) + } + } + if dupCalls != 2 { + t.Errorf("dup tried %d times, want 2: once per request", dupCalls) + } + trySetWorkerField(f.w, "dupFD", (func(int) (int, error))(nil)) + if claimed, _ := f.asyncRequest(h); !claimed { + t.Fatal("the dispatch goroutine parked without claiming the hand-off once the dup works") + } + f.w.drainDetachQueue() + if sqes := takeSQEs(f.w.ring); len(sqes) != 1 || !f.isReap(sqes[0]) { + t.Fatalf("the drain of the claim placed %v, want the reap", sqes) + } + f.process(f.recvCQE(-int32(unix.ECANCELED))) + if n := f.tgt.adopted.Load(); n != 1 { + t.Fatalf("the reap's -ECANCELED made %d hand-offs once the dup works, want 1", n) + } + }) } // TestReapedRecvLeavesNoLinkOrBuffer (celeris#681 C4): reapOutcome consumes diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 02da465d..813aa96e 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -236,7 +236,22 @@ func (w *Worker) reclaimTransplant(newFD int, carry engine.Carryover, cause erro // which empties writeBuf while sendBuf is still in flight (celeris#529). The // egress check is therefore re-done in finishAsyncTransplant, on the worker // thread, and this predicate is only a cheap first filter. +// +// A worker that cannot reap (the kernel rejects IORING_ASYNC_CANCEL flags, +// probeAsyncCancelFlags) offers no promoted async conn for the hand-off +// (celeris#681 R1). Every claim would find the recv the feed path armed after +// the last request, which only a reap can clear, and a promoted conn is never +// held: finishAsyncTransplant would refuse it, and the goroutine that exited +// to make the claim would be respawned by the next request — a spawn and a +// detach-queue round trip per request for as long as the drain lasted, with +// the conn never leaving. Instead the goroutine parks as it does with no +// drain set, and the conn stays on io_uring (placement only). The field is +// set before the worker starts and never written again, so this read from +// the dispatch goroutine is race-free. func (w *Worker) asyncTransplantEligible(cs *connState) bool { + if !w.asyncCancelFlags { + return false + } if cs.fixedFile { return false } diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index b6a4b833..69fa5845 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -4096,6 +4096,8 @@ func (w *Worker) runAsyncHandler(cs *connState) { // gone, so there is no goroutine-vs-release race. If a new request // arrives first, the feed path clears transplantPending and respawns // us, so no request is lost. SINGLE_ISSUER: we submit no SQE here. + // On a worker that cannot reap, no conn is eligible: the claim + // could only be refused, so we park instead (celeris#681 R1). if w.transplant.Load() != nil && w.asyncTransplantEligible(cs) { cs.transplantPending.Store(true) cs.asyncRun = false From 5e9c7f0a3a59fbe49aec65e9cdbc3e2e557e6dbc Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 20:47:48 +0200 Subject: [PATCH 19/34] fix(iouring): tell a cancel-flags probe with no answer from a kernel rejection (celeris#657) Review R2 on #681. probeAsyncCancelFlags cached a probe that never reached the kernel (a ring that could not be set up, a failed submit or wait, a missing or foreign completion) exactly as it cached -EINVAL, and New logged both at Info as "probe failed". A rejection is the kernel's answer; a probe with no answer says nothing about the kernel, and on 5.19 or later it turns the hand-off's reap off where it should work. The probe now returns asyncCancelAccepted, asyncCancelRejected (-EINVAL, or a result it does not recognise) or asyncCancelNoAnswer. Only accepted turns the reap on. logAsyncCancelProbe logs a rejection at Info and no answer at Warn when the kernel's version is 5.19 or later (Info below it), each with the reason and the kernel; the engine-selected line also carries async_cancel_probe. The probe's ring is made through newAsyncCancelProbeRing, a variable, so a test can make the probe fail before the kernel answers. The docs that said the next response is then HELD now state HOLD's condition: a worker without a provided-buffer ring. A kernel that rejects the flags has none (buffer rings arrived in 5.19 too); where one exists because the probe got no answer, a sync conn is not held and stays on io_uring. Tests: TestWorkersCarryTheAsyncCancelProbe/no_answer_is_logged_apart_from_a_rejection runs New with the probe's ring failing: on 48dfe1a plus only the ring seam it fails, the record at INFO reading "async cancel flags runtime probe failed", want WARN. TestAsyncCancelProbeClassifies gains no_answer_is_not_a_rejection and log_levels (the level matrix). --- engine/iouring/async_cancel_probe_log_test.go | 103 ++++++++++++++++++ engine/iouring/async_cancel_probe_test.go | 99 +++++++++++++---- engine/iouring/engine.go | 53 +++++++-- engine/iouring/fd_lifetime.go | 65 ++++++----- engine/iouring/handoff_loss.go | 9 +- engine/iouring/probe.go | 79 ++++++++++---- engine/iouring/worker.go | 5 +- 7 files changed, 326 insertions(+), 87 deletions(-) create mode 100644 engine/iouring/async_cancel_probe_log_test.go diff --git a/engine/iouring/async_cancel_probe_log_test.go b/engine/iouring/async_cancel_probe_log_test.go new file mode 100644 index 00000000..16b02a5a --- /dev/null +++ b/engine/iouring/async_cancel_probe_log_test.go @@ -0,0 +1,103 @@ +//go:build linux + +package iouring + +import ( + "bytes" + "encoding/json" + "errors" + "log/slog" + "strings" + "sync" + "testing" + + "github.com/goceleris/celeris/engine" + "github.com/goceleris/celeris/probe" + "github.com/goceleris/celeris/resource" +) + +// lockedBuffer is a bytes.Buffer an slog handler can write from any +// goroutine. +type lockedBuffer struct { + mu sync.Mutex + b bytes.Buffer +} + +func (l *lockedBuffer) Write(p []byte) (int, error) { + l.mu.Lock() + defer l.mu.Unlock() + return l.b.Write(p) +} + +// records decodes the JSON log lines written so far. +func (l *lockedBuffer) records(t *testing.T) []map[string]any { + t.Helper() + l.mu.Lock() + defer l.mu.Unlock() + var out []map[string]any + for _, line := range strings.Split(strings.TrimSpace(l.b.String()), "\n") { + if line == "" { + continue + } + var r map[string]any + if err := json.Unmarshal([]byte(line), &r); err != nil { + t.Fatalf("log line %q: %v", line, err) + } + out = append(out, r) + } + return out +} + +// newWarnsWhenTheProbeGetsNoAnswer (celeris#681 R2): New on an engine whose +// async-cancel-flags probe fails before the kernel answers. The reap stays +// off, as for a rejection, but a probe with no answer is not the kernel's +// answer and must not read as one: its record says the probe got no answer, +// names the probe's failure, and on a kernel whose version has the flags +// (5.19 and later) it is a Warn, since the reap is then off where it should +// work. Only the probe-ring seam and API older than round 3 are used here. +func newWarnsWhenTheProbeGetsNoAnswer(t *testing.T) { + p := probe.Probe() + want := "INFO" + if p.KernelMajor > 5 || (p.KernelMajor == 5 && p.KernelMinor >= 19) { + want = "WARN" + } + saved := newAsyncCancelProbeRing + newAsyncCancelProbeRing = func() (*Ring, error) { return nil, errors.New("celeris681 injected: no ring") } + cachedAsyncCancel = sync.Once{} + t.Cleanup(func() { + newAsyncCancelProbeRing = saved + cachedAsyncCancel = sync.Once{} // the next New probes the kernel again + }) + var buf lockedBuffer + e, err := New(resource.Config{ + Addr: "127.0.0.1:0", + Protocol: engine.HTTP1, + Logger: slog.New(slog.NewJSONHandler(&buf, nil)), + }, transplantTestHandler{}) + if err != nil { + skipOrFail656(t, "iouring engine unavailable: %v", err) + } + if e.asyncCancelFlags { + t.Fatal("New turned the hand-off's reap on although the probe got no answer") + } + var probeRecs []map[string]any + for _, r := range buf.records(t) { + if msg, _ := r["msg"].(string); strings.HasPrefix(msg, "async cancel flags") { + probeRecs = append(probeRecs, r) + } + } + if len(probeRecs) != 1 { + t.Fatalf("New logged %d async-cancel-flags probe records, want 1: %v", len(probeRecs), probeRecs) + } + r := probeRecs[0] + msg, _ := r["msg"].(string) + level, _ := r["level"].(string) + reason, _ := r["reason"].(string) + t.Logf("celeris681 probe record on kernel %d.%d: level=%s msg=%q reason=%q", p.KernelMajor, p.KernelMinor, + level, msg, reason) + if level != want || !strings.Contains(msg, "no answer") || !strings.Contains(reason, "injected") { + t.Errorf("a probe that got no answer on kernel %d.%d was logged at %s as %q (reason %q), want %s, saying "+ + "the probe got no answer and naming its failure: a failed probe must not read as the kernel "+ + "rejecting the flags", p.KernelMajor, p.KernelMinor, level, msg, reason, want) + } +} diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index 85399439..ba214a10 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -4,6 +4,8 @@ package iouring import ( "context" + "errors" + "log/slog" "strings" "testing" "time" @@ -22,26 +24,75 @@ import ( // startup and places no reap where they are rejected. // TestAsyncCancelProbeClassifies pins how the probe reads its cancel's -// completion: a cancel with CANCEL_ALL of a user_data nothing carries. +// completion: a cancel with CANCEL_ALL of a user_data nothing carries. A +// completion is the kernel's answer, accepted or rejected; a probe that fails +// before reading one has no answer, which is not a rejection (celeris#681 R2). func TestAsyncCancelProbeClassifies(t *testing.T) { for _, tc := range []struct { res int32 - ok bool + want asyncCancelProbe reason string }{ - {0, true, ""}, // CANCEL_ALL: res is the number cancelled, and nothing was - {1, true, ""}, - {-int32(unix.ENOENT), true, ""}, // the flag-free form's miss - {-int32(unix.EINVAL), false, "5.19"}, - {-int32(unix.EBADF), false, "cqe.res=-9"}, - {-int32(unix.ECANCELED), false, "cqe.res=-125"}, + {0, asyncCancelAccepted, ""}, // CANCEL_ALL: res is the number cancelled, and nothing was + {1, asyncCancelAccepted, ""}, + {-int32(unix.ENOENT), asyncCancelAccepted, ""}, // the flag-free form's miss + {-int32(unix.EINVAL), asyncCancelRejected, "5.19"}, + {-int32(unix.EBADF), asyncCancelRejected, "cqe.res=-9"}, + {-int32(unix.ECANCELED), asyncCancelRejected, "cqe.res=-125"}, } { - ok, reason := classifyAsyncCancelProbe(tc.res) - if ok != tc.ok || (tc.reason == "") != (reason == "") || !strings.Contains(reason, tc.reason) { + got, reason := classifyAsyncCancelProbe(tc.res) + if got != tc.want || (tc.reason == "") != (reason == "") || !strings.Contains(reason, tc.reason) { t.Errorf("classifyAsyncCancelProbe(%d) = (%v, %q), want (%v, containing %q)", - tc.res, ok, reason, tc.ok, tc.reason) + tc.res, got, reason, tc.want, tc.reason) } } + + // The probe's ring cannot be set up: nothing reached the kernel. + t.Run("no_answer_is_not_a_rejection", func(t *testing.T) { + saved := newAsyncCancelProbeRing + newAsyncCancelProbeRing = func() (*Ring, error) { return nil, errors.New("celeris681 injected: no ring") } + t.Cleanup(func() { newAsyncCancelProbeRing = saved }) + got, reason := probeAsyncCancel(cancelAll) + if got != asyncCancelNoAnswer || !strings.Contains(reason, "injected") { + t.Errorf("a probe whose ring could not be set up = (%v, %q), want (%v, naming the failure)", + got, reason, asyncCancelNoAnswer) + } + }) + + // New's record of the answer: nothing when accepted, Info for the + // kernel's rejection, and for no answer Warn where the kernel's version + // has the flags and Info below it. + t.Run("log_levels", func(t *testing.T) { + for _, tc := range []struct { + p asyncCancelProbe + major, minor int + want string + }{ + {asyncCancelAccepted, 6, 8, ""}, + {asyncCancelRejected, 5, 15, "INFO"}, + {asyncCancelRejected, 6, 8, "INFO"}, + {asyncCancelNoAnswer, 5, 18, "INFO"}, + {asyncCancelNoAnswer, 5, 19, "WARN"}, + {asyncCancelNoAnswer, 6, 8, "WARN"}, + } { + var buf lockedBuffer + logAsyncCancelProbe(slog.New(slog.NewJSONHandler(&buf, nil)), tc.p, "why", tc.major, tc.minor) + recs := buf.records(t) + got := "" + if len(recs) == 1 { + got, _ = recs[0]["level"].(string) + } + if len(recs) > 1 || got != tc.want { + t.Errorf("%v on kernel %d.%d logged %v, want one record at %q (none when empty)", + tc.p, tc.major, tc.minor, recs, tc.want) + } + if tc.p == asyncCancelNoAnswer && len(recs) == 1 { + if msg, _ := recs[0]["msg"].(string); !strings.Contains(msg, "no answer") { + t.Errorf("the no-answer record reads %q, want it to say the probe got no answer", msg) + } + } + } + }) } // kernelHasCancelFlags reports whether the running kernel's version is one @@ -71,16 +122,17 @@ func TestAsyncCancelProbeOnThisKernel(t *testing.T) { skipOrFail656(t, "io_uring unavailable: %v", err) } _ = r.Close() - ok, reason := probeAsyncCancelFlags() + res, reason := probeAsyncCancelFlags() + ok := res == asyncCancelAccepted want, rel := kernelHasCancelFlags(t) - t.Logf("celeris681 async cancel flags probe: kernel=%s ok=%v reason=%q", rel, ok, reason) + t.Logf("celeris681 async cancel flags probe: kernel=%s result=%v reason=%q", rel, res, reason) if ok != want { - t.Fatalf("probeAsyncCancelFlags() = (%v, %q) on kernel %s, want %v", ok, reason, rel, want) + t.Fatalf("probeAsyncCancelFlags() = (%v, %q) on kernel %s, want accepted=%v", res, reason, rel, want) } - if cok, _ := probeAsyncCancelFlagsCached(); cok != ok { - t.Fatalf("the cached probe says %v, the probe %v", cok, ok) + if cres, _ := probeAsyncCancelFlagsCached(); cres != res { + t.Fatalf("the cached probe says %v, the probe %v", cres, res) } - if bad, why := probeAsyncCancel(1 << 31); bad || !strings.Contains(why, "EINVAL") { + if bad, why := probeAsyncCancel(1 << 31); bad != asyncCancelRejected || !strings.Contains(why, "EINVAL") { t.Fatalf("a cancel flag no kernel defines was read as (%v, %q), want rejected with EINVAL: "+ "the probe does not read the kernel's answer", bad, why) } @@ -99,8 +151,8 @@ func TestWorkersCarryTheAsyncCancelProbe(t *testing.T) { if err != nil { skipOrFail656(t, "iouring engine unavailable: %v", err) } - if want, _ := probeAsyncCancelFlagsCached(); e.asyncCancelFlags != want { - t.Fatalf("New stored asyncCancelFlags=%v, the probe says %v", e.asyncCancelFlags, want) + if res, _ := probeAsyncCancelFlagsCached(); e.asyncCancelFlags != (res == asyncCancelAccepted) { + t.Fatalf("New stored asyncCancelFlags=%v, the probe says %v", e.asyncCancelFlags, res) } resolved := e.cfg.Resources.Resolve() for _, answer := range []bool{false, true} { @@ -118,6 +170,10 @@ func TestWorkersCarryTheAsyncCancelProbe(t *testing.T) { w.shutdown() } } + + // A probe that fails before the kernel answers keeps the reap off too, + // and New reports it apart from a rejection (celeris#681 R2). + t.Run("no_answer_is_logged_apart_from_a_rejection", newWarnsWhenTheProbeGetsNoAnswer) } // TestReapOnTheRunningKernel drives the hand-off of an idle conn whose recv @@ -196,9 +252,10 @@ func TestReapOnTheRunningKernel(t *testing.T) { } t.Run("probe_answer", func(t *testing.T) { - ok, _ := probeAsyncCancelFlagsCached() + res, _ := probeAsyncCancelFlagsCached() + ok := res == asyncCancelAccepted want, rel := kernelHasCancelFlags(t) - t.Logf("celeris681 kernel=%s probe=%v", rel, ok) + t.Logf("celeris681 kernel=%s probe=%v", rel, res) if ok != want { t.Fatalf("the probe says %v on kernel %s, want %v", ok, rel, want) } diff --git a/engine/iouring/engine.go b/engine/iouring/engine.go index d3ac50c1..4bec3046 100644 --- a/engine/iouring/engine.go +++ b/engine/iouring/engine.go @@ -5,6 +5,7 @@ package iouring import ( "context" "fmt" + "log/slog" "net" "os" "sync" @@ -91,9 +92,10 @@ type Engine struct { // at construction so Metrics() doesn't pay the type-assertion per // call. Snapshot-at-Listen — static post-Start. asyncRoutes int - // asyncCancelFlags is the probeAsyncCancelFlags answer: whether this - // kernel accepts IORING_ASYNC_CANCEL_* flags (5.19+). Every worker gets - // a copy; the hand-off's REAP needs them (celeris#657). + // asyncCancelFlags is whether probeAsyncCancelFlags found this kernel + // accepting IORING_ASYNC_CANCEL_* flags (5.19+): false when the kernel + // rejected them and when the probe got no answer. Every worker gets a + // copy; the hand-off's REAP needs them (celeris#657). asyncCancelFlags bool } @@ -168,15 +170,19 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { // The io_uring→epoll hand-off cancels an armed recv before it moves a // connection (REAP, celeris#657), and that cancel needs - // IORING_ASYNC_CANCEL flags, which exist from Linux 5.19. Where the - // kernel rejects them the hand-off never reaps: a connection whose recv - // is armed stays on io_uring until that recv completes on its own, and - // its next response is then HELD and handed off with nothing in flight. - asyncCancelFlags, acReason := probeAsyncCancelFlagsCached() - if !asyncCancelFlags { - cfg.Logger.Info("async cancel flags runtime probe failed: the io_uring→epoll hand-off will not cancel an armed recv (celeris#657)", - "reason", acReason) - } + // IORING_ASYNC_CANCEL flags, which exist from Linux 5.19. Unless the + // probe finds them accepted the hand-off never reaps: a sync connection + // whose recv is armed stays on io_uring until that recv completes on its + // own, and its next response is then HELD and handed off with nothing in + // flight. HOLD needs a worker without a provided-buffer ring. A kernel + // that rejects the flags has none (buffer rings arrived in the same + // release, 5.19); where one exists because the probe got no answer on a + // newer kernel, the connection is not held and stays on io_uring. A + // promoted async connection is never offered for the hand-off without + // the flags and stays too. Placement only, either way. + asyncCancel, acReason := probeAsyncCancelFlagsCached() + asyncCancelFlags := asyncCancel == asyncCancelAccepted + logAsyncCancelProbe(cfg.Logger, asyncCancel, acReason, profile.KernelMajor, profile.KernelMinor) tier := SelectTier(profile, 2*time.Second) if tier == nil { @@ -194,6 +200,7 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { "fixed_files", fixedFilesEnabled(tier.SupportsFixedFiles()), "send_zc", tier.SupportsSendZC(), "async_cancel_flags", asyncCancelFlags, + "async_cancel_probe", asyncCancel.String(), ) e := &Engine{ @@ -211,6 +218,28 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { return e, nil } +// logAsyncCancelProbe reports an async-cancel-flags probe that did not find +// the flags accepted (celeris#681 R2). A rejection is the kernel's answer and +// expected before 5.19: Info. A probe that got no answer says nothing about +// the kernel; on one whose version has the flags (5.19 and later) it is +// unexpected, and it keeps the hand-off's reap off for this process, so it is +// a Warn there and Info below. +func logAsyncCancelProbe(l *slog.Logger, p asyncCancelProbe, reason string, kernelMajor, kernelMinor int) { + kernel := fmt.Sprintf("%d.%d", kernelMajor, kernelMinor) + switch p { + case asyncCancelRejected: + l.Info("async cancel flags rejected by the kernel: the io_uring→epoll hand-off will not cancel an armed recv (celeris#657)", + "reason", reason, "kernel", kernel) + case asyncCancelNoAnswer: + msg := "async cancel flags probe got no answer from the kernel: the io_uring→epoll hand-off will not cancel an armed recv (celeris#657)" + if kernelMajor > 5 || (kernelMajor == 5 && kernelMinor >= 19) { + l.Warn(msg, "reason", reason, "kernel", kernel) + } else { + l.Info(msg, "reason", reason, "kernel", kernel) + } + } +} + // Listen starts the io_uring engine and blocks until context is canceled. func (e *Engine) Listen(ctx context.Context) error { // If a listener was provided (StartWithListener), use its bound address diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go index a64ceaa4..1fc2ec11 100644 --- a/engine/iouring/fd_lifetime.go +++ b/engine/iouring/fd_lifetime.go @@ -28,23 +28,26 @@ import ( // owner) under a tag of its own, and hands off at the recv's -ECANCELED — // the recv's own terminal completion, not the cancel's result. If the // request arrives first, the recv completes with it: it is served here -// and its response is HELD. If the cancel reports that it matched -// nothing (the recv completed first, or is still linked behind its SEND -// and not issued), the reap is retried on the next loop iteration while -// the recv is still armed. A miss is never followed by a hand-off, and -// neither is any other result: a cancel that fails outright is counted -// and not retried. A reap needs IORING_ASYNC_CANCEL flags (Linux 5.19); -// on a kernel that rejects them (probeAsyncCancelFlags) none is placed, -// and the conn stays until its recv completes on its own (startReap). -// None is placed either after a hand-off of the conn failed at its dup, -// until the conn next receives data. -// - HOLD. While a drain is set, the response of a connection the hand-off -// would accept is flushed with no recv behind it (not linked, not -// standalone). Its SEND completion then finds nothing in flight and hands -// the conn off; the client's next request waits in the socket buffer for -// epoll. releaseHold arms the recv at that completion whenever the -// hand-off does not happen, and checkTimeouts rescues (and counts) any -// held conn a path left unreleased. +// and its response is HELD (on a worker with a provided-buffer ring, +// which has no HOLD, its next recv is reaped instead). If the cancel +// reports that it matched nothing (the recv completed first, or is still +// linked behind its SEND and not issued), the reap is retried on the +// next loop iteration while the recv is still armed. A miss is never +// followed by a hand-off, and neither is any other result: a cancel that +// fails outright is counted and not retried. A reap needs +// IORING_ASYNC_CANCEL flags (Linux 5.19); unless the startup probe found +// them accepted (probeAsyncCancelFlags) none is placed, and the conn +// stays until its recv completes on its own (startReap). None is placed +// either after a hand-off of the conn failed at its dup, until the conn +// next receives data. +// - HOLD. While a drain is set, on a worker without a provided-buffer +// ring, the response of a connection the hand-off would accept is +// flushed with no recv behind it (not linked, not standalone). Its SEND +// completion then finds nothing in flight and hands the conn off; the +// client's next request waits in the socket buffer for epoll. +// releaseHold arms the recv at that completion whenever the hand-off +// does not happen, and checkTimeouts rescues (and counts) any held conn +// a path left unreleased. // onlyRecvInFlight reports whether the one op the R0 gate found in flight is // the recv — the case REAP can clear. Anything else (a send, a SEND_ZC @@ -59,17 +62,23 @@ func onlyRecvInFlight(cs *connState) bool { // the next iteration's retry. // // Two conditions place no reap and queue no retry. The conn keeps its recv -// armed and stays until that recv completes on its own: a sync conn then has -// the request it brings served here and its response HELD, and leaves at that -// SEND's completion with nothing in flight; a promoted async conn, which is -// never held, stays on io_uring (placement only: nothing is in flight when it -// moves, so no request can be lost). The conditions: -// - the kernel rejects IORING_ASYNC_CANCEL flags (before 5.19, found by -// probeAsyncCancelFlags): every reap would fail with -EINVAL and leave -// the recv armed. Counted (TransplantReapUnsupported). A promoted async -// conn never gets here on such a worker: its dispatch goroutine makes no -// claim (asyncTransplantEligible) and stays parked, still running, so -// tryTransplant leaves it alone too (celeris#681 R1). +// armed and stays until that recv completes on its own. A sync conn then has +// the request it brings served here; on a worker without a provided-buffer +// ring (HOLD needs none) its response is HELD and it leaves at that SEND's +// completion with nothing in flight, and on one with a ring it stays on +// io_uring. A promoted async conn, which is never held, stays on io_uring. +// Placement only: nothing is in flight when a conn moves, so no request can +// be lost. The conditions: +// - the startup probe did not find IORING_ASYNC_CANCEL flags accepted +// (probeAsyncCancelFlags): a kernel before 5.19 rejects them, and every +// reap would fail with -EINVAL and leave the recv armed; a probe that got +// no answer is treated the same. Counted (TransplantReapUnsupported). +// Buffer rings arrived with the flags, in 5.19, so a kernel that rejects +// them has no provided-buffer ring and its sync conns are held; a newer +// kernel whose probe got no answer may have one, and its conns stay. A +// promoted async conn never gets here on such a worker: its dispatch +// goroutine makes no claim (asyncTransplantEligible) and stays parked, +// still running, so tryTransplant leaves it alone too (celeris#681 R1). // - the conn's last hand-off failed at its dup (reapSuppressed): reaping // the recv re-armed after that failure would fail the same way at once. func (w *Worker) startReap(cs *connState) { diff --git a/engine/iouring/handoff_loss.go b/engine/iouring/handoff_loss.go index 368b5d6f..769469de 100644 --- a/engine/iouring/handoff_loss.go +++ b/engine/iouring/handoff_loss.go @@ -62,10 +62,11 @@ import "sync/atomic" // example -EINVAL from a kernel that rejects the cancel flags the // startup probe found accepted). Not retried and never followed by a // hand-off. Must stay 0. -// - reapUnsupported: reaps not placed because this kernel rejects the -// IORING_ASYNC_CANCEL flags a reap needs (probeAsyncCancelFlags; they -// exist from 5.19). The conn stays until its recv completes on its own. -// A rate, 0 on every kernel from 5.19. +// - reapUnsupported: reaps not placed because the startup probe did not +// find the IORING_ASYNC_CANCEL flags a reap needs accepted +// (probeAsyncCancelFlags; they exist from 5.19). The conn stays until +// its recv completes on its own. A rate, 0 wherever the probe finds the +// flags, which it does on every kernel from 5.19 measured. // - holdRescued: held conns the timeout sweep found with their response // sent, not handed off and no recv armed — a path that skipped the // release. The belt under releaseHold. Must stay 0. diff --git a/engine/iouring/probe.go b/engine/iouring/probe.go index 293b22a2..f5777fa1 100644 --- a/engine/iouring/probe.go +++ b/engine/iouring/probe.go @@ -34,7 +34,7 @@ var ( cachedMultiAcceptOK bool cachedMultiAcceptReas string cachedAsyncCancel sync.Once - cachedAsyncCancelOK bool + cachedAsyncCancelRes asyncCancelProbe cachedAsyncCancelReas string ) @@ -73,11 +73,11 @@ func probeMultishotAcceptCached() (bool, string) { // probeAsyncCancelFlagsCached returns the cached async-cancel-flags probe // result. -func probeAsyncCancelFlagsCached() (bool, string) { +func probeAsyncCancelFlagsCached() (asyncCancelProbe, string) { cachedAsyncCancel.Do(func() { - cachedAsyncCancelOK, cachedAsyncCancelReas = probeAsyncCancelFlags() + cachedAsyncCancelRes, cachedAsyncCancelReas = probeAsyncCancelFlags() }) - return cachedAsyncCancelOK, cachedAsyncCancelReas + return cachedAsyncCancelRes, cachedAsyncCancelReas } // SEND_ZC ioprio flags and notification result values. @@ -469,6 +469,43 @@ const ( asyncCancelProbeTag uint64 = 0xCA_4CE1_7A6 ) +// asyncCancelProbe is what probeAsyncCancelFlags learned. Only +// asyncCancelAccepted turns the hand-off's reap on; the other two keep it +// off, and they are told apart because they mean different things +// (celeris#681 R2): a rejection is the kernel's answer, a probe that got no +// answer says nothing about the kernel. +type asyncCancelProbe uint8 + +const ( + // asyncCancelAccepted: the kernel completed the probe's cancel and + // accepted its IORING_ASYNC_CANCEL flags. + asyncCancelAccepted asyncCancelProbe = iota + // asyncCancelRejected: the kernel completed the probe's cancel without + // accepting the flags: -EINVAL, as every kernel before 5.19 answers, or + // a result the probe does not recognise. + asyncCancelRejected + // asyncCancelNoAnswer: the probe failed before the kernel answered. Its + // ring could not be set up, the submit or the wait failed, no completion + // came, or the completion was not the probe's cancel. + asyncCancelNoAnswer +) + +func (p asyncCancelProbe) String() string { + switch p { + case asyncCancelAccepted: + return "accepted" + case asyncCancelRejected: + return "rejected" + case asyncCancelNoAnswer: + return "no answer" + } + return fmt.Sprintf("asyncCancelProbe(%d)", uint8(p)) +} + +// newAsyncCancelProbeRing makes the probe's private ring. A variable so a +// test can make the probe fail before the kernel answers. +var newAsyncCancelProbeRing = func() (*Ring, error) { return NewRing(8, 0, 0) } + // probeAsyncCancelFlags tests whether the kernel accepts the // IORING_ASYNC_CANCEL_* flags, which exist from Linux 5.19. The io_uring→epoll // hand-off's REAP (celeris#657) cancels an armed recv with @@ -483,8 +520,10 @@ const ( // nothing carries and reads the cancel's own completion; see // classifyAsyncCancelProbe for how the result reads. The cancel runs inline // at submit on every kernel measured, so its completion is normally there -// when Submit returns; a short wait covers any that is not. -func probeAsyncCancelFlags() (bool, string) { +// when Submit returns; a short wait covers any that is not. A probe that +// fails before that completion is read reports asyncCancelNoAnswer, never a +// rejection. +func probeAsyncCancelFlags() (asyncCancelProbe, string) { return probeAsyncCancel(cancelAll) } @@ -492,38 +531,38 @@ func probeAsyncCancelFlags() (bool, string) { // given: the probe passes the reap's own (IORING_ASYNC_CANCEL_ALL), and a // test passes a bit no kernel defines, which every kernel rejects, to drive // the rejection through the same submit and completion path. -func probeAsyncCancel(cancelFlags uint32) (bool, string) { - ring, err := NewRing(8, 0, 0) +func probeAsyncCancel(cancelFlags uint32) (asyncCancelProbe, string) { + ring, err := newAsyncCancelProbeRing() if err != nil { - return false, "NewRing failed: " + err.Error() + return asyncCancelNoAnswer, "NewRing failed: " + err.Error() } defer func() { _ = ring.Close() }() sqe := ring.GetSQE() if sqe == nil { - return false, "GetSQE returned nil" + return asyncCancelNoAnswer, "GetSQE returned nil" } prepCancelUserDataReported(sqe, asyncCancelProbeTarget) *(*uint32)(unsafe.Pointer(&(*[sqeSize]byte)(sqe)[28])) = cancelFlags setSQEUserData(sqe, asyncCancelProbeTag) if _, err := ring.Submit(); err != nil { - return false, "Submit failed: " + err.Error() + return asyncCancelNoAnswer, "Submit failed: " + err.Error() } head, tail := ring.BeginCQ() if head == tail { if err := ring.SubmitAndWaitTimeout(500 * time.Millisecond); err != nil { - return false, "SubmitAndWaitTimeout failed: " + err.Error() + return asyncCancelNoAnswer, "SubmitAndWaitTimeout failed: " + err.Error() } head, tail = ring.BeginCQ() if head == tail { - return false, "no CQE produced for the cancel (waited 500ms)" + return asyncCancelNoAnswer, "no CQE produced for the cancel (waited 500ms)" } } cqe := ring.cqeAt(head) ud, res := cqe.UserData, cqe.Res ring.EndCQ(head + 1) if ud != asyncCancelProbeTag { - return false, fmt.Sprintf("unexpected CQE user_data %#x (want the cancel's %#x)", ud, asyncCancelProbeTag) + return asyncCancelNoAnswer, fmt.Sprintf("unexpected CQE user_data %#x (want the cancel's %#x)", ud, asyncCancelProbeTag) } return classifyAsyncCancelProbe(res) } @@ -537,18 +576,18 @@ func probeAsyncCancel(cancelFlags uint32) (bool, string) { // miss; no kernel measured answers the probe with it. // - -EINVAL: rejected. Measured on 5.15.0-191: every cancel form celeris // builds returns -EINVAL there and leaves its target running. -// - anything else: not an answer the probe understands, so the flags are -// treated as rejected and the value is reported. +// - anything else: an answer the probe does not understand, so the flags +// are treated as rejected and the value is reported. // // Split out of probeAsyncCancelFlags so every outcome can be checked against // a synthetic result. -func classifyAsyncCancelProbe(res int32) (bool, string) { +func classifyAsyncCancelProbe(res int32) (asyncCancelProbe, string) { switch { case res >= 0, res == -int32(unix.ENOENT): - return true, "" + return asyncCancelAccepted, "" case res == -int32(unix.EINVAL): - return false, "IORING_ASYNC_CANCEL flags rejected: cqe.res=-22 (EINVAL); the kernel predates Linux 5.19" + return asyncCancelRejected, "IORING_ASYNC_CANCEL flags rejected: cqe.res=-22 (EINVAL); the kernel predates Linux 5.19" default: - return false, fmt.Sprintf("the cancel completed with cqe.res=%d, which the probe does not recognise", res) + return asyncCancelRejected, fmt.Sprintf("the cancel completed with cqe.res=%d, which the probe does not recognise", res) } } diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index 69fa5845..cb1e9d8e 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -488,8 +488,9 @@ type Worker struct { // reapRetrySpare is the second buffer of the swap. Worker thread only. reapRetry []uint64 reapRetrySpare []uint64 - // asyncCancelFlags: this kernel accepts IORING_ASYNC_CANCEL_* flags - // (probeAsyncCancelFlags, 5.19+). A reap is only placed when it does; + // asyncCancelFlags: probeAsyncCancelFlags found this kernel accepting + // IORING_ASYNC_CANCEL_* flags (5.19+); false when it rejected them or the + // probe got no answer. A reap is only placed when it is true; // createWorkers copies the engine's answer. Read-only after init. asyncCancelFlags bool // dupFD duplicates the descriptor a hand-off moves: unix.Dup when nil. From c15c686fbde936b3718f1f25705c7d60b0b0a1e5 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 20:48:44 +0200 Subject: [PATCH 20/34] test(iouring): check the cancel-flags probe against what the kernel does, not its version (celeris#657) Review R3 on #681. TestAsyncCancelProbeOnThisKernel and TestReapOnTheRunningKernel/probe_answer failed when the probe's answer differed from uname >= 5.19, and the version helper failed on a release it could not parse. That contradicts why the probe exists: a vendor kernel can claim a version its feature surface does not match. The version is now only logged (kernelRelease). The bit-31 check stays as the proof the probe reads the kernel's answer. What the version check used to catch, a probe that answers wrongly on this kernel, is now checked against the kernel itself: TestReapOnTheRunningKernel/probe_matches_the_kernel forces the flags on, lets the running kernel answer the real reap (cancels the recv, or fails it with -EINVAL), and requires the probe's answer to be what the kernel did. --- engine/iouring/async_cancel_probe_test.go | 73 ++++++++++++++++------- 1 file changed, 51 insertions(+), 22 deletions(-) diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index ba214a10..6c2aa989 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -5,6 +5,7 @@ package iouring import ( "context" "errors" + "fmt" "log/slog" "strings" "testing" @@ -95,27 +96,31 @@ func TestAsyncCancelProbeClassifies(t *testing.T) { }) } -// kernelHasCancelFlags reports whether the running kernel's version is one -// that has IORING_ASYNC_CANCEL flags (5.19 or later). -func kernelHasCancelFlags(t *testing.T) (bool, string) { - t.Helper() +// kernelRelease describes the running kernel for a log line: its release and +// whether that version is one that has IORING_ASYNC_CANCEL flags (5.19 or +// later). It is never a verdict (celeris#681 R3): a vendor kernel can claim a +// version its feature surface does not match, which is why the engine probes +// for the flags at all, so the tests compare the probe with what the running +// kernel does, not with its version. +func kernelRelease() string { var u unix.Utsname if err := unix.Uname(&u); err != nil { - t.Fatalf("uname: %v", err) + return "uname: " + err.Error() } rel := unix.ByteSliceToString(u.Release[:]) kv, err := probe.ParseKernelVersion(rel) if err != nil { - t.Fatalf("parse kernel release %q: %v", rel, err) + return fmt.Sprintf("%s (unparsed: %v)", rel, err) } - return kv.AtLeast(5, 19), rel + return fmt.Sprintf("%s (a version with the flags: %v)", rel, kv.AtLeast(5, 19)) } // TestAsyncCancelProbeOnThisKernel runs the probe itself against the running -// kernel and checks its answer against the kernel's version, and checks that -// the probe reads the kernel's answer: a cancel flag no kernel defines -// (bit 31) is rejected with -EINVAL everywhere, so the same path must then -// report the flags unsupported. +// kernel, logs its answer beside the kernel's version, and checks that the +// probe reads the kernel's answer: a cancel flag no kernel defines (bit 31) +// is rejected with -EINVAL everywhere, so the same path must then report a +// rejection. Whether the answer is right for this kernel is +// TestReapOnTheRunningKernel/probe_matches_the_kernel's check. func TestAsyncCancelProbeOnThisKernel(t *testing.T) { r, err := NewRing(8, 0, 0) if err != nil { @@ -123,12 +128,7 @@ func TestAsyncCancelProbeOnThisKernel(t *testing.T) { } _ = r.Close() res, reason := probeAsyncCancelFlags() - ok := res == asyncCancelAccepted - want, rel := kernelHasCancelFlags(t) - t.Logf("celeris681 async cancel flags probe: kernel=%s result=%v reason=%q", rel, res, reason) - if ok != want { - t.Fatalf("probeAsyncCancelFlags() = (%v, %q) on kernel %s, want accepted=%v", res, reason, rel, want) - } + t.Logf("celeris681 async cancel flags probe: kernel=%s result=%v reason=%q", kernelRelease(), res, reason) if cres, _ := probeAsyncCancelFlagsCached(); cres != res { t.Fatalf("the cached probe says %v, the probe %v", cres, res) } @@ -254,11 +254,7 @@ func TestReapOnTheRunningKernel(t *testing.T) { t.Run("probe_answer", func(t *testing.T) { res, _ := probeAsyncCancelFlagsCached() ok := res == asyncCancelAccepted - want, rel := kernelHasCancelFlags(t) - t.Logf("celeris681 kernel=%s probe=%v", rel, res) - if ok != want { - t.Fatalf("the probe says %v on kernel %s, want %v", ok, rel, want) - } + t.Logf("celeris681 kernel=%s probe=%v", kernelRelease(), res) f := idleArmed(t, ok) if !ok { withoutFlags(t, f) @@ -276,4 +272,37 @@ func TestReapOnTheRunningKernel(t *testing.T) { t.Run("flags_rejected", func(t *testing.T) { withoutFlags(t, idleArmed(t, false)) }) + + // The probe's answer checked against the running kernel itself rather + // than its version (celeris#681 R3): with the flags forced on, the reap + // goes to the kernel, and what the kernel does with it must be what the + // probe said. A kernel that accepts the flags cancels the recv, and the + // conn is handed off at its -ECANCELED; one that rejects them fails the + // reap (TransplantReapFailed) and leaves the recv armed, and the conn + // then leaves after its next, held, response. + t.Run("probe_matches_the_kernel", func(t *testing.T) { + res, reason := probeAsyncCancelFlagsCached() + f := idleArmed(t, true) + if n := f.w.ring.Pending(); n != 1 { + t.Fatalf("tryTransplant placed %d SQE(s) with the flags forced on, want the reap", n) + } + run(t, f, "the kernel's answer to the reap", func() bool { + return f.tgt.adopted.Load() == 1 || f.e.metrics.handoffLoss.reapFailed.Load() == 1 + }) + accepts := f.tgt.adopted.Load() == 1 + t.Logf("celeris681 kernel=%s executed the reap: %v; the probe says %v (%q)", kernelRelease(), accepts, res, reason) + if accepts != (res == asyncCancelAccepted) { + t.Fatalf("the running kernel executed the reap: %v, but the probe says %v (%q): the probe's answer "+ + "is not what this kernel does", accepts, res, reason) + } + if !accepts { + if _, err := unix.Write(f.peer, []byte(fdlGET)); err != nil { + t.Fatalf("client write: %v", err) + } + run(t, f, "the next request", func() bool { return f.tgt.adopted.Load() == 1 }) + } + if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { + t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) + } + }) } From 356fbe25c6d4cd939976fc2dd1d038476a2915fc Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 20:51:20 +0200 Subject: [PATCH 21/34] test(adaptive): make T4 assert the loss witnesses, the must-stay-0 counters and every switch's move (celeris#657) Review R4 on #681. TestFlapConnsPerRing asserted only its clients' errors and the worker counts, and its header still said PR-1's counter was absent. T4 now also requires, from the adaptive engine's Metrics() (both sub-engines summed): W1 (StaleRecvDataTransplanted + Unattributed) = 0, W2 (TransplantHandoffInFlight) = 0, and TransplantDoubleClaim, TransplantHoldRescued and TransplantReapFailed = 0, each with its own S0T4 message. Every switch must have moved the conns: the engine switched away from detached at least all 256 and holds none, and the one switched to adopted at least all 256 and holds all of them; each SWITCH line logs moved= and adopted=. The RESULT line gains w1, w2, the three gate counters, held and reaps after err=, so the CI tally's pattern still matches. Its io_uring-availability and adaptive.New skips now honour CELERIS_REQUIRE_UPSWITCH=1 like its memlock skip: fail instead of skip (s0SkipUnlessRequired). --- adaptive/flap_conns_per_ring_test.go | 91 +++++++++++++++++++++++----- 1 file changed, 75 insertions(+), 16 deletions(-) diff --git a/adaptive/flap_conns_per_ring_test.go b/adaptive/flap_conns_per_ring_test.go index 7c1b0d60..159d337c 100644 --- a/adaptive/flap_conns_per_ring_test.go +++ b/adaptive/flap_conns_per_ring_test.go @@ -13,7 +13,15 @@ package adaptive // Shape (DECISION.md step 0; red-team.md section 7, T4): controller frozen, THREE promote/revert cycles under // continuous back-to-back load, 2.5 s after each switch (more than a 2 s client read deadline, so a request lost // at switch k is counted before switch k+1). It requires ZERO client errors: a client error here is a request the -// server never answered. (The StaleRecvData half of T4's assertion needs PR-1's counter, absent on 985a386.) +// server never answered. +// +// It also requires, from the adaptive engine's Metrics() (both sub-engines summed; celeris#681 R4), the two +// celeris#657 loss witnesses at 0 -- W1, a request read by a recv that outlived its hand-off +// (StaleRecvDataTransplanted + StaleRecvDataUnattributed), and W2, a hand-off made with an op in flight +// (TransplantHandoffInFlight) -- and the three hand-off counters that must stay 0 (TransplantDoubleClaim, +// TransplantHoldRescued, TransplantReapFailed). And every switch must have moved the connections: the engine +// switched away from detached at least all of them and holds none, and the engine switched to adopted at least +// all of them and holds all of them. A switch that moved nothing would pass every loss check vacuously. import ( "bufio" @@ -52,13 +60,10 @@ func s0FlapConnsPerRing(t *testing.T, cycles int) { ) // Two io_uring workers need 24 MiB of RLIMIT_MEMLOCK (12 MiB each, engine/iouring minMemlockPerWorker). Below // that io_uring caps itself to one ring and the PREMISE check below fails on the environment, not the engine. - // The adaptive CI job raises memlock and sets CELERIS_REQUIRE_UPSWITCH=1, which turns this skip into a failure. + // The adaptive CI job raises memlock and sets CELERIS_REQUIRE_UPSWITCH=1, which turns this skip, and s0Bind's, + // into a failure. if m := maxWorkersForMemlock(); m >= 0 && m < workers { - msg := fmt.Sprintf("RLIMIT_MEMLOCK funds %d io_uring worker(s), T4 needs %d", m, workers) - if os.Getenv("CELERIS_REQUIRE_UPSWITCH") == "1" { - t.Fatal(msg + " -- CELERIS_REQUIRE_UPSWITCH=1 forbids skipping") - } - t.Skip(msg) + s0SkipUnlessRequired(t, "RLIMIT_MEMLOCK funds %d io_uring worker(s), T4 needs %d", m, workers) } e, addr, stop := s0Bind(t, resource.Config{Addr: "127.0.0.1:0", Protocol: engine.HTTP1, Resources: resource.Resources{Workers: workers}}, respHandler{}) @@ -72,26 +77,52 @@ func s0FlapConnsPerRing(t *testing.T, cycles int) { t.Logf("S0T4 START cycles=%d conns=%d epoll=%d ok=%d err=%d", cycles, conns, e.primary.Metrics().ActiveConnections, okCount.Load(), errCount.Load()) + // side is one sub-engine's metrics: io_uring (secondary) or epoll (primary). Before the first promotion the + // lazy standby is not built yet, and reads as zero. + side := func(ioUring bool) engine.EngineMetrics { + en := e.primary + if ioUring { + en = e.secondary + } + if en == nil { + return engine.EngineMetrics{} + } + return en.Metrics() + } for sw := 1; sw <= 2*cycles; sw++ { - dir := "promote" - if sw%2 == 0 { + dir, toIOUring := "promote", sw%2 == 1 + if !toIOUring { dir = "revert" } errBefore := errCount.Load() + srcBefore, dstBefore := side(!toIOUring), side(toIOUring) t0 := time.Now() e.ForceSwitch() took := time.Since(t0) time.Sleep(dwell) pm, sm := e.primary.Metrics(), e.secondary.Metrics() - t.Logf("S0T4 SWITCH n=%d dir=%s took_ms=%d epoll=%d io_uring=%d wE=%d wI=%d ok=%d err=%d err_this=%d census=%s", + src, dst := side(!toIOUring), side(toIOUring) + moved := src.TransplantDetached - srcBefore.TransplantDetached + adopted := dst.TransplantAdopted - dstBefore.TransplantAdopted + t.Logf("S0T4 SWITCH n=%d dir=%s took_ms=%d epoll=%d io_uring=%d wE=%d wI=%d moved=%d adopted=%d ok=%d err=%d err_this=%d census=%s", sw, dir, took.Milliseconds(), pm.ActiveConnections, sm.ActiveConnections, pm.Workers, sm.Workers, - okCount.Load(), errCount.Load(), errCount.Load()-errBefore, cen.String()) + moved, adopted, okCount.Load(), errCount.Load(), errCount.Load()-errBefore, cen.String()) + if moved < conns || adopted < conns || src.ActiveConnections != 0 || dst.ActiveConnections != conns { + t.Errorf("S0T4 MOVE: switch %d (%s) detached %d and adopted %d of %d conns, and left %d on the engine "+ + "switched away from and %d on the one switched to: want every conn moved", sw, dir, moved, adopted, + conns, src.ActiveConnections, dst.ActiveConnections) + } } close(stopLoad) wg.Wait() ew, iw := e.primary.Metrics().Workers, e.secondary.Metrics().Workers - t.Logf("S0T4 RESULT cycles=%d conns=%d wE=%d wI=%d conns_per_ring=%d ok=%d err=%d census=%s", - cycles, conns, ew, iw, conns/max(iw, 1), okCount.Load(), errCount.Load(), cen.String()) + m := e.Metrics() + w1 := m.StaleRecvDataTransplanted + m.StaleRecvDataUnattributed + t.Logf("S0T4 RESULT cycles=%d conns=%d wE=%d wI=%d conns_per_ring=%d ok=%d err=%d w1=%d w2=%d doubleclaim=%d "+ + "holdrescued=%d reapfailed=%d held=%d reaps=%d census=%s", + cycles, conns, ew, iw, conns/max(iw, 1), okCount.Load(), errCount.Load(), w1, m.TransplantHandoffInFlight, + m.TransplantDoubleClaim, m.TransplantHoldRescued, m.TransplantReapFailed, m.TransplantHeld, m.TransplantReaps, + cen.String()) if ew != workers || iw != workers { t.Errorf("S0T4 PREMISE: want %d workers on both engines, got epoll=%d io_uring=%d", workers, ew, iw) } @@ -99,6 +130,33 @@ func s0FlapConnsPerRing(t *testing.T, cycles int) { t.Errorf("S0T4 LOSS: %d of %d keep-alive clients lost a request across %d promote/revert cycles at %d conns per ring (census %s)", n, conns, cycles, conns/max(iw, 1), cen.String()) } + if w1 != 0 { + t.Errorf("S0T4 W1: %d requests were read by a recv that outlived its hand-off (StaleRecvDataTransplanted=%d "+ + "StaleRecvDataUnattributed=%d), want 0", w1, m.StaleRecvDataTransplanted, m.StaleRecvDataUnattributed) + } + if n := m.TransplantHandoffInFlight; n != 0 { + t.Errorf("S0T4 W2: %d hand-offs were made with an op in flight (TransplantHandoffInFlight), want 0", n) + } + if n := m.TransplantDoubleClaim; n != 0 { + t.Errorf("S0T4 DOUBLECLAIM: TransplantDoubleClaim = %d, want 0", n) + } + if n := m.TransplantHoldRescued; n != 0 { + t.Errorf("S0T4 HOLDRESCUED: TransplantHoldRescued = %d, want 0", n) + } + if n := m.TransplantReapFailed; n != 0 { + t.Errorf("S0T4 REAPFAILED: TransplantReapFailed = %d, want 0", n) + } +} + +// s0SkipUnlessRequired skips T4 for an environment that cannot run it, and fails it instead when +// CELERIS_REQUIRE_UPSWITCH=1, as the adaptive CI job sets it: that job must not go green by skipping T4. +func s0SkipUnlessRequired(t *testing.T, format string, args ...any) { + t.Helper() + msg := fmt.Sprintf(format, args...) + if os.Getenv("CELERIS_REQUIRE_UPSWITCH") == "1" { + t.Fatal(msg + " -- CELERIS_REQUIRE_UPSWITCH=1 forbids skipping") + } + t.Skip(msg) } // The client driver and bind helper of TestFlapConnsPerRing (from the celeris#657 step-0 overlay). The driver keeps @@ -194,15 +252,16 @@ func s0Drive(addr string, conns int, stop <-chan struct{}, ok, errc *atomic.Int6 // s0Bind builds an adaptive engine from cfg, disables the switch cooldown, FREEZES the controller (so the // only switches are the test's ForceSwitch calls; performSwitch ignores the freeze), starts Listen and waits -// for the bind. +// for the bind. Where io_uring or the adaptive engine is unavailable it skips, or fails under +// CELERIS_REQUIRE_UPSWITCH=1 (s0SkipUnlessRequired). func s0Bind(t *testing.T, cfg resource.Config, h stream.Handler) (*Engine, string, func()) { t.Helper() if !probe.Probe().IOUringTier.Available() { - t.Skip("io_uring unavailable: needs both sub-engines") + s0SkipUnlessRequired(t, "io_uring unavailable: needs both sub-engines") } e, err := New(cfg, h, nil) if err != nil { - t.Skipf("adaptive.New unsupported here: %v", err) + s0SkipUnlessRequired(t, "adaptive.New unsupported here: %v", err) } e.ctrl.cooldown = 0 e.FreezeSwitching() From 1037b66aab50e420cb5bd85663d5027e304db849 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 20:51:20 +0200 Subject: [PATCH 22/34] test(iouring): assert the must-stay-0 hand-off counters in the engine-level tests (celeris#657) Review R4 on #681. transplantUnderLoad logged TransplantDoubleClaim and TransplantReapFailed but no engine-level test asserted them. It now fails the test if TransplantDoubleClaim, TransplantReapFailed or TransplantHoldRescued moved, for every caller (TestHandoffHasNothingInFlight sync and async, TestStaleRecvDataCounted, target_refuses), and the drain_stops_while_held check asserts the first two as well (it already asserted TransplantHoldRescued). --- engine/iouring/fd_lifetime_engine_test.go | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/engine/iouring/fd_lifetime_engine_test.go b/engine/iouring/fd_lifetime_engine_test.go index ebc5bad6..12539117 100644 --- a/engine/iouring/fd_lifetime_engine_test.go +++ b/engine/iouring/fd_lifetime_engine_test.go @@ -247,6 +247,9 @@ func classes(m map[string]int64) string { // warm, keeps the load up for after, stops it and returns what the clients // saw. The stale CQEs of any stolen request arrive within the load window; // the final wait lets the last of them land before the counters are read. +// It logs every celeris#657 counter, and fails the test if one of the three +// that must stay 0 moved (TransplantDoubleClaim, TransplantReapFailed, +// TransplantHoldRescued; celeris#681 R4). func transplantUnderLoad(t *testing.T, e *Engine, addr string, n int, tgt engine.TransplantTarget, warm, after time.Duration, ) loadResult { @@ -273,6 +276,11 @@ func transplantUnderLoad(t *testing.T, e *Engine, addr string, n int, tgt engine e.metrics.transplantDetached.Load(), metricOr(e, "TransplantClaimDeferred"), metricOr(e, "TransplantReapFailed"), metricOr(e, "TransplantReapUnsupported")) + for _, name := range []string{"TransplantDoubleClaim", "TransplantReapFailed", "TransplantHoldRescued"} { + if n := metric(t, e, name); n != 0 { + t.Errorf("%s = %d, want 0 (a hand-off counter that must stay 0)", name, n) + } + } return res } @@ -387,6 +395,11 @@ func TestHeldRecvIsReArmedWhenTheHandOffDoesNotHappen(t *testing.T) { t.Errorf("TransplantHoldRescued = %d, want 0: a held conn was stranded until the "+ "timeout sweep found it", n) } + for _, name := range []string{"TransplantDoubleClaim", "TransplantReapFailed"} { + if n := metric(t, e, name); n != 0 { + t.Errorf("%s = %d, want 0", name, n) + } + } if res.ok == 0 { t.Error("clients completed no requests") } From c1737111e2c6171678dbc304f33701d088a6ffef Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 20:52:12 +0200 Subject: [PATCH 23/34] docs(iouring): TransplantReapUnsupported counts no promoted async conn, and a dup failure's suppression outlives its drain (celeris#657) Review R5 on #681. EngineMetrics.TransplantReapUnsupported said every connection it counts is handed off after its next response. A promoted async connection is never offered for the hand-off where the cancel flags are missing (since R1), stays on io_uring and is not counted; a connection the worker serves itself is held and handed off only where the worker has no provided-buffer ring. The doc now says so, and that it is placement only. reapSuppressed is set at a failed dup and cleared only by the conn's next data and its release, so it outlives the drain it was set in. The field's doc now states that and what it does across drains: a conn that has received nothing since meets a hand-off attempt in a later drain only at a completion that is not data, and no reap is placed for it there until its next data. Doc-only. --- engine/engine.go | 19 +++++++++++++------ engine/iouring/conn.go | 12 ++++++++++++ engine/iouring/fd_lifetime.go | 2 ++ engine/iouring/handoff_loss.go | 6 ++++-- 4 files changed, 31 insertions(+), 8 deletions(-) diff --git a/engine/engine.go b/engine/engine.go index b662f2bc..53e661c0 100644 --- a/engine/engine.go +++ b/engine/engine.go @@ -540,12 +540,19 @@ type EngineMetrics struct { //nolint:revive // user-approved name // sub-engines. TransplantReapFailed uint64 // TransplantReapUnsupported counts hand-off recv cancels not placed - // because the kernel rejects the IORING_ASYNC_CANCEL flags they need - // (Linux 5.19 added them; the io_uring engine probes for them at - // startup). The connection stays on io_uring until its armed recv - // completes on its own, and is handed off after its next response. - // A rate, 0 on every kernel from 5.19. io_uring-only; on the adaptive - // engine the sum over both sub-engines. + // because the io_uring engine's startup probe did not find the + // IORING_ASYNC_CANCEL flags they need accepted (Linux 5.19 added them; + // a probe that got no answer counts the same). It counts only + // connections the worker serves itself (every connection in sync mode, + // and in async mode those not promoted to a dispatch goroutine): such a + // connection stays on io_uring until its armed recv completes on its + // own, and is handed off after the response to the request that recv + // brings, which is held, where the worker has no provided-buffer ring (a + // kernel without the flags has none); with one it stays. A promoted + // async connection is never offered for the hand-off where the flags are + // missing: it stays on io_uring and is not counted. Placement only, + // never a lost request. A rate, 0 wherever the probe finds the flags. + // io_uring-only; on the adaptive engine the sum over both sub-engines. TransplantReapUnsupported uint64 } diff --git a/engine/iouring/conn.go b/engine/iouring/conn.go index 04cae5e6..e97bc697 100644 --- a/engine/iouring/conn.go +++ b/engine/iouring/conn.go @@ -222,6 +222,18 @@ type connState struct { // RECV, a cancel and two completions per loop iteration for as long as // the failure and the drain both lasted. // + // Only the conn's next data (handleRecv) and its release clear it, not + // the end of the drain it was set in or the start of the next, so it + // can outlive that drain (celeris#681 R5). Across drains the effect is + // placement only, and narrow: a conn that has received nothing since + // its dup failed meets a hand-off attempt in a later drain only at a + // completion that is not data, for example a provided-buffer recv ended + // by -ENOBUFS and re-armed, and no reap is placed for it there; its next + // data clears the flag and the attempt after that data proceeds as + // usual. A promoted async conn never has it set at a park: the data + // that respawns its dispatch goroutine clears it first, so its hand-off + // is retried at that park. + // // All four worker-thread only, like recvCancelPending. transplantReap uint16 reapStale bool diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go index 1fc2ec11..3d37f35b 100644 --- a/engine/iouring/fd_lifetime.go +++ b/engine/iouring/fd_lifetime.go @@ -81,6 +81,8 @@ func onlyRecvInFlight(cs *connState) bool { // still running, so tryTransplant leaves it alone too (celeris#681 R1). // - the conn's last hand-off failed at its dup (reapSuppressed): reaping // the recv re-armed after that failure would fail the same way at once. +// Only the conn's next data lifts it, not the end of the drain; the +// field's doc says what that does across drains. func (w *Worker) startReap(cs *connState) { if !w.asyncCancelFlags { w.handoffLoss.noteReapUnsupported() diff --git a/engine/iouring/handoff_loss.go b/engine/iouring/handoff_loss.go index 769469de..e75ffb3f 100644 --- a/engine/iouring/handoff_loss.go +++ b/engine/iouring/handoff_loss.go @@ -65,8 +65,10 @@ import "sync/atomic" // - reapUnsupported: reaps not placed because the startup probe did not // find the IORING_ASYNC_CANCEL flags a reap needs accepted // (probeAsyncCancelFlags; they exist from 5.19). The conn stays until -// its recv completes on its own. A rate, 0 wherever the probe finds the -// flags, which it does on every kernel from 5.19 measured. +// its recv completes on its own. A promoted async conn is never offered +// for the hand-off on such a worker, so it is not counted. A rate, 0 +// wherever the probe finds the flags, which it does on every kernel from +// 5.19 measured. // - holdRescued: held conns the timeout sweep found with their response // sent, not handed off and no recv armed — a path that skipped the // release. The belt under releaseHold. Must stay 0. From 1f883e2c9340ca3356cd7de365d511475bdd2501 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 21:03:54 +0200 Subject: [PATCH 24/34] test(iouring): count an async claim once it is queued, and retry the client poll on EINTR (celeris#657) Two races in the async park rig of the R1 tests, both in the test: - asyncRequest took transplantPending as the claim, but the dispatch goroutine sets it before it enqueues itself, so a drain run at once could find the detach queue still empty (control_async_park_claims_with_flags failed once, "the drain of the claim placed []", in a local run of the unit job's witness step). The claim now counts once detachQPending is set, the last step of making it. - poll(2) on the client socket can return EINTR on the runtime's preemption signal (1 of 600 stressed runs); it is retried until its deadline. 1000 of 1000 runs of the two tests then passed under -race (count=500), and the 29-test witness list 290 of 290 (count=10). --- engine/iouring/fd_lifetime_fixture_test.go | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/engine/iouring/fd_lifetime_fixture_test.go b/engine/iouring/fd_lifetime_fixture_test.go index 7d8b2cce..af3de4b2 100644 --- a/engine/iouring/fd_lifetime_fixture_test.go +++ b/engine/iouring/fd_lifetime_fixture_test.go @@ -365,18 +365,27 @@ func asyncParkFixture(t *testing.T, flags bool) (*fdlFixture, *goidHandler) { // asyncRequest delivers one request to f's promoted conn, reads its response // at the client, and waits until the dispatch goroutine that answered it has -// reached its park: there it either claimed its own hand-off -// (transplantPending, enqueued for the worker, and exited) or waits on its +// reached its park: there it either claimed its own hand-off or waits on its // cond for the next request. It reports which, and the SQEs the delivery -// placed. +// placed. A claim counts once it is on the worker's detach queue +// (detachQPending), the last step of making it: transplantPending is set +// before the enqueue, so a drain run on seeing only that flag could find the +// queue still empty. func (f *fdlFixture) asyncRequest(h *goidHandler) (claimed bool, placed []sqeRec) { f.t.Helper() n := len(h.served()) f.deliver(fdlGET) placed = takeSQEs(f.w.ring) fds := []unix.PollFd{{Fd: int32(f.peer), Events: unix.POLLIN}} - if k, err := unix.Poll(fds, 10000); err != nil || k != 1 { - f.t.Fatalf("no response at the client within 10s (poll %d, %v)", k, err) + for deadline := time.Now().Add(10 * time.Second); ; { + k, err := unix.Poll(fds, int(max(time.Until(deadline).Milliseconds(), 0))) + if err == unix.EINTR { // the runtime's preemption signal + continue + } + if err != nil || k != 1 { + f.t.Fatalf("no response at the client within 10s (poll %d, %v)", k, err) + } + break } var resp [512]byte if k, err := unix.Read(f.peer, resp[:]); err != nil || !strings.HasPrefix(string(resp[:max(k, 0)]), "HTTP/1.1 200") { @@ -387,7 +396,7 @@ func (f *fdlFixture) asyncRequest(h *goidHandler) (claimed bool, placed []sqeRec f.t.Fatalf("the handler ran %d times for one request", len(ids)-n) } for deadline := time.Now().Add(10 * time.Second); ; time.Sleep(time.Millisecond) { - if f.cs.transplantPending.Load() || f.w.detachQPending.Load() != 0 { + if f.w.detachQPending.Load() != 0 { return true, placed } if goroutineWaitsOnCond(ids[n]) { From 683c7bb2790c6f7dd7d29df8fdbf4ce462ddae6c Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 21:55:09 +0200 Subject: [PATCH 25/34] DO NOT MERGE: plant the no-HOLD/REAP engine and neutralise the T4 and revert-test verdicts, for one deliberate red CI run (celeris#681 R6) Review R6 on #681: T4's "fails on bad input" had been proven only in local containers. This commit makes the GitHub runner show it, and the next commit reverts it; the branch after that is byte-identical to the commit before this one. Planted (tools/mk_red_r3.py; round 2's interlock arms T-BLIND-NHR and A-BLIND-NHR, re-derived on this head): - the no-HOLD/REAP engine: the R0 gate removed at both hand-off sites (transplant_source.go) and `hold := false` (worker.go), the campaign's negative control, which loses requests at a revert; - every t.Errorf in T4's s0FlapConnsPerRing and in the revert tests' reverseScenario and flapScenario becomes t.Logf, so go test passes whatever the engine does and the adaptive steps can only go red through their own tallies: T4's RESULT line (err=0 required) and the one-worker leg's summary lines (err=0 required). Expected red: the adaptive race step on T4's RESULT line and the one-worker leg on its summaries; also the unit job (the engine-level fd-lifetime tests fail on this engine) and the lint job (startReap and onlyRecvInFlight lose their callers). --- adaptive/flap_conns_per_ring_test.go | 16 ++++++++-------- adaptive/reverse_transplant_test.go | 16 ++++++++-------- engine/iouring/transplant_source.go | 12 ------------ engine/iouring/worker.go | 2 +- 4 files changed, 17 insertions(+), 29 deletions(-) diff --git a/adaptive/flap_conns_per_ring_test.go b/adaptive/flap_conns_per_ring_test.go index 159d337c..e286c88d 100644 --- a/adaptive/flap_conns_per_ring_test.go +++ b/adaptive/flap_conns_per_ring_test.go @@ -108,7 +108,7 @@ func s0FlapConnsPerRing(t *testing.T, cycles int) { sw, dir, took.Milliseconds(), pm.ActiveConnections, sm.ActiveConnections, pm.Workers, sm.Workers, moved, adopted, okCount.Load(), errCount.Load(), errCount.Load()-errBefore, cen.String()) if moved < conns || adopted < conns || src.ActiveConnections != 0 || dst.ActiveConnections != conns { - t.Errorf("S0T4 MOVE: switch %d (%s) detached %d and adopted %d of %d conns, and left %d on the engine "+ + t.Logf("S0T4 MOVE: switch %d (%s) detached %d and adopted %d of %d conns, and left %d on the engine "+ "switched away from and %d on the one switched to: want every conn moved", sw, dir, moved, adopted, conns, src.ActiveConnections, dst.ActiveConnections) } @@ -124,27 +124,27 @@ func s0FlapConnsPerRing(t *testing.T, cycles int) { m.TransplantDoubleClaim, m.TransplantHoldRescued, m.TransplantReapFailed, m.TransplantHeld, m.TransplantReaps, cen.String()) if ew != workers || iw != workers { - t.Errorf("S0T4 PREMISE: want %d workers on both engines, got epoll=%d io_uring=%d", workers, ew, iw) + t.Logf("S0T4 PREMISE: want %d workers on both engines, got epoll=%d io_uring=%d", workers, ew, iw) } if n := errCount.Load(); n != 0 { - t.Errorf("S0T4 LOSS: %d of %d keep-alive clients lost a request across %d promote/revert cycles at %d conns per ring (census %s)", + t.Logf("S0T4 LOSS: %d of %d keep-alive clients lost a request across %d promote/revert cycles at %d conns per ring (census %s)", n, conns, cycles, conns/max(iw, 1), cen.String()) } if w1 != 0 { - t.Errorf("S0T4 W1: %d requests were read by a recv that outlived its hand-off (StaleRecvDataTransplanted=%d "+ + t.Logf("S0T4 W1: %d requests were read by a recv that outlived its hand-off (StaleRecvDataTransplanted=%d "+ "StaleRecvDataUnattributed=%d), want 0", w1, m.StaleRecvDataTransplanted, m.StaleRecvDataUnattributed) } if n := m.TransplantHandoffInFlight; n != 0 { - t.Errorf("S0T4 W2: %d hand-offs were made with an op in flight (TransplantHandoffInFlight), want 0", n) + t.Logf("S0T4 W2: %d hand-offs were made with an op in flight (TransplantHandoffInFlight), want 0", n) } if n := m.TransplantDoubleClaim; n != 0 { - t.Errorf("S0T4 DOUBLECLAIM: TransplantDoubleClaim = %d, want 0", n) + t.Logf("S0T4 DOUBLECLAIM: TransplantDoubleClaim = %d, want 0", n) } if n := m.TransplantHoldRescued; n != 0 { - t.Errorf("S0T4 HOLDRESCUED: TransplantHoldRescued = %d, want 0", n) + t.Logf("S0T4 HOLDRESCUED: TransplantHoldRescued = %d, want 0", n) } if n := m.TransplantReapFailed; n != 0 { - t.Errorf("S0T4 REAPFAILED: TransplantReapFailed = %d, want 0", n) + t.Logf("S0T4 REAPFAILED: TransplantReapFailed = %d, want 0", n) } } diff --git a/adaptive/reverse_transplant_test.go b/adaptive/reverse_transplant_test.go index 3dad2ca3..2d0cf77a 100644 --- a/adaptive/reverse_transplant_test.go +++ b/adaptive/reverse_transplant_test.go @@ -162,17 +162,17 @@ func reverseScenario(t *testing.T, h stream.Handler, async bool) { async, epoBefore, iouBefore, promoted, epoInflight, iouInflight, epoConverged, iouConverged, okCount.Load(), errCount.Load()) if async && promoted == 0 { - t.Errorf("async test never promoted any conn to the dispatch goroutine — async path not exercised") + t.Logf("async test never promoted any conn to the dispatch goroutine — async path not exercised") } // Converged: essentially all conns must have migrated to epoll. if epoConverged < conns-2 { - t.Errorf("reverse transplant did not converge: epoll=%d (want ~%d)", epoConverged, conns) + t.Logf("reverse transplant did not converge: epoll=%d (want ~%d)", epoConverged, conns) } if iouConverged > 2 { - t.Errorf("io_uring did not drain on convergence: io_uring=%d (want ~0)", iouConverged) + t.Logf("io_uring did not drain on convergence: io_uring=%d (want ~0)", iouConverged) } if errCount.Load() > int64(conns) { - t.Errorf("too many request errors across the revert: %d (conns=%d)", errCount.Load(), conns) + t.Logf("too many request errors across the revert: %d (conns=%d)", errCount.Load(), conns) } } @@ -208,10 +208,10 @@ func flapScenario(t *testing.T, h stream.Handler, async bool) { t.Logf("[async=%v] flap %d -> active=%s: epoll=%d io_uring=%d (ok=%d err=%d)", async, flap, activeName, epo, iou, okCount.Load(), errCount.Load()) if active < conns/2 { - t.Errorf("flap %d: transplant did not migrate to %s: active=%d (want ~%d)", flap, activeName, active, conns) + t.Logf("flap %d: transplant did not migrate to %s: active=%d (want ~%d)", flap, activeName, active, conns) } if standby > conns/4 { - t.Errorf("flap %d: standby did not drain: standby=%d (want ~0)", flap, standby) + t.Logf("flap %d: standby did not drain: standby=%d (want ~0)", flap, standby) } } @@ -221,10 +221,10 @@ func flapScenario(t *testing.T, h stream.Handler, async bool) { async, okCount.Load(), errCount.Load(), e.primary.Metrics().AsyncPromotedConns, e.secondary.Metrics().AsyncPromotedConns) if async && e.primary.Metrics().AsyncPromotedConns == 0 && e.secondary.Metrics().AsyncPromotedConns == 0 { - t.Errorf("async flap never promoted any conn — async path not exercised") + t.Logf("async flap never promoted any conn — async path not exercised") } if errCount.Load() > int64(conns) { - t.Errorf("too many request errors across flaps: %d (conns=%d)", errCount.Load(), conns) + t.Logf("too many request errors across flaps: %d (conns=%d)", errCount.Load(), conns) } } diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 813aa96e..f8a18633 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -123,12 +123,6 @@ func (w *Worker) tryTransplant(fd int) { // this descriptor may be in flight when it moves. The usual case is the // recv armed after the last response (its linked RECV, or the idle // conn's standalone one): reap it and hand off at its -ECANCELED. - if cs.recvArmed || cs.kernelInflight != 0 || cs.zcNotifPending { - if onlyRecvInFlight(cs) { - w.startReap(cs) - } - return - } w.handOff(cs, fd, h, false) } @@ -344,11 +338,5 @@ func (w *Worker) finishAsyncTransplant(cs *connState) { // has exited, so the recv the feed path armed for it is the one op the // conn can have: reap it, and come back here at its -ECANCELED. A // request that beats the cancel respawns the goroutine as usual. - if cs.recvArmed || cs.kernelInflight != 0 || cs.zcNotifPending { - if onlyRecvInFlight(cs) { - w.startReap(cs) - } - return - } w.handOff(cs, cs.fd, h, true) } diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index cb1e9d8e..b956f34a 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -3042,7 +3042,7 @@ func (w *Worker) respondAndArm(cs *connState, fd int, c *completionEntry, link, w.closeConn(fd) return } - hold := w.bufRing == nil && w.transplant.Load() != nil && w.holdEligible(cs) + hold := false switch { case hold: cs.transplantHold = true From 272bcba1791067fb758ced8b253f5145be4ab0d3 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 21:58:55 +0200 Subject: [PATCH 26/34] Revert the deliberate red commit: restore HOLD/REAP and the T4 and revert-test verdicts (celeris#681 R6) This reverts commit 683c7bb2790c6f7dd7d29df8fdbf4ce462ddae6c, the no-HOLD/REAP plant with neutralised verdicts that made CI run 35465814564 red on purpose: the adaptive race step on T4's RESULT line (err=188, w1=188, w2=398) and the one-worker leg on its summaries (err 13 and 36), plus the unit and lint jobs. The tree is now byte-identical to 1f883e2. --- adaptive/flap_conns_per_ring_test.go | 16 ++++++++-------- adaptive/reverse_transplant_test.go | 16 ++++++++-------- engine/iouring/transplant_source.go | 12 ++++++++++++ engine/iouring/worker.go | 2 +- 4 files changed, 29 insertions(+), 17 deletions(-) diff --git a/adaptive/flap_conns_per_ring_test.go b/adaptive/flap_conns_per_ring_test.go index e286c88d..159d337c 100644 --- a/adaptive/flap_conns_per_ring_test.go +++ b/adaptive/flap_conns_per_ring_test.go @@ -108,7 +108,7 @@ func s0FlapConnsPerRing(t *testing.T, cycles int) { sw, dir, took.Milliseconds(), pm.ActiveConnections, sm.ActiveConnections, pm.Workers, sm.Workers, moved, adopted, okCount.Load(), errCount.Load(), errCount.Load()-errBefore, cen.String()) if moved < conns || adopted < conns || src.ActiveConnections != 0 || dst.ActiveConnections != conns { - t.Logf("S0T4 MOVE: switch %d (%s) detached %d and adopted %d of %d conns, and left %d on the engine "+ + t.Errorf("S0T4 MOVE: switch %d (%s) detached %d and adopted %d of %d conns, and left %d on the engine "+ "switched away from and %d on the one switched to: want every conn moved", sw, dir, moved, adopted, conns, src.ActiveConnections, dst.ActiveConnections) } @@ -124,27 +124,27 @@ func s0FlapConnsPerRing(t *testing.T, cycles int) { m.TransplantDoubleClaim, m.TransplantHoldRescued, m.TransplantReapFailed, m.TransplantHeld, m.TransplantReaps, cen.String()) if ew != workers || iw != workers { - t.Logf("S0T4 PREMISE: want %d workers on both engines, got epoll=%d io_uring=%d", workers, ew, iw) + t.Errorf("S0T4 PREMISE: want %d workers on both engines, got epoll=%d io_uring=%d", workers, ew, iw) } if n := errCount.Load(); n != 0 { - t.Logf("S0T4 LOSS: %d of %d keep-alive clients lost a request across %d promote/revert cycles at %d conns per ring (census %s)", + t.Errorf("S0T4 LOSS: %d of %d keep-alive clients lost a request across %d promote/revert cycles at %d conns per ring (census %s)", n, conns, cycles, conns/max(iw, 1), cen.String()) } if w1 != 0 { - t.Logf("S0T4 W1: %d requests were read by a recv that outlived its hand-off (StaleRecvDataTransplanted=%d "+ + t.Errorf("S0T4 W1: %d requests were read by a recv that outlived its hand-off (StaleRecvDataTransplanted=%d "+ "StaleRecvDataUnattributed=%d), want 0", w1, m.StaleRecvDataTransplanted, m.StaleRecvDataUnattributed) } if n := m.TransplantHandoffInFlight; n != 0 { - t.Logf("S0T4 W2: %d hand-offs were made with an op in flight (TransplantHandoffInFlight), want 0", n) + t.Errorf("S0T4 W2: %d hand-offs were made with an op in flight (TransplantHandoffInFlight), want 0", n) } if n := m.TransplantDoubleClaim; n != 0 { - t.Logf("S0T4 DOUBLECLAIM: TransplantDoubleClaim = %d, want 0", n) + t.Errorf("S0T4 DOUBLECLAIM: TransplantDoubleClaim = %d, want 0", n) } if n := m.TransplantHoldRescued; n != 0 { - t.Logf("S0T4 HOLDRESCUED: TransplantHoldRescued = %d, want 0", n) + t.Errorf("S0T4 HOLDRESCUED: TransplantHoldRescued = %d, want 0", n) } if n := m.TransplantReapFailed; n != 0 { - t.Logf("S0T4 REAPFAILED: TransplantReapFailed = %d, want 0", n) + t.Errorf("S0T4 REAPFAILED: TransplantReapFailed = %d, want 0", n) } } diff --git a/adaptive/reverse_transplant_test.go b/adaptive/reverse_transplant_test.go index 2d0cf77a..3dad2ca3 100644 --- a/adaptive/reverse_transplant_test.go +++ b/adaptive/reverse_transplant_test.go @@ -162,17 +162,17 @@ func reverseScenario(t *testing.T, h stream.Handler, async bool) { async, epoBefore, iouBefore, promoted, epoInflight, iouInflight, epoConverged, iouConverged, okCount.Load(), errCount.Load()) if async && promoted == 0 { - t.Logf("async test never promoted any conn to the dispatch goroutine — async path not exercised") + t.Errorf("async test never promoted any conn to the dispatch goroutine — async path not exercised") } // Converged: essentially all conns must have migrated to epoll. if epoConverged < conns-2 { - t.Logf("reverse transplant did not converge: epoll=%d (want ~%d)", epoConverged, conns) + t.Errorf("reverse transplant did not converge: epoll=%d (want ~%d)", epoConverged, conns) } if iouConverged > 2 { - t.Logf("io_uring did not drain on convergence: io_uring=%d (want ~0)", iouConverged) + t.Errorf("io_uring did not drain on convergence: io_uring=%d (want ~0)", iouConverged) } if errCount.Load() > int64(conns) { - t.Logf("too many request errors across the revert: %d (conns=%d)", errCount.Load(), conns) + t.Errorf("too many request errors across the revert: %d (conns=%d)", errCount.Load(), conns) } } @@ -208,10 +208,10 @@ func flapScenario(t *testing.T, h stream.Handler, async bool) { t.Logf("[async=%v] flap %d -> active=%s: epoll=%d io_uring=%d (ok=%d err=%d)", async, flap, activeName, epo, iou, okCount.Load(), errCount.Load()) if active < conns/2 { - t.Logf("flap %d: transplant did not migrate to %s: active=%d (want ~%d)", flap, activeName, active, conns) + t.Errorf("flap %d: transplant did not migrate to %s: active=%d (want ~%d)", flap, activeName, active, conns) } if standby > conns/4 { - t.Logf("flap %d: standby did not drain: standby=%d (want ~0)", flap, standby) + t.Errorf("flap %d: standby did not drain: standby=%d (want ~0)", flap, standby) } } @@ -221,10 +221,10 @@ func flapScenario(t *testing.T, h stream.Handler, async bool) { async, okCount.Load(), errCount.Load(), e.primary.Metrics().AsyncPromotedConns, e.secondary.Metrics().AsyncPromotedConns) if async && e.primary.Metrics().AsyncPromotedConns == 0 && e.secondary.Metrics().AsyncPromotedConns == 0 { - t.Logf("async flap never promoted any conn — async path not exercised") + t.Errorf("async flap never promoted any conn — async path not exercised") } if errCount.Load() > int64(conns) { - t.Logf("too many request errors across flaps: %d (conns=%d)", errCount.Load(), conns) + t.Errorf("too many request errors across flaps: %d (conns=%d)", errCount.Load(), conns) } } diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index f8a18633..813aa96e 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -123,6 +123,12 @@ func (w *Worker) tryTransplant(fd int) { // this descriptor may be in flight when it moves. The usual case is the // recv armed after the last response (its linked RECV, or the idle // conn's standalone one): reap it and hand off at its -ECANCELED. + if cs.recvArmed || cs.kernelInflight != 0 || cs.zcNotifPending { + if onlyRecvInFlight(cs) { + w.startReap(cs) + } + return + } w.handOff(cs, fd, h, false) } @@ -338,5 +344,11 @@ func (w *Worker) finishAsyncTransplant(cs *connState) { // has exited, so the recv the feed path armed for it is the one op the // conn can have: reap it, and come back here at its -ECANCELED. A // request that beats the cancel respawns the goroutine as usual. + if cs.recvArmed || cs.kernelInflight != 0 || cs.zcNotifPending { + if onlyRecvInFlight(cs) { + w.startReap(cs) + } + return + } w.handOff(cs, cs.fd, h, true) } diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index b956f34a..cb1e9d8e 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -3042,7 +3042,7 @@ func (w *Worker) respondAndArm(cs *connState, fd int, c *completionEntry, link, w.closeConn(fd) return } - hold := false + hold := w.bufRing == nil && w.transplant.Load() != nil && w.holdEligible(cs) switch { case hold: cs.transplantHold = true From 30f9f7096cf24612f55ed95ffa69310c7d9048f2 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 22:31:00 +0200 Subject: [PATCH 27/34] fix(iouring): make an unset cancel-flags probe answer read as no answer (celeris#681 N3) The zero value of asyncCancelProbe was asyncCancelAccepted, the one answer that turns the hand-off's reap on, so an answer that was never set read as ON. The zero value is now asyncCancelNoAnswer, which keeps the reap off. TestAsyncCancelProbeClassifies/zero_value_is_no_answer pins it. asyncTransplantEligible's doc now says that a probe with no answer has the same effect there as a rejection. --- .../iouring/async_cancel_probe_answer_test.go | 26 +++++++++++++++++++ engine/iouring/async_cancel_probe_test.go | 3 +++ engine/iouring/probe.go | 26 +++++++++++-------- engine/iouring/transplant_source.go | 7 ++--- 4 files changed, 48 insertions(+), 14 deletions(-) create mode 100644 engine/iouring/async_cancel_probe_answer_test.go diff --git a/engine/iouring/async_cancel_probe_answer_test.go b/engine/iouring/async_cancel_probe_answer_test.go new file mode 100644 index 00000000..c2e00c52 --- /dev/null +++ b/engine/iouring/async_cancel_probe_answer_test.go @@ -0,0 +1,26 @@ +//go:build linux + +package iouring + +import "testing" + +// The cancel-flags probe's answer (celeris#681 round 4). Each check here is a +// function run as a subtest of one of the probe tests the CI witness step +// names, so it runs there with skipping forbidden. They use only the probe's +// API as it stood before round 4, plus the test seams round 4 adds, so each +// can also be run against the round-3 head. + +// probeZeroValueIsNoAnswer (N3): an answer that was never set must keep the +// reap off. The zero value of asyncCancelProbe is therefore no answer; it +// used to be accepted, which is the one value that turns the reap on. +func probeZeroValueIsNoAnswer(t *testing.T) { + var unset asyncCancelProbe + if unset == asyncCancelAccepted { + t.Fatalf("the zero value of asyncCancelProbe reads as %v: an answer that was never set would turn "+ + "the hand-off's reap on", unset) + } + if unset != asyncCancelNoAnswer || unset.String() != "no answer" { + t.Errorf("the zero value of asyncCancelProbe is %v (%q), want %v: an unset answer reads as no answer", + unset, unset.String(), asyncCancelNoAnswer) + } +} diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index 6c2aa989..869d86df 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -48,6 +48,9 @@ func TestAsyncCancelProbeClassifies(t *testing.T) { } } + // An answer never set keeps the reap off (celeris#681 N3). + t.Run("zero_value_is_no_answer", probeZeroValueIsNoAnswer) + // The probe's ring cannot be set up: nothing reached the kernel. t.Run("no_answer_is_not_a_rejection", func(t *testing.T) { saved := newAsyncCancelProbeRing diff --git a/engine/iouring/probe.go b/engine/iouring/probe.go index f5777fa1..1e904243 100644 --- a/engine/iouring/probe.go +++ b/engine/iouring/probe.go @@ -470,34 +470,38 @@ const ( ) // asyncCancelProbe is what probeAsyncCancelFlags learned. Only -// asyncCancelAccepted turns the hand-off's reap on; the other two keep it -// off, and they are told apart because they mean different things -// (celeris#681 R2): a rejection is the kernel's answer, a probe that got no -// answer says nothing about the kernel. +// asyncCancelAccepted turns the hand-off's reap on; the others keep it off, +// and they are told apart because they mean different things (celeris#681 +// R2): a rejection is the kernel's answer, a probe that got no answer says +// nothing about the kernel. +// +// The zero value is asyncCancelNoAnswer (celeris#681 N3), so an answer that +// was never set reads as no answer and keeps the reap off, never as +// accepted. type asyncCancelProbe uint8 const ( + // asyncCancelNoAnswer: the probe failed before the kernel answered. Its + // ring could not be set up, the submit or the wait failed, no completion + // came, or the completion was not the probe's cancel. The zero value. + asyncCancelNoAnswer asyncCancelProbe = iota // asyncCancelAccepted: the kernel completed the probe's cancel and // accepted its IORING_ASYNC_CANCEL flags. - asyncCancelAccepted asyncCancelProbe = iota + asyncCancelAccepted // asyncCancelRejected: the kernel completed the probe's cancel without // accepting the flags: -EINVAL, as every kernel before 5.19 answers, or // a result the probe does not recognise. asyncCancelRejected - // asyncCancelNoAnswer: the probe failed before the kernel answered. Its - // ring could not be set up, the submit or the wait failed, no completion - // came, or the completion was not the probe's cancel. - asyncCancelNoAnswer ) func (p asyncCancelProbe) String() string { switch p { + case asyncCancelNoAnswer: + return "no answer" case asyncCancelAccepted: return "accepted" case asyncCancelRejected: return "rejected" - case asyncCancelNoAnswer: - return "no answer" } return fmt.Sprintf("asyncCancelProbe(%d)", uint8(p)) } diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 813aa96e..8354cf78 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -237,9 +237,10 @@ func (w *Worker) reclaimTransplant(newFD int, carry engine.Carryover, cause erro // egress check is therefore re-done in finishAsyncTransplant, on the worker // thread, and this predicate is only a cheap first filter. // -// A worker that cannot reap (the kernel rejects IORING_ASYNC_CANCEL flags, -// probeAsyncCancelFlags) offers no promoted async conn for the hand-off -// (celeris#681 R1). Every claim would find the recv the feed path armed after +// A worker that cannot reap (its engine's probeAsyncCancelFlags did not find +// IORING_ASYNC_CANCEL flags accepted: the kernel rejected them, or the probe +// got no answer, which has the same effect here) offers no promoted async +// conn for the hand-off (celeris#681 R1). Every claim would find the recv the feed path armed after // the last request, which only a reap can clear, and a promoted conn is never // held: finishAsyncTransplant would refuse it, and the goroutine that exited // to make the claim would be respawned by the next request — a spawn and a From 0c01c1cd9f8e5d8cb2282ca1e8326d2c0f998ee1 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 22:32:38 +0200 Subject: [PATCH 28/34] fix(iouring): give a cancel-flags probe answer it does not recognise its own class, and warn with the errno (celeris#681 N2) classifyAsyncCancelProbe read any completion other than an acceptance (res >= 0, -ENOENT) or -EINVAL as a rejection, and New logged it at Info, even on a 5.19+ kernel where the flags exist. It is now its own class, asyncCancelUnexpected: the reap stays off, the reason names the errno (for example "cqe.res=-9 (EBADF)"), and New logs it at Warn on every kernel. TestAsyncCancelProbeClassifies/unrecognised_answer_is_its_own_class pins the class, the errno and the Warn; the classification table and log_levels gain the new class. The docs that list what keeps the reap off name it. --- engine/engine.go | 3 +- .../iouring/async_cancel_probe_answer_test.go | 46 ++++++++++++++++++- engine/iouring/async_cancel_probe_test.go | 26 ++++++++--- engine/iouring/engine.go | 14 ++++-- engine/iouring/fd_lifetime.go | 9 ++-- engine/iouring/probe.go | 33 ++++++++++--- engine/iouring/transplant_source.go | 4 +- engine/iouring/worker.go | 5 +- 8 files changed, 113 insertions(+), 27 deletions(-) diff --git a/engine/engine.go b/engine/engine.go index 53e661c0..de40d661 100644 --- a/engine/engine.go +++ b/engine/engine.go @@ -542,7 +542,8 @@ type EngineMetrics struct { //nolint:revive // user-approved name // TransplantReapUnsupported counts hand-off recv cancels not placed // because the io_uring engine's startup probe did not find the // IORING_ASYNC_CANCEL flags they need accepted (Linux 5.19 added them; - // a probe that got no answer counts the same). It counts only + // a probe that got no answer, or an answer it does not recognise, counts + // the same). It counts only // connections the worker serves itself (every connection in sync mode, // and in async mode those not promoted to a dispatch goroutine): such a // connection stays on io_uring until its armed recv completes on its diff --git a/engine/iouring/async_cancel_probe_answer_test.go b/engine/iouring/async_cancel_probe_answer_test.go index c2e00c52..c958be48 100644 --- a/engine/iouring/async_cancel_probe_answer_test.go +++ b/engine/iouring/async_cancel_probe_answer_test.go @@ -2,7 +2,13 @@ package iouring -import "testing" +import ( + "log/slog" + "strings" + "testing" + + "golang.org/x/sys/unix" +) // The cancel-flags probe's answer (celeris#681 round 4). Each check here is a // function run as a subtest of one of the probe tests the CI witness step @@ -24,3 +30,41 @@ func probeZeroValueIsNoAnswer(t *testing.T) { unset, unset.String(), asyncCancelNoAnswer) } } + +// probeUnrecognisedAnswerIsItsOwnClass (N2): a completion that is neither an +// acceptance (res >= 0, -ENOENT) nor the rejection every kernel before 5.19 +// gives (-EINVAL) is an answer the probe does not recognise. It used to be +// read as a rejection and logged at Info, even on a 5.19+ kernel where the +// flags exist. It is a class of its own ("unexpected"): the reap stays off, +// the reason names the errno, and New logs it at Warn on every kernel. +// Written against String(), so it runs against the round-3 head too. +func probeUnrecognisedAnswerIsItsOwnClass(t *testing.T) { + for _, res := range []int32{-int32(unix.EBADF), -int32(unix.ECANCELED), -int32(unix.EPERM)} { + errno := unix.ErrnoName(unix.Errno(-res)) + p, reason := classifyAsyncCancelProbe(res) + if p == asyncCancelAccepted { + t.Fatalf("classifyAsyncCancelProbe(%d) = %v: an answer the probe does not recognise turned the reap on", res, p) + } + if p == asyncCancelRejected || p == asyncCancelNoAnswer || p.String() != "unexpected" { + t.Errorf("classifyAsyncCancelProbe(%d) = %v (%q), want a class of its own, \"unexpected\": "+ + "an answer the probe does not recognise is neither the kernel's rejection nor no answer", res, p, reason) + } + if !strings.Contains(reason, errno) { + t.Errorf("classifyAsyncCancelProbe(%d) reason %q does not name the errno %s", res, reason, errno) + } + for _, k := range [][2]int{{5, 15}, {6, 8}} { + var buf lockedBuffer + logAsyncCancelProbe(slog.New(slog.NewJSONHandler(&buf, nil)), p, reason, k[0], k[1]) + recs := buf.records(t) + level, logged := "", "" + if len(recs) == 1 { + level, _ = recs[0]["level"].(string) + logged, _ = recs[0]["reason"].(string) + } + if len(recs) != 1 || level != "WARN" || !strings.Contains(logged, errno) { + t.Errorf("cqe.res=%d on kernel %d.%d was logged as %v, want one WARN record whose reason names %s", + res, k[0], k[1], recs, errno) + } + } + } +} diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index 869d86df..eb1f245d 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -26,8 +26,10 @@ import ( // TestAsyncCancelProbeClassifies pins how the probe reads its cancel's // completion: a cancel with CANCEL_ALL of a user_data nothing carries. A -// completion is the kernel's answer, accepted or rejected; a probe that fails -// before reading one has no answer, which is not a rejection (celeris#681 R2). +// completion is the kernel's answer: accepted, rejected, or (celeris#681 N2) +// one the probe does not recognise, which is a class of its own. A probe that +// fails before reading one has no answer, which is not a rejection +// (celeris#681 R2). func TestAsyncCancelProbeClassifies(t *testing.T) { for _, tc := range []struct { res int32 @@ -38,8 +40,8 @@ func TestAsyncCancelProbeClassifies(t *testing.T) { {1, asyncCancelAccepted, ""}, {-int32(unix.ENOENT), asyncCancelAccepted, ""}, // the flag-free form's miss {-int32(unix.EINVAL), asyncCancelRejected, "5.19"}, - {-int32(unix.EBADF), asyncCancelRejected, "cqe.res=-9"}, - {-int32(unix.ECANCELED), asyncCancelRejected, "cqe.res=-125"}, + {-int32(unix.EBADF), asyncCancelUnexpected, "cqe.res=-9 (EBADF)"}, + {-int32(unix.ECANCELED), asyncCancelUnexpected, "cqe.res=-125 (ECANCELED)"}, } { got, reason := classifyAsyncCancelProbe(tc.res) if got != tc.want || (tc.reason == "") != (reason == "") || !strings.Contains(reason, tc.reason) { @@ -51,6 +53,10 @@ func TestAsyncCancelProbeClassifies(t *testing.T) { // An answer never set keeps the reap off (celeris#681 N3). t.Run("zero_value_is_no_answer", probeZeroValueIsNoAnswer) + // An answer the probe does not recognise is not a rejection, and New + // warns about it with the errno (celeris#681 N2). + t.Run("unrecognised_answer_is_its_own_class", probeUnrecognisedAnswerIsItsOwnClass) + // The probe's ring cannot be set up: nothing reached the kernel. t.Run("no_answer_is_not_a_rejection", func(t *testing.T) { saved := newAsyncCancelProbeRing @@ -64,8 +70,9 @@ func TestAsyncCancelProbeClassifies(t *testing.T) { }) // New's record of the answer: nothing when accepted, Info for the - // kernel's rejection, and for no answer Warn where the kernel's version - // has the flags and Info below it. + // kernel's rejection, Warn on every kernel for an answer the probe does + // not recognise, and for no answer Warn where the kernel's version has + // the flags and Info below it. t.Run("log_levels", func(t *testing.T) { for _, tc := range []struct { p asyncCancelProbe @@ -75,6 +82,8 @@ func TestAsyncCancelProbeClassifies(t *testing.T) { {asyncCancelAccepted, 6, 8, ""}, {asyncCancelRejected, 5, 15, "INFO"}, {asyncCancelRejected, 6, 8, "INFO"}, + {asyncCancelUnexpected, 5, 15, "WARN"}, + {asyncCancelUnexpected, 6, 8, "WARN"}, {asyncCancelNoAnswer, 5, 18, "INFO"}, {asyncCancelNoAnswer, 5, 19, "WARN"}, {asyncCancelNoAnswer, 6, 8, "WARN"}, @@ -95,6 +104,11 @@ func TestAsyncCancelProbeClassifies(t *testing.T) { t.Errorf("the no-answer record reads %q, want it to say the probe got no answer", msg) } } + if tc.p == asyncCancelUnexpected && len(recs) == 1 { + if msg, _ := recs[0]["msg"].(string); !strings.Contains(msg, "does not recognise") { + t.Errorf("the unexpected-answer record reads %q, want it to say the probe does not recognise the answer", msg) + } + } } }) } diff --git a/engine/iouring/engine.go b/engine/iouring/engine.go index 4bec3046..73127986 100644 --- a/engine/iouring/engine.go +++ b/engine/iouring/engine.go @@ -94,8 +94,9 @@ type Engine struct { asyncRoutes int // asyncCancelFlags is whether probeAsyncCancelFlags found this kernel // accepting IORING_ASYNC_CANCEL_* flags (5.19+): false when the kernel - // rejected them and when the probe got no answer. Every worker gets a - // copy; the hand-off's REAP needs them (celeris#657). + // rejected them, when it answered in a way the probe does not recognise + // and when the probe got no answer. Every worker gets a copy; the + // hand-off's REAP needs them (celeris#657). asyncCancelFlags bool } @@ -220,8 +221,10 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { // logAsyncCancelProbe reports an async-cancel-flags probe that did not find // the flags accepted (celeris#681 R2). A rejection is the kernel's answer and -// expected before 5.19: Info. A probe that got no answer says nothing about -// the kernel; on one whose version has the flags (5.19 and later) it is +// expected before 5.19: Info. An answer the probe does not recognise is one +// no kernel measured gives, on any version: Warn, with the reason, which +// names the errno (celeris#681 N2). A probe that got no answer says nothing +// about the kernel; on one whose version has the flags (5.19 and later) it is // unexpected, and it keeps the hand-off's reap off for this process, so it is // a Warn there and Info below. func logAsyncCancelProbe(l *slog.Logger, p asyncCancelProbe, reason string, kernelMajor, kernelMinor int) { @@ -230,6 +233,9 @@ func logAsyncCancelProbe(l *slog.Logger, p asyncCancelProbe, reason string, kern case asyncCancelRejected: l.Info("async cancel flags rejected by the kernel: the io_uring→epoll hand-off will not cancel an armed recv (celeris#657)", "reason", reason, "kernel", kernel) + case asyncCancelUnexpected: + l.Warn("async cancel flags probe got an answer it does not recognise: the io_uring→epoll hand-off will not cancel an armed recv (celeris#657)", + "reason", reason, "kernel", kernel) case asyncCancelNoAnswer: msg := "async cancel flags probe got no answer from the kernel: the io_uring→epoll hand-off will not cancel an armed recv (celeris#657)" if kernelMajor > 5 || (kernelMajor == 5 && kernelMinor >= 19) { diff --git a/engine/iouring/fd_lifetime.go b/engine/iouring/fd_lifetime.go index 3d37f35b..060ea03c 100644 --- a/engine/iouring/fd_lifetime.go +++ b/engine/iouring/fd_lifetime.go @@ -72,10 +72,11 @@ func onlyRecvInFlight(cs *connState) bool { // - the startup probe did not find IORING_ASYNC_CANCEL flags accepted // (probeAsyncCancelFlags): a kernel before 5.19 rejects them, and every // reap would fail with -EINVAL and leave the recv armed; a probe that got -// no answer is treated the same. Counted (TransplantReapUnsupported). -// Buffer rings arrived with the flags, in 5.19, so a kernel that rejects -// them has no provided-buffer ring and its sync conns are held; a newer -// kernel whose probe got no answer may have one, and its conns stay. A +// no answer, or an answer it does not recognise, is treated the same. +// Counted (TransplantReapUnsupported). Buffer rings arrived with the +// flags, in 5.19, so a kernel that rejects them has no provided-buffer +// ring and its sync conns are held; a newer kernel whose probe got no +// answer (or an unrecognised one) may have one, and its conns stay. A // promoted async conn never gets here on such a worker: its dispatch // goroutine makes no claim (asyncTransplantEligible) and stays parked, // still running, so tryTransplant leaves it alone too (celeris#681 R1). diff --git a/engine/iouring/probe.go b/engine/iouring/probe.go index 1e904243..bbed7dbd 100644 --- a/engine/iouring/probe.go +++ b/engine/iouring/probe.go @@ -472,8 +472,9 @@ const ( // asyncCancelProbe is what probeAsyncCancelFlags learned. Only // asyncCancelAccepted turns the hand-off's reap on; the others keep it off, // and they are told apart because they mean different things (celeris#681 -// R2): a rejection is the kernel's answer, a probe that got no answer says -// nothing about the kernel. +// R2, N2): a rejection is the kernel's answer, a probe that got no answer +// says nothing about the kernel, and an answer the probe does not recognise +// is one no kernel measured gives. // // The zero value is asyncCancelNoAnswer (celeris#681 N3), so an answer that // was never set reads as no answer and keeps the reap off, never as @@ -489,9 +490,13 @@ const ( // accepted its IORING_ASYNC_CANCEL flags. asyncCancelAccepted // asyncCancelRejected: the kernel completed the probe's cancel without - // accepting the flags: -EINVAL, as every kernel before 5.19 answers, or - // a result the probe does not recognise. + // accepting the flags: -EINVAL, as every kernel before 5.19 answers. asyncCancelRejected + // asyncCancelUnexpected: the kernel completed the probe's cancel with a + // result the probe does not recognise (neither an acceptance nor + // -EINVAL). No kernel measured answers so; it keeps the reap off and New + // warns with the errno (celeris#681 N2). + asyncCancelUnexpected ) func (p asyncCancelProbe) String() string { @@ -502,6 +507,8 @@ func (p asyncCancelProbe) String() string { return "accepted" case asyncCancelRejected: return "rejected" + case asyncCancelUnexpected: + return "unexpected" } return fmt.Sprintf("asyncCancelProbe(%d)", uint8(p)) } @@ -580,8 +587,10 @@ func probeAsyncCancel(cancelFlags uint32) (asyncCancelProbe, string) { // miss; no kernel measured answers the probe with it. // - -EINVAL: rejected. Measured on 5.15.0-191: every cancel form celeris // builds returns -EINVAL there and leaves its target running. -// - anything else: an answer the probe does not understand, so the flags -// are treated as rejected and the value is reported. +// - anything else: unexpected, an answer the probe does not understand +// (celeris#681 N2). It is neither the kernel's acceptance nor its +// rejection, so it is a class of its own: the reap stays off, and the +// reason names the errno, which New logs at Warn on every kernel. // // Split out of probeAsyncCancelFlags so every outcome can be checked against // a synthetic result. @@ -592,6 +601,16 @@ func classifyAsyncCancelProbe(res int32) (asyncCancelProbe, string) { case res == -int32(unix.EINVAL): return asyncCancelRejected, "IORING_ASYNC_CANCEL flags rejected: cqe.res=-22 (EINVAL); the kernel predates Linux 5.19" default: - return asyncCancelRejected, fmt.Sprintf("the cancel completed with cqe.res=%d, which the probe does not recognise", res) + return asyncCancelUnexpected, fmt.Sprintf("the cancel completed with cqe.res=%d (%s), which the probe does not recognise", + res, errnoName(-res)) } } + +// errnoName names errno e for a log line: its symbol (EBADF), or its number +// when the platform has no name for it. +func errnoName(e int32) string { + if name := unix.ErrnoName(unix.Errno(e)); name != "" { + return name + } + return fmt.Sprintf("errno %d", e) +} diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 8354cf78..1e4fe761 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -239,8 +239,8 @@ func (w *Worker) reclaimTransplant(newFD int, carry engine.Carryover, cause erro // // A worker that cannot reap (its engine's probeAsyncCancelFlags did not find // IORING_ASYNC_CANCEL flags accepted: the kernel rejected them, or the probe -// got no answer, which has the same effect here) offers no promoted async -// conn for the hand-off (celeris#681 R1). Every claim would find the recv the feed path armed after +// got no answer or one it does not recognise, each with the same effect +// here) offers no promoted async conn for the hand-off (celeris#681 R1). Every claim would find the recv the feed path armed after // the last request, which only a reap can clear, and a promoted conn is never // held: finishAsyncTransplant would refuse it, and the goroutine that exited // to make the claim would be respawned by the next request — a spawn and a diff --git a/engine/iouring/worker.go b/engine/iouring/worker.go index cb1e9d8e..b813521b 100644 --- a/engine/iouring/worker.go +++ b/engine/iouring/worker.go @@ -489,8 +489,9 @@ type Worker struct { reapRetry []uint64 reapRetrySpare []uint64 // asyncCancelFlags: probeAsyncCancelFlags found this kernel accepting - // IORING_ASYNC_CANCEL_* flags (5.19+); false when it rejected them or the - // probe got no answer. A reap is only placed when it is true; + // IORING_ASYNC_CANCEL_* flags (5.19+); false when it rejected them, when + // it gave an answer the probe does not recognise, or when the probe got + // no answer. A reap is only placed when it is true; // createWorkers copies the engine's answer. Read-only after init. asyncCancelFlags bool // dupFD duplicates the descriptor a hand-off moves: unix.Dup when nil. From 0265e82c603c74bb0cbbeadb07181cc917b20be7 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 22:34:32 +0200 Subject: [PATCH 29/34] fix(iouring): cache only the kernel's answer to the cancel-flags probe, and wait again after a cut-short wait (celeris#681 N1) probeAsyncCancelFlagsCached kept whatever the probe returned for the life of the process, a probe that got no answer included. One transient failure, such as EMFILE or ENOMEM setting up the probe's ring while the adaptive engine builds its io_uring engine under load, kept the hand-off's reap off from then on. Now only the kernel's answer, accepted or rejected, is cached. After no answer, or an answer the probe does not recognise, the next New probes again. Calls are serialised by a mutex. The probe waits for its cancel's completion only when it is not there after the submit. SubmitAndWaitTimeout returns nil both when that wait times out and when a signal cuts it short (EINTR), so the probe used to read an interrupted wait as no answer. A wait that returns early with nothing to read is now repeated once, for the rest of the 500 ms. Test seams: runAsyncCancelProbe (the probe the cache runs), and the probe's submit, wait and wait timeout. New subtests: TestWorkersCarryTheAsyncCancelProbe/only_an_answer_is_cached and /unexpected_answer_keeps_the_reap_off_and_is_not_cached, and TestAsyncCancelProbeOnThisKernel/a_cut_short_wait_is_retried_once. --- .../iouring/async_cancel_probe_answer_test.go | 133 ++++++++++++++++++ engine/iouring/async_cancel_probe_log_test.go | 7 +- .../iouring/async_cancel_probe_reset_test.go | 13 ++ engine/iouring/async_cancel_probe_test.go | 58 ++++++++ engine/iouring/engine.go | 9 +- engine/iouring/probe.go | 79 ++++++++--- 6 files changed, 275 insertions(+), 24 deletions(-) create mode 100644 engine/iouring/async_cancel_probe_reset_test.go diff --git a/engine/iouring/async_cancel_probe_answer_test.go b/engine/iouring/async_cancel_probe_answer_test.go index c958be48..6ce7340f 100644 --- a/engine/iouring/async_cancel_probe_answer_test.go +++ b/engine/iouring/async_cancel_probe_answer_test.go @@ -3,11 +3,16 @@ package iouring import ( + "io" "log/slog" "strings" "testing" + "time" "golang.org/x/sys/unix" + + "github.com/goceleris/celeris/engine" + "github.com/goceleris/celeris/resource" ) // The cancel-flags probe's answer (celeris#681 round 4). Each check here is a @@ -68,3 +73,131 @@ func probeUnrecognisedAnswerIsItsOwnClass(t *testing.T) { } } } + +// probeOnlyAnAnswerIsCached (N1): the kernel's answer, accepted or rejected, +// is kept for the process. A probe that got no answer is not, so the next +// call, and the next New, probe again. It used to be cached like an answer: +// one transient failure, such as EMFILE or ENOMEM setting up the probe's ring +// while the adaptive engine builds its io_uring engine under load, kept the +// reap off for the life of the process. +func probeOnlyAnAnswerIsCached(t *testing.T) { + saved := runAsyncCancelProbe + var calls int + var give asyncCancelProbe + runAsyncCancelProbe = func() (asyncCancelProbe, string) { + calls++ + return give, "celeris681 injected: " + give.String() + } + t.Cleanup(func() { + runAsyncCancelProbe = saved + resetAsyncCancelProbeCache() // the next New probes the kernel again + }) + for _, tc := range []struct { + p asyncCancelProbe + probes int // probes run by three calls + }{ + {asyncCancelAccepted, 1}, + {asyncCancelRejected, 1}, + {asyncCancelNoAnswer, 3}, + } { + resetAsyncCancelProbeCache() + calls, give = 0, tc.p + for i := 0; i < 3; i++ { + if got, _ := probeAsyncCancelFlagsCached(); got != tc.p { + t.Fatalf("call %d with the probe answering %v returned %v", i+1, tc.p, got) + } + } + if calls != tc.probes { + t.Errorf("three calls with the probe answering %v ran it %d time(s), want %d: only the kernel's answer "+ + "(accepted or rejected) is kept, a probe that got no answer is run again", tc.p, calls, tc.probes) + } + } + + // Through New: the probe gets no answer when the first engine is built + // and is accepted when the second is. + resetAsyncCancelProbeCache() + calls, give = 0, asyncCancelNoAnswer + cfg := resource.Config{ + Addr: "127.0.0.1:0", + Protocol: engine.HTTP1, + Logger: slog.New(slog.NewTextHandler(io.Discard, nil)), + } + first, err := New(cfg, transplantTestHandler{}) + if err != nil { + skipOrFail656(t, "iouring engine unavailable: %v", err) + } + give = asyncCancelAccepted + second, err := New(cfg, transplantTestHandler{}) + if err != nil { + skipOrFail656(t, "iouring engine unavailable: %v", err) + } + if first.asyncCancelFlags || !second.asyncCancelFlags || calls != 2 { + t.Errorf("the first New's probe got no answer, the second's is accepted: asyncCancelFlags %v then %v, "+ + "probes run %d; want false, true and 2: a probe with no answer kept the reap off for the next engine", + first.asyncCancelFlags, second.asyncCancelFlags, calls) + } +} + +// probeCutShortWaitIsRetriedOnce (N1): the probe waits for its cancel's +// completion only when the completion is not there after the submit, and +// SubmitAndWaitTimeout returns nil both when that wait times out and when a +// signal cuts it short (EINTR). A wait that comes back early with nothing to +// read was cut short, and is repeated once for the rest of the time; a second +// early return, or a wait that ran its full time, is no answer. The probe +// used to give up after the first wait, so an EINTR read as no answer. +func probeCutShortWaitIsRetriedOnce(t *testing.T) { + r, err := NewRing(8, 0, 0) + if err != nil { + skipOrFail656(t, "io_uring unavailable: %v", err) + } + _ = r.Close() + want, why := probeAsyncCancel(cancelAll) + if want != asyncCancelAccepted && want != asyncCancelRejected { + t.Fatalf("with nothing held back the probe got %v (%q): there is no kernel answer to compare with", want, why) + } + savedSubmit, savedWait, savedTimeout := asyncCancelProbeSubmit, asyncCancelProbeWait, asyncCancelProbeTimeout + t.Cleanup(func() { + asyncCancelProbeSubmit, asyncCancelProbeWait, asyncCancelProbeTimeout = savedSubmit, savedWait, savedTimeout + }) + // Hold the cancel back at the submit, so nothing has completed when the + // probe first looks; the first real wait submits it. + asyncCancelProbeSubmit = func(*Ring) (int, error) { return 0, nil } + var waits int + cutShort := func(cuts int) func(*Ring, time.Duration) error { + return func(r *Ring, d time.Duration) error { + waits++ + if waits <= cuts { + return nil // what SubmitAndWaitTimeout returns for a wait a signal cut short (EINTR) + } + return r.SubmitAndWaitTimeout(d) + } + } + for _, tc := range []struct { + name string + cuts int + want asyncCancelProbe + waits int + }{ + {"cut short once", 1, want, 2}, + {"cut short twice", 2, asyncCancelNoAnswer, 2}, + } { + waits, asyncCancelProbeWait = 0, cutShort(tc.cuts) + if got, reason := probeAsyncCancel(cancelAll); got != tc.want || waits != tc.waits { + t.Errorf("wait %s: the probe gave (%v, %q) after %d wait(s), want %v after %d: a wait cut short is "+ + "repeated once, and only once", tc.name, got, reason, waits, tc.want, tc.waits) + } + } + + // A wait that ran its full time is not repeated. + asyncCancelProbeTimeout = 50 * time.Millisecond + waits = 0 + asyncCancelProbeWait = func(_ *Ring, d time.Duration) error { + waits++ + time.Sleep(d) + return nil + } + if got, reason := probeAsyncCancel(cancelAll); got != asyncCancelNoAnswer || waits != 1 { + t.Errorf("a full-length wait with no completion: the probe gave (%v, %q) after %d wait(s), want %v after 1", + got, reason, waits, asyncCancelNoAnswer) + } +} diff --git a/engine/iouring/async_cancel_probe_log_test.go b/engine/iouring/async_cancel_probe_log_test.go index 16b02a5a..90896536 100644 --- a/engine/iouring/async_cancel_probe_log_test.go +++ b/engine/iouring/async_cancel_probe_log_test.go @@ -54,7 +54,8 @@ func (l *lockedBuffer) records(t *testing.T) []map[string]any { // answer and must not read as one: its record says the probe got no answer, // names the probe's failure, and on a kernel whose version has the flags // (5.19 and later) it is a Warn, since the reap is then off where it should -// work. Only the probe-ring seam and API older than round 3 are used here. +// work. Only the probe-ring seam, the cache reset (async_cancel_probe_reset_test.go) +// and API older than round 3 are used here. func newWarnsWhenTheProbeGetsNoAnswer(t *testing.T) { p := probe.Probe() want := "INFO" @@ -63,10 +64,10 @@ func newWarnsWhenTheProbeGetsNoAnswer(t *testing.T) { } saved := newAsyncCancelProbeRing newAsyncCancelProbeRing = func() (*Ring, error) { return nil, errors.New("celeris681 injected: no ring") } - cachedAsyncCancel = sync.Once{} + resetAsyncCancelProbeCache() t.Cleanup(func() { newAsyncCancelProbeRing = saved - cachedAsyncCancel = sync.Once{} // the next New probes the kernel again + resetAsyncCancelProbeCache() // the next New probes the kernel again }) var buf lockedBuffer e, err := New(resource.Config{ diff --git a/engine/iouring/async_cancel_probe_reset_test.go b/engine/iouring/async_cancel_probe_reset_test.go new file mode 100644 index 00000000..f355e4de --- /dev/null +++ b/engine/iouring/async_cancel_probe_reset_test.go @@ -0,0 +1,13 @@ +//go:build linux + +package iouring + +// resetAsyncCancelProbeCache forgets the async-cancel-flags probe's cached +// answer, so the next probeAsyncCancelFlagsCached (and so the next New) +// probes again. Tests only. +func resetAsyncCancelProbeCache() { + asyncCancelMu.Lock() + defer asyncCancelMu.Unlock() + asyncCancelKnown = false + cachedAsyncCancelRes, cachedAsyncCancelReas = asyncCancelNoAnswer, "" +} diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index eb1f245d..ffdf7c00 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -153,6 +153,10 @@ func TestAsyncCancelProbeOnThisKernel(t *testing.T) { t.Fatalf("a cancel flag no kernel defines was read as (%v, %q), want rejected with EINVAL: "+ "the probe does not read the kernel's answer", bad, why) } + + // A wait for the cancel's completion that a signal cuts short is + // repeated once (celeris#681 N1). + t.Run("a_cut_short_wait_is_retried_once", probeCutShortWaitIsRetriedOnce) } // TestWorkersCarryTheAsyncCancelProbe: New stores the probe's answer and @@ -191,6 +195,60 @@ func TestWorkersCarryTheAsyncCancelProbe(t *testing.T) { // A probe that fails before the kernel answers keeps the reap off too, // and New reports it apart from a rejection (celeris#681 R2). t.Run("no_answer_is_logged_apart_from_a_rejection", newWarnsWhenTheProbeGetsNoAnswer) + + // Only the kernel's answer is kept for the process: after a probe with no + // answer the next New probes again (celeris#681 N1). + t.Run("only_an_answer_is_cached", probeOnlyAnAnswerIsCached) + + // An answer the probe does not recognise is not cached either, keeps the + // reap off in every engine New builds, and each New warns with its errno + // (celeris#681 N1, N2). + t.Run("unexpected_answer_keeps_the_reap_off_and_is_not_cached", func(t *testing.T) { + saved := runAsyncCancelProbe + calls := 0 + const reason = "the cancel completed with cqe.res=-9 (EBADF), which the probe does not recognise" + runAsyncCancelProbe = func() (asyncCancelProbe, string) { + calls++ + return asyncCancelUnexpected, reason + } + resetAsyncCancelProbeCache() + t.Cleanup(func() { + runAsyncCancelProbe = saved + resetAsyncCancelProbeCache() + }) + var buf lockedBuffer + for i := 1; i <= 2; i++ { + e, err := New(resource.Config{ + Addr: "127.0.0.1:0", + Protocol: engine.HTTP1, + Logger: slog.New(slog.NewJSONHandler(&buf, nil)), + }, transplantTestHandler{}) + if err != nil { + skipOrFail656(t, "iouring engine unavailable: %v", err) + } + if e.asyncCancelFlags { + t.Fatalf("New %d turned the hand-off's reap on for an answer the probe does not recognise", i) + } + } + if calls != 2 { + t.Errorf("two News ran the probe %d time(s), want 2: an answer the probe does not recognise was cached", calls) + } + var warns int + for _, r := range buf.records(t) { + if msg, _ := r["msg"].(string); !strings.HasPrefix(msg, "async cancel flags") { + continue + } + level, _ := r["level"].(string) + why, _ := r["reason"].(string) + if level != "WARN" || !strings.Contains(why, "EBADF") { + t.Errorf("an unexpected answer was logged at %s with reason %q, want WARN naming EBADF", level, why) + } + warns++ + } + if warns != 2 { + t.Errorf("two News logged %d probe records, want one each", warns) + } + }) } // TestReapOnTheRunningKernel drives the hand-off of an idle conn whose recv diff --git a/engine/iouring/engine.go b/engine/iouring/engine.go index 73127986..64cbd046 100644 --- a/engine/iouring/engine.go +++ b/engine/iouring/engine.go @@ -180,7 +180,9 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { // release, 5.19); where one exists because the probe got no answer on a // newer kernel, the connection is not held and stays on io_uring. A // promoted async connection is never offered for the hand-off without - // the flags and stays too. Placement only, either way. + // the flags and stays too. Placement only, either way. Only the kernel's + // answer is cached: after a probe with no answer (or one it does not + // recognise) the next New probes again (celeris#681 N1). asyncCancel, acReason := probeAsyncCancelFlagsCached() asyncCancelFlags := asyncCancel == asyncCancelAccepted logAsyncCancelProbe(cfg.Logger, asyncCancel, acReason, profile.KernelMajor, profile.KernelMinor) @@ -225,8 +227,9 @@ func New(cfg resource.Config, handler stream.Handler) (*Engine, error) { // no kernel measured gives, on any version: Warn, with the reason, which // names the errno (celeris#681 N2). A probe that got no answer says nothing // about the kernel; on one whose version has the flags (5.19 and later) it is -// unexpected, and it keeps the hand-off's reap off for this process, so it is -// a Warn there and Info below. +// unexpected, and it keeps the hand-off's reap off for the engine being built +// (the next New probes again: probeAsyncCancelFlagsCached keeps only an +// answer), so it is a Warn there and Info below. func logAsyncCancelProbe(l *slog.Logger, p asyncCancelProbe, reason string, kernelMajor, kernelMinor int) { kernel := fmt.Sprintf("%d.%d", kernelMajor, kernelMinor) switch p { diff --git a/engine/iouring/probe.go b/engine/iouring/probe.go index bbed7dbd..a6ef1f04 100644 --- a/engine/iouring/probe.go +++ b/engine/iouring/probe.go @@ -19,7 +19,8 @@ import ( // matrix can run thousands of cells in one process with -race), and // re-running every probe // (each opens a temp ring + a TCP listener + a dial) on every New() is -// pure overhead. +// pure overhead. The async-cancel-flags probe is cached only once the kernel +// has answered it (probeAsyncCancelFlagsCached). var ( cachedSendZC sync.Once cachedSendZCResult SendZCProbeResult @@ -33,7 +34,11 @@ var ( cachedMultiAccept sync.Once cachedMultiAcceptOK bool cachedMultiAcceptReas string - cachedAsyncCancel sync.Once + // The async-cancel-flags probe's answer, once the kernel has given one: + // asyncCancelKnown says whether cachedAsyncCancelRes/Reas hold it. All + // three are guarded by asyncCancelMu. + asyncCancelMu sync.Mutex + asyncCancelKnown bool cachedAsyncCancelRes asyncCancelProbe cachedAsyncCancelReas string ) @@ -71,15 +76,33 @@ func probeMultishotAcceptCached() (bool, string) { return cachedMultiAcceptOK, cachedMultiAcceptReas } -// probeAsyncCancelFlagsCached returns the cached async-cancel-flags probe -// result. +// probeAsyncCancelFlagsCached returns the async-cancel-flags probe's answer. +// Only the kernel's answer, accepted or rejected, is cached for the process +// (celeris#681 N1). A probe that got no answer says nothing lasting about the +// kernel: its private ring can fail to set up with EMFILE or ENOMEM when the +// adaptive engine builds its io_uring engine lazily under load, and its wait +// can come back without the cancel's completion. Cached, one such failure kept +// the hand-off's reap off for the life of the process. So a probe with no +// answer is not cached, and neither is an answer the probe does not recognise: +// the next New probes again, and logs again. Calls are serialised, so +// concurrent News run one probe at a time and never overwrite an answer. func probeAsyncCancelFlagsCached() (asyncCancelProbe, string) { - cachedAsyncCancel.Do(func() { - cachedAsyncCancelRes, cachedAsyncCancelReas = probeAsyncCancelFlags() - }) - return cachedAsyncCancelRes, cachedAsyncCancelReas + asyncCancelMu.Lock() + defer asyncCancelMu.Unlock() + if asyncCancelKnown { + return cachedAsyncCancelRes, cachedAsyncCancelReas + } + res, reason := runAsyncCancelProbe() + if res == asyncCancelAccepted || res == asyncCancelRejected { + cachedAsyncCancelRes, cachedAsyncCancelReas, asyncCancelKnown = res, reason, true + } + return res, reason } +// runAsyncCancelProbe is the probe probeAsyncCancelFlagsCached runs. A +// variable so a test can give each class of answer to the cache and to New. +var runAsyncCancelProbe = probeAsyncCancelFlags + // SEND_ZC ioprio flags and notification result values. const ( sendZCReportUsage = 1 << 3 // IORING_SEND_ZC_REPORT_USAGE: request ZC usage info in notification @@ -517,6 +540,15 @@ func (p asyncCancelProbe) String() string { // test can make the probe fail before the kernel answers. var newAsyncCancelProbeRing = func() (*Ring, error) { return NewRing(8, 0, 0) } +// The probe's submit and its wait for the cancel's completion, and how long +// that wait is. Variables so a test can hold the cancel back and cut the wait +// short (celeris#681 N1). +var ( + asyncCancelProbeSubmit = func(r *Ring) (int, error) { return r.Submit() } + asyncCancelProbeWait = func(r *Ring, d time.Duration) error { return r.SubmitAndWaitTimeout(d) } + asyncCancelProbeTimeout = 500 * time.Millisecond +) + // probeAsyncCancelFlags tests whether the kernel accepts the // IORING_ASYNC_CANCEL_* flags, which exist from Linux 5.19. The io_uring→epoll // hand-off's REAP (celeris#657) cancels an armed recv with @@ -531,9 +563,12 @@ var newAsyncCancelProbeRing = func() (*Ring, error) { return NewRing(8, 0, 0) } // nothing carries and reads the cancel's own completion; see // classifyAsyncCancelProbe for how the result reads. The cancel runs inline // at submit on every kernel measured, so its completion is normally there -// when Submit returns; a short wait covers any that is not. A probe that -// fails before that completion is read reports asyncCancelNoAnswer, never a -// rejection. +// when Submit returns; a short wait covers any that is not. That wait is +// retried once if it comes back early with nothing to read: SubmitAndWaitTimeout +// returns nil both when its timeout expires and when a signal cuts the wait +// short (EINTR), and only an early return can be the latter (celeris#681 N1). +// A probe that fails before that completion is read reports +// asyncCancelNoAnswer, never a rejection. func probeAsyncCancelFlags() (asyncCancelProbe, string) { return probeAsyncCancel(cancelAll) } @@ -556,17 +591,25 @@ func probeAsyncCancel(cancelFlags uint32) (asyncCancelProbe, string) { prepCancelUserDataReported(sqe, asyncCancelProbeTarget) *(*uint32)(unsafe.Pointer(&(*[sqeSize]byte)(sqe)[28])) = cancelFlags setSQEUserData(sqe, asyncCancelProbeTag) - if _, err := ring.Submit(); err != nil { + if _, err := asyncCancelProbeSubmit(ring); err != nil { return asyncCancelNoAnswer, "Submit failed: " + err.Error() } head, tail := ring.BeginCQ() if head == tail { - if err := ring.SubmitAndWaitTimeout(500 * time.Millisecond); err != nil { - return asyncCancelNoAnswer, "SubmitAndWaitTimeout failed: " + err.Error() - } - head, tail = ring.BeginCQ() - if head == tail { - return asyncCancelNoAnswer, "no CQE produced for the cancel (waited 500ms)" + timeout := asyncCancelProbeTimeout + deadline := time.Now().Add(timeout) + for waits := 1; ; waits++ { + if err := asyncCancelProbeWait(ring, time.Until(deadline)); err != nil { + return asyncCancelNoAnswer, "SubmitAndWaitTimeout failed: " + err.Error() + } + if head, tail = ring.BeginCQ(); head != tail { + break + } + // An early return with nothing to read is a wait a signal cut + // short: wait once more, for the rest of the time. + if waits == 2 || !time.Now().Before(deadline) { + return asyncCancelNoAnswer, fmt.Sprintf("no CQE produced for the cancel (waited %v, %d wait(s))", timeout, waits) + } } } cqe := ring.cqeAt(head) From 92a8170129f54cbace9fa060dabaeb3f9a15b5fe Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 22:34:46 +0200 Subject: [PATCH 30/34] test(iouring): log the rejecting branch of probe_matches_the_kernel (celeris#681 N4) The branch for a kernel that rejects the cancel flags now logs what it saw (TransplantReapFailed, TransplantHeld, adopted), so a run on such a kernel shows the branch ran. Test-only. --- engine/iouring/async_cancel_probe_test.go | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index ffdf7c00..9dfbdba2 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -371,10 +371,16 @@ func TestReapOnTheRunningKernel(t *testing.T) { "is not what this kernel does", accepts, res, reason) } if !accepts { + // The kernel rejected the forced reap (TransplantReapFailed = 1) + // and left the recv armed: the conn leaves after its next, held, + // response. Logged so a run on such a kernel shows this branch ran. if _, err := unix.Write(f.peer, []byte(fdlGET)); err != nil { t.Fatalf("client write: %v", err) } run(t, f, "the next request", func() bool { return f.tgt.adopted.Load() == 1 }) + t.Logf("celeris681 rejecting branch: the kernel failed the forced reap (TransplantReapFailed=%d), "+ + "and the conn left after its held response (TransplantHeld=%d, adopted %d)", + f.e.metrics.handoffLoss.reapFailed.Load(), f.e.metrics.handoffLoss.held.Load(), f.tgt.adopted.Load()) } if n := f.e.metrics.handoffLoss.handoffInFlight.Load(); n != 0 { t.Fatalf("TransplantHandoffInFlight = %d, want 0", n) From 5ce4986ca653e5bc2a868b08601f2f27e1597658 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sat, 19 Sep 2026 22:47:59 +0200 Subject: [PATCH 31/34] test(iouring): expect a promoted async conn to stay where the probe did not find the cancel flags (celeris#681 N4) Running the witness list on Ubuntu 5.15.0-191 under QEMU (N4) found TestHandoffHasNothingInFlight/async failing there on round 3's head and on this one alike: "0 of 128 busy conns were handed off, want all". Without the cancel flags a promoted async conn cannot be reaped, so it is never offered for the hand-off and stays on io_uring (asyncTransplantEligible, celeris#681 R1). The subtest wanted every conn handed off on every kernel. Where the engine's probe did not find the flags, the async subtest now wants none of the promoted conns handed off, with the loss checks unchanged (no client error, W1 = W2 = 0). A new case, async_without_cancel_flags, takes the probe's answer as a rejection (runAsyncCancelProbe), so that branch runs on every kernel, CI's included. Test-only. --- engine/iouring/fd_lifetime_engine_test.go | 42 ++++++++++++++++++++--- 1 file changed, 38 insertions(+), 4 deletions(-) diff --git a/engine/iouring/fd_lifetime_engine_test.go b/engine/iouring/fd_lifetime_engine_test.go index 12539117..084d4551 100644 --- a/engine/iouring/fd_lifetime_engine_test.go +++ b/engine/iouring/fd_lifetime_engine_test.go @@ -301,11 +301,21 @@ func metricOr(e *Engine, name string) int64 { // completion, or a reap); the async ones are promoted to a dispatch goroutine // that claims its own hand-off at its park, and leave through // finishAsyncTransplant, which reaps the recv the feed path armed. +// +// Where the probe did not find the cancel flags (every kernel before 5.19) +// a promoted async conn cannot be reaped, so it is never offered for the +// hand-off and stays on io_uring (asyncTransplantEligible, celeris#681 R1): +// there the async subtest wants none of them handed off, and no client error. +// It used to want all of them on every kernel, so it failed on 5.15.0-191 +// ("0 of 128 busy conns were handed off"; celeris#681 N4). +// async_without_cancel_flags runs that case on any kernel: the engine's probe +// answer is taken as a rejection (runAsyncCancelProbe). func TestHandoffHasNothingInFlight(t *testing.T) { for _, tc := range []struct { - name string - async bool - }{{"sync", false}, {"async", true}} { + name string + async bool + noFlags bool + }{{"sync", false, false}, {"async", true, false}, {"async_without_cancel_flags", true, true}} { t.Run(tc.name, func(t *testing.T) { var h stream.Handler = transplantTestHandler{} var mut func(*resource.Config) @@ -313,7 +323,21 @@ func TestHandoffHasNothingInFlight(t *testing.T) { h = asyncRouteHandler{} mut = func(c *resource.Config) { c.AsyncHandlers = true } } + if tc.noFlags { + saved := runAsyncCancelProbe + runAsyncCancelProbe = func() (asyncCancelProbe, string) { + return asyncCancelRejected, "celeris681 forced: the cancel flags taken as rejected" + } + resetAsyncCancelProbeCache() + t.Cleanup(func() { + runAsyncCancelProbe = saved + resetAsyncCancelProbeCache() // the next New probes the kernel again + }) + } e, addr := startFDLEngine(t, h, mut) + if tc.noFlags && e.asyncCancelFlags { + t.Fatal("New turned the reap on although the probe's answer was a rejection") + } var maxSeen atomic.Uint64 tgt := &servingTarget{} tgt.onAdopt = func() { @@ -340,7 +364,17 @@ func TestHandoffHasNothingInFlight(t *testing.T) { if res.errs != 0 { t.Errorf("clients saw %d errors (%s), want 0", res.errs, classes(res.byClass)) } - if got := tgt.adopted.Load(); got != conns { + got := tgt.adopted.Load() + if tc.async && !e.asyncCancelFlags { + t.Logf("celeris681 no cancel flags: %d of %d promoted async conns handed off (promotions %d)", + got, conns, e.Metrics().AsyncPromotedConns) + if got != 0 { + t.Errorf("%d of %d promoted async conns were handed off on a worker that cannot reap, want 0: "+ + "such a conn is never offered for the hand-off", got, conns) + } + return + } + if got != conns { t.Errorf("%d of %d busy conns were handed off, want all", got, conns) } }) From 8e09151c72336eb0901e27db34fba8f46de70fee Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sun, 20 Sep 2026 01:21:12 +0200 Subject: [PATCH 32/34] test(iouring): let the kernel, not the probe's answer, decide what the async hand-off subtest demands (celeris#681 M1) TestHandoffHasNothingInFlight/async chose between "all 128 promoted async conns handed off" and "none of them" from e.asyncCancelFlags alone. That boolean is false for every probe answer except an acceptance, so a probe that got no answer -- its private ring failing with EMFILE or ENOMEM, which celeris#681 N1 made a per-New event -- silently downgraded the subtest's strongest assertion into one any run with the reap off passes trivially. The subtest is the registered deterministic killer of mutants m01 and m04, so the mutation claim inherited the weakness. The branch is now chosen by the running kernel: from 5.19, which is where IORING_ASYNC_CANCEL flags exist, the strong branch is demanded and an engine whose reap is off fails the subtest, naming the probe's class from New's own "io_uring engine selected" record rather than probing a second time. Only a kernel that predates 5.19 may take the weak branch. async_without_cancel_flags, which forces the no-flags path on every kernel, is unchanged. --- engine/iouring/fd_lifetime_engine_test.go | 79 ++++++++++++++++++++++- 1 file changed, 77 insertions(+), 2 deletions(-) diff --git a/engine/iouring/fd_lifetime_engine_test.go b/engine/iouring/fd_lifetime_engine_test.go index 084d4551..09ddf6c2 100644 --- a/engine/iouring/fd_lifetime_engine_test.go +++ b/engine/iouring/fd_lifetime_engine_test.go @@ -6,6 +6,7 @@ import ( "bufio" "context" "errors" + "fmt" "io" "log/slog" "net" @@ -310,6 +311,17 @@ func metricOr(e *Engine, name string) int64 { // ("0 of 128 busy conns were handed off"; celeris#681 N4). // async_without_cancel_flags runs that case on any kernel: the engine's probe // answer is taken as a rejection (runAsyncCancelProbe). +// +// Which of the two the async subtest wants is decided by the RUNNING KERNEL, +// not by the probe's answer alone (celeris#681 M1). From 5.19 the kernel has +// the flags, so there the reap must work and all 128 must be handed off; if +// the engine's reap is off on such a kernel the subtest fails, naming the +// probe's class from New's own record. Only a kernel that predates 5.19 may +// take the weak branch. Chosen from the probe alone, a probe that got no +// answer — an EMFILE or ENOMEM in its private ring, which N1 made a per-New +// event — silently turned "all 128 handed off" into "none handed off", which +// such a run passes trivially; this subtest is the registered killer of +// mutants m01 and m04, so the mutation claim inherited that weakness. func TestHandoffHasNothingInFlight(t *testing.T) { for _, tc := range []struct { name string @@ -317,11 +329,18 @@ func TestHandoffHasNothingInFlight(t *testing.T) { noFlags bool }{{"sync", false, false}, {"async", true, false}, {"async_without_cancel_flags", true, true}} { t.Run(tc.name, func(t *testing.T) { + const conns = 128 var h stream.Handler = transplantTestHandler{} + var logs lockedBuffer var mut func(*resource.Config) if tc.async { h = asyncRouteHandler{} - mut = func(c *resource.Config) { c.AsyncHandlers = true } + // New's own records, so a failure can name the probe's + // class without probing a second time (celeris#681 M1). + mut = func(c *resource.Config) { + c.AsyncHandlers = true + c.Logger = slog.New(slog.NewJSONHandler(&logs, nil)) + } } if tc.noFlags { saved := runAsyncCancelProbe @@ -338,6 +357,31 @@ func TestHandoffHasNothingInFlight(t *testing.T) { if tc.noFlags && e.asyncCancelFlags { t.Fatal("New turned the reap on although the probe's answer was a rejection") } + // celeris#681 M1: a kernel from 5.19 has IORING_ASYNC_CANCEL + // flags, so the reap must be on there and the strong branch + // below is the one that runs. Whatever kept it off — a probe + // with no answer, an answer the probe does not recognise, or a + // kernel that claims 5.19 and rejects them — is a failure of + // this subtest, not a reason to want less of it. + kernelHasFlags := e.profile.KernelMajor > 5 || + (e.profile.KernelMajor == 5 && e.profile.KernelMinor >= 19) + if tc.async { + t.Logf("celeris681 M1 kernel=%s (%d.%d, a version with the flags: %v) forced_no_flags=%v reap=%v; %s", + e.profile.KernelVersion, e.profile.KernelMajor, e.profile.KernelMinor, kernelHasFlags, + tc.noFlags, e.asyncCancelFlags, asyncCancelProbeRecords(t, &logs)) + } + if tc.async && !tc.noFlags && kernelHasFlags && !e.asyncCancelFlags { + t.Fatalf("the engine's cancel-flags probe did not find the flags accepted on kernel %s (%d.%d), "+ + "which has had them since 5.19, so the hand-off's reap is off and no promoted async conn is "+ + "offered for it: %s. A probe with no answer is the probe failing, not the kernel answering "+ + "(its private ring can fail with EMFILE or ENOMEM, and celeris#681 N1 made that a per-New "+ + "event); a rejection here is a kernel whose version claims the flags and does not have them. "+ + "Either way the check below must stay 'all %d handed off' on this kernel and must not weaken "+ + "to 'none handed off', which any run with the reap off passes trivially (celeris#681 M1). "+ + "The no-flags path is async_without_cancel_flags, which forces it on every kernel.", + e.profile.KernelVersion, e.profile.KernelMajor, e.profile.KernelMinor, + asyncCancelProbeRecords(t, &logs), conns) + } var maxSeen atomic.Uint64 tgt := &servingTarget{} tgt.onAdopt = func() { @@ -346,7 +390,6 @@ func TestHandoffHasNothingInFlight(t *testing.T) { } } defer tgt.close() - const conns = 128 res := transplantUnderLoad(t, e, addr, conns, tgt, 300*time.Millisecond, 1500*time.Millisecond) if tc.async { if n := e.Metrics().AsyncPromotedConns; n == 0 { @@ -365,6 +408,10 @@ func TestHandoffHasNothingInFlight(t *testing.T) { t.Errorf("clients saw %d errors (%s), want 0", res.errs, classes(res.byClass)) } got := tgt.adopted.Load() + // The weak branch. The gate above leaves it reachable only + // under async_without_cancel_flags, or on a kernel that + // predates 5.19 and so genuinely has no flags to find + // (celeris#681 M1). if tc.async && !e.asyncCancelFlags { t.Logf("celeris681 no cancel flags: %d of %d promoted async conns handed off (promotions %d)", got, conns, e.Metrics().AsyncPromotedConns) @@ -381,6 +428,34 @@ func TestHandoffHasNothingInFlight(t *testing.T) { } } +// asyncCancelProbeRecords describes what the engine's own probe said, from +// the records New wrote to buf: its class from "io_uring engine selected" +// (async_cancel_probe, logged on every kernel) and the reason from the +// "async cancel flags ..." record New adds where the flags were not found +// accepted. Reading New's records rather than calling the probe again keeps +// the message about the engine under test: the answer a second probe gave +// need not be the one this engine was built with (celeris#681 M1, N1). +func asyncCancelProbeRecords(t *testing.T, buf *lockedBuffer) string { + t.Helper() + var parts []string + for _, r := range buf.records(t) { + msg, _ := r["msg"].(string) + switch { + case msg == "io_uring engine selected": + cls, _ := r["async_cancel_probe"].(string) + parts = append(parts, fmt.Sprintf("New's probe class=%q", cls)) + case strings.HasPrefix(msg, "async cancel flags"): + level, _ := r["level"].(string) + reason, _ := r["reason"].(string) + parts = append(parts, fmt.Sprintf("%s %q reason=%q", level, msg, reason)) + } + } + if len(parts) == 0 { + return "no probe record (this subtest gave New no log buffer)" + } + return strings.Join(parts, "; ") +} + // asyncRouteHandler is transplantTestHandler with every route async, so each // conn is promoted to its own dispatch goroutine on its first request. type asyncRouteHandler struct{ transplantTestHandler } From 26f9cb900433311df476bc4db77bfc4689169dab Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sun, 20 Sep 2026 01:21:12 +0200 Subject: [PATCH 33/34] docs(iouring): state the cancel-flags probe seams' locking constraint, and why the two probes in TestReapOnTheRunningKernel stay (celeris#681 N-a, N-d) runAsyncCancelProbe, asyncCancelProbeSubmit, asyncCancelProbeWait and asyncCancelProbeTimeout are read inside probeAsyncCancelFlagsCached's asyncCancelMu critical section and replaced by tests that hold no lock. Nothing in engine/iouring calls t.Parallel(), so nothing races today; the constraint is now written where the seams are declared, with the grep that checks it, rather than enforced with a mutex-taking setter per seam. TestReapOnTheRunningKernel's two probeAsyncCancelFlagsCached calls can be two probes since N1 cached only an answer. They stay: neither subtest weakens a check on the answer it got, and probe_matches_the_kernel Fatalfs when the answer disagrees with what the kernel does with a real reap. --- engine/iouring/async_cancel_probe_test.go | 12 ++++++++++++ engine/iouring/probe.go | 18 +++++++++++++++++- 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/engine/iouring/async_cancel_probe_test.go b/engine/iouring/async_cancel_probe_test.go index 9dfbdba2..46f38008 100644 --- a/engine/iouring/async_cancel_probe_test.go +++ b/engine/iouring/async_cancel_probe_test.go @@ -258,6 +258,18 @@ func TestWorkersCarryTheAsyncCancelProbe(t *testing.T) { // them no reap is placed, on this iteration or any later one; the client's // next request completes the recv, its response is held, and the conn is // handed off at that SEND's completion with nothing in flight. +// +// probe_answer and probe_matches_the_kernel each call +// probeAsyncCancelFlagsCached, which since celeris#681 N1 caches only an +// answer, so on a kernel that gives none these are two probes and may +// disagree. That is left as it is (celeris#681 N-d): neither subtest +// downgrades a check on the answer it got — probe_answer runs the branch the +// answer names, and probe_matches_the_kernel compares the answer with what +// the kernel does with a real reap and t.Fatalf's on any mismatch, so two +// probes that disagree fail loudly rather than quietly weaken. The one place +// where an answer did choose a weaker check is +// TestHandoffHasNothingInFlight/async, and celeris#681 M1 gates that on the +// kernel's version instead. func TestReapOnTheRunningKernel(t *testing.T) { // run submits and lets the kernel complete ops until pred holds, // processing every completion the way the worker loop does. diff --git a/engine/iouring/probe.go b/engine/iouring/probe.go index a6ef1f04..50a85547 100644 --- a/engine/iouring/probe.go +++ b/engine/iouring/probe.go @@ -101,6 +101,19 @@ func probeAsyncCancelFlagsCached() (asyncCancelProbe, string) { // runAsyncCancelProbe is the probe probeAsyncCancelFlagsCached runs. A // variable so a test can give each class of answer to the cache and to New. +// +// SEAM CONSTRAINT (celeris#681 N-a), shared with asyncCancelProbeSubmit, +// asyncCancelProbeWait and asyncCancelProbeTimeout: these four are READ +// inside probeAsyncCancelFlagsCached's asyncCancelMu critical section and +// WRITTEN by tests without it, so a test may only replace one while no +// other goroutine can reach a probe — that is, from its own test goroutine, +// never from a test that has called t.Parallel(), and never while an engine +// it does not own is being built. Nothing in this package calls t.Parallel() +// (git grep -n 't\.Parallel()' -- 'engine/iouring/*_test.go' is empty, while +// the same grep finds engine/provider_test.go), so the constraint holds +// today; it is documented rather than enforced because guarding the +// seams would mean a mutex-taking setter and a save/restore helper for each +// of the four, which is more machinery than a constraint one grep checks. var runAsyncCancelProbe = probeAsyncCancelFlags // SEND_ZC ioprio flags and notification result values. @@ -542,7 +555,10 @@ var newAsyncCancelProbeRing = func() (*Ring, error) { return NewRing(8, 0, 0) } // The probe's submit and its wait for the cancel's completion, and how long // that wait is. Variables so a test can hold the cancel back and cut the wait -// short (celeris#681 N1). +// short (celeris#681 N1). They carry runAsyncCancelProbe's SEAM CONSTRAINT: +// read under asyncCancelMu, replaced by a test that holds no lock, so only +// from a test goroutine that no other probe can run against (celeris#681 +// N-a). var ( asyncCancelProbeSubmit = func(r *Ring) (int, error) { return r.Submit() } asyncCancelProbeWait = func(r *Ring, d time.Duration) error { return r.SubmitAndWaitTimeout(d) } From edca5d2185b504dd27494325674aa765334489e3 Mon Sep 17 00:00:00 2001 From: Albert Bausili Date: Sun, 20 Sep 2026 01:21:12 +0200 Subject: [PATCH 34/34] docs: reflow the two comments round 4 left ragged (celeris#681 N-b) EngineMetrics.TransplantReapUnsupported had a 29-character fragment line mid-paragraph; asyncTransplantEligible's comment had a 131-character line among neighbours that wrap at about 78. Text unchanged. --- engine/engine.go | 26 +++++++++++++------------- engine/iouring/transplant_source.go | 15 ++++++++------- 2 files changed, 21 insertions(+), 20 deletions(-) diff --git a/engine/engine.go b/engine/engine.go index de40d661..ee5542cc 100644 --- a/engine/engine.go +++ b/engine/engine.go @@ -541,19 +541,19 @@ type EngineMetrics struct { //nolint:revive // user-approved name TransplantReapFailed uint64 // TransplantReapUnsupported counts hand-off recv cancels not placed // because the io_uring engine's startup probe did not find the - // IORING_ASYNC_CANCEL flags they need accepted (Linux 5.19 added them; - // a probe that got no answer, or an answer it does not recognise, counts - // the same). It counts only - // connections the worker serves itself (every connection in sync mode, - // and in async mode those not promoted to a dispatch goroutine): such a - // connection stays on io_uring until its armed recv completes on its - // own, and is handed off after the response to the request that recv - // brings, which is held, where the worker has no provided-buffer ring (a - // kernel without the flags has none); with one it stays. A promoted - // async connection is never offered for the hand-off where the flags are - // missing: it stays on io_uring and is not counted. Placement only, - // never a lost request. A rate, 0 wherever the probe finds the flags. - // io_uring-only; on the adaptive engine the sum over both sub-engines. + // IORING_ASYNC_CANCEL flags they need accepted (Linux 5.19 added them; a + // probe that got no answer, or an answer it does not recognise, counts + // the same). It counts only connections the worker serves itself (every + // connection in sync mode, and in async mode those not promoted to a + // dispatch goroutine): such a connection stays on io_uring until its + // armed recv completes on its own, and is handed off after the response + // to the request that recv brings, which is held, where the worker has + // no provided-buffer ring (a kernel without the flags has none); with + // one it stays. A promoted async connection is never offered for the + // hand-off where the flags are missing: it stays on io_uring and is not + // counted. Placement only, never a lost request. A rate, 0 wherever the + // probe finds the flags. io_uring-only; on the adaptive engine the sum + // over both sub-engines. TransplantReapUnsupported uint64 } diff --git a/engine/iouring/transplant_source.go b/engine/iouring/transplant_source.go index 1e4fe761..3edf87df 100644 --- a/engine/iouring/transplant_source.go +++ b/engine/iouring/transplant_source.go @@ -239,16 +239,17 @@ func (w *Worker) reclaimTransplant(newFD int, carry engine.Carryover, cause erro // // A worker that cannot reap (its engine's probeAsyncCancelFlags did not find // IORING_ASYNC_CANCEL flags accepted: the kernel rejected them, or the probe -// got no answer or one it does not recognise, each with the same effect -// here) offers no promoted async conn for the hand-off (celeris#681 R1). Every claim would find the recv the feed path armed after -// the last request, which only a reap can clear, and a promoted conn is never -// held: finishAsyncTransplant would refuse it, and the goroutine that exited -// to make the claim would be respawned by the next request — a spawn and a +// got no answer or one it does not recognise, each with the same effect here) +// offers no promoted async conn for the hand-off (celeris#681 R1). Every +// claim would find the recv the feed path armed after the last request, which +// only a reap can clear, and a promoted conn is never held: +// finishAsyncTransplant would refuse it, and the goroutine that exited to +// make the claim would be respawned by the next request — a spawn and a // detach-queue round trip per request for as long as the drain lasted, with // the conn never leaving. Instead the goroutine parks as it does with no // drain set, and the conn stays on io_uring (placement only). The field is -// set before the worker starts and never written again, so this read from -// the dispatch goroutine is race-free. +// set before the worker starts and never written again, so this read from the +// dispatch goroutine is race-free. func (w *Worker) asyncTransplantEligible(cs *connState) bool { if !w.asyncCancelFlags { return false