From 7bf23d10a8fede7bfa716f32888b668ce71f65ea Mon Sep 17 00:00:00 2001 From: Diogo Martins Date: Sat, 19 Sep 2026 15:18:38 +0100 Subject: [PATCH 1/2] tcp: idle and send-stall timeouts, so a connection finally has a clock There was none anywhere in the TCP connection lifecycle. A peer that connected and went quiet held an fd, a pooled TcpConnection with its native write slab and a recv queue for as long as it liked - and in incremental mode a registered buffer ring plus a gid, which is capped, so an idle connection at the cap converts straight into shed accepts. A peer that stopped READING was worse: its window shuts, the SEND never completes, and FlushAsync parks forever. TCP will not end that either, because a zero window is legitimate and holdable indefinitely, and the only per-accepted-socket option set here is TCP_NODELAY. Two knobs on TcpOptions, both 60s by default, 0 disables: IdleTimeoutMs - nothing received or sent for this long. SendTimeoutMs - a flush outstanding for this long. Both ride the reactor's existing ~250ms ticker, the way TlsService sweeps handshakes and the QUIC transport sweeps idle connections, so a connection closes at the first tick past its deadline rather than exactly on it. They are deliberately two clocks rather than one. A connection with a flush outstanding is not idle, it is sending, so the send clock governs it and the idle one does not - otherwise a large response to a slow peer is reaped for making no INBOUND progress while working perfectly. And the idle clock cannot cover a stall on its own: a peer that keeps SENDING while it has stopped READING refreshes the idle stamp on every inbound completion, so the sweep never fires while that connection's send is wedged. A websocket written from a background task - the shape reported in #234 - is exactly that, and is covered only by the send clock. Teardown is shutdown() + MarkClosed() and nothing else, as in TlsService.SweepHandshakes. shutdown() is what the peer sees and what completes the operation the reactor's ref is waiting on; MarkClosed wakes the handler now, parked on a read or on the very flush being timed out. It deliberately does not clear the table slot or DecRef: the teardown the resulting completions already run is the one that gets the refcount right, and it runs only once the kernel is done with the connection's slab. Releasing the reactor's ref here would let the connection reach zero and be recycled - slab freed or resized - with a SEND the kernel has not given back still pointing into it. The activity stamp reads a clock the reactor caches once per loop pass. Reading Environment.TickCount64 per completion instead measured -8.7% on Tcp/Raw and -4.8% on Tcp/Pipe at 4 reactors: three vDSO calls per request against a 2.4us budget. The sweep runs four times a second, so per-batch granularity is already far finer than anything consuming it. Tests: five, in E2E, with all three negative controls (an active connection is left alone; 0 disables each clock). Proven by disabling the sweep, where the idle connection never closes and the stalled flush stays parked. All suites green: E2E 185, Unit 43, Http 44, Tls 142, Chaos 47, File 4. Bench, Tcp/Raw at 4 reactors, nine alternating A/B pairs: no separable difference. Per-pair sign flips both ways and the medians sit ~1% apart, inside a baseline arm that itself spanned 1.49-1.74M req/s across the session; the swept arm was the tighter of the two (1.53-1.66M). Scope: the idle/keep-alive half of #97. The shared-mode connection cap - Track grows the table unbounded and MaxConnections is incremental-only - is the other half and is not here. --- .../Tcp/TcpConnection.Write.Flush.cs | 19 ++ src/ioxide/Connection/Tcp/TcpConnection.cs | 21 ++ src/ioxide/Native/Native.Socket.cs | 6 + .../Loop/Reactor.Loop.DispatchCompletions.cs | 1 + .../Reactor/Loop/Reactor.Loop.Incremental.cs | 2 + .../Reactor/Loop/Reactor.Loop.SharedRing.cs | 2 + src/ioxide/Reactor/Reactor.Runner.cs | 8 + src/ioxide/Reactor/Reactor.cs | 2 + .../Transport/Tcp/Reactor.Tcp.Sweep.cs | 102 ++++++ .../Reactor/Transport/Tcp/Reactor.Tcp.cs | 7 + .../Reactor/Transport/Tcp/TcpOptions.cs | 46 +++ src/ioxide/Tls/Interop/Sockets.cs | 11 +- .../Ioxide.Tests.E2E/Core/TcpTimeoutTests.cs | 297 ++++++++++++++++++ tests/Ioxide.Tests.E2E/Program.cs | 1 + 14 files changed, 516 insertions(+), 9 deletions(-) create mode 100644 src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.Sweep.cs create mode 100644 tests/Ioxide.Tests.E2E/Core/TcpTimeoutTests.cs diff --git a/src/ioxide/Connection/Tcp/TcpConnection.Write.Flush.cs b/src/ioxide/Connection/Tcp/TcpConnection.Write.Flush.cs index 060e3d2d..cbf03293 100644 --- a/src/ioxide/Connection/Tcp/TcpConnection.Write.Flush.cs +++ b/src/ioxide/Connection/Tcp/TcpConnection.Write.Flush.cs @@ -18,6 +18,24 @@ public sealed unsafe partial class TcpConnection : IValueTaskSource private int _flushArmed; private int _flushInProgress; + /// + /// The reactor's cached clock at the moment a flush was handed over, read by the sweep against + /// . + /// + /// Only meaningful while - it is written on every arm and never + /// cleared, because clearing it would put a store on CompleteFlush, which is the hottest path + /// in the server, to maintain a value nothing reads in that state. + /// + internal long FlushArmedMs; + + /// + /// Whether a flush is outstanding - which is what tells the reactor's sweep that this + /// connection is sending rather than idle, so the send clock governs it and the idle one does + /// not. Read rather than being non-zero, so the stamp never has to + /// double as a flag. + /// + internal bool FlushOutstanding => Volatile.Read(ref _flushInProgress) != 0; + public ValueTask FlushAsync() { if (Volatile.Read(ref _closed) == 1) @@ -123,6 +141,7 @@ private ValueTask FlushCore() _flushSignal.Reset(); WriteInFlight = target; + Volatile.Write(ref FlushArmedMs, _reactor.NowMs); // A segmented response that spilled past the primary slab is gathered into one SENDMSG. _flushVectored = _inOverflow; diff --git a/src/ioxide/Connection/Tcp/TcpConnection.cs b/src/ioxide/Connection/Tcp/TcpConnection.cs index 58b40f78..367bf1f7 100644 --- a/src/ioxide/Connection/Tcp/TcpConnection.cs +++ b/src/ioxide/Connection/Tcp/TcpConnection.cs @@ -12,6 +12,24 @@ public sealed unsafe partial class TcpConnection /// The listener port this connection was accepted on; set per accept. public ushort ListenerPort { get; internal set; } + /// + /// Environment.TickCount64 at the last completion this connection saw in either direction - + /// stamped at accept and on every recv and send completion, read by the reactor's sweep + /// (Reactor.Tcp.Sweep.cs) against . + /// + /// + /// A coarse tick rather than a precise clock on purpose: the sweep runs at ~250 ms and the + /// timeouts it serves are second-scale, so the cheapest read that cannot fall back is enough. + /// Reactor thread only, like the rest of the connection. + /// + internal long LastActivityMs; + + /// + /// Set once the sweep has shut this connection down, so the next tick skips it instead of + /// re-issuing shutdown() every 250 ms until the teardown completions land. + /// + internal bool SweepClosed; + /// /// Whether this connection sends with SEND_ZC (zero-copy). Bound at accept from /// ; kTLS forces it back to plain via the @@ -124,6 +142,9 @@ internal void Clear() Volatile.Write(ref _closed, 0); Volatile.Write(ref _flushArmed, 0); Volatile.Write(ref _flushInProgress, 0); + Volatile.Write(ref FlushArmedMs, 0); + LastActivityMs = _reactor.NowMs; + SweepClosed = false; WriteHead = 0; WriteTail = 0; diff --git a/src/ioxide/Native/Native.Socket.cs b/src/ioxide/Native/Native.Socket.cs index 16788cde..d83bcbfa 100644 --- a/src/ioxide/Native/Native.Socket.cs +++ b/src/ioxide/Native/Native.Socket.cs @@ -16,6 +16,9 @@ public static unsafe partial class Native { public const int SO_RCVBUF = 8; public const int SO_REUSEPORT = 15; + /// shutdown(2) how: both directions. + public const int SHUT_RDWR = 2; + public const int AF_INET6 = 10; public const int IPPROTO_IPV6 = 41; public const int IPV6_V6ONLY = 26; @@ -27,6 +30,9 @@ public static unsafe partial class Native { /// bind to port 0 (QUIC client sockets take an ephemeral port). [DllImport("libc")] public static extern int getsockname(int fd, void* addr, uint* len); [DllImport("libc")] public static extern int listen(int fd, int backlog); + /// End a connection at the socket: the peer gets a FIN and any outstanding io_uring recv or + /// send completes, which is what releases a connection the reactor still holds a ref to. + [DllImport("libc")] public static extern int shutdown(int fd, int how); [DllImport("libc")] public static extern int setsockopt(int fd, int level, int optname, void* optval, uint optlen); [DllImport("libc")] public static extern int getsockopt(int fd, int level, int optname, void* optval, uint* optlen); diff --git a/src/ioxide/Reactor/Loop/Reactor.Loop.DispatchCompletions.cs b/src/ioxide/Reactor/Loop/Reactor.Loop.DispatchCompletions.cs index 18742f84..f61eb2b4 100644 --- a/src/ioxide/Reactor/Loop/Reactor.Loop.DispatchCompletions.cs +++ b/src/ioxide/Reactor/Loop/Reactor.Loop.DispatchCompletions.cs @@ -38,6 +38,7 @@ private void OnSendCompletion(int fd, ushort gen, int res, uint cqeFlags) return; } conn.WriteHead += res; + conn.LastActivityMs = NowMs; // A zero-copy send posts its data CQE with F_MORE and a notif will follow; hold the slab until // that notif arrives. Plain SEND never sets F_MORE, so this is a no-op for it. diff --git a/src/ioxide/Reactor/Loop/Reactor.Loop.Incremental.cs b/src/ioxide/Reactor/Loop/Reactor.Loop.Incremental.cs index 3fc1b423..c4006138 100644 --- a/src/ioxide/Reactor/Loop/Reactor.Loop.Incremental.cs +++ b/src/ioxide/Reactor/Loop/Reactor.Loop.Incremental.cs @@ -185,6 +185,8 @@ private void LoopIncremental() break; } + NowMs = Environment.TickCount64; // one read per batch; see Reactor.Tcp.Sweep.cs + uint ready = _ring.CqReady(); for (uint i = 0; i < ready; i++) { diff --git a/src/ioxide/Reactor/Loop/Reactor.Loop.SharedRing.cs b/src/ioxide/Reactor/Loop/Reactor.Loop.SharedRing.cs index 9b2699b8..5f52901f 100644 --- a/src/ioxide/Reactor/Loop/Reactor.Loop.SharedRing.cs +++ b/src/ioxide/Reactor/Loop/Reactor.Loop.SharedRing.cs @@ -63,6 +63,8 @@ private void LoopSharedRing() break; } + NowMs = Environment.TickCount64; // one read per batch; see Reactor.Tcp.Sweep.cs + uint ready = _ring.CqReady(); for (uint i = 0; i < ready; i++) { diff --git a/src/ioxide/Reactor/Reactor.Runner.cs b/src/ioxide/Reactor/Reactor.Runner.cs index ac603b5b..edfbcd69 100644 --- a/src/ioxide/Reactor/Reactor.Runner.cs +++ b/src/ioxide/Reactor/Reactor.Runner.cs @@ -43,6 +43,14 @@ public void Run() AnnounceListening(); ArmTcpAccepts(); ArmWakePoll(); + + // After OnStart, so a reactor that serves no TCP - or has both clocks off - registers + // nothing and the sweep costs it not even a table walk. + if (TcpSweepEnabled) + { + AddTicker(TcpSweep); + } + StartTicker(); if (_incremental) LoopIncremental(); diff --git a/src/ioxide/Reactor/Reactor.cs b/src/ioxide/Reactor/Reactor.cs index 96cf766f..60d33c6e 100644 --- a/src/ioxide/Reactor/Reactor.cs +++ b/src/ioxide/Reactor/Reactor.cs @@ -129,6 +129,8 @@ public Reactor(int id, ServerConfig config) _incRecvBufferSize = (uint)inc.RecvBufferSize; _pool = new Stack(_tcp.PoolMax); _zeroCopySend = _tcp.ZeroCopySend; + _idleTimeoutMs = _tcp.IdleTimeoutMs; + _sendTimeoutMs = _tcp.SendTimeoutMs; } [MethodImpl(MethodImplOptions.AggressiveInlining)] diff --git a/src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.Sweep.cs b/src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.Sweep.cs new file mode 100644 index 00000000..5bd29c15 --- /dev/null +++ b/src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.Sweep.cs @@ -0,0 +1,102 @@ +using static ioxide.Native; + +namespace ioxide; + +/// +/// The clocks on a TCP connection's lifecycle, which until now had none: an idle connection is +/// reaped, and so is one whose send the peer stopped draining. +/// +/// +/// Rides the reactor's existing ~250 ms ticker rather than arming anything of its own, which is +/// what does for handshakes and the QUIC transport does for +/// idle connections. Second-scale timeouts do not need better granularity than that, and a +/// connection closes at the first tick past its deadline rather than exactly on it. +/// +public sealed unsafe partial class Reactor +{ + private readonly int _idleTimeoutMs; + private readonly int _sendTimeoutMs; + + /// + /// Environment.TickCount64, refreshed once per loop pass rather than read per completion. + /// + /// The stamps this feeds are read by a sweep that runs four times a second, so a clock good to + /// one batch of completions is far finer than anything that consumes it - while reading the + /// real one per CQE put a vDSO call on both the recv and the send hot path, three per request, + /// and measured as a 4-9% throughput cost on the small-response samples. Refreshing here is one + /// read per io_uring_enter, amortised over the whole batch it returned. + /// + internal long NowMs = Environment.TickCount64; + + private bool TcpSweepEnabled => _tcpEnabled && (_idleTimeoutMs > 0 || _sendTimeoutMs > 0); + + /// + /// One pass over the connection table. Runs on the reactor thread from the ticker, so it owns + /// the table outright and can touch a connection directly. + /// + private void TcpSweep() + { + long now = Environment.TickCount64; + TcpConnection?[] conns = _connections; + + for (int fd = 0; fd < conns.Length; fd++) + { + TcpConnection? conn = conns[fd]; + if (conn is null || conn.SweepClosed) + { + continue; + } + + // A connection with a flush outstanding is not idle, it is sending - so the send clock + // governs it and the idle one does not apply. Without this split a large response to a + // slow peer would be reaped for making no INBOUND progress while it was working + // perfectly: under MSG_WAITALL the whole flush is a single completion, so nothing + // refreshes the activity stamp for as long as the send legitimately takes. + if (conn.FlushOutstanding) + { + if (_sendTimeoutMs > 0 && now - Volatile.Read(ref conn.FlushArmedMs) > _sendTimeoutMs) + { + TcpSweepClose(conn); + } + continue; + } + + if (_idleTimeoutMs > 0 && now - conn.LastActivityMs > _idleTimeoutMs) + { + TcpSweepClose(conn); + } + } + } + + /// + /// End one connection the sweep has condemned - and nothing more than that. + /// + /// + /// Both halves are needed and neither is redundant, exactly as in TlsService.SweepHandshakes. + /// + /// shutdown() is what the PEER sees, and it is also what releases the connection: a + /// TcpConnection is held by two refs, the handler's and the reactor's, and the reactor's is + /// given up only when its outstanding operation completes. For an idle connection that is a + /// multishot recv against a peer saying nothing, which otherwise never completes; for a stalled + /// one it is a SEND the peer's closed window is holding, which otherwise never completes + /// either. Shutting the socket down ends both. + /// + /// MarkClosed is what wakes the handler NOW - parked on a read, or on the very flush being + /// timed out - with the closed state its loop already knows how to handle, rather than one + /// io_uring round trip later. + /// + /// What this deliberately does NOT do is clear the table slot, cancel, or DecRef. The teardown + /// those completions already run (CloseFromRecv, and the send path's res <= 0 branch) is the + /// one that gets the refcount right, and it only runs once the kernel has finished with the + /// connection's slab. Releasing the reactor's ref here instead would let the connection reach + /// zero - and be recycled, with its slab freed or resized - while a SEND the kernel has not + /// given back still points into it. + /// + private void TcpSweepClose(TcpConnection conn) + { + conn.SweepClosed = true; // one shutdown per connection, not one per tick until it lands + + shutdown(conn.ClientFd, SHUT_RDWR); + conn.MarkClosed(); + } +} diff --git a/src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.cs b/src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.cs index a610e9a1..cbde5d09 100644 --- a/src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.cs +++ b/src/ioxide/Reactor/Transport/Tcp/Reactor.Tcp.cs @@ -112,6 +112,7 @@ private void OnTcpRecvCompletionShared(int fd, ushort gen, int res, uint flags) // return to the group. if (conn != null) { + conn.LastActivityMs = NowMs; _recvStarved.Add(((ulong)gen << 32) | (uint)fd); } return; @@ -141,6 +142,8 @@ private void OnTcpRecvCompletionShared(int fd, ushort gen, int res, uint flags) return; } + conn.LastActivityMs = NowMs; + byte* ptr = hasBuf ? _bufSlab + (nuint)bid * (nuint)_recvBufferSize : null; if (!conn.Complete(res, bid, hasBuf, ptr)) { @@ -172,6 +175,7 @@ private void OnTcpRecvCompletionIncremental(int fd, ushort gen, int res, uint fl // still holds buffers (#93). Park; the loop re-arms once a buffer recycles. if (conn != null) { + conn.LastActivityMs = NowMs; _recvStarved.Add(((ulong)gen << 32) | (uint)fd); } return; @@ -192,6 +196,8 @@ private void OnTcpRecvCompletionIncremental(int fd, ushort gen, int res, uint fl return; // stale CQE; its ring is already gone } + conn.LastActivityMs = NowMs; + // Data lands at the buffer's running offset; the kernel keeps appending // to this bid until the buffer is full (F_BUF_MORE clear). byte* ptr = conn.BufSlab + (nuint)bid * (nuint)_incRecvBufferSize + (nuint)conn.CumOffset![bid]; @@ -243,6 +249,7 @@ private void OnTcpAcceptCompletion(int listenFd, int res, bool more) Track(clientFd, conn); conn.InitRefs(); conn.ListenerPort = PortOf(listenFd); + conn.LastActivityMs = NowMs; // the idle clock starts at accept if (_incremental) { diff --git a/src/ioxide/Reactor/Transport/Tcp/TcpOptions.cs b/src/ioxide/Reactor/Transport/Tcp/TcpOptions.cs index f68dbf8f..bc5246c0 100644 --- a/src/ioxide/Reactor/Transport/Tcp/TcpOptions.cs +++ b/src/ioxide/Reactor/Transport/Tcp/TcpOptions.cs @@ -45,4 +45,50 @@ public sealed record TcpOptions // Per-connection SPSC recv queue depth (power of two); overflow closes the connection. public int RecvQueueEntries { get; init; } = 64; + + /// + /// Close a connection that has neither received nor sent anything for this long. 0 disables. + /// + /// What it defends: a peer that connects and goes quiet holds an fd, a pooled + /// with its native write slab, and a recv queue - and in + /// incremental mode a registered buffer ring plus a gid, which is capped, so an idle + /// connection at the cap converts directly into shed accepts. + /// + /// Enforced on the reactor's ticker, so the granularity is the tick (~250 ms) and a connection + /// closes at the first tick after its deadline rather than exactly on it. + /// + /// + /// A connection with a flush in flight is NOT idle - it is sending, and + /// governs it instead. Otherwise a large response to a slow peer + /// would be reaped for making no INBOUND progress while it was working perfectly. + /// + /// The shape to check before deploying this: a protocol that legitimately goes quiet for + /// longer than the timeout in both directions - an idle websocket, a long-poll - is closed by + /// it. Raise it past the protocol's own keep-alive interval, or set 0 and bound those + /// connections some other way. + /// + public int IdleTimeoutMs { get; init; } = 60_000; + + /// + /// Close a connection whose flush has been in flight for this long. 0 disables. + /// + /// What it defends: a peer that stops reading. Its window shuts, the socket send buffer fills, + /// and the SEND never completes - so FlushAsync parks forever, holding the connection, + /// its slab and the handler's state. TCP will not end it either: a zero window is legitimate + /// and a peer can hold one indefinitely. Nothing else in the stack bounds this. + /// + /// + /// This is the deadline for the WHOLE flush, not for progress within it, because MSG_WAITALL + /// (the default - see ) coalesces a flush into a single + /// completion: there is no per-chunk signal to measure progress against. So set it against the + /// slowest legitimate full response, not against a stall - a large body over a slow link is + /// the false positive to watch for. + /// + /// It is deliberately not folded into : a peer that keeps SENDING + /// while it has stopped READING refreshes the idle stamp on every inbound completion, so an + /// idle sweep never fires while that connection's send is wedged. Duplex protocols - a + /// websocket written from a background task is the reported case - need this clock and are not + /// covered by the other one. + /// + public int SendTimeoutMs { get; init; } = 60_000; } diff --git a/src/ioxide/Tls/Interop/Sockets.cs b/src/ioxide/Tls/Interop/Sockets.cs index b8f07f12..9aba08cb 100644 --- a/src/ioxide/Tls/Interop/Sockets.cs +++ b/src/ioxide/Tls/Interop/Sockets.cs @@ -1,15 +1,8 @@ -using System.Runtime.InteropServices; - namespace ioxide.tls; /// The one socket call the TLS module needs that is not about TLS itself. -internal static partial class Sockets +internal static class Sockets { - private const int SHUT_RDWR = 2; - - [LibraryImport("libc", SetLastError = true)] - private static partial int shutdown(int fd, int how); - /// /// Ends a connection at the socket, so the peer gets a FIN and any outstanding io_uring recv /// completes with EOF - which is what actually releases a connection the reactor is still @@ -24,7 +17,7 @@ public static void Shutdown(int fd) { if (fd >= 0) { - _ = shutdown(fd, SHUT_RDWR); + _ = Native.shutdown(fd, Native.SHUT_RDWR); } } } diff --git a/tests/Ioxide.Tests.E2E/Core/TcpTimeoutTests.cs b/tests/Ioxide.Tests.E2E/Core/TcpTimeoutTests.cs new file mode 100644 index 00000000..4979ac7f --- /dev/null +++ b/tests/Ioxide.Tests.E2E/Core/TcpTimeoutTests.cs @@ -0,0 +1,297 @@ +using System.Net.Sockets; +using System.Text; +using ioxide; +using ioxide.utils; + +namespace Ioxide.Tests; + +/// +/// The two clocks on a TCP connection: reaps one that has +/// gone quiet, reaps one whose peer stopped draining. +/// Before these there was no clock anywhere in the TCP connection lifecycle, so both shapes held +/// an fd, a pooled connection and its native write slab for as long as the peer cared to. +/// +/// +/// Timeouts here are hundreds of milliseconds rather than the second-scale defaults, because the +/// sweep's granularity is the reactor's ~250 ms ticker and a test should not pay a real one. Every +/// deadline below is generous against that tick, not tight against it - the assertion is that the +/// connection closes at all, never that it closed at a particular moment. +/// +internal static class TcpTimeoutTests +{ + /// Two ticks plus slack: what "the sweep has certainly run" costs. + private const int SweepGraceMs = 4_000; + + public static void Register(Runner runner) + { + runner.Test("tcp/idle: a connection that goes quiet is closed at the idle timeout", () => + { + var closed = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + + int port = StartWith(idleMs: 500, sendMs: 0, async (_, conn) => + { + try + { + if (!await IsThisTestsConnection(conn)) + { + return; + } + + // Parked here with a peer that has gone quiet. Only the sweep ends this - which + // is the point: nothing else was ever going to. + conn.ResetRead(); + RecvSnapshot snapshot = await conn.ReadAsync(); + closed.TrySetResult(snapshot.IsClosed); + } + finally + { + conn.DecRef(); + } + }); + + using var client = new TcpClient(); + client.Connect("127.0.0.1", port); + client.ReceiveTimeout = SweepGraceMs; + + // One byte, then silence. It marks this connection as the test's - and it means the + // assertion is that an ACTIVE connection going quiet is reaped, not merely that a + // connection which never said anything was. + client.GetStream().Write("hi"u8); + + // The peer's side of it: shutdown() reaches the client as a FIN, so a read returns 0. + int n = client.GetStream().Read(new byte[16], 0, 16); + + Assert.Equal(0, n); + Assert.True(closed.Task.Wait(SweepGraceMs), "the handler was never woken by the sweep"); + Assert.True(closed.Task.Result, "the handler woke, but not with a closed snapshot"); + }); + + runner.Test("tcp/idle: a connection still talking is left alone", () => + { + // The false positive that would make the whole feature unusable. Same timeout as the + // test above, driven for well over three times its length - if activity did not refresh + // the stamp, this closes long before the loop finishes. + int port = StartWith(idleMs: 500, sendMs: 0, EchoHandler); + + using var client = new TcpClient(); + client.Connect("127.0.0.1", port); + client.ReceiveTimeout = 2_000; + NetworkStream stream = client.GetStream(); + + var reply = new byte[2]; + for (int i = 0; i < 10; i++) + { + stream.Write("ping"u8); + Assert.Equal(2, stream.Read(reply, 0, 2)); + Thread.Sleep(200); // 2s of traffic at 200ms intervals, against a 500ms timeout + } + + // Still usable after the loop: the sweep never touched it. + stream.Write("ping"u8); + Assert.Equal(2, stream.Read(reply, 0, 2)); + }); + + runner.Test("tcp/idle: 0 disables the sweep", () => + { + int port = StartWith(idleMs: 0, sendMs: 0, EchoHandler); + + using var client = new TcpClient(); + client.Connect("127.0.0.1", port); + client.ReceiveTimeout = 2_000; + NetworkStream stream = client.GetStream(); + + // Quiet for several ticks. With the sweep off nothing may reap this, so the connection + // must still answer afterwards. + Thread.Sleep(1_500); + + stream.Write("ping"u8); + var reply = new byte[2]; + Assert.Equal(2, stream.Read(reply, 0, 2)); + }); + + runner.Test("tcp/send: a flush the peer stopped draining is released, not parked forever", () => + { + // The reported shape (#234's workload, and the reason the idle clock alone is not + // enough): the peer keeps the connection open and simply stops reading. Its window + // shuts, the SEND never completes, and FlushAsync parks with no bound at all. + var report = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + + int port = StartWith(idleMs: 0, sendMs: 700, StalledFlushHandler(report)); + + using var client = new TcpClient(); + + // Set before the connect so it is what gets advertised: the server then parks within a + // megabyte or so instead of after however much this box's autotuning decides to buffer. + client.ReceiveBufferSize = 4096; + client.Connect("127.0.0.1", port); + client.GetStream().Write("GET / HTTP/1.1\r\n\r\n"u8); + + // Deliberately never reads. Holding the socket open is the whole reproduction. + Assert.True(report.Task.Wait(20_000), "the handler never reported - its flush is still parked"); + Assert.Equal("", report.Task.Result); + }); + + runner.Test("tcp/send: 0 leaves a stalled flush parked", () => + { + // The control, and the proof that the test above measures the sweep rather than some + // other teardown: with the clock off, the same reproduction must NOT come back. + var report = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + + int port = StartWith(idleMs: 0, sendMs: 0, StalledFlushHandler(report)); + + using var client = new TcpClient(); + client.ReceiveBufferSize = 4096; + client.Connect("127.0.0.1", port); + client.GetStream().Write("GET / HTTP/1.1\r\n\r\n"u8); + + Assert.True(!report.Task.Wait(3_000), + "a flush was released with no send timeout configured - something else is reaping it"); + }); + } + + /// + /// Reads until this connection delivers actual bytes, and says whether it ever did. + /// + /// + /// The harness proves a server is listening by connecting a TcpClient and dropping it + /// (TestServer.WaitForListen), so every server here serves one connection that sends nothing + /// and closes at once. A handler that reported on that one was not measuring the test's + /// connection at all - and on the send path it was worse than useless: a flush on an + /// already-closed connection takes FlushAsync's _closed early-out and returns instantly, so the + /// probe's handler "absorbed" 128 MiB without a single byte reaching a socket, and reported + /// that no flush ever parked. + /// + private static async Task IsThisTestsConnection(TcpConnection conn) + { + while (true) + { + RecvSnapshot snapshot = await conn.ReadAsync(); + + bool received = false; + while (conn.TryGetItem(snapshot, out SpscRecvRing.Item item)) + { + if (item.HasBuffer) + { + received = true; + conn.ReturnBuffer(in item); + } + } + + if (received) + { + return true; + } + if (snapshot.IsClosed) + { + return false; // the probe + } + conn.ResetRead(); + } + } + + /// Answers every read with two bytes, so a test can keep a connection demonstrably alive. + private static async Task EchoHandler(Reactor reactor, TcpConnection conn) + { + try + { + while (true) + { + RecvSnapshot snapshot = await conn.ReadAsync(); + + while (conn.TryGetItem(snapshot, out SpscRecvRing.Item item)) + { + if (item.HasBuffer) + { + conn.ReturnBuffer(in item); + } + } + + conn.Write("ok"u8); + await conn.FlushAsync(); + + if (snapshot.IsClosed) + { + return; + } + conn.ResetRead(); + } + } + finally + { + conn.DecRef(); + } + } + + /// + /// Writes until one flush stops coming back, then waits on it. Reports "" once that flush is + /// released, or whatever went wrong instead - and never reports at all while it stays parked, + /// which is what the control test asserts. + /// + private static Func StalledFlushHandler(TaskCompletionSource report) + => async (_, conn) => + { + try + { + if (!await IsThisTestsConnection(conn)) + { + return; + } + + // Chunked rather than one guessed-at size: how much a loopback pair absorbs before + // the send stops completing is a property of the box, not a constant. + byte[] chunk = new byte[256 * 1024]; + Task? parked = null; + long written = 0; + + for (int attempt = 0; attempt < 128 && parked is null; attempt++) + { + conn.Write(chunk); + Task flush = conn.FlushAsync().AsTask(); + + // A deadline, not a timing assertion: a flush that has not come back is the + // state under test, and one that has simply costs another chunk. + if (await Task.WhenAny(flush, Task.Delay(500)) != flush) + { + parked = flush; + } + else + { + written += chunk.Length; + } + } + + if (parked is null) + { + report.TrySetResult( + $"the peer drained {written / (1024 * 1024)} MiB; no flush ever stayed in flight"); + return; + } + + await parked; + report.TrySetResult(""); + } + catch (Exception e) + { + report.TrySetResult($"{e.GetType().Name}: {e.Message}"); + } + finally + { + conn.DecRef(); + } + }; + + private static int StartWith(int idleMs, int sendMs, Func handle) + => TestServer.StartConfigured(handle, new ServerConfig + { + RecvBufferSize = 4096, + RecvSlots = 256, + Tcp = new TcpOptions + { + WriteSlabSize = 256 * 1024, + PoolMax = 64, + RecvQueueEntries = 64, + IdleTimeoutMs = idleMs, + SendTimeoutMs = sendMs, + }, + }).Port; +} diff --git a/tests/Ioxide.Tests.E2E/Program.cs b/tests/Ioxide.Tests.E2E/Program.cs index 00f35769..8207d4d0 100644 --- a/tests/Ioxide.Tests.E2E/Program.cs +++ b/tests/Ioxide.Tests.E2E/Program.cs @@ -15,6 +15,7 @@ private static int Main() TimerTests.Register(runner); AffinityTests.Register(runner); HardeningTests.Register(runner); + TcpTimeoutTests.Register(runner); UdpTests.Register(runner); QuicTests.Register(runner); QuicEngineTests.Register(runner); From fe9431759809d65fbd022e5e91f4328c6ee03b4b Mon Sep 17 00:00:00 2001 From: Diogo Martins Date: Sun, 20 Sep 2026 14:06:56 +0100 Subject: [PATCH 2/2] release: 0.14.236 All twelve published packages share one version, as they always have. Also aligns the two in-repo statements of that version, which had drifted far enough to be actively misleading (#224): IoxideRuntime.Version said "0.0.17" and the README badge line said 0.4.169, against packages on 0.13.233. Both are the version of the same thing, so a release that moved one and left the others is what produced that spread in the first place. Research/* keeps its own versions - those are separate experiments, not published from this set. --- README.md | 2 +- src/clients/ioxide.file/ioxide.file.csproj | 2 +- src/clients/ioxide.httpclient/ioxide.httpclient.csproj | 2 +- src/clients/ioxide.pg/ioxide.pg.csproj | 2 +- src/clients/ioxide.redis/ioxide.redis.csproj | 2 +- src/clients/ioxide.timer/ioxide.timer.csproj | 2 +- src/ioxide/IoxideRuntime.cs | 2 +- src/ioxide/ioxide.csproj | 2 +- src/protocols/ioxide.http2/ioxide.http2.csproj | 2 +- src/protocols/ioxide.http3/ioxide.http3.csproj | 2 +- src/protocols/ioxide.nghttp2/ioxide.nghttp2.csproj | 2 +- src/protocols/ioxide.nghttp3/ioxide.nghttp3.csproj | 2 +- src/protocols/ioxide.ngtcp2/ioxide.ngtcp2.csproj | 2 +- src/serving/ioxide.Kestrel/ioxide.Kestrel.csproj | 2 +- 14 files changed, 14 insertions(+), 14 deletions(-) diff --git a/README.md b/README.md index 366225a0..4ebae3aa 100644 --- a/README.md +++ b/README.md @@ -20,7 +20,7 @@ ioxide hands you raw bytes and stays out of HTTP; when you want a framework on t `ioxide.Kestrel` swaps the transport under an existing ASP.NET Core app with your endpoints unchanged. -> Linux 6.1+ · .NET 10 / .NET 11 · `0.4.169` - experimental +> Linux 6.1+ · .NET 10 / .NET 11 · `0.14.236` - experimental **[Documentation](https://mda2av.github.io/ioxide/)** - architecture, guides, and every example as runnable code side by side. diff --git a/src/clients/ioxide.file/ioxide.file.csproj b/src/clients/ioxide.file/ioxide.file.csproj index b305908c..eb3f4454 100644 --- a/src/clients/ioxide.file/ioxide.file.csproj +++ b/src/clients/ioxide.file/ioxide.file.csproj @@ -8,7 +8,7 @@ ioxide.file ioxide.file - 0.13.233 + 0.14.236 MDA2AV File serving for the ioxide io_uring runtime: immutable asset snapshots with baked responses, pooled positional ring reads, atomic reloads. MIT diff --git a/src/clients/ioxide.httpclient/ioxide.httpclient.csproj b/src/clients/ioxide.httpclient/ioxide.httpclient.csproj index 6882d04c..193f6993 100644 --- a/src/clients/ioxide.httpclient/ioxide.httpclient.csproj +++ b/src/clients/ioxide.httpclient/ioxide.httpclient.csproj @@ -8,7 +8,7 @@ ioxide.httpclient ioxide.httpclient - 0.13.233 + 0.14.236 MDA2AV The ring-native HTTP/1.1 client for the ioxide io_uring runtime - the upstream leg between a proxy and an origin. Connections are opened on the reactor thread that will use them, so a request never crosses a thread on its way out or back, and every response resumes the awaiting handler inline on its own reactor. Includes client-side TLS (SNI, ALPN, certificate verification and client certificates for mutual TLS) for https:// origins. Depends on ioxide core alone: no protocol package, no native asset. MIT diff --git a/src/clients/ioxide.pg/ioxide.pg.csproj b/src/clients/ioxide.pg/ioxide.pg.csproj index 43318884..b5e57bf8 100644 --- a/src/clients/ioxide.pg/ioxide.pg.csproj +++ b/src/clients/ioxide.pg/ioxide.pg.csproj @@ -8,7 +8,7 @@ ioxide.pg ioxide.pg - 0.13.233 + 0.14.236 MDA2AV Postgres driver for the ioxide io_uring runtime: pooled ring-native connections per reactor, ring-native connect and handshake, inline completion resume. MIT diff --git a/src/clients/ioxide.redis/ioxide.redis.csproj b/src/clients/ioxide.redis/ioxide.redis.csproj index 6b77a7a3..a87204b3 100644 --- a/src/clients/ioxide.redis/ioxide.redis.csproj +++ b/src/clients/ioxide.redis/ioxide.redis.csproj @@ -8,7 +8,7 @@ ioxide.redis ioxide.redis - 0.13.233 + 0.14.236 MDA2AV Redis client for the ioxide io_uring runtime: pooled ring-native connections per reactor, full RESP2 protocol, a generic command API plus typed helpers (strings, keys, hashes, lists, sets, sorted sets, pub/sub, transactions, scripting), and pipelining. Inline completion resume. MIT diff --git a/src/clients/ioxide.timer/ioxide.timer.csproj b/src/clients/ioxide.timer/ioxide.timer.csproj index f592d1a8..aeb7ffa8 100644 --- a/src/clients/ioxide.timer/ioxide.timer.csproj +++ b/src/clients/ioxide.timer/ioxide.timer.csproj @@ -8,7 +8,7 @@ ioxide.timer ioxide.timer - 0.13.233 + 0.14.236 MDA2AV Deadlines for the ioxide io_uring runtime: waits submitted to the reactor's own ring with IORING_OP_TIMEOUT, completing inline on the reactor that owns the caller, with no timer thread and nothing allocated per wait. MIT diff --git a/src/ioxide/IoxideRuntime.cs b/src/ioxide/IoxideRuntime.cs index a08e551e..b7dbb96d 100644 --- a/src/ioxide/IoxideRuntime.cs +++ b/src/ioxide/IoxideRuntime.cs @@ -7,7 +7,7 @@ namespace ioxide; /// public static class IoxideRuntime { - public const string Version = "0.0.17"; + public const string Version = "0.14.236"; // Wiring (a builder API will eventually wrap this): // var reactor = new Reactor(id, config); // implements IRingHost diff --git a/src/ioxide/ioxide.csproj b/src/ioxide/ioxide.csproj index d98b1460..f4d525b8 100644 --- a/src/ioxide/ioxide.csproj +++ b/src/ioxide/ioxide.csproj @@ -8,7 +8,7 @@ ioxide ioxide - 0.13.233 + 0.14.236 MDA2AV A shared-nothing io_uring runtime for .NET: one ring per reactor thread, inline completions, zero native dependencies. The engine - reactor, connection, and the IRingHost client seam. Includes TLS termination: the OpenSSL handshake driven over the ring, then kernel TLS (kTLS) transmit offload, so handlers keep writing plaintext. TLS needs OpenSSL 3 and the Linux tls module; nothing else does, and neither is loaded unless you use it. MIT diff --git a/src/protocols/ioxide.http2/ioxide.http2.csproj b/src/protocols/ioxide.http2/ioxide.http2.csproj index 96de3ead..d042d1ad 100644 --- a/src/protocols/ioxide.http2/ioxide.http2.csproj +++ b/src/protocols/ioxide.http2/ioxide.http2.csproj @@ -8,7 +8,7 @@ ioxide.http2 ioxide.http2 - 0.13.233 + 0.14.236 MDA2AV Pure-C# HTTP/2 for the ioxide io_uring runtime: framing, HPACK (static and dynamic tables, Huffman) and flow control, with zero native code. Serves h2c with prior knowledge and h2 over TLS by ALPN, buffered or streamed in either direction. MIT diff --git a/src/protocols/ioxide.http3/ioxide.http3.csproj b/src/protocols/ioxide.http3/ioxide.http3.csproj index 2dc495fa..787424c3 100644 --- a/src/protocols/ioxide.http3/ioxide.http3.csproj +++ b/src/protocols/ioxide.http3/ioxide.http3.csproj @@ -8,7 +8,7 @@ ioxide.http3 ioxide.http3 - 0.13.233 + 0.14.236 MDA2AV Pure C# HTTP/3 for the ioxide io_uring runtime: frame parsing, QPACK (static table + Huffman) and request dispatch with zero native dependencies. Rides any QuicConnection via its stream read surface - engine-agnostic, drop-in alternative to ioxide.nghttp3. MIT diff --git a/src/protocols/ioxide.nghttp2/ioxide.nghttp2.csproj b/src/protocols/ioxide.nghttp2/ioxide.nghttp2.csproj index 667ee8b2..e3340802 100644 --- a/src/protocols/ioxide.nghttp2/ioxide.nghttp2.csproj +++ b/src/protocols/ioxide.nghttp2/ioxide.nghttp2.csproj @@ -8,7 +8,7 @@ ioxide.nghttp2 ioxide.nghttp2 - 0.13.233 + 0.14.236 MDA2AV HTTP/2 for the ioxide io_uring runtime: framing, HPACK and flow control from nghttp2, statically linked behind a small shim with no external dependencies beyond libc. Serves HTTP/2 over any TcpConnection - h2c with prior knowledge, or h2 over TLS via ALPN. nghttp2 is sans-I/O, so ioxide keeps the ring and the loop. MIT diff --git a/src/protocols/ioxide.nghttp3/ioxide.nghttp3.csproj b/src/protocols/ioxide.nghttp3/ioxide.nghttp3.csproj index b6f7bffa..7483ed4e 100644 --- a/src/protocols/ioxide.nghttp3/ioxide.nghttp3.csproj +++ b/src/protocols/ioxide.nghttp3/ioxide.nghttp3.csproj @@ -8,7 +8,7 @@ ioxide.nghttp3 ioxide.nghttp3 - 0.13.233 + 0.14.236 MDA2AV HTTP/3 layer for the ioxide io_uring runtime: nghttp3 (H3 + QPACK) bundled as a single self-contained native library with no external dependencies. Rides any QuicConnection via its stream read surface - engine-agnostic, no ioxide.ngtcp2 dependency. MIT diff --git a/src/protocols/ioxide.ngtcp2/ioxide.ngtcp2.csproj b/src/protocols/ioxide.ngtcp2/ioxide.ngtcp2.csproj index 95a14681..0ba4b3f4 100644 --- a/src/protocols/ioxide.ngtcp2/ioxide.ngtcp2.csproj +++ b/src/protocols/ioxide.ngtcp2/ioxide.ngtcp2.csproj @@ -8,7 +8,7 @@ ioxide.ngtcp2 ioxide.ngtcp2 - 0.13.233 + 0.14.236 MDA2AV QUIC engine for the ioxide io_uring runtime: ngtcp2 + picotls bundled as a single self-contained native library (only system dependency: libcrypto.so.3 / OpenSSL 3.x). Plugs into the reactor's QUIC transport via QuicConnection. Server side; engine bindings in progress. MIT diff --git a/src/serving/ioxide.Kestrel/ioxide.Kestrel.csproj b/src/serving/ioxide.Kestrel/ioxide.Kestrel.csproj index 055a7683..eab74167 100644 --- a/src/serving/ioxide.Kestrel/ioxide.Kestrel.csproj +++ b/src/serving/ioxide.Kestrel/ioxide.Kestrel.csproj @@ -8,7 +8,7 @@ ioxide.Kestrel ioxide.Kestrel - 0.13.233 + 0.14.236 MDA2AV ASP.NET Core Kestrel transport backed by the ioxide io_uring runtime: one reactor (ring) per core, SO_REUSEPORT load-balanced, with Kestrel's HTTP request loop pinned to the reactor thread. Drop-in via UseIoxide(). MIT