session: widen the replay window to 4096 packets

Data sequence numbers are shared by every carrier, and the receiver took
a packet only within 64 of the newest one. Carriers of equal priority
share flows; with one 150ms slower than the other, 1532 of 3000 packets
arrived at all, since everything on the slow one came in more than 64
behind. cupsonline reorders whole batches (up to 64 packets each) across
its rooms, so one late frame was enough.

The window is now a 4096-bit ring. Receiver-only, no wire change.

Also: TestSessionReplayWindow fed the exit packets as soon as Start
returned, but the exit returns before it is ready; it now waits, which
fixes the failure seen under -race.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
p1neappleXpress
2026-09-26 19:43:31 +03:00
co-authored by Claude Opus 5.5
parent bef7cecd63
commit f62f1913a2
3 changed files with 173 additions and 11 deletions
+42 -11
View File
@@ -27,7 +27,7 @@ type Session struct {
stopped bool
sequence uint64
highest uint64
window uint64
window replayWindow
handshakeTimeout time.Duration
helloInterval time.Duration
@@ -432,7 +432,7 @@ func (s *Session) resetLocked() {
s.ready = false
s.peer = [32]byte{}
s.remote = PeerParameters{}
s.sequence, s.highest, s.window = 0, 0, 0
s.sequence, s.highest, s.window = 0, 0, replayWindow{}
s.peerKeepalive = false
s.candidate = nil
if _, err := rand.Read(s.local[:]); err != nil {
@@ -905,7 +905,7 @@ func (s *Session) offerReplacementLocked(link *transportLink, sender [32]byte, p
Capabilities: params.Capabilities & s.params.Capabilities,
MaxPacketSize: minInt(params.MaxPacketSize, s.params.MaxPacketSize),
}
s.sequence, s.highest, s.window = 0, 0, 0
s.sequence, s.highest, s.window = 0, 0, replayWindow{}
s.peerKeepalive = false
s.candidate = nil
for _, l := range s.links {
@@ -1066,26 +1066,57 @@ func (s *Session) receiveControl(link *transportLink, p []byte, env *control.Env
go cb(sub, payload)
}
// replayWindowSize is how far behind the newest sequence a data packet may
// arrive and still be accepted once. Carriers of equal priority share
// flows and can differ in latency by hundreds of milliseconds, thousands
// of packets at speed, and cupsonline reorders whole batches across its
// rooms; a 64-packet window dropped nearly all a slower carrier delivered.
const replayWindowSize = 4096
// replayWindow is a ring of bits, one per sequence number within
// replayWindowSize of the newest one: set once that sequence was accepted.
type replayWindow [replayWindowSize / 64]uint64
func (w *replayWindow) has(seq uint64) bool {
i := seq % replayWindowSize
return w[i/64]&(1<<(i%64)) != 0
}
func (w *replayWindow) set(seq uint64) {
i := seq % replayWindowSize
w[i/64] |= 1 << (i % 64)
}
func (w *replayWindow) clear(seq uint64) {
i := seq % replayWindowSize
w[i/64] &^= 1 << (i % 64)
}
// acceptSequenceLocked accepts each sequence number at most once, and only
// within replayWindowSize of the newest one seen. Caller holds s.mu.
func (s *Session) acceptSequenceLocked(seq uint64) bool {
if seq == 0 {
return false
}
if seq > s.highest {
gap := seq - s.highest
if gap >= 64 {
s.window = 0
// The slots of the numbers skipped over still hold bits from a
// full turn of the ring ago.
if seq-s.highest >= replayWindowSize {
s.window = replayWindow{}
} else {
s.window <<= gap
for n := s.highest + 1; n < seq; n++ {
s.window.clear(n)
}
}
s.highest = seq
s.window |= 1
s.window.clear(seq)
s.window.set(seq)
return true
}
gap := s.highest - seq
if gap >= 64 || s.window&(uint64(1)<<gap) != 0 {
if s.highest-seq >= replayWindowSize || s.window.has(seq) {
return false
}
s.window |= uint64(1) << gap
s.window.set(seq)
return true
}
+127
View File
@@ -0,0 +1,127 @@
package transport
import (
"encoding/binary"
"sync"
"sync/atomic"
"testing"
"time"
"openflux/transport/control"
)
// delayedWire delivers to its peer after a fixed delay, in order, like a
// carrier with that much latency.
type delayedWire struct {
negotiationWire
delay time.Duration
once sync.Once
q chan delayed
}
type delayed struct {
due time.Time
p []byte
}
func (w *delayedWire) Send(p []byte) error {
w.once.Do(func() {
w.q = make(chan delayed, 1<<16)
go func() {
for d := range w.q {
time.Sleep(time.Until(d.due))
_ = w.negotiationWire.Send(d.p)
}
}()
})
w.q <- delayed{time.Now().Add(w.delay), append([]byte(nil), p...)}
return nil
}
// Carriers of equal priority share flows between them. When one is much
// slower, its packets arrive long after later-numbered ones came through
// the fast one, and every one of them must still be accepted.
func TestSessionEqualPriorityCarriersWithLatencySkew(t *testing.T) {
params := PeerParameters{
Capabilities: control.CapabilityIPv4 | control.CapabilityTCP | control.CapabilityUDP,
MaxPacketSize: 1500,
}
client, _ := NewSession(params, false)
exit, _ := NewSession(params, true)
t.Cleanup(func() { _ = client.Stop(); _ = exit.Stop() })
for _, c := range []struct {
name string
delay time.Duration
}{{"yandex", 0}, {"vyandex", 150 * time.Millisecond}} {
cw := &delayedWire{delay: c.delay}
ew := &delayedWire{delay: c.delay}
cw.peer, ew.peer = &ew.negotiationWire, &cw.negotiationWire
if err := client.AddTransport(c.name, cw, testSessionSecret, testSessionCtx, 50); err != nil {
t.Fatal(err)
}
if err := exit.AddTransport(c.name, ew, testSessionSecret, testSessionCtx, 50); err != nil {
t.Fatal(err)
}
}
var got atomic.Int64
exit.Receive(func([]byte) { got.Add(1) })
startPair(t, client, exit)
eventually(t, "both carriers heard", func() bool {
client.mu.Lock()
defer client.mu.Unlock()
return len(client.liveLinksLocked()) == 2
})
const total = 3000
sent := 0
for i := 0; i < total; i++ {
p := testIPv4(40, 17)
copy(p[12:], []byte{10, 10, 10, 2})
copy(p[16:], []byte{8, 8, 8, 8})
binary.BigEndian.PutUint16(p[20:], uint16(10000+i)) // one flow per packet
binary.BigEndian.PutUint16(p[22:], 53)
if client.Send(p) == nil {
sent++
}
if i%100 == 99 {
time.Sleep(2 * time.Millisecond) // ~50k packets/s, within the batch queue
}
}
deadline := time.Now().Add(3 * time.Second)
for got.Load() < int64(sent) && time.Now().Before(deadline) {
time.Sleep(10 * time.Millisecond)
}
if n := got.Load(); n < int64(sent) {
t.Fatalf("delivered %d of %d packets", n, sent)
}
if sent < total*9/10 {
t.Fatalf("only %d of %d packets could be sent", sent, total)
}
}
func TestReplayWindowRing(t *testing.T) {
s := &Session{}
for seq := uint64(1); seq <= 10000; seq++ {
if !s.acceptSequenceLocked(seq) {
t.Fatalf("fresh seq %d rejected", seq)
}
}
for _, c := range []struct {
seq uint64
want bool
}{
{10000, false}, // newest, again
{10000 - replayWindowSize + 1, false}, // oldest still in the window, again
{10000 - replayWindowSize, false}, // out of the window
{0, false},
{10000 + replayWindowSize - 1, true}, // jump: its slot held seq 10000-1
{10000 - 1, false}, // same slot as the jump target
{10000 + 5, true}, // skipped over by the jump, fresh
{10000 + 5, false},
} {
if got := s.acceptSequenceLocked(c.seq); got != c.want {
t.Fatalf("seq %d accepted=%v, want %v", c.seq, got, c.want)
}
}
}
+4
View File
@@ -180,6 +180,10 @@ func TestSessionReplayWindow(t *testing.T) {
}
}
// The exit returns from Start before it is ready: it becomes ready
// when the client's echo arrives, possibly after the client returned.
eventually(t, "exit ready", b.IsConnected)
count := 0
b.Receive(func([]byte) { count++ })
data := func(seq uint64) []byte {