Compare commits

..
Author SHA1 Message Date
Leency a8f285ec86 kernel/net: lock the socket in the retransmission timer
Check kernel codestyle / Check kernel codestyle (pull_request) Successful in 37s
Test PR / Build (en_US) (pull_request) Successful in 2m47s
Test PR / Build (es_ES) (pull_request) Successful in 2m4s
Test PR / Build (ru_RU) (pull_request) Successful in 1m58s
- take SOCKET.mutex while rewriting SND_NXT/cwnd/ssthresh; skip if an ACK
  stopped or re-armed the timer meanwhile
- back off from t_rxtcur: the first timeout was 3.2 s, the second 25.6 s
2026-09-30 10:14:17 +03:00
Leency 772031e6b6 kernel/net: fix TCP RTT measurement
- tcp_output: the "timing anything?" test was inverted, no segment was timed
- tcp_timer: t_rtt was never incremented
- tcp_xmit_timer: first sample keyed on t_srtt, not t_rtt; signed jg on
  srtt/rttvar updates (ja reset them to 1 whenever RTT dropped); t_rtt = 0
- timestamp RTT: 1/100 s -> 640 ms ticks (RTO came out 64x too long);
  skip it when TSecr is 0
2026-09-30 10:14:15 +03:00
Leency b1110dc03b kernel/net: start persist probes at the retransmission floor
BSD waits 5.12s before the first zero-window probe and doubles from
there. A peer that closes its window and never volunteers an update --
QEMU 0.10's slirp, or any stack applying silly-window avoidance to its
updates -- is rediscovered only by our probe, so 5, 10, 20s of silence
swallowed most of a 15s upload stage. Start at TCP_time_re_min (1.28s),
as Linux starts at the RTO; still exponential, still capped.
2026-09-30 10:14:12 +03:00
Leency 6c30c66aa3 kernel/net: implement the TCP retransmission timeout
t_rxtcur, the timeout the retransmission timer is armed with, was never
assigned anywhere: not at socket creation, not in tcp_xmit_timer. Every
socket armed the timer with 0, the 640 ms tick took it to 0xFFFFFFFF,
and it never expired. And when it did (in principle) expire, the handler
was a bare call tcp_output, which sends from SND_NXT: the unacknowledged
segment before it was never resent, nothing was backed off, the timer
was not re-armed. A segment the peer dropped -- a window that shrank
under data in flight, any loss on the uplink -- stalled the connection
until the application gave up. 23 duplicate ACKs from the peer did not
help either.

Observed as every upload through QEMU's slirp freezing the moment the
host side paused: pcap of an 8 MB POST to a sink that stops reading for
12 s showed the peer's window go 328, the guest's 1400-byte segment
beyond it dropped, then nothing but empty ACKs, and after the window
reopened the guest sent the NEXT 32 KB with the hole still there and
sat. speedtest.net's upload stage lost 3 of 4 connections this way on
the 2009 QEMU.

- tcp_init_socket: t_rxtcur = TCP_time_rtt_default (BSD TCPTV_RTOBASE)
  until the first RTT sample.
- tcp_xmit_timer: t_rxtcur = srtt + 4*rttvar (BSD TCP_REXMTVAL) within
  [re_min, re_max]; a fresh sample resets t_rxtshift.
- tcp_timer, retransmission expiry: BSD TCPT_REXMT -- drop the
  connection with ETIMEDOUT past TCP_max_rxtshift; re-arm with the
  backed-off timeout (shift capped at 6); SND_NXT = SND_UNA; no RTT
  sample from the resent segment; ssthresh = max(2, min(wnd,cwnd)/2/mss)
  segments, cwnd = one segment, dupacks = 0; then tcp_output.

With the patch the same fixture completes: 8 MB, 12.36 s of which 12 s
is the sink's deliberate pause, 205 Mbps once it reads again.
2026-09-30 10:14:12 +03:00
8 changed files with 119 additions and 79 deletions

No files matched your search

+1 -8
View File
@@ -463,7 +463,7 @@ proc kernel_alloc stdcall, size:dword
and eax, -PAGE_SIZE;
mov [size], eax
and eax, eax
jz .err0
jz .err
mov ebx, eax
shr ebx, 12
mov [pages_count], ebx
@@ -509,13 +509,6 @@ proc kernel_alloc stdcall, size:dword
pop ebx
ret
.err:
mov eax, [lin_addr]
test eax, eax
jz .err0
mov ecx, [pages_count]
call release_pages
stdcall free_kernel_space, [lin_addr]
.err0:
xor eax, eax
pop edi
pop ebx
+1 -59
View File
@@ -1397,9 +1397,7 @@ proc create_ring_buffer stdcall, size:dword, flags:dword
mov [page_tabs + edi], eax
mov [page_tabs + edi + edx], eax
invlpg [ebx]
add ebx, [size]
invlpg [ebx] ; the mirror half
sub ebx, [size]
invlpg [ebx+0x10000]
add eax, 0x1000
add ebx, 0x1000
add edi, 4
@@ -1418,62 +1416,6 @@ proc create_ring_buffer stdcall, size:dword, flags:dword
ret
endp
; Same as create_ring_buffer, but from non-contiguous pages.
; Not for DMA. Free with kernel_free.
align 4
proc create_ring_buffer_sparse stdcall, size:dword, flags:dword
locals
buf_ptr dd ?
endl
mov eax, [size]
test eax, eax
jz .fail
add eax, eax
stdcall alloc_kernel_space, eax
test eax, eax
jz .fail
mov [buf_ptr], eax
push ebx esi edi
mov ebx, eax ; page being mapped
mov esi, [size] ; distance to its mirror
mov edi, esi
shr edi, 12 ; pages left
.page:
call alloc_page
test eax, eax
jz .mm_fail
or eax, [flags]
mov edx, ebx
shr edx, 12
mov [page_tabs + edx*4], eax
invlpg [ebx]
lea edx, [ebx + esi]
shr edx, 12
mov [page_tabs + edx*4], eax
invlpg [ebx + esi]
add ebx, PAGE_SIZE
dec edi
jnz .page
mov eax, [buf_ptr]
pop edi esi ebx
ret
.mm_fail:
mov eax, [buf_ptr]
mov ecx, [size]
shr ecx, 11 ; both halves, in pages
call release_pages
stdcall free_kernel_space, [buf_ptr]
pop edi esi ebx
xor eax, eax
.fail:
ret
endp
align 4
proc print_mem
+1 -5
View File
@@ -198,10 +198,6 @@ ends
SOCKET_STRUCT_SIZE = 4096 ; in bytes
if sizeof.STREAM_SOCKET > SOCKET_STRUCT_SIZE
err "STREAM_SOCKET does not fit in SOCKET_STRUCT_SIZE"
end if
SOCKET_QUEUE_SIZE = 10 ; maximum number of incoming packets queued for 1 socket
; the incoming packet queue for sockets is placed in the socket struct itself, at this location from start
SOCKET_QUEUE_LOCATION = (SOCKET_STRUCT_SIZE - SOCKET_QUEUE_SIZE*sizeof.socket_queue_entry - sizeof.queue)
@@ -1651,7 +1647,7 @@ socket_ring_create:
mov esi, eax
push edx
stdcall create_ring_buffer_sparse, SOCKET_BUFFER_SIZE, PG_SWR
stdcall create_ring_buffer, SOCKET_BUFFER_SIZE, PG_SWR
pop edx
test eax, eax
jz .fail
+1 -1
View File
@@ -63,7 +63,7 @@ TCP_OPT_TIMESTAMP = 8
TCP_time_MSL = 47 ; max segment lifetime (30s)
TCP_time_re_min = 2 ; min retransmission (1,28s)
TCP_time_re_max = 100 ; max retransmission (64s)
TCP_time_pers_min = 8 ; min persist (5,12s)
TCP_time_pers_min = 2 ; min persist (1,28s)
TCP_time_pers_max = 94 ; max persist (60,16s)
TCP_time_keep_init = 118 ; connection establishment (75,52s)
TCP_time_keep_idle = 4608 ; idle time before 1st probe (2h)
+6
View File
@@ -525,8 +525,11 @@ endl
test [temp_bits], TCP_BIT_TIMESTAMP
jz .no_timestamp_rtt
cmp [ebx + TCP_SOCKET.ts_ecr], 0 ; nothing echoed
je .no_timestamp_rtt
mov eax, [timestamp]
sub eax, [ebx + TCP_SOCKET.ts_ecr]
shr eax, 6 ; 1/100 s -> 640 ms ticks
inc eax
call tcp_xmit_timer
jmp .rtt_done
@@ -1114,8 +1117,11 @@ endl
test [temp_bits], TCP_BIT_TIMESTAMP
jz .timestamp_not_present
cmp [ebx + TCP_SOCKET.ts_ecr], 0 ; nothing echoed
je .timestamp_not_present
mov eax, [timestamp]
sub eax, [ebx + TCP_SOCKET.ts_ecr]
shr eax, 6 ; 1/100 s -> 640 ms ticks
inc eax
call tcp_xmit_timer
jmp .rtt_done_
+1 -1
View File
@@ -626,7 +626,7 @@ endl
mov [eax + TCP_SOCKET.SND_MAX], edx ; [eax + TCP_SOCKET.SND_NXT] from before we updated it
cmp [eax + TCP_SOCKET.t_rtt], 0 ; are we currently timing anything?
je @f
jne @f
mov [eax + TCP_SOCKET.t_rtt], 1 ; nope, start transmission timer
mov [eax + TCP_SOCKET.t_rtseq], edi
inc [TCPS_segstimed]
+15 -5
View File
@@ -96,7 +96,7 @@ macro tcp_init_socket socket {
mov [socket + TCP_SOCKET.t_srtt], TCP_time_srtt_default
mov [socket + TCP_SOCKET.t_rttvar], TCP_time_rtt_default * 4
mov [socket + TCP_SOCKET.t_rttmin], TCP_time_re_min
;;; TODO: TCP_time_rangeset
mov [socket + TCP_SOCKET.t_rxtcur], TCP_time_rtt_default
mov [socket + TCP_SOCKET.SND_CWND], TCP_max_win shl TCP_max_winshift
mov [socket + TCP_SOCKET.SND_SSTHRESH], TCP_max_win shl TCP_max_winshift
@@ -518,7 +518,7 @@ tcp_xmit_timer:
inc [TCPS_rttupdated]
cmp [ebx + TCP_SOCKET.t_rtt], 0
cmp [ebx + TCP_SOCKET.t_srtt], 0 ; first sample?
je .no_rtt_yet
; srtt is stored as a fixed point with 3 bits after the binary point.
@@ -534,7 +534,7 @@ tcp_xmit_timer:
pop ecx
add [ebx + TCP_SOCKET.t_srtt], eax
ja @f
jg @f ; signed: delta may be negative
mov [ebx + TCP_SOCKET.t_srtt], 1
@@:
@@ -556,10 +556,10 @@ tcp_xmit_timer:
pop edx
add [ebx + TCP_SOCKET.t_rttvar], eax
ja @f
jg @f
mov [ebx + TCP_SOCKET.t_rttvar], 1
@@:
ret
jmp .rto
.no_rtt_yet:
@@ -572,6 +572,16 @@ tcp_xmit_timer:
mov [ebx + TCP_SOCKET.t_rttvar], eax
pop ecx
.rto:
; Retransmit timeout = srtt + 4*rttvar, reset the backoff
push ecx
mov ecx, [ebx + TCP_SOCKET.t_srtt]
shr ecx, TCP_RTT_SHIFT
add ecx, [ebx + TCP_SOCKET.t_rttvar]
tcpt_rangeset [ebx + TCP_SOCKET.t_rxtcur], ecx, TCP_time_re_min, TCP_time_re_max
pop ecx
mov [ebx + TCP_SOCKET.t_rxtshift], 0
mov [ebx + TCP_SOCKET.t_rtt], 0 ; the timed segment is done
ret
+93
View File
@@ -91,6 +91,10 @@ proc tcp_timer_640ms
jne .loop
inc [eax + TCP_SOCKET.t_idle]
cmp [eax + TCP_SOCKET.t_rtt], 0 ; timing a segment?
je @f
inc [eax + TCP_SOCKET.t_rtt]
@@:
test [eax + TCP_SOCKET.timer_flags], timer_flag_retransmission
jz .check_more2
@@ -99,6 +103,95 @@ proc tcp_timer_640ms
DEBUGF DEBUG_NETWORK_VERBOSE, "socket %x: Retransmission timer expired\n", eax
; Lock socket, an ACK may have stopped or restarted the timer meanwhile
pusha
lea ecx, [eax + SOCKET.mutex]
call mutex_lock
popa
test [eax + TCP_SOCKET.timer_flags], timer_flag_retransmission
jz .rexmt_cancelled
cmp [eax + TCP_SOCKET.timer_retransmission], 0
jne .rexmt_cancelled
; Too many retransmissions? Drop the connection
inc [eax + TCP_SOCKET.t_rxtshift]
cmp [eax + TCP_SOCKET.t_rxtshift], TCP_max_rxtshift
jbe .rexmt
pusha
lea ecx, [eax + SOCKET.mutex]
call mutex_unlock
popa
DEBUGF DEBUG_NETWORK_VERBOSE, "socket %x: too many retransmissions, dropping\n", eax
push [eax + SOCKET.NextPtr]
mov ebx, ETIMEDOUT
call tcp_drop
pop eax
jmp .check_only
.rexmt_cancelled:
pusha
lea ecx, [eax + SOCKET.mutex]
call mutex_unlock
popa
jmp .check_more2
.rexmt:
push ebx ecx edx
; Restart timer with backoff: t_rxtcur << min(t_rxtshift, 6)
mov ebx, [eax + TCP_SOCKET.t_rxtcur]
mov cl, [eax + TCP_SOCKET.t_rxtshift]
cmp cl, 6
jbe @f
mov cl, 6
@@:
shl ebx, cl
cmp ebx, TCP_time_re_min
jae @f
mov ebx, TCP_time_re_min
@@:
cmp ebx, TCP_time_re_max
jbe @f
mov ebx, TCP_time_re_max
@@:
mov [eax + TCP_SOCKET.timer_retransmission], ebx
; Resend from the last acknowledged byte, don't time it
push [eax + TCP_SOCKET.SND_UNA]
pop [eax + TCP_SOCKET.SND_NXT]
mov [eax + TCP_SOCKET.t_rtt], 0
mov [eax + TCP_SOCKET.t_dupacks], 0
; Slow start: ssthresh = max(2, min(wnd, cwnd) / 2 / mss) * mss, cwnd = mss
mov ecx, [eax + TCP_SOCKET.t_maxseg]
mov edx, [eax + TCP_SOCKET.SND_WND]
cmp edx, [eax + TCP_SOCKET.SND_CWND]
jbe @f
mov edx, [eax + TCP_SOCKET.SND_CWND]
@@:
mov ebx, eax ; socket ptr
mov eax, edx
shr eax, 1
xor edx, edx
div ecx
cmp eax, 2
jae @f
mov eax, 2
@@:
mul ecx
mov [ebx + TCP_SOCKET.SND_SSTHRESH], eax
mov [ebx + TCP_SOCKET.SND_CWND], ecx
mov eax, ebx
pop edx ecx ebx
pusha
lea ecx, [eax + SOCKET.mutex]
call mutex_unlock
popa
push eax
call tcp_output
pop eax