[PATCH 1/2] tcp: restore RACK-lost skbs to tsorted_sent_queue on undo
From: Neal Cardwell
Date: Fri Oct 09 2026 - 10:18:41 EST
RACK unlinks each skb that it marks lost from tp->tsorted_sent_queue, so
that later scans do not walk it again. The skb only returns to the list
when it is retransmitted. If an undo clears TCPCB_LOST before that, the
skb is on no list at all: RACK can never mark it lost again,
tcp_xmit_retransmit_queue() skips it, and only an RTO repairs it. This
happens for example after a partial undo in fast recovery, when PRR has
not yet retransmitted all the skbs that RACK marked lost.
As suggested by Eric, move the skbs that RACK marks lost to a new list,
tp->tsorted_lost_queue, instead of unlinking them. RACK scans the sent
queue in send order, and the skbs that a later scan marks lost were sent
later, so the lost queue stays sorted by send time without any sorting.
Nothing else needs to change: a retransmit moves the skb to the tail of
the sent queue, a SACK or cumulative ACK unlinks it, and tcp_fragment()
links the new skb next to the old one, in whichever list that is.
(Today, for an skb that RACK unlinked, tcp_fragment() builds a two-node
ring that is on no list.)
On undo, put the lost queue back into the sent queue. Usually every lost
skb was sent before every skb that is still in the sent queue, and one
list_splice_init() at the head does it. But the skbs that
tcp_timeout_mark_lost() marks stay in the sent queue, and RACK may later
move skbs sent after them to the lost queue. In that case, merge the two
sorted lists in one pass.
Unlike skipping skbs with TCPCB_EVER_RETRANS, this also restores skbs
that a TLP probed, or whose retransmit was lost, so that they do not
have to wait for an RTO either.
The cost is one list_head in tcp_sock, outside the fast path cache line
groups.
Fixes: 043b87d7599e ("tcp: more efficient RACK loss detection")
Reported-by: Neil Ramaswamy <nramaswamy@xxxxxxxxxx>
Closes: https://lore.kernel.org/netdev/cover.1791506907.git.nramaswamy@xxxxxxxxxx/
Suggested-by: Eric Dumazet <edumazet@xxxxxxxxxx>
Link: https://lore.kernel.org/netdev/CAL4Wiip9RyA-2=p0SVwBcn-oe0864FqqF3QMTCX_7=+ErcJp6Q@xxxxxxxxxxxxxx/
---
.../networking/net_cachelines/tcp_sock.rst | 1 +
include/linux/tcp.h | 3 ++
net/ipv4/tcp.c | 2 +
net/ipv4/tcp_input.c | 51 +++++++++++++++++++
net/ipv4/tcp_minisocks.c | 1 +
net/ipv4/tcp_recovery.c | 9 +++-
6 files changed, 66 insertions(+), 1 deletion(-)
diff --git a/Documentation/networking/net_cachelines/tcp_sock.rst b/Documentation/networking/net_cachelines/tcp_sock.rst
index 0f6088c4ab8bb..c5a4c06a5765d 100644
--- a/Documentation/networking/net_cachelines/tcp_sock.rst
+++ b/Documentation/networking/net_cachelines/tcp_sock.rst
@@ -34,6 +34,7 @@ u32 compressed_ack_rcv_nxt
u32 tsoffset read_mostly read_mostly tcp_established_options(tx);tcp_fast_parse_options(rx)
struct list_head tsq_node
struct list_head tsorted_sent_queue read_write tcp_update_skb_after_send
+struct list_head tsorted_lost_queue
u32 snd_wl1 read_mostly tcp_may_update_window
u32 snd_wnd read_mostly read_mostly tcp_wnd_end,tcp_tso_should_defer(tx);tcp_fast_path_on(rx)
u32 max_window read_mostly tcp_bound_to_half_wnd,forced_push
diff --git a/include/linux/tcp.h b/include/linux/tcp.h
index 6a8c77719322f..98699d599f11e 100644
--- a/include/linux/tcp.h
+++ b/include/linux/tcp.h
@@ -377,6 +377,9 @@ struct tcp_sock {
*/
u32 compressed_ack_rcv_nxt;
struct list_head tsq_node; /* anchor in tsq_tasklet.head list */
+ struct list_head tsorted_lost_queue; /* time-sorted skbs that RACK
+ * marked lost, until resent
+ */
/* Information of the most recently (s)acked skb */
struct tcp_rack {
diff --git a/net/ipv4/tcp.c b/net/ipv4/tcp.c
index 562752352afe4..6cbf4d79824e2 100644
--- a/net/ipv4/tcp.c
+++ b/net/ipv4/tcp.c
@@ -428,6 +428,7 @@ void tcp_init_sock(struct sock *sk)
tcp_init_xmit_timers(sk);
INIT_LIST_HEAD(&tp->tsq_node);
INIT_LIST_HEAD(&tp->tsorted_sent_queue);
+ INIT_LIST_HEAD(&tp->tsorted_lost_queue);
icsk->icsk_rto = TCP_TIMEOUT_INIT;
@@ -3352,6 +3353,7 @@ void tcp_write_queue_purge(struct sock *sk)
}
tcp_rtx_queue_purge(sk);
INIT_LIST_HEAD(&tcp_sk(sk)->tsorted_sent_queue);
+ INIT_LIST_HEAD(&tcp_sk(sk)->tsorted_lost_queue);
tcp_clear_all_retrans_hints(tcp_sk(sk));
tcp_sk(sk)->packets_out = 0;
inet_csk(sk)->icsk_backoff = 0;
diff --git a/net/ipv4/tcp_input.c b/net/ipv4/tcp_input.c
index 92bc60716f33d..a18ee1bcf0138 100644
--- a/net/ipv4/tcp_input.c
+++ b/net/ipv4/tcp_input.c
@@ -2840,6 +2840,56 @@ static void DBGUNDO(struct sock *sk, const char *msg)
#endif
}
+/* Was @a sent after @b, in the order of tp->tsorted_sent_queue? */
+static bool tcp_tsorted_after(const struct sk_buff *a, const struct sk_buff *b)
+{
+ return tcp_skb_sent_after(a->skb_mstamp_ns, b->skb_mstamp_ns,
+ TCP_SKB_CB(a)->end_seq,
+ TCP_SKB_CB(b)->end_seq);
+}
+
+/* Undo cleared the lost marks, so move the skbs that RACK marked lost back
+ * to tp->tsorted_sent_queue, in send order, where RACK can detect their
+ * loss again.
+ */
+static void tcp_tsorted_restore_lost(struct tcp_sock *tp)
+{
+ struct list_head *sent = &tp->tsorted_sent_queue;
+ struct list_head *lost = &tp->tsorted_lost_queue;
+ struct sk_buff *skb, *tmp, *pos;
+
+ if (list_empty(lost))
+ return;
+
+ /* Usually every lost skb was sent before every skb that is still in
+ * the sent queue, so the lost queue goes back at the head. But the
+ * skbs that tcp_timeout_mark_lost() marks stay in the sent queue,
+ * and RACK may later move skbs sent after them to the lost queue.
+ * Then merge the two queues, which are both sorted by send time.
+ */
+ pos = list_first_entry(sent, struct sk_buff, tcp_tsorted_anchor);
+ if (list_empty(sent) ||
+ !tcp_tsorted_after(list_last_entry(lost, struct sk_buff,
+ tcp_tsorted_anchor), pos)) {
+ list_splice_init(lost, sent);
+ return;
+ }
+
+ list_for_each_entry_safe(skb, tmp, lost, tcp_tsorted_anchor) {
+ /* Find the first skb in the sent queue sent after skb. */
+ list_for_each_entry_from(pos, sent, tcp_tsorted_anchor) {
+ if (tcp_tsorted_after(pos, skb))
+ break;
+ }
+ if (list_entry_is_head(pos, sent, tcp_tsorted_anchor)) {
+ list_splice_tail_init(lost, sent);
+ return;
+ }
+ list_move_tail(&skb->tcp_tsorted_anchor,
+ &pos->tcp_tsorted_anchor);
+ }
+}
+
static void tcp_undo_cwnd_reduction(struct sock *sk, bool unmark_loss)
{
struct tcp_sock *tp = tcp_sk(sk);
@@ -2850,6 +2900,7 @@ static void tcp_undo_cwnd_reduction(struct sock *sk, bool unmark_loss)
skb_rbtree_walk(skb, &sk->tcp_rtx_queue) {
TCP_SKB_CB(skb)->sacked &= ~TCPCB_LOST;
}
+ tcp_tsorted_restore_lost(tp);
tp->lost_out = 0;
tcp_clear_all_retrans_hints(tp);
}
diff --git a/net/ipv4/tcp_minisocks.c b/net/ipv4/tcp_minisocks.c
index 0ddfd5af6e58f..e73db6b70d079 100644
--- a/net/ipv4/tcp_minisocks.c
+++ b/net/ipv4/tcp_minisocks.c
@@ -580,6 +580,7 @@ struct sock *tcp_create_openreq_child(const struct sock *sk,
INIT_LIST_HEAD(&newtp->tsq_node);
INIT_LIST_HEAD(&newtp->tsorted_sent_queue);
+ INIT_LIST_HEAD(&newtp->tsorted_lost_queue);
tcp_init_wl(newtp, treq->rcv_isn);
diff --git a/net/ipv4/tcp_recovery.c b/net/ipv4/tcp_recovery.c
index 1396467510736..4258b00e28570 100644
--- a/net/ipv4/tcp_recovery.c
+++ b/net/ipv4/tcp_recovery.c
@@ -84,7 +84,14 @@ static void tcp_rack_detect_loss(struct sock *sk, u32 *reo_timeout)
remaining = tcp_rack_skb_timeout(tp, skb, reo_wnd);
if (remaining <= 0) {
tcp_mark_skb_lost(sk, skb);
- list_del_init(&skb->tcp_tsorted_anchor);
+ /* Move the skb to the lost queue instead of just
+ * unlinking it, so that undo can put it back. Each
+ * scan marks skbs in send order, and the skbs a later
+ * scan marks were sent later, so the lost queue stays
+ * sorted by send time.
+ */
+ list_move_tail(&skb->tcp_tsorted_anchor,
+ &tp->tsorted_lost_queue);
} else {
/* Record maximum wait time */
*reo_timeout = max_t(u32, *reo_timeout, remaining);
--
2.56.0.385.gd3acb90ef8-goog