Skip to content

Commit 8cfb2e7

Browse files
Zihan Xigregkh
authored andcommitted
packet: synchronize pressure clearing with ring reconfiguration
[ Upstream commit 1a35da3 ] packet_set_ring() updates the RX ring state under sk_receive_queue.lock, but used to publish the tpacket receive mode through po->prot_hook.func after releasing that lock. packet_poll() and packet_recvmsg() can then run the pressure clearing path after the ring has been cleared while still seeing tpacket_rcv, causing __packet_rcv_has_room() to dereference stale or NULL ring storage. Move the existing receive hook assignment into the same sk_receive_queue.lock section as the ring state update. Keep the assignment otherwise unchanged, including on TX ring reconfiguration, to avoid adding behavior changes that are not required for the fix. Serialize packet_recvmsg() pressure clearing with the same queue lock only after PACKET_SOCK_PRESSURE has been observed. If the flag is clear and the socket has moved away from tpacket_rcv, packet_set_ring() has already detached the socket and waited for synchronize_net(), so no new packet input can set the flag again. packet_poll() already holds sk_receive_queue.lock, so it uses the new unlocked helper directly. Fixes: 2ccdbaa ("packet: rollover lock contention avoidance") Cc: stable@vger.kernel.org Reported-by: Vega <vega@nebusec.ai> Assisted-by: Codex:gpt-5.4 Signed-off-by: Zihan Xi <zihanx@nebusec.ai> Link: https://patch.msgid.link/f90b5688311fa278d1361ea8c6be0bf25967d591.1785247446.git.zihanx@nebusec.ai Signed-off-by: Paolo Abeni <pabeni@redhat.com> [ Replaced `packet_sock_flag(po, PACKET_SOCK_PRESSURE)` with `READ_ONCE(po->pressure)` since the flag conversion isn't in this tree. ] Signed-off-by: Sasha Levin <sashal@kernel.org> Signed-off-by: Greg Kroah-Hartman <gregkh@linuxfoundation.org>
1 parent a627d36 commit 8cfb2e7

1 file changed

Lines changed: 16 additions & 4 deletions

File tree

‎net/packet/af_packet.c‎

Lines changed: 16 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -1333,13 +1333,25 @@ static int packet_rcv_has_room(struct packet_sock *po, struct sk_buff *skb)
13331333
return ret;
13341334
}
13351335

1336-
static void packet_rcv_try_clear_pressure(struct packet_sock *po)
1336+
static void __packet_rcv_try_clear_pressure(struct packet_sock *po)
13371337
{
13381338
if (READ_ONCE(po->pressure) &&
13391339
__packet_rcv_has_room(po, NULL) == ROOM_NORMAL)
13401340
WRITE_ONCE(po->pressure, 0);
13411341
}
13421342

1343+
static void packet_rcv_try_clear_pressure(struct packet_sock *po)
1344+
{
1345+
struct sock *sk = &po->sk;
1346+
1347+
if (!READ_ONCE(po->pressure))
1348+
return;
1349+
1350+
spin_lock_bh(&sk->sk_receive_queue.lock);
1351+
__packet_rcv_try_clear_pressure(po);
1352+
spin_unlock_bh(&sk->sk_receive_queue.lock);
1353+
}
1354+
13431355
static void packet_sock_destruct(struct sock *sk)
13441356
{
13451357
skb_queue_purge(&sk->sk_error_queue);
@@ -4304,7 +4316,7 @@ static __poll_t packet_poll(struct file *file, struct socket *sock,
43044316
TP_STATUS_KERNEL))
43054317
mask |= EPOLLIN | EPOLLRDNORM;
43064318
}
4307-
packet_rcv_try_clear_pressure(po);
4319+
__packet_rcv_try_clear_pressure(po);
43084320
spin_unlock_bh(&sk->sk_receive_queue.lock);
43094321
spin_lock_bh(&sk->sk_write_queue.lock);
43104322
if (po->tx_ring.pg_vec) {
@@ -4544,14 +4556,14 @@ static int packet_set_ring(struct sock *sk, union tpacket_req_u *req_u,
45444556
rb->frame_max = (req->tp_frame_nr - 1);
45454557
rb->head = 0;
45464558
rb->frame_size = req->tp_frame_size;
4559+
po->prot_hook.func = (po->rx_ring.pg_vec) ?
4560+
tpacket_rcv : packet_rcv;
45474561
spin_unlock_bh(&rb_queue->lock);
45484562

45494563
swap(rb->pg_vec_order, order);
45504564
swap(rb->pg_vec_len, req->tp_block_nr);
45514565

45524566
rb->pg_vec_pages = req->tp_block_size/PAGE_SIZE;
4553-
po->prot_hook.func = (po->rx_ring.pg_vec) ?
4554-
tpacket_rcv : packet_rcv;
45554567
skb_queue_purge(rb_queue);
45564568
if (atomic_long_read(&po->mapped))
45574569
pr_err("packet_mmap: vma is busy: %ld\n",

0 commit comments

Comments
 (0)