On Sun, Aug 09, 2026 at 02:14:52PM +0200, Xin Xie wrote:
> Commit 06afd2c31d33 ("hsr: Synchronize sending frames to have always
> incremented outgoing seq nr.") and commit 430d67bdcb04 ("net: hsr: Use
> the seqnr lock for frames received via interlink port.") hold
> seqnr_lock across the whole forwarding path. Transmitting while
> holding the lock can create lock-dependency issues when HSR devices
> are stacked with other net devices, up to a real deadlock (see
> Reported-by/Closes).
>
> Since commit aae9d6b616b5 ("hsr: Implement more robust duplicate
> discard for HSR"), duplicate discard is order-independent (sparse
> bitmaps), so only the sequence counter updates need serialization.
> Limit seqnr_lock to the counter updates in handle_std_frame(); the
> master TX path (hsr_dev_xmit()) and the interlink RX path
> (hsr_handle_frame()) drop their outer lock, and the supervision frame
> builders release it right after their counter update.
>
> Concurrent forwarding may emit frames out of allocation order, which
> sparse-bitmap discard tolerates. The master/interlink tx statistics are
> no longer serialized; consistent per-cpu/per-queue statistics for all
> HSR paths are handled in a separate series.
>
> Fixes: 06afd2c31d33 ("hsr: Synchronize sending frames to have always
> incremented outgoing seq nr.")
> Fixes: 430d67bdcb04 ("net: hsr: Use the seqnr lock for frames received via
> interlink port.")
> Reported-by: [email protected]
> Closes: https://syzkaller.appspot.com/bug?extid=fbf74291c3b7e753b481
> Cc: <[email protected]> # aae9d6b616b5: hsr: Implement more robust
> duplicate discard for HSR
> Signed-off-by: Xin Xie <[email protected]>
> ---
> net/hsr/hsr_device.c | 15 ++++-----------
> net/hsr/hsr_forward.c | 3 ++-
> net/hsr/hsr_slave.c | 11 +----------
> 3 files changed, 7 insertions(+), 22 deletions(-)
>
> diff --git a/net/hsr/hsr_device.c b/net/hsr/hsr_device.c
> index 5555b71ab19b..3fd1762d8916 100644
> --- a/net/hsr/hsr_device.c
> +++ b/net/hsr/hsr_device.c
> @@ -232,9 +232,7 @@ static netdev_tx_t hsr_dev_xmit(struct sk_buff *skb,
> struct net_device *dev)
> skb->dev = master->dev;
> skb_reset_mac_header(skb);
> skb_reset_mac_len(skb);
> - spin_lock_bh(&hsr->seqnr_lock);
> hsr_forward_skb(skb, master);
> - spin_unlock_bh(&hsr->seqnr_lock);
> } else {
> dev_core_stats_tx_dropped_inc(dev);
> dev_kfree_skb_any(skb);
> @@ -335,6 +333,7 @@ static void send_hsr_supervision_frame(struct hsr_port
> *port,
> hsr_stag->sequence_nr = htons(hsr->sequence_nr);
> hsr->sequence_nr++;
> }
> + spin_unlock_bh(&hsr->seqnr_lock);
>
> hsr_stag->tlv.HSR_TLV_type = type;
> /* HSRv0 has 6 unused bytes after the MAC */
> @@ -356,14 +355,10 @@ static void send_hsr_supervision_frame(struct hsr_port
> *port,
> ether_addr_copy(hsr_sp->macaddress_A, hsr->macaddress_redbox);
> }
>
> - if (skb_put_padto(skb, ETH_ZLEN)) {
> - spin_unlock_bh(&hsr->seqnr_lock);
> + if (skb_put_padto(skb, ETH_ZLEN))
> return;
> - }
>
> hsr_forward_skb(skb, port);
> - spin_unlock_bh(&hsr->seqnr_lock);
> - return;
> }
>
> static void send_prp_supervision_frame(struct hsr_port *master,
> @@ -390,6 +385,7 @@ static void send_prp_supervision_frame(struct hsr_port
> *master,
> spin_lock_bh(&hsr->seqnr_lock);
> hsr_stag->sequence_nr = htons(hsr->sup_sequence_nr);
> hsr->sup_sequence_nr++;
> + spin_unlock_bh(&hsr->seqnr_lock);
> hsr_stag->tlv.HSR_TLV_type = PRP_TLV_LIFE_CHECK_DD;
> hsr_stag->tlv.HSR_TLV_length = sizeof(struct hsr_sup_payload);
>
> @@ -397,13 +393,10 @@ static void send_prp_supervision_frame(struct hsr_port
> *master,
> hsr_sp = skb_put(skb, sizeof(struct hsr_sup_payload));
> ether_addr_copy(hsr_sp->macaddress_A, master->dev->dev_addr);
>
> - if (skb_put_padto(skb, ETH_ZLEN)) {
> - spin_unlock_bh(&hsr->seqnr_lock);
> + if (skb_put_padto(skb, ETH_ZLEN))
> return;
> - }
>
> hsr_forward_skb(skb, master);
> - spin_unlock_bh(&hsr->seqnr_lock);
> }
>
> /* Announce (supervision frame) timer function
> diff --git a/net/hsr/hsr_forward.c b/net/hsr/hsr_forward.c
> index 0774981a65c1..8e4158a9b57c 100644
> --- a/net/hsr/hsr_forward.c
> +++ b/net/hsr/hsr_forward.c
> @@ -621,9 +621,10 @@ static void handle_std_frame(struct sk_buff *skb,
> if (port->type == HSR_PT_MASTER ||
> port->type == HSR_PT_INTERLINK) {
> /* Sequence nr for the master/interlink node */
> - lockdep_assert_held(&hsr->seqnr_lock);
> + spin_lock_bh(&hsr->seqnr_lock);
> frame->sequence_nr = hsr->sequence_nr;
> hsr->sequence_nr++;
> + spin_unlock_bh(&hsr->seqnr_lock);
> }
> }
>
> diff --git a/net/hsr/hsr_slave.c b/net/hsr/hsr_slave.c
> index bb2182a169a3..8b96eafe15b3 100644
> --- a/net/hsr/hsr_slave.c
> +++ b/net/hsr/hsr_slave.c
> @@ -73,16 +73,7 @@ static rx_handler_result_t hsr_handle_frame(struct sk_buff
> **pskb)
> }
> skb_reset_mac_len(skb);
>
> - /* Only the frames received over the interlink port will assign a
> - * sequence number and require synchronisation vs other sender.
> - */
> - if (port->type == HSR_PT_INTERLINK) {
> - spin_lock_bh(&hsr->seqnr_lock);
> - hsr_forward_skb(skb, port);
> - spin_unlock_bh(&hsr->seqnr_lock);
> - } else {
> - hsr_forward_skb(skb, port);
> - }
> + hsr_forward_skb(skb, port);
>
> finish_consume:
> return RX_HANDLER_CONSUMED;
> --
> 2.43.0
>
Thanks for the fix.
Reviewed-by: Hangbin Liu <[email protected]>