can: bcm: defer rx_op deallocation to workqueue to fix thrtimer UAF

JIRA: https://redhat.atlassian.net/browse/RHEL-212684

Conflicts: In context only as CentOS/RHEL do not have upstream:
    96ea3a1e2d3 ("can: add CAN skb extension infrastructure")
    c2aba69d0c3 ("can: bcm: add locking for bcm_op runtime updates")
These commits are not dependencies.

commit 68973f9db76144825e4f35dfdc80fb8279eb2d57
Author: Lee Jones <lee@kernel.org>
Date:   Tue Jul 14 18:55:23 2026 +0200

    can: bcm: defer rx_op deallocation to workqueue to fix thrtimer UAF

    Commit f1b4e32aca08 ("can: bcm: use call_rcu() instead of costly
    synchronize_rcu()") replaced synchronize_rcu() in bcm_delete_rx_op()
    with call_rcu() and introduced the RX_NO_AUTOTIMER flag.

    However, this flag check was omitted for thrtimer in the packet rx
    fast-path. During BCM RX operation teardown, a concurrent RCU reader
    (bcm_rx_handler) can race and re-arm thrtimer via
    bcm_rx_update_and_send() after call_rcu() has been scheduled.  Once
    the RCU grace period elapses, bcm_op is freed.  The subsequently
    firing thrtimer then dereferences the deallocated op, causing a UAF.

    Adding flag checks to the rx fast-path (bcm_rx_update_and_send) does not
    fully close the TOCTOU race and introduces latency for every CAN frame.
    Conversely, calling hrtimer_cancel() directly inside the RCU callback
    (softirq context) is fatal as hrtimer_cancel() can sleep, triggering
    a "scheduling while atomic" panic.

    Resolve this by deferring the timer cancellation and memory free to a
    dedicated unbound workqueue (bcm_wq).  The RCU callback now queues a
    work item to bcm_wq, which safely cancels both timers and deallocates
    memory in sleepable process context.  A dedicated workqueue is used to
    prevent system-wide WQ saturation and is cleanly flushed/destroyed
    on module unload to avoid rmmod page faults.

    Since the deferred work can now outlive the calling context by an
    unbounded amount, also take a reference on op->sk when it is assigned
    and drop it only once the deferred work has cancelled both timers, so a
    socket can no longer be freed out from under a still-armed timer whose
    callback (bcm_send_to_user()) dereferences op->sk.

    Fixes: f1b4e32aca08 ("can: bcm: use call_rcu() instead of costly synchronize_rcu()")
    Tested-by: Feng Xue <feng.xue@outlook.com>
    Tested-by: Oliver Hartkopp <socketcan@hartkopp.net>
    Signed-off-by: Lee Jones <lee@kernel.org>
    Signed-off-by: Oliver Hartkopp <socketcan@hartkopp.net>
    Link: https://patch.msgid.link/20260714-bcm_fixes-v15-1-562f7e3e42da@hartkopp.net
    Cc: stable@kernel.org
    Signed-off-by: Marc Kleine-Budde <mkl@pengutronix.de>

Signed-off-by: Jamie Bainbridge <jbainbri@redhat.com>
This commit is contained in:
Jamie Bainbridge
2026-07-21 11:52:51 +10:00
parent 49953a2444
commit 124be4fba8
+34 -3
View File
@@ -58,6 +58,7 @@
#include <linux/can/skb.h>
#include <linux/can/bcm.h>
#include <linux/slab.h>
#include <linux/workqueue.h>
#include <net/sock.h>
#include <net/net_namespace.h>
@@ -90,6 +91,8 @@ MODULE_ALIAS("can-proto-2");
#define BCM_MIN_NAMELEN CAN_REQUIRED_SIZE(struct sockaddr_can, can_ifindex)
static struct workqueue_struct *bcm_wq;
/*
* easy access to the first 64 bit of can(fd)_frame payload. cp->data is
* 64 bit aligned so the offset has to be multiples of 8 which is ensured
@@ -103,6 +106,7 @@ static inline u64 get_u64(const struct canfd_frame *cp, int offset)
struct bcm_op {
struct list_head list;
struct rcu_head rcu;
struct work_struct work;
int ifindex;
canid_t can_id;
u32 flags;
@@ -768,9 +772,12 @@ static struct bcm_op *bcm_find_op(struct list_head *ops,
return NULL;
}
static void bcm_free_op_rcu(struct rcu_head *rcu_head)
static void bcm_free_op_work(struct work_struct *work)
{
struct bcm_op *op = container_of(rcu_head, struct bcm_op, rcu);
struct bcm_op *op = container_of(work, struct bcm_op, work);
hrtimer_cancel(&op->timer);
hrtimer_cancel(&op->thrtimer);
if ((op->frames) && (op->frames != &op->sframe))
kfree(op->frames);
@@ -778,9 +785,23 @@ static void bcm_free_op_rcu(struct rcu_head *rcu_head)
if ((op->last_frames) && (op->last_frames != &op->last_sframe))
kfree(op->last_frames);
/* the last possible access to op->timer/op->thrtimer has now
* happened above via hrtimer_cancel() - op->sk is no longer
* needed by any pending timer callback, so drop our reference
*/
sock_put(op->sk);
kfree(op);
}
static void bcm_free_op_rcu(struct rcu_head *rcu_head)
{
struct bcm_op *op = container_of(rcu_head, struct bcm_op, rcu);
INIT_WORK(&op->work, bcm_free_op_work);
queue_work(bcm_wq, &op->work);
}
static void bcm_remove_op(struct bcm_op *op)
{
hrtimer_cancel(&op->timer);
@@ -1009,6 +1030,7 @@ static int bcm_tx_setup(struct bcm_msg_head *msg_head, struct msghdr *msg,
/* bcm_can_tx / bcm_tx_timeout_handler needs this */
op->sk = sk;
sock_hold(sk);
op->ifindex = ifindex;
/* initialize uninitialized (kzalloc) structure */
@@ -1187,6 +1209,7 @@ static int bcm_rx_setup(struct bcm_msg_head *msg_head, struct msghdr *msg,
/* bcm_can_tx / bcm_tx_timeout_handler needs this */
op->sk = sk;
sock_hold(sk);
op->ifindex = ifindex;
/* ifindex for timeout events w/o previous frame reception */
@@ -1798,11 +1821,15 @@ static int __init bcm_module_init(void)
{
int err;
bcm_wq = alloc_workqueue("can-bcm-wq", WQ_UNBOUND, 0);
if (!bcm_wq)
return -ENOMEM;
pr_info("can: broadcast manager protocol\n");
err = register_pernet_subsys(&canbcm_pernet_ops);
if (err)
return err;
goto register_pernet_failed;
err = register_netdevice_notifier(&canbcm_notifier);
if (err)
@@ -1820,6 +1847,8 @@ register_proto_failed:
unregister_netdevice_notifier(&canbcm_notifier);
register_notifier_failed:
unregister_pernet_subsys(&canbcm_pernet_ops);
register_pernet_failed:
destroy_workqueue(bcm_wq);
return err;
}
@@ -1828,6 +1857,8 @@ static void __exit bcm_module_exit(void)
can_proto_unregister(&bcm_can_proto);
unregister_netdevice_notifier(&canbcm_notifier);
unregister_pernet_subsys(&canbcm_pernet_ops);
rcu_barrier();
destroy_workqueue(bcm_wq);
}
module_init(bcm_module_init);