From 6693d18ae0232da2c6febc0decf099e4fd064381 Mon Sep 17 00:00:00 2001 From: Denys Fedoryshchenko Date: Sat, 8 Aug 2026 17:47:49 +0300 Subject: ipoe: flush sessions left by a previous instance with a single command On startup accel-pppd is expected to drop everything a previous instance left behind in the kernel. For sessions it did so by dumping them with IPOE_CMD_GET and sending one IPOE_CMD_DELETE per ifindex. That dump was written for the session backup code removed in 1972a7e5c and was never adapted to its new, destructive role: - it was issued on the socket subscribed to the packet multicast group, so it competed with the notifications that the still attached stale rx handlers keep generating; on a loaded box the receive buffer overruns and rtnl_dump_filter() gives up with ENOBUFS, - its return value was discarded, so such a failure was silent, - it ran before the interfaces were detached, which is what produces those notifications in the first place, - it was skipped entirely when the multicast group could not be resolved, again with nothing but a warning about packet handling, - and every session cost a socket, a round trip, a grace period and a full unregister_netdev(). Whatever it missed stays in the kernel forever: the daemon has no record of those sessions, so nothing ever deletes them, and every subscriber later assigned one of their addresses is refused with EEXIST by IPOE_CMD_MODIFY. Add IPOE_CMD_FLUSH, which unlinks all sessions in one go, waits for a single grace period and unregisters the devices with unregister_netdevice_many(), and use it instead. The old path is kept as a fallback for a module predating the command, which is reported as EOPNOTSUPP, and now checks its return value and uses a private socket. Reorder init() so the interfaces are detached before the multicast group is joined, and so the flush also runs when only the group lookup failed. --- drivers/ipoe/ipoe.c | 54 +++++++++++++++++++++++++++++++++++++++++++++++++++++ drivers/ipoe/ipoe.h | 1 + 2 files changed, 55 insertions(+) (limited to 'drivers/ipoe') diff --git a/drivers/ipoe/ipoe.c b/drivers/ipoe/ipoe.c index 73926994..8f1c852b 100644 --- a/drivers/ipoe/ipoe.c +++ b/drivers/ipoe/ipoe.c @@ -1420,6 +1420,52 @@ out_unlock: return ret; } +static int ipoe_nl_cmd_flush(struct sk_buff *skb, struct genl_info *info) +{ + struct ipoe_session *ses; + LIST_HEAD(list); + LIST_HEAD(kill_list); + + down(&ipoe_wlock); + + list_splice_init(&ipoe_list2, &list); + + list_for_each_entry(ses, &list, entry2) { + if (ses->peer_addr) + list_del_rcu(&ses->entry); + if (ses->u.hwaddr_u) + list_del_rcu(&ses->entry3); + } + + up(&ipoe_wlock); + + if (list_empty(&list)) + return 0; + + /* a single grace period covers the whole batch */ + synchronize_rcu(); + + list_for_each_entry(ses, &list, entry2) { + while (atomic_read(&ses->refs)) + schedule_timeout_uninterruptible(1); + + if (ses->link_dev) { + dev_put(ses->link_dev); + ses->link_dev = NULL; + } + } + + rtnl_lock(); + list_for_each_entry(ses, &list, entry2) + unregister_netdevice_queue(ses->dev, &kill_list); + unregister_netdevice_many(&kill_list); + rtnl_unlock(); + + /* the sessions are freed by now, do not touch 'list' again */ + + return 0; +} + static int ipoe_nl_cmd_modify(struct sk_buff *skb, struct genl_info *info) { int ret = -EINVAL, r = 0; @@ -1894,6 +1940,14 @@ static const struct genl_ops ipoe_nl_ops[] = { .flags = GENL_ADMIN_PERM, #if LINUX_VERSION_CODE < KERNEL_VERSION(5,2,0) .policy = ipoe_nl_policy, +#endif + }, + { + .cmd = IPOE_CMD_FLUSH, + .doit = ipoe_nl_cmd_flush, + .flags = GENL_ADMIN_PERM, +#if LINUX_VERSION_CODE < KERNEL_VERSION(5,2,0) + .policy = ipoe_nl_policy, #endif }, }; diff --git a/drivers/ipoe/ipoe.h b/drivers/ipoe/ipoe.h index 4097e2da..2d041c50 100644 --- a/drivers/ipoe/ipoe.h +++ b/drivers/ipoe/ipoe.h @@ -16,6 +16,7 @@ enum { IPOE_CMD_DEL_EXCLUDE, IPOE_CMD_ADD_NET, IPOE_CMD_DEL_NET, + IPOE_CMD_FLUSH, __IPOE_CMD_MAX, }; -- cgit v1.2.3