* [PATCH net-next v7 1/5] net: pse-pd: add notifier chain for controller lifecycle events
2026-09-27 19:18 [PATCH net-next v7 0/5] net: pse-pd: decouple controller lookup from MDIO probe Carlo Szelinsky
@ 2026-09-27 19:18 ` Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 2/5] net: pse-pd: fire lifecycle events on controller register/unregister Carlo Szelinsky
` (3 subsequent siblings)
4 siblings, 0 replies; 6+ messages in thread
From: Carlo Szelinsky @ 2026-09-27 19:18 UTC (permalink / raw)
To: Oleksij Rempel, Kory Maincent, Andrew Lunn, Heiner Kallweit,
Russell King, David S . Miller, Eric Dumazet, Jakub Kicinski,
Paolo Abeni
Cc: Corey Leavitt, Jonas Jelonek, Simon Horman,
Aleksander Jan Bajkowski, Mark Brown, Liam Girdwood,
netdev-bot+sashiko, netdev, linux-kernel, Carlo Szelinsky
From: Corey Leavitt <corey@leavitt.info>
Introduce a blocking notifier chain that allows other subsystems to be
informed when a PSE controller is registered or unregistered, and
provide pse_register_notifier() / pse_unregister_notifier() as the
subscriber interface.
Subsequent patches will use this to let the phy subsystem own the
phydev->psec lifecycle directly, decoupling PSE lookup from
fwnode_mdiobus_register_phy() and removing the probe-time
-EPROBE_DEFER coupling that currently exists between mdio, phy and
pse-pd when the PSE controller driver is modular.
A blocking chain (rather than atomic) is used because callbacks sleep.
The phy subscriber added later takes a mutex, and of_pse_control_get()
reaches the controller over i2c. The contract also lets a subscriber
take rtnl, which is why a controller driver must not hold it across
pse_controller_register() or pse_controller_unregister().
The enum pse_controller_event is placed outside the
IS_ENABLED(CONFIG_PSE_CONTROLLER) guard so that subscribers compiled
into a kernel without PSE support can still reference the event
values in dead-code paths without breaking the build.
This patch is pure infrastructure: nothing fires events yet, and
nothing subscribes. No observable behavior change.
Signed-off-by: Corey Leavitt <corey@leavitt.info>
Signed-off-by: Carlo Szelinsky <github@szelinsky.de>
Tested-by: Jonas Jelonek <jelonek.jonas@gmail.com>
---
drivers/net/pse-pd/pse_core.c | 34 ++++++++++++++++++++++++++++++++++
include/linux/pse-pd/pse.h | 33 +++++++++++++++++++++++++++++++++
2 files changed, 67 insertions(+)
diff --git a/drivers/net/pse-pd/pse_core.c b/drivers/net/pse-pd/pse_core.c
index a5e6d7b26b9f..84c734ed4553 100644
--- a/drivers/net/pse-pd/pse_core.c
+++ b/drivers/net/pse-pd/pse_core.c
@@ -8,6 +8,7 @@
#include <linux/device.h>
#include <linux/ethtool.h>
#include <linux/ethtool_netlink.h>
+#include <linux/notifier.h>
#include <linux/of.h>
#include <linux/phy.h>
#include <linux/pse-pd/pse.h>
@@ -23,6 +24,39 @@ static LIST_HEAD(pse_controller_list);
static DEFINE_XARRAY_ALLOC(pse_pw_d_map);
static DEFINE_MUTEX(pse_pw_d_mutex);
+static BLOCKING_NOTIFIER_HEAD(pse_controller_notifier);
+
+/**
+ * pse_register_notifier - register a callback for PSE controller events
+ * @nb: notifier block to register
+ *
+ * See enum pse_controller_event for events fired and their subscriber
+ * contract. Callbacks run in process context; they may sleep, take
+ * rtnl, and call of_pse_control_get(). The chain fires synchronously,
+ * so a PSE controller driver's probe/unbind path must not hold any
+ * such lock when calling pse_controller_register() or
+ * pse_controller_unregister().
+ *
+ * Return: 0 on success, negative error code otherwise.
+ */
+int pse_register_notifier(struct notifier_block *nb)
+{
+ return blocking_notifier_chain_register(&pse_controller_notifier, nb);
+}
+EXPORT_SYMBOL_GPL(pse_register_notifier);
+
+/**
+ * pse_unregister_notifier - unregister a previously registered callback
+ * @nb: notifier block previously passed to pse_register_notifier()
+ *
+ * Return: 0 on success, negative error code otherwise.
+ */
+int pse_unregister_notifier(struct notifier_block *nb)
+{
+ return blocking_notifier_chain_unregister(&pse_controller_notifier, nb);
+}
+EXPORT_SYMBOL_GPL(pse_unregister_notifier);
+
/**
* struct pse_control - a PSE control
* @pcdev: a pointer to the PSE controller device
diff --git a/include/linux/pse-pd/pse.h b/include/linux/pse-pd/pse.h
index 4e5696cfade7..bc5d36bcd993 100644
--- a/include/linux/pse-pd/pse.h
+++ b/include/linux/pse-pd/pse.h
@@ -21,6 +21,7 @@ struct net_device;
struct phy_device;
struct pse_controller_dev;
struct netlink_ext_ack;
+struct notifier_block;
/* C33 PSE extended state and substate. */
struct ethtool_c33_pse_ext_state_info {
@@ -337,6 +338,25 @@ enum pse_budget_eval_strategies {
PSE_BUDGET_EVAL_STRAT_DYNAMIC = 1 << 2,
};
+/**
+ * enum pse_controller_event - PSE controller lifecycle events
+ *
+ * Event data in callbacks is always a pointer to the struct
+ * pse_controller_dev firing the event.
+ *
+ * @PSE_REGISTERED: controller added to pse_controller_list and
+ * resolvable by of_pse_control_get().
+ * @PSE_UNREGISTERED: controller already taken off pse_controller_list, so
+ * no longer resolvable, but still valid to dereference for the
+ * duration of the callback. Subscribers holding pse_control
+ * references targeting it must drop them before returning and must
+ * not acquire new references for it.
+ */
+enum pse_controller_event {
+ PSE_REGISTERED,
+ PSE_UNREGISTERED,
+};
+
#if IS_ENABLED(CONFIG_PSE_CONTROLLER)
int pse_controller_register(struct pse_controller_dev *pcdev);
void pse_controller_unregister(struct pse_controller_dev *pcdev);
@@ -366,6 +386,9 @@ int pse_ethtool_set_prio(struct pse_control *psec,
bool pse_has_podl(struct pse_control *psec);
bool pse_has_c33(struct pse_control *psec);
+int pse_register_notifier(struct notifier_block *nb);
+int pse_unregister_notifier(struct notifier_block *nb);
+
#else
static inline struct pse_control *of_pse_control_get(struct device_node *node,
@@ -416,6 +439,16 @@ static inline bool pse_has_c33(struct pse_control *psec)
return false;
}
+static inline int pse_register_notifier(struct notifier_block *nb)
+{
+ return 0;
+}
+
+static inline int pse_unregister_notifier(struct notifier_block *nb)
+{
+ return 0;
+}
+
#endif
#endif
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH net-next v7 2/5] net: pse-pd: fire lifecycle events on controller register/unregister
2026-09-27 19:18 [PATCH net-next v7 0/5] net: pse-pd: decouple controller lookup from MDIO probe Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 1/5] net: pse-pd: add notifier chain for controller lifecycle events Carlo Szelinsky
@ 2026-09-27 19:18 ` Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 3/5] net: pse-pd: unwind allocations when controller registration fails Carlo Szelinsky
` (2 subsequent siblings)
4 siblings, 0 replies; 6+ messages in thread
From: Carlo Szelinsky @ 2026-09-27 19:18 UTC (permalink / raw)
To: Oleksij Rempel, Kory Maincent, Andrew Lunn, Heiner Kallweit,
Russell King, David S . Miller, Eric Dumazet, Jakub Kicinski,
Paolo Abeni
Cc: Corey Leavitt, Jonas Jelonek, Simon Horman,
Aleksander Jan Bajkowski, Mark Brown, Liam Girdwood,
netdev-bot+sashiko, netdev, linux-kernel, Carlo Szelinsky
From: Corey Leavitt <corey@leavitt.info>
Hook the newly-introduced pse_controller_notifier chain so that
pse_controller_register() fires PSE_REGISTERED after the controller
has been added to pse_controller_list (i.e. is now resolvable by
of_pse_control_get()), and pse_controller_unregister() fires
PSE_UNREGISTERED once it has been taken back off, while pcdev and
everything a subscriber's pse_control points at are still valid to
dereference.
No subscriber exists yet, so the event itself does nothing; a later
change wires the phy subsystem in as the first one. The reordering below
is not a no-op, though - it closes a teardown race that is reachable
today, with no subscriber involved.
Unregistration is reordered around that event, because a subscriber runs
arbitrary teardown inside it.
The controller is unlinked first. of_pse_control_get() walks
pse_controller_list and dereferences pcdev->pi[] through
of_pse_match_pi(), so a lookup racing the teardown could otherwise reach
an array that pse_release_pis() has already freed. Subscribers are handed
pcdev as the event data and do not need it on the list, so taking it off
first costs nothing and closes that race for every caller, not just the
ones this series adds.
disable_irq() moves up for the same reason: pse_isr() queues
notifications and reaches pcdev->pi, and nothing after it re-enables the
interrupt.
cancel_work_sync() moves up, above the frees, but stays below the event.
A subscriber dropping the last pse_control reference reaches
__pse_control_release(), which calls regulator_disable() if the PI is
still on. With the static budget strategy that retries any port on the
same power domain waiting for power, and if the domain is still over
budget it sheds a lower priority port through pse_disable_pi_pol(),
which queues a notification and calls schedule_work(). Draining the
worker before the walk would therefore leave work queued behind it,
racing the kfifo_free() below.
That retry does more than queue work: _pse_pi_delivery_power_sw_pw_ctrl()
calls ops->pi_enable(), so a port on the controller being torn down can
be energised from inside the event. The driver is still bound at that
point - the walk runs before pse_release_pis() and before devres
unwinds the PI regulators - so the call is legal, but it is worth
naming rather than leaving to be discovered.
Draining after the walk also keeps the worker's own transient
pse_control reference - taken by pse_control_find_by_id() - from becoming
the last one after pse_release_pis() has freed the array.
The frees move the other way. pse_flush_pw_ds() and pse_release_pis()
are the first two statements today, ahead of disable_irq() and
cancel_work_sync(); they end up last here. That ordering is what the
race fix consists of: until now pse_isr() and the worker could both
reach pcdev->pi[] after pse_release_pis() had freed it. They also have
to stay below the event, because the release path it runs reads
pcdev->pi[] and pi->pw_d->supply.
The series at
https://lore.kernel.org/netdev/20260813200653.980170-1-github@szelinsky.de/
makes a related reordering for net, independently of any subscriber, and
the two will conflict when it back-merges. The order there is not
identical: it leaves the unlink below cancel_work_sync() and
pse_flush_pw_ds(), having no event to place. The merged function wants
the order here, which contains that fix.
Signed-off-by: Corey Leavitt <corey@leavitt.info>
Signed-off-by: Carlo Szelinsky <github@szelinsky.de>
Tested-by: Jonas Jelonek <jelonek.jonas@gmail.com>
---
drivers/net/pse-pd/pse_core.c | 32 ++++++++++++++++++++++++++++----
1 file changed, 28 insertions(+), 4 deletions(-)
diff --git a/drivers/net/pse-pd/pse_core.c b/drivers/net/pse-pd/pse_core.c
index 84c734ed4553..56cecf60c5c4 100644
--- a/drivers/net/pse-pd/pse_core.c
+++ b/drivers/net/pse-pd/pse_core.c
@@ -1138,6 +1138,9 @@ int pse_controller_register(struct pse_controller_dev *pcdev)
list_add(&pcdev->list, &pse_controller_list);
mutex_unlock(&pse_list_mutex);
+ blocking_notifier_call_chain(&pse_controller_notifier,
+ PSE_REGISTERED, pcdev);
+
return 0;
}
EXPORT_SYMBOL_GPL(pse_controller_register);
@@ -1148,15 +1151,36 @@ EXPORT_SYMBOL_GPL(pse_controller_register);
*/
void pse_controller_unregister(struct pse_controller_dev *pcdev)
{
- pse_flush_pw_ds(pcdev);
- pse_release_pis(pcdev);
+ /* Stop the interrupt first: pse_isr() queues notifications and
+ * reaches pcdev->pi, and nothing below re-enables it.
+ */
if (pcdev->irq)
disable_irq(pcdev->irq);
- cancel_work_sync(&pcdev->ntf_work);
- kfifo_free(&pcdev->ntf_fifo);
+
+ /* Unlink before the event: of_pse_control_get() walks
+ * pse_controller_list and dereferences pcdev->pi[] through
+ * of_pse_match_pi(), so no lookup may still reach this controller
+ * once its teardown starts. Subscribers are handed pcdev as the
+ * event data, so the notifier does not need it on the list.
+ */
mutex_lock(&pse_list_mutex);
list_del(&pcdev->list);
mutex_unlock(&pse_list_mutex);
+
+ blocking_notifier_call_chain(&pse_controller_notifier,
+ PSE_UNREGISTERED, pcdev);
+
+ /* After the event, not before. A subscriber dropping the last
+ * pse_control reference reaches __pse_control_release() ->
+ * regulator_disable() -> _pse_pi_disable(), which can end up in
+ * pse_disable_pi_pol() and queue a notification of its own, so a
+ * cancel_work_sync() placed above the walk would not stay drained.
+ */
+ cancel_work_sync(&pcdev->ntf_work);
+
+ pse_flush_pw_ds(pcdev);
+ pse_release_pis(pcdev);
+ kfifo_free(&pcdev->ntf_fifo);
}
EXPORT_SYMBOL_GPL(pse_controller_unregister);
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH net-next v7 3/5] net: pse-pd: unwind allocations when controller registration fails
2026-09-27 19:18 [PATCH net-next v7 0/5] net: pse-pd: decouple controller lookup from MDIO probe Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 1/5] net: pse-pd: add notifier chain for controller lifecycle events Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 2/5] net: pse-pd: fire lifecycle events on controller register/unregister Carlo Szelinsky
@ 2026-09-27 19:18 ` Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 4/5] net: pse-pd: check the PI vpwr supply before registering the controller Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 5/5] net: phy: own phydev->psec via PSE notifier and remove fwnode_mdio hook Carlo Szelinsky
4 siblings, 0 replies; 6+ messages in thread
From: Carlo Szelinsky @ 2026-09-27 19:18 UTC (permalink / raw)
To: Oleksij Rempel, Kory Maincent, Andrew Lunn, Heiner Kallweit,
Russell King, David S . Miller, Eric Dumazet, Jakub Kicinski,
Paolo Abeni
Cc: Corey Leavitt, Jonas Jelonek, Simon Horman,
Aleksander Jan Bajkowski, Mark Brown, Liam Girdwood,
netdev-bot+sashiko, netdev, linux-kernel, Carlo Szelinsky
pse_controller_register() allocates the notification kfifo and, through
of_load_pse_pis(), the PI array plus an OF reference per described PI.
Every failure after that point simply returns: none of it is freed.
kfifo_free() and pse_release_pis() only run from
pse_controller_unregister(), which a failed registration never reaches,
and devm_pse_controller_register() drops only its own devres cookie.
pcdev->pi is a plain allocation, so nothing else will ever free it, and
all five in-tree controller drivers keep pse_controller_dev inside their
private data, whose last pointer goes away with the failed probe.
A partial pse_register_pw_ds() is worse than a leak. The power domains it
already created are devm-allocated but live in the global pse_pw_d_map,
so the failed probe frees them while that xarray still points at them,
and the next controller to register walks into freed memory in
regulator_is_equal().
Unwind at two depths, because the PI array cannot always be freed here.
pse_pi_ops.is_enabled(), .enable() and .disable() all index pcdev->pi[],
and the PI regulators are devm-registered on pcdev->dev, so from the
first successful devm_pse_pi_regulator_register() until devres unwinds
the failed probe there are live regulators whose ops would follow a freed
pointer. regulator_late_cleanup() and the "state" class attribute both
reach those ops. So the array is released on the failures that happen
while it exists and before the first PI regulator does: setup_pi_matrix()
and the supply check. A driver's setup_pi_matrix() may well have
registered devm regulators of its own by the time it fails - pd692x0
registers its managers there - but those do not index pcdev->pi[], so
they do not constrain this. Earlier than that there is nothing to release -
pcdev->pi is still NULL, or of_load_pse_pis() has already freed it - and
from the registration loop onwards the array has to stay, so those
failures only free the kfifo and flush the power domains, leaving it
leaked exactly as it is today rather than handing those regulators a
dangling pointer. An allocation failure in the loop's first iteration
leaks it too, where releasing would still have been safe, but the rule
stays simple enough to read. Freeing it safely needs the NULL-and-guard
treatment the pending net teardown fix adds to the regulator ops, which
is not in net-next.
of_load_pse_pis() already releases the PI array on its own failures, so
that path only needs the kfifo.
pcdev->pi is cleared at the release_pis label and deliberately not
inside pse_release_pis() itself. The label is the one place the array
is freed while no PI regulator exists. pse_controller_unregister()
calls the same helper with every PI regulator still registered - the
devres node for devm_pse_controller_register() is added after them, so
it is released first - and pse_pi_is_enabled() indexes pcdev->pi[]
unguarded behind the regulator "state" attribute. Clearing the pointer
in the helper would turn that pre-existing read of freed memory into a
NULL dereference.
Signed-off-by: Carlo Szelinsky <github@szelinsky.de>
---
drivers/net/pse-pd/pse_core.c | 37 +++++++++++++++++++++++++++--------
1 file changed, 29 insertions(+), 8 deletions(-)
diff --git a/drivers/net/pse-pd/pse_core.c b/drivers/net/pse-pd/pse_core.c
index 56cecf60c5c4..16d75b4babf3 100644
--- a/drivers/net/pse-pd/pse_core.c
+++ b/drivers/net/pse-pd/pse_core.c
@@ -1092,17 +1092,18 @@ int pse_controller_register(struct pse_controller_dev *pcdev)
!pcdev->ops->pi_get_pw_status) {
dev_err(pcdev->dev,
"Mandatory status report callbacks are missing");
- return -EINVAL;
+ ret = -EINVAL;
+ goto free_kfifo;
}
ret = of_load_pse_pis(pcdev);
if (ret)
- return ret;
+ goto free_kfifo;
if (pcdev->ops->setup_pi_matrix) {
ret = pcdev->ops->setup_pi_matrix(pcdev);
if (ret)
- return ret;
+ goto release_pis;
}
/* Each regulator name len is pcdev dev name + 7 char +
@@ -1110,7 +1111,12 @@ int pse_controller_register(struct pse_controller_dev *pcdev)
*/
reg_name_len = strlen(dev_name(pcdev->dev)) + 18;
- /* Register PI regulators */
+ /* Register PI regulators. Once one of these exists, pse_pi_ops index
+ * pcdev->pi[] and nothing here can unregister it again, so the array
+ * must outlive this function. Failures below therefore unwind to
+ * free_kfifo and deliberately leak it, as they already do today,
+ * rather than hand the live regulators a freed pointer.
+ */
for (i = 0; i < pcdev->nr_lines; i++) {
char *reg_name;
@@ -1119,20 +1125,22 @@ int pse_controller_register(struct pse_controller_dev *pcdev)
continue;
reg_name = devm_kzalloc(pcdev->dev, reg_name_len, GFP_KERNEL);
- if (!reg_name)
- return -ENOMEM;
+ if (!reg_name) {
+ ret = -ENOMEM;
+ goto free_kfifo;
+ }
snprintf(reg_name, reg_name_len, "pse-%s_pi%d",
dev_name(pcdev->dev), i);
ret = devm_pse_pi_regulator_register(pcdev, reg_name, i);
if (ret)
- return ret;
+ goto free_kfifo;
}
ret = pse_register_pw_ds(pcdev);
if (ret)
- return ret;
+ goto flush_pw_ds;
mutex_lock(&pse_list_mutex);
list_add(&pcdev->list, &pse_controller_list);
@@ -1142,6 +1150,19 @@ int pse_controller_register(struct pse_controller_dev *pcdev)
PSE_REGISTERED, pcdev);
return 0;
+
+flush_pw_ds:
+ pse_flush_pw_ds(pcdev);
+ goto free_kfifo;
+
+release_pis:
+ pse_release_pis(pcdev);
+ pcdev->pi = NULL;
+
+free_kfifo:
+ kfifo_free(&pcdev->ntf_fifo);
+
+ return ret;
}
EXPORT_SYMBOL_GPL(pse_controller_register);
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH net-next v7 4/5] net: pse-pd: check the PI vpwr supply before registering the controller
2026-09-27 19:18 [PATCH net-next v7 0/5] net: pse-pd: decouple controller lookup from MDIO probe Carlo Szelinsky
` (2 preceding siblings ...)
2026-09-27 19:18 ` [PATCH net-next v7 3/5] net: pse-pd: unwind allocations when controller registration fails Carlo Szelinsky
@ 2026-09-27 19:18 ` Carlo Szelinsky
2026-09-27 19:18 ` [PATCH net-next v7 5/5] net: phy: own phydev->psec via PSE notifier and remove fwnode_mdio hook Carlo Szelinsky
4 siblings, 0 replies; 6+ messages in thread
From: Carlo Szelinsky @ 2026-09-27 19:18 UTC (permalink / raw)
To: Oleksij Rempel, Kory Maincent, Andrew Lunn, Heiner Kallweit,
Russell King, David S . Miller, Eric Dumazet, Jakub Kicinski,
Paolo Abeni
Cc: Corey Leavitt, Jonas Jelonek, Simon Horman,
Aleksander Jan Bajkowski, Mark Brown, Liam Girdwood,
netdev-bot+sashiko, netdev, linux-kernel, Carlo Szelinsky
Each PSE PI regulator is registered with a "vpwr" supply, which the
bindings expect the board to provide via vpwr-supply in the PI node. The
regulator core deliberately treats an unresolved supply at registration
time as non-fatal: it logs, registers a bus device so the supply can be
picked up later, and lets registration succeed.
That leaves pse_controller_register() completing for a controller whose
PIs cannot be handed out. regulator_get_exclusive() in
pse_control_get_internal() resolves the supply itself and returns
-EPROBE_DEFER until the provider appears, so of_pse_control_get() keeps
failing for this PI even though the controller is registered and
discoverable. A consumer that only retries on controller registration -
as the phy layer does after the next patch - then never gets its PI.
Check every PI's supply first, before any regulator of ours is
registered, so the error surfaces in the PSE driver's own probe and
deferred probe retries it there. The check skips exactly the PIs the
registration loop skips, so a controller with no pse-pis node - where
every PI has a NULL np but a regulator is still created for it - is
covered too, through the controller device rather than a PI node.
The check follows both stages regulator_resolve_supply() uses: the PI's
own node, then the controller device. Both are needed. of_get_regulator()
reads the node it is handed and otherwise walks that device's children,
so a vpwr-supply written once on the controller node - covering every PI
- is invisible from the PI node, while the core finds it at stage two.
Checking only the PI node would let exactly the case this patch exists
to prevent slip through unreported.
It is still not a full replica. The core also falls back to a dummy
under have_full_constraints(), which does not matter here because it
bails on an explicit -EPROBE_DEFER first, and it defers when the
provider's parent is not yet bound, which regulator_get_optional() does
not check. So a PSE probe interleaving with the provider's own probe can
still pass the check and resolve late. That window is narrower than the
one being closed - a provider that has not registered at all - and
closing it properly means asking the core about rdev->supply after
registration, which the unwind constraints above do not allow.
Doing all the checks before the registration loop also keeps the common
deferral on the unwind path that can release the PI array: a controller
can describe its PIs on different providers - ti,tps23881.yaml puts
pse-pi@0 on vpwr1 and pse-pi@1 on vpwr2 - so "one PI resolves, the next
defers" is the ordinary case, and checking inside the registration loop
would leave registered regulators behind on every retry.
This widens what a missing provider costs. Before, only the PI on that
supply failed, and only when a consumer asked for it; now the controller
itself defers, so a board whose PIs sit on different providers loses PSE
on all of them until the last one appears. That is the normal contract
for a regulator consumer, and the alternative - a controller that
advertises PIs it cannot hand out - is the bug being fixed. A provider
that never appears leaves the controller unbound rather than half
working, which deferred probe reports in the usual way.
Signed-off-by: Carlo Szelinsky <github@szelinsky.de>
---
drivers/net/pse-pd/pse_core.c | 72 +++++++++++++++++++++++++++++++++++
1 file changed, 72 insertions(+)
diff --git a/drivers/net/pse-pd/pse_core.c b/drivers/net/pse-pd/pse_core.c
index 16d75b4babf3..457eef5784f8 100644
--- a/drivers/net/pse-pd/pse_core.c
+++ b/drivers/net/pse-pd/pse_core.c
@@ -12,6 +12,7 @@
#include <linux/of.h>
#include <linux/phy.h>
#include <linux/pse-pd/pse.h>
+#include <linux/regulator/consumer.h>
#include <linux/regulator/driver.h>
#include <linux/regulator/machine.h>
#include <linux/rtnetlink.h>
@@ -860,6 +861,67 @@ static const struct regulator_ops pse_pi_ops = {
.set_current_limit = pse_pi_set_current_limit,
};
+/* The regulator core treats an unresolved "vpwr" supply as non-fatal and
+ * retries it on its own later, which would leave this controller able to
+ * register while of_pse_control_get() still fails with -EPROBE_DEFER for
+ * this PI, and nothing to retry the consumer's attach. Check the supply
+ * up front so the PSE driver's own probe defers instead.
+ *
+ * of_regulator_get_optional() reports a PI with no vpwr-supply described
+ * as -ENODEV rather than falling back to the dummy regulator, and has a
+ * stub for CONFIG_OF=n. Those PIs keep their existing behaviour: the
+ * regulator core resolves them to the dummy when the PI regulator is
+ * registered.
+ */
+static int pse_pi_check_supply(struct pse_controller_dev *pcdev, int id)
+{
+ struct regulator *supply;
+ int ret;
+
+ /* Skip exactly the PIs the registration loop below skips, so every
+ * regulator that will be created is covered.
+ */
+ if (!pcdev->no_of_pse_pi && !pcdev->pi[id].np)
+ return 0;
+
+ /* Follow both stages regulator_resolve_supply() will use for the
+ * regulator about to be registered: the PI's own node first, then
+ * the controller device. Only -ENODEV moves on to the second stage;
+ * a provider that has not registered yet gives -EPROBE_DEFER, which
+ * is the case worth catching here.
+ */
+ if (pcdev->pi[id].np) {
+ supply = of_regulator_get_optional(pcdev->dev,
+ pcdev->pi[id].np, "vpwr");
+ if (!IS_ERR(supply)) {
+ regulator_put(supply);
+ return 0;
+ }
+
+ ret = PTR_ERR(supply);
+ if (ret != -ENODEV)
+ return dev_err_probe(pcdev->dev, ret,
+ "PI %d: failed to get vpwr supply\n",
+ id);
+ }
+
+ /* A vpwr-supply on the controller node covers every PI, and is
+ * where a PI without a node of its own resolves too.
+ */
+ supply = regulator_get_optional(pcdev->dev, "vpwr");
+ if (!IS_ERR(supply)) {
+ regulator_put(supply);
+ return 0;
+ }
+
+ ret = PTR_ERR(supply);
+ if (ret == -ENODEV)
+ return 0;
+
+ return dev_err_probe(pcdev->dev, ret,
+ "PI %d: failed to get vpwr supply\n", id);
+}
+
static int
devm_pse_pi_regulator_register(struct pse_controller_dev *pcdev,
char *name, int id)
@@ -1111,6 +1173,16 @@ int pse_controller_register(struct pse_controller_dev *pcdev)
*/
reg_name_len = strlen(dev_name(pcdev->dev)) + 18;
+ /* Check every PI supply before registering any regulator: a provider
+ * that has not probed yet is the ordinary -EPROBE_DEFER case, and
+ * unwinding it must not leave PI regulators behind.
+ */
+ for (i = 0; i < pcdev->nr_lines; i++) {
+ ret = pse_pi_check_supply(pcdev, i);
+ if (ret)
+ goto release_pis;
+ }
+
/* Register PI regulators. Once one of these exists, pse_pi_ops index
* pcdev->pi[] and nothing here can unregister it again, so the array
* must outlive this function. Failures below therefore unwind to
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread* [PATCH net-next v7 5/5] net: phy: own phydev->psec via PSE notifier and remove fwnode_mdio hook
2026-09-27 19:18 [PATCH net-next v7 0/5] net: pse-pd: decouple controller lookup from MDIO probe Carlo Szelinsky
` (3 preceding siblings ...)
2026-09-27 19:18 ` [PATCH net-next v7 4/5] net: pse-pd: check the PI vpwr supply before registering the controller Carlo Szelinsky
@ 2026-09-27 19:18 ` Carlo Szelinsky
4 siblings, 0 replies; 6+ messages in thread
From: Carlo Szelinsky @ 2026-09-27 19:18 UTC (permalink / raw)
To: Oleksij Rempel, Kory Maincent, Andrew Lunn, Heiner Kallweit,
Russell King, David S . Miller, Eric Dumazet, Jakub Kicinski,
Paolo Abeni
Cc: Corey Leavitt, Jonas Jelonek, Simon Horman,
Aleksander Jan Bajkowski, Mark Brown, Liam Girdwood,
netdev-bot+sashiko, netdev, linux-kernel, Carlo Szelinsky
From: Corey Leavitt <corey@leavitt.info>
fwnode_mdiobus_register_phy() resolved the `pses` phandle during PHY
registration and stashed the result in phydev->psec. With a modular PSE
controller driver that lookup returns -EPROBE_DEFER until the module is
loaded, so the PHY - and any DSA switch behind it - keeps bouncing off
deferred probe. On some boards that shows up as a boot-time probe-retry
storm and PHYs that never register at all.
Transfer ownership of phydev->psec to phylib and drive it from the PSE
controller lifecycle notifier instead:
- On PSE_REGISTERED: walk mdio_bus_type and attach phydev->psec for
every registered phy whose psec is still NULL. This is the "phy was
enumerated before the PSE controller loaded" case, the root cause of
the retry storm.
- On PSE_UNREGISTERED: walk it again and release every phydev->psec
that targets the departing controller, before pse_release_pis()
frees pcdev->pi. Without this a phy still holding a reference would
cause a use-after-free in __pse_control_release()'s
pcdev->pi[psec->id] access.
- phy_device_register() attaches once for a controller that is already
registered. The phydev->psec check in phy_try_attach_pse() makes the
two paths idempotent.
fwnode_find_pse_control() and its call site go away, and fwnode_mdio
drops the PSE header. No PSE-originated -EPROBE_DEFER leaves the MDIO
layer any more, so the retry storm is gone.
The attach happens after device_add() has made the phy visible on
mdio_bus_type, and takes no rtnl. Two reasons it must not:
- Binding a phy that itself provides an SFP cage reaches
sfp_bus_add_upstream() via phy_probe() -> phy_setup_ports() ->
phy_sfp_probe(), and that takes rtnl_lock(). Reported on RTL8214FC.
- Some drivers register their MDIO bus from ndo_init - lantiq_etop via
ltq_etop_init(), sni_ave via ave_init() - which register_netdevice()
already calls under rtnl, and mdiobus_scan() reaches
phy_device_register() from there.
phydev->psec is therefore serialised by a dedicated mutex, taken through
pse_phy_lock(). It lives in pse_core rather than phylib because
net/ethtool is always built into vmlinux while PHYLIB is tristate, so a
phylib export would be unresolved with CONFIG_PHYLIB=m or =n;
PSE_CONTROLLER is bool, so pse_core is always reachable. The ethtool PSE
paths take the same mutex, so the use-after-free that rtnl used to close
stays closed. Lock order is rtnl -> pse_phy_mutex -> pse_list_mutex ->
pcdev->lock.
phy_device_remove() releases phydev->psec synchronously, while the phy
is still on the bus. A phy that has been device_del()'d but is still
pinned - an attached netdev, or an SFP module phy waiting for
phy_device_free() - is off the mdio_bus_type klist and so invisible to
the PSE_UNREGISTERED walk, and a deferred release would then touch a
pcdev->pi[] the controller has already freed. Releasing before
device_del() means the walk and the release contend for the same mutex,
so whichever runs second sees NULL.
phydev->psec_detached closes two windows around that, by telling
phy_try_attach_pse() to skip the phy. It is a plain bool rather than
another bit in the flags word: it is written under pse_phy_lock(), while
neighbours such as suspended and sysfs_links are written under
phydev->lock and rtnl, and a shared storage unit would make those
read-modify-write updates race. bus_for_each_dev() still reaches a
phy until bus_remove_device() takes it off the klist, which is well into
device_del(), so a PSE_REGISTERED walk could otherwise attach a fresh
handle just after the release, with nothing left to free it. The same
applies while the phy is being registered: device_add() puts it on the
bus before its own later failure points, and its unwind takes it back
off, so a handle attached in that window would be missed by the
PSE_UNREGISTERED walk as well. The flag is therefore held until
registration has actually succeeded, and set again from
phy_device_remove().
One note on behaviour. of_pse_control_get() reaches the hardware -
pse_pi_is_hw_enabled() calls pi_get_admin_state(), an i2c transaction on
tps23881, si3474, pd692x0 and realtek-pse-mcu - and that error used to
propagate out of fwnode_mdiobus_register_phy() and get retried by
deferred probe. Now a transient failure leaves the port without PSE
until the controller registers again; it is reported with phydev_warn()
rather than dropped silently. -EPROBE_DEFER stays silent because the
notifier retries it at PSE_REGISTERED time, and the preceding patch
makes sure an already registered controller cannot keep returning it.
The UNREGISTERED walk does not help rmmod either:
pse_control_get_internal() takes try_module_get(pcdev->owner) per
handle, so while a phy holds one the unload is refused before the module
exit path runs. What the walk covers is driver unbind and device
removal, where pse_controller_unregister() runs with the module loaded.
What makes the UNREGISTERED walk safe is the teardown order the preceding
patch establishes. By the time it runs the controller is already off
pse_controller_list and the interrupt is off, so nothing new can reach
the PI array; pcdev->pi and the power domains are still live, which the
release path needs because dropping the last reference can disable a PI
that is still energised; and the notification worker is drained after the
walk rather than before it, both because that release path can queue work
of its own and because the worker's own transient pse_control reference
must not become the last one once the array has been freed.
Now that the walk empties pcdev->pse_control_head on unregister, warn if
anything is still on it afterwards. Nothing can add one at that point:
the controller is off the list, the irq is off and the worker is
drained. It is quiet because of_pse_control_get() has a single caller
and the walk covers it, and it fires if a consumer that does not
subscribe is ever added.
The warning belongs here rather than with the reordering it sits in,
because until this patch the fwnode_mdio hook gives every matching phy a
handle for its lifetime and nothing releases it on controller unbind -
so an unbind between the two would trip it on any board where a phy
resolved a PI.
Reported-by: Jonas Jelonek <jelonek.jonas@gmail.com>
Closes: https://lore.kernel.org/netdev/e00048dd-1ed3-40c3-9912-59bccf015ad5@gmail.com/
Reported-by: Aleksander Jan Bajkowski <olek2@wp.pl>
Closes: https://lore.kernel.org/netdev/bac5e6e9-7358-4ccb-87fc-9c40baa33682@wp.pl/
Suggested-by: Paolo Abeni <pabeni@redhat.com>
Link: https://lore.kernel.org/netdev/20260703071025.100797-1-pabeni@redhat.com/
Signed-off-by: Corey Leavitt <corey@leavitt.info>
Co-developed-by: Carlo Szelinsky <github@szelinsky.de>
Signed-off-by: Carlo Szelinsky <github@szelinsky.de>
Tested-by: Jonas Jelonek <jelonek.jonas@gmail.com>
Tested-by: Aleksander Jan Bajkowski <olek2@wp.pl>
---
drivers/net/mdio/fwnode_mdio.c | 34 --------
drivers/net/phy/phy_device.c | 139 ++++++++++++++++++++++++++++++++-
drivers/net/pse-pd/pse_core.c | 68 ++++++++++++++++
include/linux/phy.h | 7 ++
include/linux/pse-pd/pse.h | 32 ++++++++
net/ethtool/pse-pd.c | 16 ++--
6 files changed, 256 insertions(+), 40 deletions(-)
diff --git a/drivers/net/mdio/fwnode_mdio.c b/drivers/net/mdio/fwnode_mdio.c
index ba7091518265..7bd979b59f49 100644
--- a/drivers/net/mdio/fwnode_mdio.c
+++ b/drivers/net/mdio/fwnode_mdio.c
@@ -11,33 +11,11 @@
#include <linux/fwnode_mdio.h>
#include <linux/of.h>
#include <linux/phy.h>
-#include <linux/pse-pd/pse.h>
MODULE_AUTHOR("Calvin Johnson <calvin.johnson@oss.nxp.com>");
MODULE_LICENSE("GPL");
MODULE_DESCRIPTION("FWNODE MDIO bus (Ethernet PHY) accessors");
-static struct pse_control *
-fwnode_find_pse_control(struct fwnode_handle *fwnode,
- struct phy_device *phydev)
-{
- struct pse_control *psec;
- struct device_node *np;
-
- if (!IS_ENABLED(CONFIG_PSE_CONTROLLER))
- return NULL;
-
- np = to_of_node(fwnode);
- if (!np)
- return NULL;
-
- psec = of_pse_control_get(np, phydev);
- if (PTR_ERR(psec) == -ENOENT)
- return NULL;
-
- return psec;
-}
-
static struct mii_timestamper *
fwnode_find_mii_timestamper(struct fwnode_handle *fwnode)
{
@@ -118,7 +96,6 @@ int fwnode_mdiobus_register_phy(struct mii_bus *bus,
struct fwnode_handle *child, u32 addr)
{
struct mii_timestamper *mii_ts = NULL;
- struct pse_control *psec = NULL;
struct phy_device *phy;
bool is_c45;
u32 phy_id;
@@ -159,14 +136,6 @@ int fwnode_mdiobus_register_phy(struct mii_bus *bus,
goto clean_phy;
}
- psec = fwnode_find_pse_control(child, phy);
- if (IS_ERR(psec)) {
- rc = PTR_ERR(psec);
- goto unregister_phy;
- }
-
- phy->psec = psec;
-
/* phy->mii_ts may already be defined by the PHY driver. A
* mii_timestamper probed via the device tree will still have
* precedence.
@@ -176,9 +145,6 @@ int fwnode_mdiobus_register_phy(struct mii_bus *bus,
return 0;
-unregister_phy:
- if (is_acpi_node(child) || is_of_node(child))
- phy_device_remove(phy);
clean_phy:
phy_device_free(phy);
clean_mii_ts:
diff --git a/drivers/net/phy/phy_device.c b/drivers/net/phy/phy_device.c
index 5b13a74e2fa9..971d9326d226 100644
--- a/drivers/net/phy/phy_device.c
+++ b/drivers/net/phy/phy_device.c
@@ -26,6 +26,7 @@
#include <linux/module.h>
#include <linux/of.h>
#include <linux/netdevice.h>
+#include <linux/notifier.h>
#include <linux/phy.h>
#include <linux/phylib_stubs.h>
#include <linux/phy_led_triggers.h>
@@ -1012,9 +1013,111 @@ struct phy_device *get_phy_device(struct mii_bus *bus, int addr, bool is_c45)
}
EXPORT_SYMBOL(get_phy_device);
+/* Best-effort attach of phydev->psec from a DT `pses = <&...>` phandle.
+ * Caller must hold pse_phy_lock(). A missing phandle (-ENOENT) or a
+ * not-yet-registered controller (-EPROBE_DEFER) is silent; the notifier
+ * retries the latter at PSE_REGISTERED time. Any other error means a broken
+ * binding and is warned about, but left non-fatal so the phy still registers.
+ *
+ * A phy with psec_detached set is skipped: it is either not registered yet
+ * or on its way out, and nothing would release a handle attached now.
+ */
+static void phy_try_attach_pse(struct phy_device *phydev)
+{
+ struct pse_control *psec;
+ struct device_node *np;
+
+ pse_phy_lock_assert_held();
+
+ np = phydev->mdio.dev.of_node;
+ if (!np)
+ return;
+
+ if (phydev->psec || phydev->psec_detached)
+ return;
+
+ psec = of_pse_control_get(np, phydev);
+ if (IS_ERR(psec)) {
+ if (PTR_ERR(psec) != -EPROBE_DEFER && PTR_ERR(psec) != -ENOENT)
+ phydev_warn(phydev, "failed to get PSE control: %pe\n",
+ psec);
+ return;
+ }
+
+ phydev->psec = psec;
+}
+
+static int phy_pse_attach_one(struct device *dev, void *data)
+{
+ pse_phy_lock_assert_held();
+
+ if (dev->type != &mdio_bus_phy_type)
+ return 0;
+
+ phy_try_attach_pse(to_phy_device(dev));
+ return 0;
+}
+
+static int phy_pse_detach_one(struct device *dev, void *data)
+{
+ struct pse_controller_dev *pcdev = data;
+ struct phy_device *phydev;
+ struct pse_control *psec;
+
+ pse_phy_lock_assert_held();
+
+ if (dev->type != &mdio_bus_phy_type)
+ return 0;
+
+ phydev = to_phy_device(dev);
+ psec = phydev->psec;
+ if (!psec || !pse_control_matches_pcdev(psec, pcdev))
+ return 0;
+
+ phydev->psec = NULL;
+ pse_control_put(psec);
+ return 0;
+}
+
+static int phy_pse_notifier_event(struct notifier_block *nb,
+ unsigned long event, void *data)
+{
+ switch (event) {
+ case PSE_REGISTERED:
+ pse_phy_lock();
+ bus_for_each_dev(&mdio_bus_type, NULL, NULL,
+ phy_pse_attach_one);
+ pse_phy_unlock();
+ return NOTIFY_OK;
+ case PSE_UNREGISTERED:
+ pse_phy_lock();
+ bus_for_each_dev(&mdio_bus_type, NULL, data,
+ phy_pse_detach_one);
+ pse_phy_unlock();
+ return NOTIFY_OK;
+ default:
+ return NOTIFY_DONE;
+ }
+}
+
+static struct notifier_block phy_pse_notifier __read_mostly = {
+ .notifier_call = phy_pse_notifier_event,
+};
+
/**
* phy_device_register - Register the phy device on the MDIO bus
* @phydev: phy_device structure to be added to the MDIO bus
+ *
+ * phydev->psec is attached after device_add() has made the phy visible on
+ * mdio_bus_type, so that a concurrent PSE notifier walk and the attach can
+ * never leave the phy unattached. Neither step takes rtnl: keeping
+ * device_add() out of rtnl avoids deadlocking when binding a phy that itself
+ * provides an SFP cage (phy_probe() -> phy_sfp_probe() ->
+ * sfp_bus_add_upstream() takes rtnl), and pse_phy_lock() rather than rtnl
+ * guards the attach so a bus registered from ndo_init (which already holds
+ * rtnl) does not recurse on it.
+ *
+ * Return: 0 on success, negative error code on failure.
*/
int phy_device_register(struct phy_device *phydev)
{
@@ -1034,12 +1137,25 @@ int phy_device_register(struct phy_device *phydev)
goto out;
}
+ /* Keep the PSE_REGISTERED walk off this phy until registration has
+ * actually succeeded. device_add() puts the phy on the bus before its
+ * own later failure points, and its unwind takes it back off, so a
+ * handle attached in that window would be missed by the
+ * PSE_UNREGISTERED walk as well and left with no owner.
+ */
+ phydev->psec_detached = true;
+
err = device_add(&phydev->mdio.dev);
if (err) {
phydev_err(phydev, "failed to add\n");
goto out;
}
+ pse_phy_lock();
+ phydev->psec_detached = false;
+ phy_try_attach_pse(phydev);
+ pse_phy_unlock();
+
return 0;
out:
@@ -1061,8 +1177,22 @@ EXPORT_SYMBOL(phy_device_register);
*/
void phy_device_remove(struct phy_device *phydev)
{
+ struct pse_control *psec;
+
unregister_mii_timestamper(phydev->mii_ts);
- pse_control_put(phydev->psec);
+
+ /* Detach synchronously, before the phy leaves the bus, so the put cannot
+ * outlive the PSE controller: an off-bus but still-pinned phy is missed
+ * by the PSE_UNREGISTERED walk. pse_phy_lock() serialises against that
+ * walk, and psec_detached keeps the PSE_REGISTERED walk from attaching a
+ * new handle in the window before device_del() takes the phy off the bus.
+ */
+ pse_phy_lock();
+ psec = phydev->psec;
+ phydev->psec = NULL;
+ phydev->psec_detached = true;
+ pse_control_put(psec);
+ pse_phy_unlock();
device_del(&phydev->mdio.dev);
@@ -3976,8 +4106,14 @@ static int __init phy_init(void)
if (rc)
goto err_c45;
+ rc = pse_register_notifier(&phy_pse_notifier);
+ if (rc)
+ goto err_genphy;
+
return 0;
+err_genphy:
+ phy_driver_unregister(&genphy_driver);
err_c45:
phy_driver_unregister(&genphy_c45_driver);
err_ethtool_phy_ops:
@@ -3994,6 +4130,7 @@ static int __init phy_init(void)
static void __exit phy_exit(void)
{
+ pse_unregister_notifier(&phy_pse_notifier);
phy_driver_unregister(&genphy_c45_driver);
phy_driver_unregister(&genphy_driver);
rtnl_lock();
diff --git a/drivers/net/pse-pd/pse_core.c b/drivers/net/pse-pd/pse_core.c
index 457eef5784f8..20d33f7af743 100644
--- a/drivers/net/pse-pd/pse_core.c
+++ b/drivers/net/pse-pd/pse_core.c
@@ -25,8 +25,55 @@ static LIST_HEAD(pse_controller_list);
static DEFINE_XARRAY_ALLOC(pse_pw_d_map);
static DEFINE_MUTEX(pse_pw_d_mutex);
+/* Serialises phydev->psec against the PSE controller lifecycle notifier and
+ * the ethtool PSE paths, in place of rtnl. The attach must not take rtnl: an
+ * MDIO bus registered from ndo_init (e.g. lantiq_etop) calls
+ * phy_device_register() with rtnl already held, so taking rtnl for the attach
+ * would deadlock. It lives here rather than in phylib because PSE_CONTROLLER
+ * is bool, so pse_core is always built into vmlinux and net/ethtool can call
+ * these directly; phylib is tristate and must not be linked against from
+ * built-in code. Lock order: rtnl -> pse_phy_mutex -> pse_list_mutex ->
+ * pcdev->lock.
+ */
+static DEFINE_MUTEX(pse_phy_mutex);
+
static BLOCKING_NOTIFIER_HEAD(pse_controller_notifier);
+/**
+ * pse_phy_lock - serialise access to phydev->psec
+ *
+ * Held by the PSE controller lifecycle notifier, by the phy attach and detach
+ * paths and by the ethtool PSE paths. The PSE_UNREGISTERED walk clears
+ * phydev->psec and drops the phy's reference under this lock, so anything that
+ * attaches, detaches or dereferences phydev->psec must hold it across the
+ * whole access.
+ */
+void pse_phy_lock(void)
+{
+ mutex_lock(&pse_phy_mutex);
+}
+EXPORT_SYMBOL_GPL(pse_phy_lock);
+
+/**
+ * pse_phy_unlock - release the lock taken by pse_phy_lock()
+ */
+void pse_phy_unlock(void)
+{
+ mutex_unlock(&pse_phy_mutex);
+}
+EXPORT_SYMBOL_GPL(pse_phy_unlock);
+
+#ifdef CONFIG_LOCKDEP
+/**
+ * pse_phy_lock_assert_held - assert that pse_phy_lock() is held
+ */
+void pse_phy_lock_assert_held(void)
+{
+ lockdep_assert_held(&pse_phy_mutex);
+}
+EXPORT_SYMBOL_GPL(pse_phy_lock_assert_held);
+#endif
+
/**
* pse_register_notifier - register a callback for PSE controller events
* @nb: notifier block to register
@@ -1271,6 +1318,13 @@ void pse_controller_unregister(struct pse_controller_dev *pcdev)
*/
cancel_work_sync(&pcdev->ntf_work);
+ /* Every handle should be gone here: subscribers drop theirs in the
+ * event above, and the worker's transient one goes with the drain.
+ * The controller is off the list and the irq is off, so nothing can
+ * add one. Anything left is a holder nobody accounted for.
+ */
+ WARN_ON(!list_empty(&pcdev->pse_control_head));
+
pse_flush_pw_ds(pcdev);
pse_release_pis(pcdev);
kfifo_free(&pcdev->ntf_fifo);
@@ -2132,3 +2186,17 @@ bool pse_has_c33(struct pse_control *psec)
return psec->pcdev->types & ETHTOOL_PSE_C33;
}
EXPORT_SYMBOL_GPL(pse_has_c33);
+
+/**
+ * pse_control_matches_pcdev - Test whether a pse_control targets a controller
+ * @psec: pse_control obtained from of_pse_control_get()
+ * @pcdev: PSE controller to compare against
+ *
+ * Return: %true if @psec was obtained from @pcdev, %false otherwise.
+ */
+bool pse_control_matches_pcdev(struct pse_control *psec,
+ struct pse_controller_dev *pcdev)
+{
+ return psec->pcdev == pcdev;
+}
+EXPORT_SYMBOL_GPL(pse_control_matches_pcdev);
diff --git a/include/linux/phy.h b/include/linux/phy.h
index 7c5098a0dd6c..ef5b5e6f4e0f 100644
--- a/include/linux/phy.h
+++ b/include/linux/phy.h
@@ -666,6 +666,12 @@ struct phy_oatc14_sqi_capability {
* @master_slave_state: Current master/slave configuration
* @mii_ts: Pointer to time stamper callbacks
* @psec: Pointer to Power Sourcing Equipment control struct
+ * @psec_detached: Set while phylib will not accept a PSE handle for this
+ * phy, either because registration has not completed or because it has
+ * already released @psec, so the PSE_REGISTERED notifier walk skips it.
+ * Not a bitfield: once the phy is on the bus it is written under
+ * pse_phy_lock() while other flags are written under phydev->lock or
+ * rtnl, and they must not share a storage unit
* @ports: List of PHY ports structures
* @n_ports: Number of ports currently attached to the PHY
* @max_n_ports: Max number of ports this PHY can expose
@@ -807,6 +813,7 @@ struct phy_device {
struct net_device *attached_dev;
struct mii_timestamper *mii_ts;
struct pse_control *psec;
+ bool psec_detached;
struct list_head ports;
int n_ports;
diff --git a/include/linux/pse-pd/pse.h b/include/linux/pse-pd/pse.h
index bc5d36bcd993..16181f8a2c97 100644
--- a/include/linux/pse-pd/pse.h
+++ b/include/linux/pse-pd/pse.h
@@ -386,9 +386,23 @@ int pse_ethtool_set_prio(struct pse_control *psec,
bool pse_has_podl(struct pse_control *psec);
bool pse_has_c33(struct pse_control *psec);
+bool pse_control_matches_pcdev(struct pse_control *psec,
+ struct pse_controller_dev *pcdev);
+
int pse_register_notifier(struct notifier_block *nb);
int pse_unregister_notifier(struct notifier_block *nb);
+void pse_phy_lock(void);
+void pse_phy_unlock(void);
+
+#ifdef CONFIG_LOCKDEP
+void pse_phy_lock_assert_held(void);
+#else
+static inline void pse_phy_lock_assert_held(void)
+{
+}
+#endif
+
#else
static inline struct pse_control *of_pse_control_get(struct device_node *node,
@@ -439,6 +453,12 @@ static inline bool pse_has_c33(struct pse_control *psec)
return false;
}
+static inline bool pse_control_matches_pcdev(struct pse_control *psec,
+ struct pse_controller_dev *pcdev)
+{
+ return false;
+}
+
static inline int pse_register_notifier(struct notifier_block *nb)
{
return 0;
@@ -449,6 +469,18 @@ static inline int pse_unregister_notifier(struct notifier_block *nb)
return 0;
}
+static inline void pse_phy_lock(void)
+{
+}
+
+static inline void pse_phy_unlock(void)
+{
+}
+
+static inline void pse_phy_lock_assert_held(void)
+{
+}
+
#endif
#endif
diff --git a/net/ethtool/pse-pd.c b/net/ethtool/pse-pd.c
index 757c9e0cc856..654325946aaa 100644
--- a/net/ethtool/pse-pd.c
+++ b/net/ethtool/pse-pd.c
@@ -71,7 +71,9 @@ static int pse_prepare_data(const struct ethnl_req_info *req_base,
if (ret < 0)
return ret;
+ pse_phy_lock();
ret = pse_get_pse_attributes(phydev, info->extack, data);
+ pse_phy_unlock();
ethnl_ops_complete(dev);
@@ -281,9 +283,12 @@ ethnl_set_pse(struct ethnl_req_info *req_info, struct genl_info *info)
phydev = ethnl_req_get_phydev(req_info, tb, ETHTOOL_A_PSE_HEADER,
info->extack);
+
+ pse_phy_lock();
+
ret = ethnl_set_pse_validate(phydev, info);
if (ret)
- return ret;
+ goto out;
if (tb[ETHTOOL_A_PSE_PRIO]) {
unsigned int prio;
@@ -291,7 +296,7 @@ ethnl_set_pse(struct ethnl_req_info *req_info, struct genl_info *info)
prio = nla_get_u32(tb[ETHTOOL_A_PSE_PRIO]);
ret = pse_ethtool_set_prio(phydev->psec, info->extack, prio);
if (ret)
- return ret;
+ goto out;
}
if (tb[ETHTOOL_A_C33_PSE_AVAIL_PW_LIMIT]) {
@@ -301,7 +306,7 @@ ethnl_set_pse(struct ethnl_req_info *req_info, struct genl_info *info)
ret = pse_ethtool_set_pw_limit(phydev->psec, info->extack,
pw_limit);
if (ret)
- return ret;
+ goto out;
}
/* These values are already validated by the ethnl_pse_set_policy */
@@ -319,10 +324,11 @@ ethnl_set_pse(struct ethnl_req_info *req_info, struct genl_info *info)
*/
ret = pse_ethtool_set_config(phydev->psec, info->extack,
&config);
- if (ret)
- return ret;
}
+out:
+ pse_phy_unlock();
+
/* Return errno or zero - PSE has no notification */
return ret;
}
--
2.43.0
^ permalink raw reply [flat|nested] 6+ messages in thread