mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Tariq Toukan <tariqt@nvidia.com>
To: Andrew Lunn <andrew+netdev@lunn.ch>,
	"David S. Miller" <davem@davemloft.net>,
	Eric Dumazet <edumazet@google.com>,
	Jakub Kicinski <kuba@kernel.org>, <netdev@vger.kernel.org>,
	Paolo Abeni <pabeni@redhat.com>
Cc: Carolina Jubran <cjubran@nvidia.com>,
	Gal Pressman <gal@nvidia.com>,
	"open list" <linux-kernel@vger.kernel.org>,
	Shahar Shitrit <shshitrit@nvidia.com>,
	Tariq Toukan <tariqt@nvidia.com>,
	Leon Romanovsky <leon@kernel.org>,
	"Mark Bloch" <mbloch@nvidia.com>,
	Saeed Mahameed <saeedm@nvidia.com>,
	Dragos Tatulea <dtatulea@nvidia.com>
Subject: [PATCH net-next 5/5] net/mlx5: nodnic: Add health monitoring
Date: Wed, 30 Sep 2026 15:57:08 +0300	[thread overview]
Message-ID: <20260930125708.144767-6-tariqt@nvidia.com> (raw)
In-Reply-To: <20260930125708.144767-1-tariqt@nvidia.com>

From: Shahar Shitrit <shshitrit@nvidia.com>

Monitor firmware health by periodically polling the health buffer from
the initialization segment.

Detect fatal device conditions, firmware health counter stalls, and
syndrome changes. Report health errors together with their severity
and mark the device as unhealthy on fatal conditions.

Start and stop health polling as part of the device probe and remove
flows.

Signed-off-by: Shahar Shitrit <shshitrit@nvidia.com>
Reviewed-by: Carolina Jubran <cjubran@nvidia.com>
Signed-off-by: Tariq Toukan <tariqt@nvidia.com>
---
 .../ethernet/mellanox/mlx5/nodnic/Makefile    |   2 +-
 .../net/ethernet/mellanox/mlx5/nodnic/en.h    |  12 +
 .../ethernet/mellanox/mlx5/nodnic/health.c    | 315 ++++++++++++++++++
 .../net/ethernet/mellanox/mlx5/nodnic/main.c  |   2 +
 4 files changed, 330 insertions(+), 1 deletion(-)
 create mode 100644 drivers/net/ethernet/mellanox/mlx5/nodnic/health.c

diff --git a/drivers/net/ethernet/mellanox/mlx5/nodnic/Makefile b/drivers/net/ethernet/mellanox/mlx5/nodnic/Makefile
index 25865d7244d8..1d09edeaf3af 100644
--- a/drivers/net/ethernet/mellanox/mlx5/nodnic/Makefile
+++ b/drivers/net/ethernet/mellanox/mlx5/nodnic/Makefile
@@ -5,4 +5,4 @@
 
 obj-$(CONFIG_MLX5_NODNIC) += mlx5_nodnic.o
 
-mlx5_nodnic-y := main.o en.o en_cfg.o tx.o rx.o nodnic_pci_vsc.o
+mlx5_nodnic-y := main.o en.o en_cfg.o tx.o rx.o nodnic_pci_vsc.o health.o
\ No newline at end of file
diff --git a/drivers/net/ethernet/mellanox/mlx5/nodnic/en.h b/drivers/net/ethernet/mellanox/mlx5/nodnic/en.h
index d811bfe242ac..90e9f273d454 100644
--- a/drivers/net/ethernet/mellanox/mlx5/nodnic/en.h
+++ b/drivers/net/ethernet/mellanox/mlx5/nodnic/en.h
@@ -113,6 +113,16 @@ struct mlx5_nodnic_working_buffer {
 	u32 alloc_size;
 };
 
+struct mlx5_nodnic_health {
+	struct health_buffer __iomem   *health;
+	__be32 __iomem		       *health_counter;
+	struct timer_list		timer;
+	u32				prev_counter;
+	int				miss_counter;
+	u8				synd;
+	u32				fatal_error;
+};
+
 struct mlx5_nodnic_priv {
 	struct mlx5_nodnic_core_dev *core_dev;
 	struct net_device *netdev;
@@ -128,6 +138,8 @@ struct mlx5_nodnic_priv {
 	struct mlx5_nodnic_dbr	dbr; /* one shared dbr for TX and RX */
 	struct mlx5_nodnic_cq	cq;
 	struct mlx5_nodnic_ring	working_buffer;
+
+	struct mlx5_nodnic_health health;
 };
 
 void mlx5_nodnic_build_netdev(struct net_device *netdev);
diff --git a/drivers/net/ethernet/mellanox/mlx5/nodnic/health.c b/drivers/net/ethernet/mellanox/mlx5/nodnic/health.c
new file mode 100644
index 000000000000..a981a5195c7e
--- /dev/null
+++ b/drivers/net/ethernet/mellanox/mlx5/nodnic/health.c
@@ -0,0 +1,315 @@
+// SPDX-License-Identifier: GPL-2.0 OR Linux-OpenIB
+// Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+
+#include <linux/kernel.h>
+#include <linux/random.h>
+#include <linux/timer.h>
+#include <linux/types.h>
+
+#include "en.h"
+#include "nodnic_ifc.h"
+#include "nodnic.h"
+
+#define MLX5_NODNIC_HEALTH_POLL_INTERVAL_MS 2000
+#define MLX5_NODNIC_MAX_MISSES	3
+
+enum nodnic_rfr_severity_bit_offsets {
+	NODNIC_CRR_BIT_OFFSET = 0x6,
+	NODNIC_RFR_BIT_OFFSET = 0x7,
+};
+
+enum {
+	MLX5_NODNIC_SEVERITY_MASK	= 0x7,
+	MLX5_NODNIC_SEVERITY_VALID_MASK	= 0x8,
+};
+
+enum  {
+	MLX5_NODNIC_SENSOR_NO_ERR,
+	MLX5_NODNIC_SENSOR_PCI_COMM_ERR,
+	MLX5_NODNIC_SENSOR_PCI_ERR,
+	MLX5_NODNIC_SENSOR_NIC_DISABLED,
+	MLX5_NODNIC_SENSOR_NIC_SW_RESET,
+	MLX5_NODNIC_SENSOR_FW_SYND_RFR,
+};
+
+static unsigned long get_next_poll_jiffies(void)
+{
+	return jiffies + msecs_to_jiffies(MLX5_NODNIC_HEALTH_POLL_INTERVAL_MS) +
+	       get_random_u32_below(HZ);
+}
+
+static bool sensor_pci_not_working(struct mlx5_nodnic_health *health)
+{
+	struct health_buffer __iomem *h = health->health;
+
+	return (ioread32be(&h->fw_ver) == 0xffffffff);
+}
+
+static u8 mlx5_nodnic_get_nic_state(struct mlx5_nodnic_core_dev *dev)
+{
+	return (ioread32be(&dev->iseg->cmdq_addr_l_sz) >> 8) & 7;
+}
+
+static int mlx5_nodnic_health_get_rfr(u8 rfr_severity)
+{
+	return rfr_severity >> NODNIC_RFR_BIT_OFFSET;
+}
+
+static bool sensor_fw_synd_rfr(struct mlx5_nodnic_core_dev *dev)
+{
+	struct mlx5_nodnic_health *health = &dev->priv->health;
+	struct health_buffer __iomem *h = health->health;
+	u8 synd = ioread8(&h->synd);
+	u8 rfr;
+
+	rfr = mlx5_nodnic_health_get_rfr(ioread8(&h->rfr_severity));
+
+	if (rfr && synd)
+		mlx5_nodnic_core_dbg(dev, "FW requests reset, synd: %d\n",
+				     synd);
+
+	return rfr && synd;
+}
+
+static
+u32 mlx5_nodnic_health_check_fatal_sensors(struct mlx5_nodnic_core_dev *dev)
+{
+	u32 nic_state;
+
+	if (pci_channel_offline(dev->pdev))
+		return MLX5_NODNIC_SENSOR_PCI_ERR;
+	if (sensor_pci_not_working(&dev->priv->health))
+		return MLX5_NODNIC_SENSOR_PCI_COMM_ERR;
+
+	nic_state = mlx5_nodnic_get_nic_state(dev);
+	if (nic_state == MLX5_NODNIC_ISEG_NIC_INTERFACE_DISABLED)
+		return MLX5_NODNIC_SENSOR_NIC_DISABLED;
+	if (nic_state == MLX5_NODNIC_ISEG_NIC_INTERFACE_SW_RESET)
+		return MLX5_NODNIC_SENSOR_NIC_SW_RESET;
+
+	if (sensor_fw_synd_rfr(dev))
+		return MLX5_NODNIC_SENSOR_FW_SYND_RFR;
+
+	return MLX5_NODNIC_SENSOR_NO_ERR;
+}
+
+static const char *mlx5_nodnic_fatal_error_str(u32 err)
+{
+	switch (err) {
+	case MLX5_NODNIC_SENSOR_PCI_COMM_ERR: return "PCI communication error";
+	case MLX5_NODNIC_SENSOR_PCI_ERR:      return "PCI error";
+	case MLX5_NODNIC_SENSOR_NIC_DISABLED: return "NIC disabled";
+	case MLX5_NODNIC_SENSOR_NIC_SW_RESET: return "NIC SW reset";
+	case MLX5_NODNIC_SENSOR_FW_SYND_RFR:  return "FW synd RFR";
+	default:                       return "unknown";
+	}
+}
+
+static int mlx5_nodnic_health_get_crr(u8 rfr_severity)
+{
+	return (rfr_severity >> NODNIC_CRR_BIT_OFFSET) & 0x01;
+}
+
+static int mlx5_nodnic_health_get_severity(u8 rfr_severity)
+{
+	return rfr_severity & MLX5_NODNIC_SEVERITY_VALID_MASK ?
+	       rfr_severity & MLX5_NODNIC_SEVERITY_MASK : LOGLEVEL_ERR;
+}
+
+/* Syndrome values from health buffer (same as mlx5_ifc.h) */
+static const char *mlx5_nodnic_hsynd_str(u8 synd)
+{
+	switch (synd) {
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_FW_INTERNAL_ERR:
+		return "firmware internal error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_DEAD_IRISC:
+		return "irisc not responding";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_HW_FATAL_ERR:
+		return "unrecoverable hardware error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_FW_CRC_ERR:
+		return "firmware CRC error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_ICM_FETCH_PCI_ERR:
+		return "ICM fetch PCI error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_ICM_PAGE_ERR:
+		return "HW fatal error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_ASYNCHRONOUS_EQ_BUF_OVERRUN:
+		return "async EQ buffer overrun";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_EQ_IN_ERR:
+		return "EQ error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_EQ_INV:
+		return "Invalid EQ referenced";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_FFSER_ERR:
+		return "FFSER error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_HIGH_TEMP_ERR:
+		return "High temperature";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_ICM_PCI_POISONED_ERR:
+		return "ICM fetch PCI data poisoned error";
+	case MLX5_NODNIC_ISEG_HEALTH_SYNDROME_TRUST_LOCKDOWN_ERR:
+		return "Trust lockdown error";
+	default:
+		return "unrecognized error";
+	}
+}
+
+static const char *mlx5_nodnic_loglevel_str(int level)
+{
+	switch (level) {
+	case LOGLEVEL_EMERG:
+		return "EMERGENCY";
+	case LOGLEVEL_ALERT:
+		return "ALERT";
+	case LOGLEVEL_CRIT:
+		return "CRITICAL";
+	case LOGLEVEL_ERR:
+		return "ERROR";
+	case LOGLEVEL_WARNING:
+		return "WARNING";
+	case LOGLEVEL_NOTICE:
+		return "NOTICE";
+	case LOGLEVEL_INFO:
+		return "INFO";
+	case LOGLEVEL_DEBUG:
+		return "DEBUG";
+	}
+	return "Unknown log level";
+}
+
+static const char *mlx5_nodnic_severity_to_kern(int severity)
+{
+	if (severity <= LOGLEVEL_ERR)
+		return KERN_ERR;
+
+	if (severity <= LOGLEVEL_WARNING)
+		return KERN_WARNING;
+
+	return KERN_INFO;
+}
+
+static void mlx5_nodnic_print_health_info(struct mlx5_nodnic_core_dev *dev)
+{
+	struct mlx5_nodnic_health *health = &dev->priv->health;
+	struct health_buffer __iomem *h = health->health;
+	struct device *device = dev->device;
+	u8 synd = ioread8(&h->synd);
+	const char *kern_severity;
+	u8 rfr_severity;
+	int severity;
+	int i;
+
+	if (!synd)
+		return;
+
+	if (sensor_pci_not_working(&dev->priv->health)) {
+		mlx5_nodnic_core_err(dev, "PCI slot is unavailable\n");
+		return;
+	}
+
+	rfr_severity = ioread8(&h->rfr_severity);
+	severity = mlx5_nodnic_health_get_severity(rfr_severity);
+	kern_severity = mlx5_nodnic_severity_to_kern(severity);
+
+	dev_printk(kern_severity, device,
+		   "Health issue observed, %s, severity(%d) %s:\n",
+		   mlx5_nodnic_hsynd_str(synd), severity,
+		   mlx5_nodnic_loglevel_str(severity));
+
+	for (i = 0; i < ARRAY_SIZE(h->assert_var); i++)
+		dev_printk(kern_severity, device, "assert_var[%d] 0x%08x\n", i,
+			   ioread32be(h->assert_var + i));
+
+	dev_printk(kern_severity, device, "assert_exit_ptr 0x%08x\n",
+		   ioread32be(&h->assert_exit_ptr));
+	dev_printk(kern_severity, device, "assert_callra 0x%08x\n",
+		   ioread32be(&h->assert_callra));
+	dev_printk(kern_severity, device, "raw fw_ver 0x%08x\n",
+		   ioread32be(&h->fw_ver));
+	dev_printk(kern_severity, device, "time %u\n", ioread32be(&h->time));
+	dev_printk(kern_severity, device, "hw_id 0x%08x\n",
+		   ioread32be(&h->hw_id));
+	dev_printk(kern_severity, device, "rfr %d\n",
+		   mlx5_nodnic_health_get_rfr(rfr_severity));
+	dev_printk(kern_severity, device, "crr %d\n",
+		   mlx5_nodnic_health_get_crr(rfr_severity));
+	dev_printk(kern_severity, device, "severity %d (%s)\n", severity,
+		   mlx5_nodnic_loglevel_str(severity));
+	dev_printk(kern_severity, device, "irisc_index %d\n",
+		   ioread8(&h->irisc_index));
+	dev_printk(kern_severity, device, "synd 0x%x: %s\n", synd,
+		   mlx5_nodnic_hsynd_str(synd));
+	dev_printk(kern_severity, device, "ext_synd 0x%04x\n",
+		   ioread16be(&h->ext_synd));
+
+	if (mlx5_nodnic_health_get_crr(rfr_severity))
+		mlx5_nodnic_core_warn(dev, "Cold reset is required\n");
+}
+
+static void mlx5_nodnic_poll_health(struct timer_list *t)
+{
+	struct mlx5_nodnic_priv *priv = timer_container_of(priv, t,
+							   health.timer);
+	struct mlx5_nodnic_core_dev *dev = priv->core_dev;
+	struct mlx5_nodnic_health *health = &priv->health;
+	struct health_buffer __iomem *h = health->health;
+	u32 fatal_error, count;
+	u8 prev_synd;
+
+	if (dev->state == NODNIC_DEVICE_STATE_INTERNAL_ERROR)
+		return;
+
+	fatal_error = mlx5_nodnic_health_check_fatal_sensors(dev);
+
+	if (fatal_error && !health->fatal_error) {
+		mlx5_nodnic_core_err(dev, "fatal error detected: %s (%u)\n",
+				     mlx5_nodnic_fatal_error_str(fatal_error),
+				     fatal_error);
+		priv->health.fatal_error = fatal_error;
+		dev->state = NODNIC_DEVICE_STATE_INTERNAL_ERROR;
+		mlx5_nodnic_print_health_info(dev);
+		return;
+	}
+
+	count = ioread32be(health->health_counter);
+	if (count == health->prev_counter)
+		++health->miss_counter;
+	else
+		health->miss_counter = 0;
+
+	health->prev_counter = count;
+	if (health->miss_counter == MLX5_NODNIC_MAX_MISSES) {
+		mlx5_nodnic_core_err(dev,
+				     "device's health compromised - reached miss count\n");
+		health->synd = ioread8(&h->synd);
+		mlx5_nodnic_print_health_info(dev);
+	}
+
+	prev_synd = health->synd;
+	health->synd = ioread8(&h->synd);
+	if (health->synd && health->synd != prev_synd)
+		mlx5_nodnic_print_health_info(dev);
+
+	mod_timer(&health->timer, get_next_poll_jiffies());
+}
+
+void mlx5_nodnic_start_health_poll(struct mlx5_nodnic_core_dev *dev)
+{
+	struct mlx5_nodnic_health *health = &dev->priv->health;
+
+	timer_setup(&health->timer, mlx5_nodnic_poll_health, 0);
+	health->fatal_error = MLX5_NODNIC_SENSOR_NO_ERR;
+	health->miss_counter = 0;
+	health->prev_counter = 0;
+	health->synd = 0;
+	health->health = &dev->iseg->health;
+	health->health_counter = &dev->iseg->health_counter;
+
+	health->timer.expires = jiffies +
+			msecs_to_jiffies(MLX5_NODNIC_HEALTH_POLL_INTERVAL_MS);
+	add_timer(&health->timer);
+}
+
+void mlx5_nodnic_stop_health_poll(struct mlx5_nodnic_core_dev *dev)
+{
+	struct mlx5_nodnic_health *health = &dev->priv->health;
+
+	timer_delete_sync(&health->timer);
+}
diff --git a/drivers/net/ethernet/mellanox/mlx5/nodnic/main.c b/drivers/net/ethernet/mellanox/mlx5/nodnic/main.c
index d77d9dbe3c89..d6720c72905f 100644
--- a/drivers/net/ethernet/mellanox/mlx5/nodnic/main.c
+++ b/drivers/net/ethernet/mellanox/mlx5/nodnic/main.c
@@ -814,6 +814,7 @@ static int mlx5_nodnic_probe_one(struct pci_dev *pdev,
 	dev->netdev_registered = true;
 
 	dev->state = NODNIC_DEVICE_STATE_UP;
+	mlx5_nodnic_start_health_poll(dev);
 
 	pci_save_state(pdev);
 
@@ -839,6 +840,7 @@ static void mlx5_nodnic_remove_one(struct pci_dev *pdev)
 	if (!dev)
 		return;
 
+	mlx5_nodnic_stop_health_poll(dev);
 	mlx5_nodnic_event_irq_cleanup(dev);
 	mlx5_nodnic_cleanup_netdev(dev);
 	mlx5_nodnic_dev_cleanup(dev);
-- 
2.44.0


      parent reply	other threads:[~2026-09-30 12:58 UTC|newest]

Thread overview: 6+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-30 12:57 [PATCH net-next 0/5] net/mlx5: Introduce mlx5_nodnic, a driver for DRAM-less NVIDIA Mellanox NIC functions Tariq Toukan
2026-09-30 12:57 ` [PATCH net-next 1/5] net/mlx5: nodnic: Introduce the mlx5_nodnic driver Tariq Toukan
2026-09-30 12:57 ` [PATCH net-next 2/5] net/mlx5: nodnic: Implement data path Tariq Toukan
2026-09-30 12:57 ` [PATCH net-next 3/5] net/mlx5: nodnic: Add TX/RX statistics Tariq Toukan
2026-09-30 12:57 ` [PATCH net-next 4/5] net/mlx5: nodnic: Add ethtool driver info Tariq Toukan
2026-09-30 12:57 ` Tariq Toukan [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260930125708.144767-6-tariqt@nvidia.com \
    --to=tariqt@nvidia.com \
    --cc=andrew+netdev@lunn.ch \
    --cc=cjubran@nvidia.com \
    --cc=davem@davemloft.net \
    --cc=dtatulea@nvidia.com \
    --cc=edumazet@google.com \
    --cc=gal@nvidia.com \
    --cc=kuba@kernel.org \
    --cc=leon@kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=mbloch@nvidia.com \
    --cc=netdev@vger.kernel.org \
    --cc=pabeni@redhat.com \
    --cc=saeedm@nvidia.com \
    --cc=shshitrit@nvidia.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®