[PATCH v19 09/14] cxl/pci: Thread port and dport through RAS handling helpers

From: Terry Bowman

Date: Mon Aug 03 2026 - 18:22:58 EST


From: Dan Williams <djbw@xxxxxxxxxx>

The callers of cxl_handle_ras() and cxl_handle_cor_ras() already hold
a struct cxl_port * and optionally a struct cxl_dport * for the device
being handled. Passing a generic struct device * requires is_cxl_memdev()
to distinguish Endpoints from Ports at trace emission time. Threading
Port and Downstream Port directly enables is_cxl_endpoint() and explicit
dport/port branching for cleaner trace dispatch.

Refactor cxl_handle_ras() and cxl_handle_cor_ras() to accept struct
cxl_port * and struct cxl_dport * directly. The CXL RAS trace event
emission logic is split into three branches: Endpoint events are
identified via is_cxl_endpoint() and emit with the memdev, dport events
emit with dport->dport_dev, and Upstream Port events fall back to
port->uport_dev. This branching is transitional: the follow-on patch
("cxl: Add port and dport identifiers to CXL AER trace events") unifies
the trace events on port/dport and removes it.

Update cxl_handle_rdport_errors() and cxl_handle_proto_error() to pass
Port and Downstream Port to the refactored functions.

RCH Downstream Port correctable trace events now report the dport device
(dport->dport_dev) as a consequence of threading Port and Downstream
Port through the RAS helpers. The following trace event rework ("cxl: Add
port and dport identifiers to CXL AER trace events") adds explicit memdev,
Port, Downstream Port, and host fields that provide full context for all
device types.

Co-developed-by: Terry Bowman <terry.bowman@xxxxxxx>
Signed-off-by: Terry Bowman <terry.bowman@xxxxxxx>
Signed-off-by: Dan Williams <djbw@xxxxxxxxxx>
Reviewed-by: Dave Jiang <dave.jiang@xxxxxxxxx>

---

Changes in v18 -> v19:
- Added review-by for DaveJ
- Update commit message that three-way trace branching is transitional
and removed by the following trace-event unification patch

Changes in v17 -> v18:
- New patch.
---
drivers/cxl/core/core.h | 12 ++++++++----
drivers/cxl/core/ras.c | 29 +++++++++++++++--------------
drivers/cxl/core/ras_rch.c | 2 +-
3 files changed, 24 insertions(+), 19 deletions(-)

diff --git a/drivers/cxl/core/core.h b/drivers/cxl/core/core.h
index 272634ff2615b..5ca1275fd8f35 100644
--- a/drivers/cxl/core/core.h
+++ b/drivers/cxl/core/core.h
@@ -185,10 +185,12 @@ static inline struct device *dport_to_host(struct cxl_dport *dport)
#ifdef CONFIG_CXL_RAS
void cxl_ras_init(void);
void cxl_ras_exit(void);
-bool cxl_handle_ras(struct device *dev, void __iomem *ras_base);
+bool cxl_handle_ras(struct cxl_port *port, struct cxl_dport *dport,
+ void __iomem *ras_base);
void cxl_do_recovery(struct pci_dev *pdev, struct cxl_port *port,
struct cxl_dport *dport);
-void cxl_handle_cor_ras(struct device *dev, void __iomem *ras_base);
+void cxl_handle_cor_ras(struct cxl_port *port, struct cxl_dport *dport,
+ void __iomem *ras_base);
void cxl_dport_map_rch_aer(struct cxl_dport *dport);
void cxl_disable_rch_root_ints(struct cxl_dport *dport);
void cxl_handle_rdport_errors(struct pci_dev *pdev);
@@ -197,13 +199,15 @@ void devm_cxl_dport_ras_setup(struct cxl_dport *dport);
#else
static inline void cxl_ras_init(void) { }
static inline void cxl_ras_exit(void) { }
-static inline bool cxl_handle_ras(struct device *dev, void __iomem *ras_base)
+static inline bool cxl_handle_ras(struct cxl_port *port, struct cxl_dport *dport,
+ void __iomem *ras_base)
{
return false;
}
static inline void cxl_do_recovery(struct pci_dev *pdev, struct cxl_port *port,
struct cxl_dport *dport) { }
-static inline void cxl_handle_cor_ras(struct device *dev, void __iomem *ras_base) { }
+static inline void cxl_handle_cor_ras(struct cxl_port *port, struct cxl_dport *dport,
+ void __iomem *ras_base) { }
static inline void cxl_dport_map_rch_aer(struct cxl_dport *dport) { }
static inline void cxl_disable_rch_root_ints(struct cxl_dport *dport) { }
static inline void cxl_handle_rdport_errors(struct pci_dev *pdev) { }
diff --git a/drivers/cxl/core/ras.c b/drivers/cxl/core/ras.c
index 83df544e5a656..ff4fb016ed4a1 100644
--- a/drivers/cxl/core/ras.c
+++ b/drivers/cxl/core/ras.c
@@ -227,20 +227,19 @@ void __iomem *to_ras_base(struct cxl_port *port, struct cxl_dport *dport)

void cxl_do_recovery(struct pci_dev *pdev, struct cxl_port *port, struct cxl_dport *dport)
{
- struct device *dev = dport ? dport->dport_dev : port->uport_dev;
void __iomem *ras_base = to_ras_base(port, dport);

if (!ras_base)
panic("CXL: UCE with unmapped RAS registers");

- if (cxl_handle_ras(dev, ras_base))
+ if (cxl_handle_ras(port, dport, ras_base))
panic("CXL cachemem error");

dev_dbg(&pdev->dev,
"CXL UCE signaled but no CXL RAS status bits set\n");
}

-void cxl_handle_cor_ras(struct device *dev, void __iomem *ras_base)
+void cxl_handle_cor_ras(struct cxl_port *port, struct cxl_dport *dport, void __iomem *ras_base)
{
void __iomem *addr;
u32 status;
@@ -252,10 +251,12 @@ void cxl_handle_cor_ras(struct device *dev, void __iomem *ras_base)
status = readl(addr);
if (status & CXL_RAS_CORRECTABLE_STATUS_MASK) {
writel(status & CXL_RAS_CORRECTABLE_STATUS_MASK, addr);
- if (is_cxl_memdev(dev))
- trace_cxl_aer_correctable_error(to_cxl_memdev(dev), status);
+ if (is_cxl_endpoint(port))
+ trace_cxl_aer_correctable_error(to_cxl_memdev(port->uport_dev), status);
+ else if (dport)
+ trace_cxl_port_aer_correctable_error(dport->dport_dev, status);
else
- trace_cxl_port_aer_correctable_error(dev, status);
+ trace_cxl_port_aer_correctable_error(port->uport_dev, status);
}
}

@@ -280,7 +281,7 @@ static void header_log_copy(void __iomem *ras_base, u32 *log)
* Log the state of the RAS status registers and prepare them to log the
* next error status. Return 1 if reset needed.
*/
-bool cxl_handle_ras(struct device *dev, void __iomem *ras_base)
+bool cxl_handle_ras(struct cxl_port *port, struct cxl_dport *dport, void __iomem *ras_base)
{
u32 hl[CXL_HEADERLOG_TRACE_SIZE_U32] = {};
void __iomem *addr;
@@ -307,10 +308,12 @@ bool cxl_handle_ras(struct device *dev, void __iomem *ras_base)
}

header_log_copy(ras_base, hl);
- if (is_cxl_memdev(dev))
- trace_cxl_aer_uncorrectable_error(to_cxl_memdev(dev), status, fe, hl);
+ if (is_cxl_endpoint(port))
+ trace_cxl_aer_uncorrectable_error(to_cxl_memdev(port->uport_dev), status, fe, hl);
+ else if (dport)
+ trace_cxl_port_aer_uncorrectable_error(dport->dport_dev, status, fe, hl);
else
- trace_cxl_port_aer_uncorrectable_error(dev, status, fe, hl);
+ trace_cxl_port_aer_uncorrectable_error(port->uport_dev, status, fe, hl);

writel(status & CXL_RAS_UNCORRECTABLE_STATUS_MASK, addr);

@@ -344,7 +347,7 @@ pci_ers_result_t cxl_error_detected(struct pci_dev *pdev,
* cache coherency is already lost; continuing risks silent
* data corruption.
*/
- ue = cxl_handle_ras(port->uport_dev, to_ras_base(port, NULL));
+ ue = cxl_handle_ras(port, NULL, to_ras_base(port, NULL));
}

/*
@@ -375,10 +378,8 @@ EXPORT_SYMBOL_NS_GPL(cxl_error_detected, "CXL");
static void cxl_handle_proto_error(struct pci_dev *pdev, struct cxl_port *port,
struct cxl_dport *dport, int severity)
{
- struct device *dev = dport ? dport->dport_dev : port->uport_dev;
-
if (severity == AER_CORRECTABLE)
- cxl_handle_cor_ras(dev, to_ras_base(port, dport));
+ cxl_handle_cor_ras(port, dport, to_ras_base(port, dport));
else
cxl_do_recovery(pdev, port, dport);
}
diff --git a/drivers/cxl/core/ras_rch.c b/drivers/cxl/core/ras_rch.c
index 6f95542e2e6a7..a5c62c71060d9 100644
--- a/drivers/cxl/core/ras_rch.c
+++ b/drivers/cxl/core/ras_rch.c
@@ -113,7 +113,7 @@ void cxl_handle_rdport_errors(struct pci_dev *pdev)
*/
if (aer_regs.cor_status & ~aer_regs.cor_mask) {
pci_print_aer(pdev, AER_CORRECTABLE, &aer_regs);
- cxl_handle_cor_ras(dport->dport_dev, to_ras_base(port, dport));
+ cxl_handle_cor_ras(port, dport, to_ras_base(port, dport));
}

if (aer_regs.uncor_status & ~aer_regs.uncor_mask) {
--
2.34.1