From 3799dc5d778552e20bc5894d80df54b30fe764c8 Mon Sep 17 00:00:00 2001 From: Raag Jadav Date: Thu, 30 Jul 2026 16:36:34 +0530 Subject: [PATCH] drm/xe/ras: Fix boot-time ras error processing Currently, we xe_ras_process_errors() inside xe_ras_init() to handle boot time errors. But this can potentially result in declaring the device as wedged quite early in the driver load sequence, which is problematic due to the lack of registered drm device or required wedged cleanup hooks at this point. Call xe_ras_process_errors() only after the prerequisites are available. Fixes: d9732e498f5f ("drm/xe/xe_ras: Query errors from system controller on probe") Signed-off-by: Raag Jadav Reviewed-by: Rodrigo Vivi Tested-by: Mallesh Koujalagi Link: https://patch.msgid.link/20260730110635.925537-1-raag.jadav@intel.com Signed-off-by: Riana Tauro (cherry picked from commit 20bc4883c7c0e28c3ba6c76ccc279486c349dd3e) Signed-off-by: Rodrigo Vivi --- drivers/gpu/drm/xe/xe_device.c | 6 ++++++ drivers/gpu/drm/xe/xe_ras.c | 6 ------ 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/drivers/gpu/drm/xe/xe_device.c b/drivers/gpu/drm/xe/xe_device.c index 7007b6113760..d25d02b24898 100644 --- a/drivers/gpu/drm/xe/xe_device.c +++ b/drivers/gpu/drm/xe/xe_device.c @@ -1141,6 +1141,12 @@ int xe_device_probe(struct xe_device *xe) if (err) goto err_unregister_display; + /* + * Process and log any errors detected by hardware. Possible results can + * include declaring the device as wedged, which must be done only after + * xe_device_wedged_fini() is registered. + */ + xe_ras_process_errors(xe); return devm_add_action_or_reset(xe->drm.dev, xe_device_sanitize, xe); err_unregister_display: diff --git a/drivers/gpu/drm/xe/xe_ras.c b/drivers/gpu/drm/xe/xe_ras.c index a31e06b8aa67..c8d40289b4fe 100644 --- a/drivers/gpu/drm/xe/xe_ras.c +++ b/drivers/gpu/drm/xe/xe_ras.c @@ -692,12 +692,6 @@ void xe_ras_init(struct xe_device *xe) if (IS_ENABLED(CONFIG_PCIEAER)) ras_usp_aer_init(xe); - /* - * During probe, process and log any errors detected by firmware while the driver was not - * loaded. Critical errors such as Punit and CSC are reported through Pcode init failure, - * causing the driver to enter survivability mode. - */ - xe_ras_process_errors(xe); ret = devm_device_add_group(xe->drm.dev, &gpu_health_group); if (ret) xe_err(xe, "Failed to create GPU health sysfs, err=%d\n", ret);