On Mon, Jul 13, 2026 at 12:34 PM Srinivasan Shanmugam
<[email protected]> wrote:
>
> Signal GPU_RESET EVENTFD subscriptions after successful GPU reset
> completion.
>
> GPU reset notifications allow userspace to observe that GPU recovery has
> completed and the device is operational again.
>
> GPU_RESET is a device-scoped event, so no queue object is used. All
> processes that subscribed to GPU_RESET on the device are notified.
>
> EVENTFD remains notification-only.
>
> v2:
> - Move GPU_RESET EVENTFD signaling to the path that confirms a real
>   GPU hardware reset occurred.
> - Notify only after successful hardware reset completion.
> - Do not signal reset-skip or recovery-only paths.
>
> Cc: Alex Deucher <[email protected]>
> Cc: Christian König <[email protected]>
> Signed-off-by: Srinivasan Shanmugam <[email protected]>

Reviewed-by: Alex Deucher <[email protected]>

> ---
>  drivers/gpu/drm/amd/amdgpu/amdgpu_device.c | 51 +++++++++++++++++++++-
>  1 file changed, 49 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c 
> b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c
> index c924a55fd16f..15aaceae13d1 100644
> --- a/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c
> +++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_device.c
> @@ -5390,15 +5390,43 @@ static void amdgpu_device_halt_activities(struct 
> amdgpu_device *adev,
>         }
>  }
>
> +/*
> + * Signal GPU_RESET EVENTFD subscriptions for all open DRM files.
> + *
> + * GPU reset is a device-wide event rather than being associated with a
> + * specific user queue. Notify every process that subscribed to the
> + * GPU_RESET event on this device.
> + */
> +static void amdgpu_device_eventfd_signal_gpu_reset(struct amdgpu_device 
> *adev)
> +{
> +       struct drm_device *ddev = adev_to_drm(adev);
> +       struct drm_file *file;
> +
> +       mutex_lock(&ddev->filelist_mutex);
> +
> +       list_for_each_entry(file, &ddev->filelist, lhead) {
> +               struct amdgpu_fpriv *fpriv = file->driver_priv;
> +
> +               if (fpriv)
> +                       amdgpu_eventfd_signal(&fpriv->eventfd_mgr,
> +                                             DRM_AMDGPU_EVENT_TYPE_GPU_RESET,
> +                                             NULL);
> +       }
> +
> +       mutex_unlock(&ddev->filelist_mutex);
> +}
> +
>  static int amdgpu_device_asic_reset(struct amdgpu_device *adev,
>                               struct list_head *device_list,
>                               struct amdgpu_reset_context *reset_context)
>  {
>         struct amdgpu_device *tmp_adev = NULL;
> +       bool need_full_reset, skip_hw_reset;
>         int retry_limit = AMDGPU_MAX_RETRY_LIMIT;
>         int r = 0;
>
> -retry: /* Rest of adevs pre asic reset from XGMI hive. */
> +retry:
> +       /* Rest of adevs pre asic reset from XGMI hive. */
>         list_for_each_entry(tmp_adev, device_list, reset_list) {
>                 r = amdgpu_device_pre_asic_reset(tmp_adev, reset_context);
>                 /*TODO Should we stop ?*/
> @@ -5430,10 +5458,30 @@ static int amdgpu_device_asic_reset(struct 
> amdgpu_device *adev,
>                 }
>                 if (r)
>                         adev->asic_reset_res = r;
> +               else
> +                       amdgpu_device_eventfd_signal_gpu_reset(adev);
>         } else {
> +               /*
> +                * Save the reset decision before amdgpu_do_asic_reset(),
> +                * which may clear AMDGPU_NEED_FULL_RESET before returning.
> +                */
> +               need_full_reset =
> +                       test_bit(AMDGPU_NEED_FULL_RESET, 
> &reset_context->flags);
> +               skip_hw_reset =
> +                       test_bit(AMDGPU_SKIP_HW_RESET, &reset_context->flags);
> +
>                 r = amdgpu_do_asic_reset(device_list, reset_context);
>                 if (r && r == -EAGAIN)
>                         goto retry;
> +
> +               /*
> +                * Notify only after a real hardware reset completed.
> +                * Do not notify recovery-only or reset-skip paths.
> +                */
> +               if (!r && need_full_reset && !skip_hw_reset) {
> +                       list_for_each_entry(tmp_adev, device_list, reset_list)
> +                               
> amdgpu_device_eventfd_signal_gpu_reset(tmp_adev);
> +               }
>         }
>
>         list_for_each_entry(tmp_adev, device_list, reset_list) {
> @@ -5530,7 +5578,6 @@ static void amdgpu_device_gpu_resume(struct 
> amdgpu_device *adev,
>         }
>  }
>
> -
>  /**
>   * amdgpu_device_gpu_recover - reset the asic and recover scheduler
>   *
> --
> 2.34.1
>

Reply via email to