mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git
synced 2026-08-30 12:13:51 -04:00
drm/amdgpu/userq: serialize queue map against GPU reset
Creating a user queue can race with a GPU reset. While recovery holds reset_domain->sem for write, MES is unresponsive, so the ADD_QUEUE from amdgpu_userq_map_helper() times out (-110) and an otherwise valid queue create fails: amdgpu: MES(0) failed to respond to msg=ADD_QUEUE [drm:mes_userq_map [amdgpu]] *ERROR* Failed to map queue in HW, err (-110) amdgpu: [drm] *ERROR* ... Failed to map Queue amdgpu: [drm] *ERROR* ... Failed to create usermode queue Take reset_domain->sem for read around the map so it runs only once MES is back up. This mirrors amdgpu_userq_cleanup() and honors the userq_mutex -> reset_domain->sem order; the reset path never takes userq_mutex, so there is no deadlock. Reviewed-by: Alex Deucher <alexander.deucher@amd.com> Signed-off-by: Jesse Zhang <Jesse.Zhang@amd.com> Signed-off-by: Alex Deucher <alexander.deucher@amd.com>
This commit is contained in:
committed by
Alex Deucher
parent
8a28e1519b
commit
a8e151fe62
@@ -745,7 +745,12 @@ amdgpu_userq_create(struct drm_file *filp, union drm_amdgpu_userq *args)
|
||||
if (!adev->userq_halt_for_enforce_isolation ||
|
||||
((queue->queue_type != AMDGPU_HW_IP_GFX) &&
|
||||
(queue->queue_type != AMDGPU_HW_IP_COMPUTE))) {
|
||||
/* Serialize the map against an in-progress GPU reset (MES is
|
||||
* unresponsive during recovery), matching amdgpu_userq_cleanup().
|
||||
*/
|
||||
down_read(&adev->reset_domain->sem);
|
||||
r = amdgpu_userq_map_helper(queue);
|
||||
up_read(&adev->reset_domain->sem);
|
||||
if (r) {
|
||||
drm_file_err(uq_mgr->file, "Failed to map Queue\n");
|
||||
trace_amdgpu_userq_create_end(queue, r);
|
||||
|
||||
Reference in New Issue
Block a user