drm/amdgpu: clamp the isolation index for rings outside a partition
authorXiang Liu <xiang.liu@amd.com>
Fri, 21 Aug 2026 09:41:57 +0000 (17:41 +0800)
committerAlex Deucher <alexander.deucher@amd.com>
Tue, 25 Aug 2026 22:19:30 +0000 (18:19 -0400)
adev->isolation[] has one slot per partition, but a ring that is not
assigned to one keeps AMDGPU_XCP_NO_PARTITION, which is ~0, so indexing
the array with it is out of bounds. SDMA submissions hit this on both
the isolation enforcement and the VM flush path and trip UBSAN.

Fall back to the first slot the way the cleaner shader path already
does, and stop taking the address before the ring type check that makes
it relevant.

Cc: stable@vger.kernel.org
Signed-off-by: Xiang Liu <xiang.liu@amd.com>
Reviewed-by: Hawking Zhang <Hawking.Zhang@amd.com>
Signed-off-by: Alex Deucher <alexander.deucher@amd.com>
drivers/gpu/drm/amd/amdgpu/amdgpu_device.c
drivers/gpu/drm/amd/amdgpu/amdgpu_vm.c

index 0195815..44bed0b 100644 (file)
@@ -6687,8 +6687,8 @@ struct dma_fence *amdgpu_device_enforce_isolation(struct amdgpu_device *adev,
                                                  struct amdgpu_ring *ring,
                                                  struct amdgpu_job *job)
 {
-       struct amdgpu_isolation *isolation = &adev->isolation[ring->xcp_id];
        struct drm_sched_fence *f = job->base.s_fence;
+       struct amdgpu_isolation *isolation;
        struct dma_fence *dep;
        void *owner;
        int r;
@@ -6701,6 +6701,9 @@ struct dma_fence *amdgpu_device_enforce_isolation(struct amdgpu_device *adev,
            ring->funcs->type != AMDGPU_RING_TYPE_COMPUTE)
                return NULL;
 
+       isolation = &adev->isolation[ring->xcp_id == AMDGPU_XCP_NO_PARTITION ?
+                                    0 : ring->xcp_id];
+
        /*
         * All submissions where enforce isolation is false are handled as if
         * they come from a single client. Use ~0l as the owner to distinct it
index a3758c6..aedf72c 100644 (file)
@@ -776,7 +776,9 @@ void amdgpu_vm_flush(struct amdgpu_ring *ring, struct amdgpu_job *job,
                     bool *emit_gds_needed)
 {
        struct amdgpu_device *adev = ring->adev;
-       struct amdgpu_isolation *isolation = &adev->isolation[ring->xcp_id];
+       struct amdgpu_isolation *isolation =
+               &adev->isolation[ring->xcp_id == AMDGPU_XCP_NO_PARTITION ?
+                                0 : ring->xcp_id];
        unsigned vmhub = ring->vm_hub;
        struct amdgpu_vmid_mgr *id_mgr = &adev->vm_manager.id_mgr[vmhub];
        struct amdgpu_vmid *id = &id_mgr->ids[job->vmid];