drm/amdkfd: restore userptr ignore bad address error
authorPhilip Yang <Philip.Yang@amd.com>
Tue, 26 Oct 2021 15:59:28 +0000 (11:59 -0400)
committerAlex Deucher <alexander.deucher@amd.com>
Thu, 28 Oct 2021 18:26:12 +0000 (14:26 -0400)
The userptr can be unmapped by application and still registered to
driver, restore userptr work return user pages will get -EFAULT bad
address error. Pretend this error as succeed. GPU access this userptr
will have VM fault later, it is better than application soft hangs with
stalled user mode queues.

v2: squash in warning fix (Alex)

Signed-off-by: Philip Yang <Philip.Yang@amd.com>
Reviewed-by: Felix Kuehling <Felix.Kuehling@amd.com>
Signed-off-by: Alex Deucher <alexander.deucher@amd.com>
drivers/gpu/drm/amd/amdgpu/amdgpu_amdkfd_gpuvm.c
drivers/gpu/drm/amd/amdgpu/amdgpu_ttm.c

index cdf46bd0d8d5bc4f4e4dd5a9892a99e23ac7b8f5..6f01c6145a87f5bd929ee9bf8d5d86c22dd7d273 100644 (file)
@@ -2041,19 +2041,26 @@ static int update_invalid_user_pages(struct amdkfd_process_info *process_info,
                /* Get updated user pages */
                ret = amdgpu_ttm_tt_get_user_pages(bo, bo->tbo.ttm->pages);
                if (ret) {
-                       pr_debug("%s: Failed to get user pages: %d\n",
-                               __func__, ret);
+                       pr_debug("Failed %d to get user pages\n", ret);
+
+                       /* Return -EFAULT bad address error as success. It will
+                        * fail later with a VM fault if the GPU tries to access
+                        * it. Better than hanging indefinitely with stalled
+                        * user mode queues.
+                        *
+                        * Return other error -EBUSY or -ENOMEM to retry restore
+                        */
+                       if (ret != -EFAULT)
+                               return ret;
+               } else {
 
-                       /* Return error -EBUSY or -ENOMEM, retry restore */
-                       return ret;
+                       /*
+                        * FIXME: Cannot ignore the return code, must hold
+                        * notifier_lock
+                        */
+                       amdgpu_ttm_tt_get_user_pages_done(bo->tbo.ttm);
                }
 
-               /*
-                * FIXME: Cannot ignore the return code, must hold
-                * notifier_lock
-                */
-               amdgpu_ttm_tt_get_user_pages_done(bo->tbo.ttm);
-
                /* Mark the BO as valid unless it was invalidated
                 * again concurrently.
                 */
index 590537b62a0a1e3e3706cdc71b7895c6873eaeb7..17b5dd6adf8c775a30e6cbb60fbd81470a582d6b 100644 (file)
@@ -696,6 +696,9 @@ int amdgpu_ttm_tt_get_user_pages(struct amdgpu_bo *bo, struct page **pages)
                                       true, NULL);
 out_unlock:
        mmap_read_unlock(mm);
+       if (r)
+               pr_debug("failed %d to get user pages 0x%lx\n", r, start);
+
        mmput(mm);
 
        return r;