drm/amdgpu: Add kernel parameter support for ignoring bad page threshold

author Kent Russell <kent.russell@amd.com>

Tue, 19 Oct 2021 14:05:07 +0000 (10:05 -0400)

committer Alex Deucher <alexander.deucher@amd.com>

Thu, 28 Oct 2021 18:26:12 +0000 (14:26 -0400)
author Kent Russell <kent.russell@amd.com>
Tue, 19 Oct 2021 14:05:07 +0000 (10:05 -0400)
committer Alex Deucher <alexander.deucher@amd.com>
Thu, 28 Oct 2021 18:26:12 +0000 (14:26 -0400)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu.h b/drivers/gpu/drm/amd/amdgpu/amdgpu.h

index d58e37f..b85b67a 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu.h
@@ -205,6 +205,7 @@ extern struct amdgpu_mgpu_info mgpu_info;
  extern int amdgpu_ras_enable;
  extern uint amdgpu_ras_mask;
  extern int amdgpu_bad_page_threshold;
+extern bool amdgpu_ignore_bad_page_threshold;
  extern struct amdgpu_watchdog_timer amdgpu_watchdog_timer;
  extern int amdgpu_async_gfx_ring;
  extern int amdgpu_mcbp;
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c

index c718fb5..4cefe86 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c
@@ -877,7 +877,7 @@ module_param_named(reset_method, amdgpu_reset_method, int, 0444);
   * result in the GPU entering bad status when the number of total
   * faulty pages by ECC exceeds the threshold value.
   */
-MODULE_PARM_DESC(bad_page_threshold, "Bad page threshold(-1 = auto(default value), 0 = disable bad page retirement)");
+MODULE_PARM_DESC(bad_page_threshold, "Bad page threshold(-1 = auto(default value), 0 = disable bad page retirement, -2 = ignore bad page threshold)");
  module_param_named(bad_page_threshold, amdgpu_bad_page_threshold, int, 0444);
  
  MODULE_PARM_DESC(num_kcq, "number of kernel compute queue user want to setup (8 if set to greater than 8 or less than 0, only affect gfx 8+)");
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c

index 3978152..05117ed 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c
@@ -1105,11 +1105,18 @@ int amdgpu_ras_eeprom_init(struct amdgpu_ras_eeprom_control *control,
                         res = amdgpu_ras_eeprom_correct_header_tag(control,
                                                                    RAS_TABLE_HDR_VAL);
                 } else {
-                       *exceed_err_limit = true;
-                       dev_err(adev->dev,
-                               "RAS records:%d exceed threshold:%d, "
-                               "GPU will not be initialized. Replace this GPU or increase the threshold",
+                       dev_err(adev->dev, "RAS records:%d exceed threshold:%d",
                                 control->ras_num_recs, ras->bad_page_cnt_threshold);
+                       if (amdgpu_bad_page_threshold == -2) {
+                               dev_warn(adev->dev, "GPU will be initialized due to bad_page_threshold = -2.");
+                               res = 0;
+                       } else {
+                               *exceed_err_limit = true;
+                               dev_err(adev->dev,
+                                       "RAS records:%d exceed threshold:%d, "
+                                       "GPU will not be initialized. Replace this GPU or increase the threshold",
+                                       control->ras_num_recs, ras->bad_page_cnt_threshold);
+                       }
                 }
         } else {
                 DRM_INFO("Creating a new EEPROM table");
author	Kent Russell <kent.russell@amd.com>
	Tue, 19 Oct 2021 14:05:07 +0000 (10:05 -0400)
committer	Alex Deucher <alexander.deucher@amd.com>
	Thu, 28 Oct 2021 18:26:12 +0000 (14:26 -0400)
drivers/gpu/drm/amd/amdgpu/amdgpu.h		patch \| blob \| history
drivers/gpu/drm/amd/amdgpu/amdgpu_drv.c		patch \| blob \| history
drivers/gpu/drm/amd/amdgpu/amdgpu_ras_eeprom.c		patch \| blob \| history