drm/amdgpu: handle IH ring1 overflow

author Philip Yang <Philip.Yang@amd.com>

Thu, 18 Nov 2021 20:24:55 +0000 (15:24 -0500)

committer Alex Deucher <alexander.deucher@amd.com>

Wed, 1 Dec 2021 21:03:34 +0000 (16:03 -0500)
author Philip Yang <Philip.Yang@amd.com>
Thu, 18 Nov 2021 20:24:55 +0000 (15:24 -0500)
committer Alex Deucher <alexander.deucher@amd.com>
Wed, 1 Dec 2021 21:03:34 +0000 (16:03 -0500)
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c

index 08478fce00f2d1a40fabefe09d44500e2f5296b5..2430d6223c2d732449343a0e4c8dfccbccf06509 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c
@@ -350,6 +350,7 @@ static inline uint64_t amdgpu_gmc_fault_key(uint64_t addr, uint16_t pasid)
   * amdgpu_gmc_filter_faults - filter VM faults
   *
   * @adev: amdgpu device structure
+ * @ih: interrupt ring that the fault received from
   * @addr: address of the VM fault
   * @pasid: PASID of the process causing the fault
   * @timestamp: timestamp of the fault
@@ -358,7 +359,8 @@ static inline uint64_t amdgpu_gmc_fault_key(uint64_t addr, uint16_t pasid)
   * True if the fault was filtered and should not be processed further.
   * False if the fault is a new one and needs to be handled.
   */
-bool amdgpu_gmc_filter_faults(struct amdgpu_device *adev, uint64_t addr,
+bool amdgpu_gmc_filter_faults(struct amdgpu_device *adev,
+                             struct amdgpu_ih_ring *ih, uint64_t addr,
                               uint16_t pasid, uint64_t timestamp)
  {
         struct amdgpu_gmc *gmc = &adev->gmc;
@@ -366,6 +368,10 @@ bool amdgpu_gmc_filter_faults(struct amdgpu_device *adev, uint64_t addr,
         struct amdgpu_gmc_fault *fault;
         uint32_t hash;
  
+       /* Stale retry fault if timestamp goes backward */
+       if (amdgpu_ih_ts_after(timestamp, ih->processed_timestamp))
+               return true;
+
         /* If we don't have space left in the ring buffer return immediately */
         stamp = max(timestamp, AMDGPU_GMC_FAULT_TIMEOUT + 1) -
                 AMDGPU_GMC_FAULT_TIMEOUT;
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h

index e55201134a01f2e8c0dcdf9b7b09f73f8003bd3a..8458cebc6d5b83639431ef5dbb8a349bc9494111 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h
@@ -316,7 +316,8 @@ void amdgpu_gmc_gart_location(struct amdgpu_device *adev,
                               struct amdgpu_gmc *mc);
  void amdgpu_gmc_agp_location(struct amdgpu_device *adev,
                              struct amdgpu_gmc *mc);
-bool amdgpu_gmc_filter_faults(struct amdgpu_device *adev, uint64_t addr,
+bool amdgpu_gmc_filter_faults(struct amdgpu_device *adev,
+                             struct amdgpu_ih_ring *ih, uint64_t addr,
                               uint16_t pasid, uint64_t timestamp);
  void amdgpu_gmc_filter_faults_remove(struct amdgpu_device *adev, uint64_t addr,
                                      uint16_t pasid);
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.c

index 0c7963dfacad1e4ce458ae6993b3d5fdeafe0091..8050f7ba93ad070723d95335dbeaf1d270595da0 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.c
@@ -164,52 +164,32 @@ void amdgpu_ih_ring_write(struct amdgpu_ih_ring *ih, const uint32_t *iv,
         }
  }
  
-/* Waiter helper that checks current rptr matches or passes checkpoint wptr */
-static bool amdgpu_ih_has_checkpoint_processed(struct amdgpu_device *adev,
-                                       struct amdgpu_ih_ring *ih,
-                                       uint32_t checkpoint_wptr,
-                                       uint32_t *prev_rptr)
-{
-       uint32_t cur_rptr = ih->rptr | (*prev_rptr & ~ih->ptr_mask);
-
-       /* rptr has wrapped. */
-       if (cur_rptr < *prev_rptr)
-               cur_rptr += ih->ptr_mask + 1;
-       *prev_rptr = cur_rptr;
-
-       /* check ring is empty to workaround missing wptr overflow flag */
-       return cur_rptr >= checkpoint_wptr ||
-              (cur_rptr & ih->ptr_mask) == amdgpu_ih_get_wptr(adev, ih);
-}
-
  /**
- * amdgpu_ih_wait_on_checkpoint_process - wait to process IVs up to checkpoint
+ * amdgpu_ih_wait_on_checkpoint_process_ts - wait to process IVs up to checkpoint
   *
   * @adev: amdgpu_device pointer
   * @ih: ih ring to process
   *
   * Used to ensure ring has processed IVs up to the checkpoint write pointer.
   */
-int amdgpu_ih_wait_on_checkpoint_process(struct amdgpu_device *adev,
+int amdgpu_ih_wait_on_checkpoint_process_ts(struct amdgpu_device *adev,
                                         struct amdgpu_ih_ring *ih)
  {
-       uint32_t checkpoint_wptr, rptr;
+       uint32_t checkpoint_wptr;
+       uint64_t checkpoint_ts;
+       long timeout = HZ;
  
         if (!ih->enabled || adev->shutdown)
                 return -ENODEV;
  
         checkpoint_wptr = amdgpu_ih_get_wptr(adev, ih);
-       /* Order wptr with rptr. */
+       /* Order wptr with ring data. */
         rmb();
-       rptr = READ_ONCE(ih->rptr);
-
-       /* wptr has wrapped. */
-       if (rptr > checkpoint_wptr)
-               checkpoint_wptr += ih->ptr_mask + 1;
+       checkpoint_ts = amdgpu_ih_decode_iv_ts(adev, ih, checkpoint_wptr, -1);
  
-       return wait_event_interruptible(ih->wait_process,
-                               amdgpu_ih_has_checkpoint_processed(adev, ih,
-                                               checkpoint_wptr, &rptr));
+       return wait_event_interruptible_timeout(ih->wait_process,
+                   !amdgpu_ih_ts_after(ih->processed_timestamp, checkpoint_ts),
+                   timeout);
  }
  
  /**
@@ -299,3 +279,18 @@ void amdgpu_ih_decode_iv_helper(struct amdgpu_device *adev,
         /* wptr/rptr are in bytes! */
         ih->rptr += 32;
  }
+
+uint64_t amdgpu_ih_decode_iv_ts_helper(struct amdgpu_ih_ring *ih, u32 rptr,
+                                      signed int offset)
+{
+       uint32_t iv_size = 32;
+       uint32_t ring_index;
+       uint32_t dw1, dw2;
+
+       rptr += iv_size * offset;
+       ring_index = (rptr & ih->ptr_mask) >> 2;
+
+       dw1 = le32_to_cpu(ih->ring[ring_index + 1]);
+       dw2 = le32_to_cpu(ih->ring[ring_index + 2]);
+       return dw1 | ((u64)(dw2 & 0xffff) << 32);
+}
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h b/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h

index 0649b59830a59b0203889aeef7ac1d02e0fbd615..dd1c2eded6b9d2a533fed7d9cf4354e1e52d0f2d 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h
@@ -68,20 +68,30 @@ struct amdgpu_ih_ring {
  
         /* For waiting on IH processing at checkpoint. */
         wait_queue_head_t wait_process;
+       uint64_t                processed_timestamp;
  };
  
+/* return true if time stamp t2 is after t1 with 48bit wrap around */
+#define amdgpu_ih_ts_after(t1, t2) \
+               (((int64_t)((t2) << 16) - (int64_t)((t1) << 16)) > 0LL)
+
  /* provided by the ih block */
  struct amdgpu_ih_funcs {
         /* ring read/write ptr handling, called from interrupt context */
         u32 (*get_wptr)(struct amdgpu_device *adev, struct amdgpu_ih_ring *ih);
         void (*decode_iv)(struct amdgpu_device *adev, struct amdgpu_ih_ring *ih,
                           struct amdgpu_iv_entry *entry);
+       uint64_t (*decode_iv_ts)(struct amdgpu_ih_ring *ih, u32 rptr,
+                                signed int offset);
         void (*set_rptr)(struct amdgpu_device *adev, struct amdgpu_ih_ring *ih);
  };
  
  #define amdgpu_ih_get_wptr(adev, ih) (adev)->irq.ih_funcs->get_wptr((adev), (ih))
  #define amdgpu_ih_decode_iv(adev, iv) \
         (adev)->irq.ih_funcs->decode_iv((adev), (ih), (iv))
+#define amdgpu_ih_decode_iv_ts(adev, ih, rptr, offset) \
+       (WARN_ON_ONCE(!(adev)->irq.ih_funcs->decode_iv_ts) ? 0 : \
+       (adev)->irq.ih_funcs->decode_iv_ts((ih), (rptr), (offset)))
  #define amdgpu_ih_set_rptr(adev, ih) (adev)->irq.ih_funcs->set_rptr((adev), (ih))
  
  int amdgpu_ih_ring_init(struct amdgpu_device *adev, struct amdgpu_ih_ring *ih,
@@ -89,10 +99,12 @@ int amdgpu_ih_ring_init(struct amdgpu_device *adev, struct amdgpu_ih_ring *ih,
  void amdgpu_ih_ring_fini(struct amdgpu_device *adev, struct amdgpu_ih_ring *ih);
  void amdgpu_ih_ring_write(struct amdgpu_ih_ring *ih, const uint32_t *iv,
                           unsigned int num_dw);
-int amdgpu_ih_wait_on_checkpoint_process(struct amdgpu_device *adev,
-                                       struct amdgpu_ih_ring *ih);
+int amdgpu_ih_wait_on_checkpoint_process_ts(struct amdgpu_device *adev,
+                                           struct amdgpu_ih_ring *ih);
  int amdgpu_ih_process(struct amdgpu_device *adev, struct amdgpu_ih_ring *ih);
  void amdgpu_ih_decode_iv_helper(struct amdgpu_device *adev,
                                 struct amdgpu_ih_ring *ih,
                                 struct amdgpu_iv_entry *entry);
+uint64_t amdgpu_ih_decode_iv_ts_helper(struct amdgpu_ih_ring *ih, u32 rptr,
+                                      signed int offset);
  #endif
diff --git a/drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c b/drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c

index 4f3c62adccbdef85d46b979e2db56ec3cf116dd0..3907fc726ab26a417f514902b94d25024614e0fd 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c
+++ b/drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c
@@ -528,6 +528,12 @@ void amdgpu_irq_dispatch(struct amdgpu_device *adev,
         /* Send it to amdkfd as well if it isn't already handled */
         if (!handled)
                 amdgpu_amdkfd_interrupt(adev, entry.iv_entry);
+
+       dev_WARN_ONCE(adev->dev, ih->processed_timestamp == entry.timestamp,
+                     "IH timestamps are not unique");
+
+       if (amdgpu_ih_ts_after(ih->processed_timestamp, entry.timestamp))
+               ih->processed_timestamp = entry.timestamp;
  }
  
  /**
diff --git a/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c b/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c

index 3ec5ff5a6dbe6f163e6a6cba12d5392c33831e89..d696c4754beaf7f039517a791b0e7f03d8c913b3 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c
@@ -107,7 +107,7 @@ static int gmc_v10_0_process_interrupt(struct amdgpu_device *adev,
  
                 /* Process it onyl if it's the first fault for this address */
                 if (entry->ih != &adev->irq.ih_soft &&
-                   amdgpu_gmc_filter_faults(adev, addr, entry->pasid,
+                   amdgpu_gmc_filter_faults(adev, entry->ih, addr, entry->pasid,
                                              entry->timestamp))
                         return 1;
  
diff --git a/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c b/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c

index cb82404df5342a640d0124271e578a878dfa1368..7490ce8295c146dbcccd7b47849cc40ca75a8fe3 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c
+++ b/drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c
@@ -523,7 +523,7 @@ static int gmc_v9_0_process_interrupt(struct amdgpu_device *adev,
  
                 /* Process it onyl if it's the first fault for this address */
                 if (entry->ih != &adev->irq.ih_soft &&
-                   amdgpu_gmc_filter_faults(adev, addr, entry->pasid,
+                   amdgpu_gmc_filter_faults(adev, entry->ih, addr, entry->pasid,
                                              entry->timestamp))
                         return 1;
  
diff --git a/drivers/gpu/drm/amd/amdgpu/navi10_ih.c b/drivers/gpu/drm/amd/amdgpu/navi10_ih.c

index 38241cf0e1f1639f5d1e5ac97c5f6d0d7d716706..8ce5b8ca1fd791133f738372b516b3e01a94a90d 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/navi10_ih.c
+++ b/drivers/gpu/drm/amd/amdgpu/navi10_ih.c
@@ -716,6 +716,7 @@ static const struct amd_ip_funcs navi10_ih_ip_funcs = {
  static const struct amdgpu_ih_funcs navi10_ih_funcs = {
         .get_wptr = navi10_ih_get_wptr,
         .decode_iv = amdgpu_ih_decode_iv_helper,
+       .decode_iv_ts = amdgpu_ih_decode_iv_ts_helper,
         .set_rptr = navi10_ih_set_rptr
  };
  
diff --git a/drivers/gpu/drm/amd/amdgpu/vega10_ih.c b/drivers/gpu/drm/amd/amdgpu/vega10_ih.c

index a9ca6988009e38c33e353461344fb37343c174cd..3070466f54e170de07b519b4d7663a95d9cccae3 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/vega10_ih.c
+++ b/drivers/gpu/drm/amd/amdgpu/vega10_ih.c
@@ -640,6 +640,7 @@ const struct amd_ip_funcs vega10_ih_ip_funcs = {
  static const struct amdgpu_ih_funcs vega10_ih_funcs = {
         .get_wptr = vega10_ih_get_wptr,
         .decode_iv = amdgpu_ih_decode_iv_helper,
+       .decode_iv_ts = amdgpu_ih_decode_iv_ts_helper,
         .set_rptr = vega10_ih_set_rptr
  };
  
diff --git a/drivers/gpu/drm/amd/amdgpu/vega20_ih.c b/drivers/gpu/drm/amd/amdgpu/vega20_ih.c

index f51dfc38ac656ed53a6543cf52d475844b343d5f..3b4eb8285943c1c4091d54c06eea8fb6d2966d5c 100644 (file)
--- a/drivers/gpu/drm/amd/amdgpu/vega20_ih.c
+++ b/drivers/gpu/drm/amd/amdgpu/vega20_ih.c
@@ -688,6 +688,7 @@ const struct amd_ip_funcs vega20_ih_ip_funcs = {
  static const struct amdgpu_ih_funcs vega20_ih_funcs = {
         .get_wptr = vega20_ih_get_wptr,
         .decode_iv = amdgpu_ih_decode_iv_helper,
+       .decode_iv_ts = amdgpu_ih_decode_iv_ts_helper,
         .set_rptr = vega20_ih_set_rptr
  };
  
diff --git a/drivers/gpu/drm/amd/amdkfd/kfd_svm.c b/drivers/gpu/drm/amd/amdkfd/kfd_svm.c

index 10868d5b549f53efab97563b0d4dabd86906b304..663489ae56d7060de8d44a44461e09320d04fb47 100644 (file)
--- a/drivers/gpu/drm/amd/amdkfd/kfd_svm.c
+++ b/drivers/gpu/drm/amd/amdkfd/kfd_svm.c
@@ -1974,7 +1974,7 @@ restart:
  
                 pr_debug("drain retry fault gpu %d svms %p\n", i, svms);
  
-               amdgpu_ih_wait_on_checkpoint_process(pdd->dev->adev,
+               amdgpu_ih_wait_on_checkpoint_process_ts(pdd->dev->adev,
                                                      &pdd->dev->adev->irq.ih1);
                 pr_debug("drain retry fault gpu %d svms 0x%p done\n", i, svms);
         }
author	Philip Yang <Philip.Yang@amd.com>
	Thu, 18 Nov 2021 20:24:55 +0000 (15:24 -0500)
committer	Alex Deucher <alexander.deucher@amd.com>
	Wed, 1 Dec 2021 21:03:34 +0000 (16:03 -0500)
drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/amdgpu_gmc.h		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/amdgpu_ih.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/amdgpu_ih.h		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/amdgpu_irq.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/gmc_v10_0.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/gmc_v9_0.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/navi10_ih.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/vega10_ih.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdgpu/vega20_ih.c		patch \| blob \| blame \| history
drivers/gpu/drm/amd/amdkfd/kfd_svm.c		patch \| blob \| blame \| history