[PATCH ath-next 1/8] wifi: ath10k: fix SDIO aux workqueue starvation in recovery detection
Kang Yang
kang.yang at oss.qualcomm.com
Fri Sep 25 05:26:31 PDT 2026
Recovery monitoring runs on workqueue_aux, which is also
used by SDIO RX dispatch and WMI SERVICE_READY handling.
A single firmware crash may generate multiple recovery
triggers. Recovery detection currently handles each one by
waiting for recovery completion on workqueue_aux. On SDIO,
this blocks the same workqueue needed to deliver recovery
traffic, preventing recovery from completing.
As a result, otherwise recoverable firmware crashes fail
with:
ctl_resp never came in (-110)
Treat duplicate recovery triggers as part of a single
recovery attempt by gating recovery startup in
ath10k_core_start_recovery() with an atomic_cmpxchg() on
pending_recovery, and by making the monitor path check the
elapsed time since the attempt started instead of blocking
on a completion.
Tested-on: QCA6174 hw3.2 PCI WLAN.RM.4.4.1-00288-QCARMSWPZ-1
Fixes: f35a07a4842a ("wifi: ath10k: move recovery check logic into a new work")
Signed-off-by: Kang Yang <kang.yang at oss.qualcomm.com>
---
drivers/net/wireless/ath/ath10k/core.c | 70 ++++++++++++++++++++--------------
drivers/net/wireless/ath/ath10k/core.h | 2 +
2 files changed, 43 insertions(+), 29 deletions(-)
diff --git a/drivers/net/wireless/ath/ath10k/core.c b/drivers/net/wireless/ath/ath10k/core.c
index 7c2939cbde5f..ad087e64db9f 100644
--- a/drivers/net/wireless/ath/ath10k/core.c
+++ b/drivers/net/wireless/ath/ath10k/core.c
@@ -2494,46 +2494,58 @@ static int ath10k_init_hw_params(struct ath10k *ar)
static void ath10k_core_recovery_check_work(struct work_struct *work)
{
struct ath10k *ar = container_of(work, struct ath10k, recovery_check_work);
- long time_left;
+ unsigned long recovery_start = READ_ONCE(ar->recovery_start_jiffies);
+ int fail_count;
- /* Sometimes the recovery will fail and then the next all recovery fail,
- * so avoid infinite recovery.
- */
- if (atomic_read(&ar->fail_cont_count) >= ATH10K_RECOVERY_MAX_FAIL_COUNT) {
- ath10k_err(ar, "consecutive fail %d times, will shutdown driver!",
- atomic_read(&ar->fail_cont_count));
- ar->state = ATH10K_STATE_WEDGED;
+ if (ar->state == ATH10K_STATE_WEDGED)
return;
- }
- ath10k_dbg(ar, ATH10K_DBG_BOOT, "total recovery count: %d", ++ar->recovery_count);
+ /*
+ * Pure monitor: only ath10k_core_start_recovery() starts a fresh
+ * cycle. Bail if there is nothing in flight.
+ */
+ if (!atomic_read(&ar->pending_recovery))
+ return;
- if (atomic_read(&ar->pending_recovery)) {
- /* Sometimes it happened another recovery work before the previous one
- * completed, then the second recovery work will destroy the previous
- * one, thus below is to avoid that.
- */
- time_left = wait_for_completion_timeout(&ar->driver_recovery,
- ATH10K_RECOVERY_TIMEOUT_HZ);
- if (time_left) {
- ath10k_warn(ar, "previous recovery succeeded, skip this!\n");
- return;
- }
+ if (!time_after(jiffies, recovery_start + ATH10K_RECOVERY_TIMEOUT_HZ))
+ return;
- /* Record the continuous recovery fail count when recovery failed. */
- atomic_inc(&ar->fail_cont_count);
+ /* Record the continuous recovery fail count when recovery is late. */
+ fail_count = atomic_inc_return(&ar->fail_cont_count);
- /* Avoid having multiple recoveries at the same time. */
- return;
+ if (fail_count >= ATH10K_RECOVERY_MAX_FAIL_COUNT) {
+ ath10k_err(ar, "consecutive fail %d times, will shutdown driver!",
+ fail_count);
+ ar->state = ATH10K_STATE_WEDGED;
}
-
- atomic_inc(&ar->pending_recovery);
- queue_work(ar->workqueue, &ar->restart_work);
}
void ath10k_core_start_recovery(struct ath10k *ar)
{
- /* Use workqueue_aux to avoid blocking recovery tracking */
+ if (ar->state == ATH10K_STATE_WEDGED)
+ return;
+
+ /*
+ * SDIO raises several triggers within microseconds of one crash;
+ * only the first winner starts a fresh cycle and stamps the
+ * deadline. Losers do nothing while the owner runs recovery.
+ */
+ if (atomic_cmpxchg(&ar->pending_recovery, 0, 1) != 0)
+ return;
+ ath10k_dbg(ar, ATH10K_DBG_BOOT, "total recovery count: %d",
+ ++ar->recovery_count);
+ /*
+ * Reset fail_cont_count on a fresh cycle so a stale monitor
+ * from a just-succeeded cycle -- one that inc'd between
+ * reading pending_recovery == 1 and ath10k_reconfig_complete()
+ * clearing it -- does not bleed into the new cycle's
+ * failure budget.
+ */
+ atomic_set(&ar->fail_cont_count, 0);
+ WRITE_ONCE(ar->recovery_start_jiffies, jiffies);
+ queue_work(ar->workqueue, &ar->restart_work);
+
+ /* Start monitoring after publishing the recovery timestamp. */
queue_work(ar->workqueue_aux, &ar->recovery_check_work);
}
EXPORT_SYMBOL(ath10k_core_start_recovery);
diff --git a/drivers/net/wireless/ath/ath10k/core.h b/drivers/net/wireless/ath/ath10k/core.h
index dfee432615eb..c739fbadc8bb 100644
--- a/drivers/net/wireless/ath/ath10k/core.h
+++ b/drivers/net/wireless/ath/ath10k/core.h
@@ -1215,6 +1215,8 @@ struct ath10k {
unsigned int recovery_count;
/* continuous recovery fail count */
atomic_t fail_cont_count;
+ /* jiffies when the current recovery attempt started */
+ unsigned long recovery_start_jiffies;
/* cycle count is reported twice for each visited channel during scan.
* access protected by data_lock
--
2.34.1
More information about the ath10k
mailing list