summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--drivers/spi/spi-tegra210-quad.c509
1 files changed, 441 insertions, 68 deletions
diff --git a/drivers/spi/spi-tegra210-quad.c b/drivers/spi/spi-tegra210-quad.c
index 8ede864c3d3c..d8cca3aa276b 100644
--- a/drivers/spi/spi-tegra210-quad.c
+++ b/drivers/spi/spi-tegra210-quad.c
@@ -191,6 +191,15 @@ struct tegra_qspi {
void __iomem *base;
phys_addr_t phys;
unsigned int irq;
+ struct work_struct irq_work;
+ struct workqueue_struct *wq;
+ /*
+ * Set by tegra_qspi_handle_timeout() while it drains the bottom
+ * half so tegra_qspi_isr() suppresses new queue_work() calls
+ * that would otherwise race the recovery path or the caller's
+ * cleanup of curr_xfer.
+ */
+ bool recovery_in_progress;
u32 cur_speed;
unsigned int cur_pos;
@@ -205,6 +214,18 @@ struct tegra_qspi {
unsigned int dma_buf_size;
unsigned int max_buf_size;
bool is_curr_dma_xfer;
+ /*
+ * Cached "this PIO chunk completes the whole transfer" decision,
+ * computed by tegra_qspi_start_cpu_based_transfer() before it
+ * unmasks the IRQ. Used by the hard IRQ small-PIO fastpath in
+ * place of dereferencing curr_xfer->len, so the ISR cannot touch
+ * the spi_transfer object even on a late IRQ that races with the
+ * synchronous teardown path. Multi-chunk PIO transfers always go
+ * through the workqueue (this flag is only set on the final
+ * chunk), so the fastpath cannot recurse into
+ * tegra_qspi_start_cpu_based_transfer() from hard IRQ context.
+ */
+ bool is_last_pio_chunk;
struct completion rx_dma_complete;
struct completion tx_dma_complete;
@@ -212,6 +233,7 @@ struct tegra_qspi {
u32 tx_status;
u32 rx_status;
u32 status_reg;
+ u32 trans_status;
bool is_packed;
bool use_dma;
@@ -622,6 +644,17 @@ static int tegra_qspi_start_dma_based_transfer(struct tegra_qspi *tqspi, struct
val = QSPI_DMA_BLK_SET(tqspi->curr_dma_words - 1);
tegra_qspi_writel(tqspi, val, QSPI_DMA_BLK);
+ /*
+ * Reset the cached transfer status before unmasking the IRQ for
+ * this chunk. The cache must represent only the IRQ for THIS
+ * chunk; a stale RDY from the previous chunk of a multi-chunk
+ * transfer would otherwise mislead tegra_qspi_handle_timeout()
+ * into a false-positive recovery while the new chunk is still in
+ * flight. Pairs with smp_load_acquire() in
+ * tegra_qspi_handle_timeout(). The new chunk's IRQ cannot fire
+ * until QSPI_DMA_CTL is written below.
+ */
+ smp_store_release(&tqspi->trans_status, 0);
tegra_qspi_unmask_irq(tqspi);
if (tqspi->is_packed)
@@ -713,7 +746,13 @@ static int tegra_qspi_start_dma_based_transfer(struct tegra_qspi *tqspi, struct
tegra_qspi_writel(tqspi, tqspi->command1_reg, QSPI_COMMAND1);
- tqspi->is_curr_dma_xfer = true;
+ /*
+ * WRITE_ONCE() pairs with READ_ONCE() in tegra_qspi_isr() and
+ * tegra_qspi_work_handler(); the flag is read lock-free across
+ * the hard-IRQ / process-context boundary so the annotation
+ * prevents compiler tearing and silences KCSAN.
+ */
+ WRITE_ONCE(tqspi->is_curr_dma_xfer, true);
tqspi->dma_control_reg = val;
val |= QSPI_DMA_EN;
tegra_qspi_writel(tqspi, val, QSPI_DMA_CTL);
@@ -734,9 +773,33 @@ static int tegra_qspi_start_cpu_based_transfer(struct tegra_qspi *qspi, struct s
val = QSPI_DMA_BLK_SET(cur_words - 1);
tegra_qspi_writel(qspi, val, QSPI_DMA_BLK);
+ /*
+ * Snapshot whether this PIO chunk completes the whole transfer
+ * before unmasking the IRQ, so the hard IRQ small-PIO fastpath
+ * can decide whether to drain inline without dereferencing the
+ * spi_transfer object. cur_pos / curr_dma_words / bytes_per_word
+ * are stable here: they are written by
+ * tegra_qspi_calculate_curr_xfer_param() earlier in this code
+ * path. The IRQ cannot fire until the QSPI_COMMAND1 write below
+ * kicks the transfer off, so this store happens-before any ISR
+ * that observes the unmask.
+ */
+ WRITE_ONCE(qspi->is_last_pio_chunk,
+ qspi->cur_pos + qspi->curr_dma_words * qspi->bytes_per_word >= t->len);
+
+ /*
+ * Reset the cached transfer status before unmasking the IRQ for
+ * this chunk so the cache represents only the IRQ for THIS chunk;
+ * a stale RDY from the previous chunk would otherwise mislead
+ * tegra_qspi_handle_timeout() into a false-positive recovery
+ * while the new chunk is still in flight. Pairs with
+ * smp_load_acquire() in tegra_qspi_handle_timeout(). The new
+ * chunk's IRQ cannot fire until QSPI_COMMAND1 is written below.
+ */
+ smp_store_release(&qspi->trans_status, 0);
tegra_qspi_unmask_irq(qspi);
- qspi->is_curr_dma_xfer = false;
+ WRITE_ONCE(qspi->is_curr_dma_xfer, false);
val = qspi->command1_reg;
val |= QSPI_PIO;
tegra_qspi_writel(qspi, val, QSPI_COMMAND1);
@@ -859,6 +922,13 @@ static u32 tegra_qspi_setup_transfer_one(struct spi_device *spi, struct spi_tran
tqspi->cur_rx_pos = 0;
tqspi->cur_tx_pos = 0;
tqspi->curr_xfer = t;
+ /*
+ * Pairs with smp_load_acquire() in tegra_qspi_handle_timeout().
+ * Clearing the cached trans_status before unmasking the IRQ for
+ * the new transfer prevents a stale RDY bit from the previous
+ * transfer fooling the timeout handler into a false recovery.
+ */
+ smp_store_release(&tqspi->trans_status, 0);
spin_unlock_irqrestore(&tqspi->lock, flags);
if (is_first_of_msg) {
@@ -1065,40 +1135,206 @@ static irqreturn_t handle_dma_based_xfer(struct tegra_qspi *tqspi);
* tegra_qspi_handle_timeout - Handle transfer timeout with hardware check
* @tqspi: QSPI controller instance
*
- * When a timeout occurs but hardware has completed the transfer (interrupt
- * was lost or delayed), manually trigger transfer completion processing.
- * This avoids failing transfers that actually succeeded.
+ * When wait_for_completion_timeout() expires the hardware may still have
+ * finished the current chunk. Drain the pending bottom half and, if the
+ * whole transfer really did complete during the drain, consume the
+ * completion and report success.
+ *
+ * When the bottom half advanced the transfer by only one chunk of a
+ * multi-chunk DMA/PIO transfer without signalling xfer_completion, a
+ * fallback that ran handle_{cpu,dma}_based_xfer() here would race with
+ * the DMA engine already moving the next chunk into the client buffer
+ * (spi_finalize_current_message() would then release the buffer while
+ * the controller is still writing memory). Fake completion is therefore
+ * only attempted when the current chunk is the last chunk of the
+ * transfer; multi-chunk continuation timeouts return -ETIMEDOUT and
+ * let the caller reset the controller.
*
- * Returns: 0 if transfer was completed, -ETIMEDOUT if real timeout
+ * Returns: 0 if the transfer completed, -ETIMEDOUT otherwise.
*/
static int tegra_qspi_handle_timeout(struct tegra_qspi *tqspi)
{
+ struct spi_transfer *t;
+ unsigned long flags;
+ bool is_last_chunk;
+ bool lost_irq_snapshot = false;
irqreturn_t ret;
- u32 status;
+ int retval;
+ u32 status, refreshed;
+ u32 lost_fifo_status = 0;
+ u32 lost_tx_status = 0;
+ u32 lost_rx_status = 0;
- /* Check if hardware actually completed the transfer */
- status = tegra_qspi_readl(tqspi, QSPI_TRANS_STATUS);
- if (!(status & QSPI_RDY))
- return -ETIMEDOUT;
+ /*
+ * Snapshot both the ISR cache and (if the cache is empty) the
+ * live status registers BEFORE entering recovery. The recovery
+ * path calls tegra_qspi_mask_clear_irq() below, which performs
+ * W1Cs on QSPI_TRANS_STATUS and on the QSPI_FIFO_STATUS error
+ * bits: a lost-IRQ recovery must capture the current FIFO error
+ * state before the mask erases it.
+ *
+ * Cache-live-cache retry: if the initial cache load returns zero
+ * we fall back to a live QSPI_TRANS_STATUS read, and if that also
+ * returns zero we retry the cache once more. That closes the
+ * interleaving where an ISR on another CPU publishes trans_status
+ * with release semantics and then W1Cs the hardware between our
+ * cache load and our live load: the second cache load observes
+ * the now-visible release and we correctly classify the transfer
+ * as complete rather than reporting a false timeout.
+ *
+ * The trans_status cache is reset to zero in
+ * tegra_qspi_start_{cpu,dma}_based_transfer() before unmasking
+ * the IRQ for every chunk, so a stale RDY from the previous
+ * chunk of a multi-chunk transfer cannot survive into this
+ * check.
+ */
+ status = smp_load_acquire(&tqspi->trans_status);
+ if (!status) {
+ status = tegra_qspi_readl(tqspi, QSPI_TRANS_STATUS);
+ if (!status) {
+ /* Retry cache; pairs with release in ISR post-store. */
+ status = smp_load_acquire(&tqspi->trans_status);
+ } else {
+ /*
+ * Live register shows RDY but the ISR cache is
+ * empty: either the ISR ran and cleared HW between
+ * our two loads (the cache retry above would have
+ * observed it, so we would not be here), or the IRQ
+ * was genuinely lost. Snapshot the live FIFO error
+ * status now so tegra_qspi_mask_clear_irq() below
+ * does not W1C it away before the manual handler
+ * downstream can see it.
+ */
+ lost_fifo_status = tegra_qspi_readl(tqspi,
+ QSPI_FIFO_STATUS);
+ lost_tx_status = lost_fifo_status &
+ (QSPI_TX_FIFO_UNF | QSPI_TX_FIFO_OVF);
+ lost_rx_status = lost_fifo_status &
+ (QSPI_RX_FIFO_OVF | QSPI_RX_FIFO_UNF);
+ lost_irq_snapshot = true;
+ }
+ }
/*
- * Hardware completed but interrupt was lost/delayed. Manually
- * process the completion by calling the appropriate handler.
+ * Enter recovery unconditionally. Every expired
+ * wait_for_completion_timeout() must serialise against a delayed
+ * ISR or worker before the caller runs dma_stop() +
+ * device_reset() + curr_xfer clear: publishing
+ * recovery_in_progress under tqspi->lock, masking the controller
+ * IRQ, calling synchronize_irq() to drain any in-flight ISR
+ * (including the small-PIO hard-IRQ fastpath), and finally
+ * cancel_work_sync() to drain the workqueue gives us that
+ * serialisation regardless of whether the hardware finished. A
+ * genuine hardware timeout still ends up as -ETIMEDOUT further
+ * down, but only after ISR and workqueue activity are quiesced.
+ *
+ * cancel_work_sync() cancels a pending worker without executing
+ * it and waits for a currently running one to finish; the
+ * recovery_in_progress guard checked inside tegra_qspi_isr()
+ * under tqspi->lock is atomic with its queue_work() and small-PIO
+ * fastpath dispatch decisions, so no new bottom-half work is
+ * enqueued once we publish the flag.
+ *
+ * tegra_qspi_mask_clear_irq() is idempotent: its read-modify-write
+ * of QSPI_INTR_MASK and W1C of QSPI_TRANS_STATUS / FIFO error
+ * status all tolerate a double-write, so it is safe whether or
+ * not the ISR has already run for this transfer.
*/
+ spin_lock_irqsave(&tqspi->lock, flags);
+ WRITE_ONCE(tqspi->recovery_in_progress, true);
+ spin_unlock_irqrestore(&tqspi->lock, flags);
+
+ tegra_qspi_mask_clear_irq(tqspi);
+ synchronize_irq(tqspi->irq);
+ cancel_work_sync(&tqspi->irq_work);
+
+ if (try_wait_for_completion(&tqspi->xfer_completion)) {
+ retval = 0;
+ goto out;
+ }
+
+ /*
+ * Re-check the cache after the drain: the worker we just drained
+ * may have published a completion status the entry snapshot did
+ * not observe (for example the ISR fired on another CPU after we
+ * loaded the cache but before we masked).
+ */
+ refreshed = smp_load_acquire(&tqspi->trans_status);
+ if (refreshed)
+ status = refreshed;
+
+ if (!(status & QSPI_RDY)) {
+ retval = -ETIMEDOUT;
+ goto out;
+ }
+
+ /*
+ * If the ISR never ran (lost IRQ path) publish the FIFO error
+ * snapshot we captured before mask_clear_irq() so the manual
+ * handler downstream has fresh error state rather than stale
+ * fields from a previous chunk's ISR.
+ */
+ if (lost_irq_snapshot) {
+ WRITE_ONCE(tqspi->status_reg, lost_fifo_status);
+ WRITE_ONCE(tqspi->tx_status, lost_tx_status);
+ WRITE_ONCE(tqspi->rx_status, lost_rx_status);
+ }
+
+ /*
+ * The bottom half did not signal full completion. Either the work
+ * ran and advanced the transfer by one chunk (possibly arming the
+ * next chunk of a multi-chunk transfer), or it was cancelled
+ * before it could run, or the current chunk really did not
+ * complete. Only fake completion when the current chunk is the
+ * last chunk of the transfer; otherwise the DMA engine may still
+ * be moving the next chunk into memory, and returning 0 here would
+ * let seq_xfer clear curr_xfer and finalise the message while the
+ * hardware is still writing.
+ *
+ * The last-chunk arithmetic mirrors tegra_qspi_start_cpu_based_
+ * transfer(), which uses cur_pos + curr_dma_words * bytes_per_word
+ * >= t->len to set is_last_pio_chunk before arming the IRQ.
+ */
+ spin_lock_irqsave(&tqspi->lock, flags);
+ t = tqspi->curr_xfer;
+ if (!t) {
+ /* CPU-path handler already cleared curr_xfer */
+ spin_unlock_irqrestore(&tqspi->lock, flags);
+ retval = 0;
+ goto out;
+ }
+ is_last_chunk = (tqspi->cur_pos +
+ tqspi->curr_dma_words * tqspi->bytes_per_word) >= t->len;
+ spin_unlock_irqrestore(&tqspi->lock, flags);
+
+ if (!is_last_chunk) {
+ retval = -ETIMEDOUT;
+ goto out;
+ }
+
dev_warn_ratelimited(tqspi->dev,
"QSPI interrupt timeout, but transfer complete\n");
- /* Clear the transfer status */
- status = tegra_qspi_readl(tqspi, QSPI_TRANS_STATUS);
- tegra_qspi_writel(tqspi, status, QSPI_TRANS_STATUS);
-
- /* Manually trigger completion handler */
- if (!tqspi->is_curr_dma_xfer)
+ if (!READ_ONCE(tqspi->is_curr_dma_xfer))
ret = handle_cpu_based_xfer(tqspi);
else
ret = handle_dma_based_xfer(tqspi);
- return (ret == IRQ_HANDLED) ? 0 : -EIO;
+ retval = (ret == IRQ_HANDLED) ? 0 : -EIO;
+
+out:
+ /*
+ * The drained bottom half may have unmasked the controller IRQ
+ * to arm the next chunk of a multi-chunk transfer. Re-mask and
+ * synchronize before clearing recovery_in_progress so that no
+ * lingering ISR can queue fresh work behind the caller's back
+ * (the caller's dma_stop() + device_reset() + curr_xfer clear
+ * runs immediately after we return on the error path).
+ */
+ tegra_qspi_mask_clear_irq(tqspi);
+ synchronize_irq(tqspi->irq);
+ WRITE_ONCE(tqspi->recovery_in_progress, false);
+ return retval;
}
static u32 tegra_qspi_cmd_config(bool is_ddr, u8 bus_width, u8 len)
@@ -1232,9 +1468,9 @@ static int tegra_qspi_combined_seq_xfer(struct tegra_qspi *tqspi,
if (ret == 0) {
/*
- * Check if hardware completed the transfer
- * even though interrupt was lost or delayed.
- * If so, process the completion and continue.
+ * Check if hardware completed the transfer even though
+ * workqueue was delayed. If so, process completion and
+ * continue.
*/
ret = tegra_qspi_handle_timeout(tqspi);
if (ret < 0) {
@@ -1351,8 +1587,8 @@ static int tegra_qspi_non_combined_seq_xfer(struct tegra_qspi *tqspi,
if (ret == 0) {
/*
* Check if hardware completed the transfer even though
- * interrupt was lost or delayed. If so, process the
- * completion and continue.
+ * workqueue was delayed. If so, process completion and
+ * continue.
*/
ret = tegra_qspi_handle_timeout(tqspi);
if (ret < 0) {
@@ -1506,6 +1742,19 @@ static irqreturn_t handle_dma_based_xfer(struct tegra_qspi *tqspi)
long wait_status;
int num_errors = 0;
+ /*
+ * Snapshot curr_xfer under the lock before the (potentially long)
+ * DMA waits below. The timeout path can clear tqspi->curr_xfer
+ * concurrently; using the local copy keeps the subsequent dma_unmap
+ * and FIFO-drain steps consistent with the transfer that actually
+ * started, and lets us bail safely if cleanup already happened.
+ */
+ spin_lock_irqsave(&tqspi->lock, flags);
+ t = tqspi->curr_xfer;
+ spin_unlock_irqrestore(&tqspi->lock, flags);
+ if (!t)
+ return IRQ_HANDLED;
+
if (tqspi->cur_direction & DATA_DIR_TX) {
if (tqspi->tx_status) {
if (tqspi->tx_dma_chan)
@@ -1539,12 +1788,6 @@ static irqreturn_t handle_dma_based_xfer(struct tegra_qspi *tqspi)
}
spin_lock_irqsave(&tqspi->lock, flags);
- t = tqspi->curr_xfer;
-
- if (!t) {
- spin_unlock_irqrestore(&tqspi->lock, flags);
- return IRQ_HANDLED;
- }
if (num_errors) {
tegra_qspi_dma_unmap_xfer(tqspi, t);
@@ -1581,46 +1824,41 @@ exit:
return IRQ_HANDLED;
}
-static irqreturn_t tegra_qspi_isr_thread(int irq, void *context_data)
+/**
+ * tegra_qspi_work_handler - Workqueue handler for interrupt bottom-half
+ * @work: work_struct embedded in tegra_qspi
+ *
+ * Runs in process context and can sleep (needed for DMA completion waits).
+ * Runs on any CPU in the WQ_UNBOUND pool, so the bottom half can migrate off
+ * the interrupt-taking CPU that the previous threaded IRQ pinned to
+ * (irq_thread() calls set_cpus_allowed_ptr() with the IRQ affinity mask).
+ *
+ * The hard IRQ handler has already:
+ * - Verified this is our interrupt (QSPI_RDY was set)
+ * - Cached FIFO status in tqspi->status_reg
+ * - Parsed tx_status / rx_status from FIFO status
+ * - Masked further interrupts
+ */
+static void tegra_qspi_work_handler(struct work_struct *work)
{
- struct tegra_qspi *tqspi = context_data;
+ struct tegra_qspi *tqspi = container_of(work, struct tegra_qspi, irq_work);
unsigned long flags;
- u32 status;
- /*
- * Read transfer status to check if interrupt was triggered by transfer
- * completion
- */
- status = tegra_qspi_readl(tqspi, QSPI_TRANS_STATUS);
+ spin_lock_irqsave(&tqspi->lock, flags);
/*
- * Occasionally the IRQ thread takes a long time to wake up (usually
- * when the CPU that it's running on is excessively busy) and we have
- * already reached the timeout before and cleaned up the timed out
- * transfer. Avoid any processing in that case and bail out early.
- *
- * If no transfer is in progress, check if this was a real interrupt
- * that the timeout handler already processed, or a spurious one.
+ * tegra_qspi_handle_timeout() sets recovery_in_progress under
+ * tqspi->lock and then calls cancel_work_sync(), so any running
+ * worker is drained and tegra_qspi_isr() cannot enqueue a new
+ * one while recovery runs. The curr_xfer NULL check catches the
+ * case where the timeout path already tore the transfer down
+ * before this work got a chance to run.
*/
- spin_lock_irqsave(&tqspi->lock, flags);
if (!tqspi->curr_xfer) {
spin_unlock_irqrestore(&tqspi->lock, flags);
- /* Spurious interrupt - transfer not ready */
- if (!(status & QSPI_RDY))
- return IRQ_NONE;
- /* Real interrupt, already handled by timeout path */
- return IRQ_HANDLED;
+ return;
}
- tqspi->status_reg = tegra_qspi_readl(tqspi, QSPI_FIFO_STATUS);
-
- if (tqspi->cur_direction & DATA_DIR_TX)
- tqspi->tx_status = tqspi->status_reg & (QSPI_TX_FIFO_UNF | QSPI_TX_FIFO_OVF);
-
- if (tqspi->cur_direction & DATA_DIR_RX)
- tqspi->rx_status = tqspi->status_reg & (QSPI_RX_FIFO_OVF | QSPI_RX_FIFO_UNF);
-
- tegra_qspi_mask_clear_irq(tqspi);
spin_unlock_irqrestore(&tqspi->lock, flags);
/*
@@ -1629,10 +1867,127 @@ static irqreturn_t tegra_qspi_isr_thread(int irq, void *context_data)
* DMA handler also needs to sleep in wait_for_completion_*(), which
* cannot be done while holding spinlock.
*/
- if (!tqspi->is_curr_dma_xfer)
+ if (!READ_ONCE(tqspi->is_curr_dma_xfer))
+ handle_cpu_based_xfer(tqspi);
+ else
+ handle_dma_based_xfer(tqspi);
+}
+
+/**
+ * tegra_qspi_isr - Hard IRQ handler
+ * @irq: IRQ number
+ * @context_data: QSPI controller instance
+ *
+ * Runs in hard IRQ context with minimal latency. Cannot sleep.
+ *
+ * Tegra QSPI uses a dedicated, non-shared GIC SPI line on every SoC that
+ * uses this driver. The handler always returns IRQ_HANDLED and always
+ * acknowledges/re-masks the controller IRQ, so the level-triggered line
+ * cannot stay asserted and trip the kernel spurious-IRQ detector into
+ * disabling the line. On a stray IRQ where curr_xfer is NULL (e.g. the
+ * timeout path has already torn the transfer down) the FIFO/status
+ * processing and bottom-half scheduling are skipped because there is no
+ * transfer to drive forward.
+ *
+ * Return: IRQ_HANDLED.
+ */
+static irqreturn_t tegra_qspi_isr(int irq, void *context_data)
+{
+ struct tegra_qspi *tqspi = context_data;
+ u32 status_reg, trans_status;
+ u32 tx_status = 0, rx_status = 0;
+
+ if (!READ_ONCE(tqspi->curr_xfer)) {
+ tegra_qspi_mask_clear_irq(tqspi);
+ return IRQ_HANDLED;
+ }
+
+ spin_lock(&tqspi->lock);
+ status_reg = tegra_qspi_readl(tqspi, QSPI_FIFO_STATUS);
+ trans_status = tegra_qspi_readl(tqspi, QSPI_TRANS_STATUS);
+
+ if (tqspi->cur_direction & DATA_DIR_TX) {
+ tx_status = status_reg & (QSPI_TX_FIFO_UNF | QSPI_TX_FIFO_OVF);
+ WRITE_ONCE(tqspi->tx_status, tx_status);
+ }
+
+ if (tqspi->cur_direction & DATA_DIR_RX) {
+ rx_status = status_reg & (QSPI_RX_FIFO_OVF | QSPI_RX_FIFO_UNF);
+ WRITE_ONCE(tqspi->rx_status, rx_status);
+ }
+
+ WRITE_ONCE(tqspi->status_reg, status_reg);
+ /*
+ * Publish trans_status with release semantics before we clear
+ * the hardware status in tegra_qspi_mask_clear_irq() below. That
+ * ordering matters for the lock-free cache read in
+ * tegra_qspi_handle_timeout(): if the timeout path sees the
+ * released trans_status it also observes the matching status_reg
+ * / tx_status / rx_status; if it does not yet see the released
+ * value it falls back to a live QSPI_TRANS_STATUS read, and that
+ * live read still returns QSPI_RDY because we have not cleared
+ * the register yet. Reversing this order would open a window
+ * where the cache is still zero but the hardware bit has already
+ * been cleared, making the fallback report a false timeout.
+ */
+ smp_store_release(&tqspi->trans_status, trans_status);
+
+ tegra_qspi_mask_clear_irq(tqspi);
+
+ /*
+ * If tegra_qspi_handle_timeout() is draining the bottom half,
+ * skip queueing new work. The flag is set under tqspi->lock and
+ * queue_work() below happens while we still hold the lock, so
+ * the guard is atomic with the queue decision. Any ISR that had
+ * already passed this check is drained by the synchronize_irq()
+ * call that tegra_qspi_handle_timeout() issues after publishing
+ * the flag.
+ */
+ if (READ_ONCE(tqspi->recovery_in_progress)) {
+ spin_unlock(&tqspi->lock);
+ return IRQ_HANDLED;
+ }
+
+ /*
+ * Small-PIO fastpath: drain the FIFO inline only when this chunk
+ * completes the entire outstanding transfer and no error bit was
+ * latched, to avoid workqueue scheduling latency for TPM-style
+ * short reads.
+ *
+ * The "last chunk" decision is computed and cached as a scalar by
+ * tegra_qspi_start_cpu_based_transfer() before it unmasks the IRQ,
+ * so the hard-IRQ fastpath never dereferences the spi_transfer
+ * pointer here. That keeps the ISR safe against any teardown race
+ * where the synchronous path could clear curr_xfer concurrently.
+ *
+ * The fastpath dispatch decision is made while still holding
+ * tqspi->lock, so the recovery_in_progress guard above covers it
+ * atomically with queue_work() below: an ISR that reaches the
+ * fastpath cannot race a tegra_qspi_handle_timeout() that
+ * subsequently observes recovery_in_progress == true, because
+ * that path calls synchronize_irq() before proceeding. We drop
+ * the lock before calling handle_cpu_based_xfer() so it can take
+ * tqspi->lock internally without deadlocking.
+ *
+ * Multi-chunk PIO continuation stays on the workqueue so that
+ * tegra_qspi_start_cpu_based_transfer() can re-arm the IRQ from
+ * process context. DMA transfers also stay on the workqueue
+ * because their completion path sleeps on the DMA engine.
+ * tegra_qspi_handle_error() -> device_reset() can sleep, so the
+ * fastpath only runs when both status words are clean.
+ */
+ if (!READ_ONCE(tqspi->is_curr_dma_xfer) &&
+ READ_ONCE(tqspi->is_last_pio_chunk) &&
+ !tx_status && !rx_status) {
+ spin_unlock(&tqspi->lock);
return handle_cpu_based_xfer(tqspi);
+ }
+
+ queue_work(tqspi->wq, &tqspi->irq_work);
- return handle_dma_based_xfer(tqspi);
+ spin_unlock(&tqspi->lock);
+
+ return IRQ_HANDLED;
}
static struct tegra_qspi_soc_data tegra210_qspi_soc_data = {
@@ -1800,12 +2155,21 @@ static int tegra_qspi_probe(struct platform_device *pdev)
pm_runtime_put_autosuspend(&pdev->dev);
- ret = request_threaded_irq(tqspi->irq, NULL,
- tegra_qspi_isr_thread, IRQF_ONESHOT,
- dev_name(&pdev->dev), tqspi);
+ tqspi->wq = alloc_workqueue("%s", WQ_HIGHPRI | WQ_UNBOUND, 0,
+ dev_name(&pdev->dev));
+ if (!tqspi->wq) {
+ dev_err(&pdev->dev, "failed to allocate workqueue\n");
+ ret = -ENOMEM;
+ goto exit_pm_disable;
+ }
+
+ INIT_WORK(&tqspi->irq_work, tegra_qspi_work_handler);
+
+ ret = request_irq(tqspi->irq, tegra_qspi_isr, 0,
+ dev_name(&pdev->dev), tqspi);
if (ret < 0) {
dev_err(&pdev->dev, "failed to request IRQ#%u: %d\n", tqspi->irq, ret);
- goto exit_pm_disable;
+ goto exit_destroy_wq;
}
ret = spi_register_controller(host);
@@ -1817,7 +2181,9 @@ static int tegra_qspi_probe(struct platform_device *pdev)
return 0;
exit_free_irq:
- free_irq(qspi_irq, tqspi);
+ free_irq(tqspi->irq, tqspi);
+exit_destroy_wq:
+ destroy_workqueue(tqspi->wq);
exit_pm_disable:
pm_runtime_dont_use_autosuspend(&pdev->dev);
pm_runtime_force_suspend(&pdev->dev);
@@ -1830,8 +2196,15 @@ static void tegra_qspi_remove(struct platform_device *pdev)
struct spi_controller *host = platform_get_drvdata(pdev);
struct tegra_qspi *tqspi = spi_controller_get_devdata(host);
+ /*
+ * Tear down in reverse order of probe() so that the controller stops
+ * accepting transfers before the IRQ is released, no new work can be
+ * queued after the IRQ is freed, and any work already queued is
+ * drained while the clocks are still running.
+ */
spi_unregister_controller(host);
free_irq(tqspi->irq, tqspi);
+ destroy_workqueue(tqspi->wq);
pm_runtime_dont_use_autosuspend(&pdev->dev);
pm_runtime_force_suspend(&pdev->dev);
tegra_qspi_deinit_dma(tqspi);