From 38d11bb2cd1d902d7ab178f0a6d78d2bafd2d720 Mon Sep 17 00:00:00 2001 From: Han Xu Date: Wed, 22 Jul 2026 17:00:43 -0500 Subject: [PATCH] spi: nxp-xspi: Improve AHB read performance with DMA Use eDMA for XSPI AHB read to improve the performance. Read with CPU root@imx943evk:~# dd if=/dev/mtd0 of=/dev/null bs=32M count=1 1+0 records in 1+0 records out 33554432 bytes (34 MB, 32 MiB) copied, 0.479719 s, 69.9 MB/s Read with DMA root@imx943evk:~# dd if=/dev/mtd0 of=/dev/null bs=32M count=1 1+0 records in 1+0 records out 33554432 bytes (34 MB, 32 MiB) copied, 0.115788 s, 290 MB/s Signed-off-by: Han Xu Reviewed-by: Frank Li Link: https://patch.msgid.link/20260722-xspi_dma_upstream-v3-2-6b7ff50cae8e@nxp.com Signed-off-by: Mark Brown --- drivers/spi/spi-nxp-xspi.c | 143 ++++++++++++++++++++++++++++++++++++- 1 file changed, 140 insertions(+), 3 deletions(-) diff --git a/drivers/spi/spi-nxp-xspi.c b/drivers/spi/spi-nxp-xspi.c index 037eac24e6fd..70cbc3ef658b 100644 --- a/drivers/spi/spi-nxp-xspi.c +++ b/drivers/spi/spi-nxp-xspi.c @@ -35,6 +35,8 @@ #include #include #include +#include +#include #include #include #include @@ -303,6 +305,12 @@ (opr)) << (((idx) % 2) * OPRND_SHIFT)) #define NXP_XSPI_MIN_IOMAP SZ_4M +/* + * It takes 0.5s to access the entire area with DMA, + * leaving a 10x safety margin + */ +#define NXP_XSPI_DMA_TOUT_MS 5000 +#define NXP_XSPI_DMA_TOUT msecs_to_jiffies(NXP_XSPI_DMA_TOUT_MS) #define NXP_XSPI_MAX_CHIPSELECT 2 #define POLL_TOUT_US 5000 @@ -336,6 +344,9 @@ struct nxp_xspi { /* mutex lock for each operation */ struct mutex lock; int selected; + int dma_copy_align; + struct dma_chan *dma_rx; + struct completion dma_rx_c; #define XSPI_DTR_PROTO BIT(0) int flags; /* Save the previous operation clock rate */ @@ -796,6 +807,95 @@ static int nxp_xspi_ahb_read(struct nxp_xspi *xspi, const struct spi_mem_op *op) return 0; } +static int nxp_xspi_dma_init(struct device *dev, struct nxp_xspi *xspi) +{ + struct dma_chan *rx; + + rx = dma_request_chan(dev, "rx"); + if (IS_ERR(rx)) { + int ret = PTR_ERR(rx); + + if (ret == -EPROBE_DEFER) + return ret; + dev_dbg(dev, "NO DMA RX channel, falling back to CPU read\n"); + xspi->dma_copy_align = 1; + return 0; + } + + xspi->dma_rx = rx; + init_completion(&xspi->dma_rx_c); + xspi->dma_copy_align = 1 << rx->device->copy_align; + + return 0; +} + +static void nxp_xspi_dma_rx_callback(void *data) +{ + struct nxp_xspi *xspi = data; + + complete(&xspi->dma_rx_c); +} + +static int nxp_xspi_ahb_dma_read(struct nxp_xspi *xspi, + const struct spi_mem_op *op) +{ + struct dma_async_tx_descriptor *desc; + struct dma_chan *chan = xspi->dma_rx; + unsigned long timeout; + dma_addr_t src_addr, dst_addr; + int ret = 0; + + dst_addr = dma_map_single(chan->device->dev, + op->data.buf.in, op->data.nbytes, + DMA_FROM_DEVICE); + if (dma_mapping_error(chan->device->dev, dst_addr)) { + dev_err(xspi->dev, "failed to map DMA dst buffer\n"); + return -ENOMEM; + } + + src_addr = dma_map_resource(chan->device->dev, + (phys_addr_t)xspi->memmap_phy + op->addr.val, + op->data.nbytes, + DMA_TO_DEVICE, 0); + if (dma_mapping_error(chan->device->dev, src_addr)) { + dev_err(xspi->dev, "failed to map DMA src resource\n"); + ret = -ENOMEM; + goto err_unmap_dst; + } + + desc = dmaengine_prep_dma_memcpy(chan, dst_addr, src_addr, + op->data.nbytes, + DMA_PREP_INTERRUPT | DMA_CTRL_ACK); + if (!desc) { + dev_err(xspi->dev, "failed to prepare AHB RX DMA descriptor\n"); + ret = -EIO; + goto err_unmap_src; + } + + desc->callback = nxp_xspi_dma_rx_callback; + desc->callback_param = xspi; + reinit_completion(&xspi->dma_rx_c); + dmaengine_submit(desc); + dma_async_issue_pending(chan); + + timeout = wait_for_completion_timeout(&xspi->dma_rx_c, NXP_XSPI_DMA_TOUT); + if (!timeout) { + dev_err(xspi->dev, "AHB RX DMA timeout\n"); + dmaengine_terminate_sync(chan); + ret = -ETIMEDOUT; + } + +err_unmap_src: + dma_unmap_resource(chan->device->dev, src_addr, + op->data.nbytes, DMA_TO_DEVICE, 0); + +err_unmap_dst: + dma_unmap_single(chan->device->dev, dst_addr, + op->data.nbytes, DMA_FROM_DEVICE); + + return ret; +} + static int nxp_xspi_fill_txfifo(struct nxp_xspi *xspi, const struct spi_mem_op *op) { @@ -1016,10 +1116,21 @@ static int nxp_xspi_exec_op(struct spi_mem *mem, const struct spi_mem_op *op) * all use IP write. */ if ((op->data.dir == SPI_MEM_DATA_IN) && !needs_ip_only(xspi) - && ((op->addr.val + op->data.nbytes) <= xspi->memmap_phy_size)) - err = nxp_xspi_ahb_read(xspi, op); - else + && ((op->addr.val + op->data.nbytes) <= xspi->memmap_phy_size)) { + /* use DMA for transfers no less than ahb_buf_size, when DMA is available */ + if (xspi->dma_rx && op->data.nbytes >= xspi->devtype_data->ahb_buf_size) { + err = nxp_xspi_ahb_dma_read(xspi, op); + if (err) { + dev_warn_ratelimited(xspi->dev, + "DMA read failed (%d), falling back to CPU read\n", err); + err = nxp_xspi_ahb_read(xspi, op); + } + } else { + err = nxp_xspi_ahb_read(xspi, op); + } + } else { err = nxp_xspi_do_op(xspi, op); + } nxp_xspi_sw_reset(xspi); @@ -1029,6 +1140,7 @@ static int nxp_xspi_exec_op(struct spi_mem *mem, const struct spi_mem_op *op) static int nxp_xspi_adjust_op_size(struct spi_mem *mem, struct spi_mem_op *op) { struct nxp_xspi *xspi = spi_controller_get_devdata(mem->spi->controller); + int alignment = xspi->dma_copy_align; if (op->data.dir == SPI_MEM_DATA_OUT) { if (op->data.nbytes > xspi->devtype_data->txfifo) @@ -1042,6 +1154,22 @@ static int nxp_xspi_adjust_op_size(struct spi_mem *mem, struct spi_mem_op *op) if (!needs_ip_only(xspi) && (op->addr.val < xspi->memmap_phy_size) && ((op->addr.val + op->data.nbytes) > xspi->memmap_phy_size)) op->data.nbytes = xspi->memmap_phy_size - op->addr.val; + + /* + * For AHB DMA read, align the transfer to the DMA alignment + * boundaries. If the start address is unaligned, shorten this + * transfer so the next one starts on an aligned boundary. + * Otherwise, if the length is unaligned, round it down. + */ + if (xspi->dma_rx && !needs_ip_only(xspi) && + (op->data.nbytes >= xspi->devtype_data->ahb_buf_size)) { + if (op->addr.val & (alignment - 1)) + op->data.nbytes = + ALIGN(op->addr.val, alignment) - op->addr.val; + else if (op->data.nbytes & (alignment - 1)) + op->data.nbytes = + ALIGN_DOWN(op->data.nbytes, alignment); + } } return 0; @@ -1203,6 +1331,10 @@ static void nxp_xspi_cleanup(void *data) if (xspi->ahb_addr) iounmap(xspi->ahb_addr); + if (xspi->dma_rx) { + dmaengine_terminate_sync(xspi->dma_rx); + dma_release_channel(xspi->dma_rx); + } } static int nxp_xspi_probe(struct platform_device *pdev) @@ -1283,6 +1415,11 @@ static int nxp_xspi_probe(struct platform_device *pdev) if (ret) return ret; + /* DMA is optional, failure(other than -EPROBE_DEFER) falls back to CPU */ + ret = nxp_xspi_dma_init(dev, xspi); + if (ret == -EPROBE_DEFER) + return ret; + ctlr->bus_num = -1; ctlr->num_chipselect = NXP_XSPI_MAX_CHIPSELECT; ctlr->mem_ops = &nxp_xspi_mem_ops;