From 77df9ad00b92a26ac434cacb9e8c3c850b76a6a5 Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Thu, 1 Oct 2026 15:02:31 +0200 Subject: [PATCH 1/8] misc: rp1-pio: name the DMA direction with a tx bool rp1_pio_sm_config_xfer_internal() tested dir == RP1_PIO_DIR_TO_SM in half a dozen places, which is both repetitive and awkward to read where the two arms of the resulting conditionals are not in the natural tx/rx order. Derive a tx bool once and use it throughout, putting the tx arm first consistently. Record it in struct dma_info as well, so that code holding only a dma_info does not have to recover the direction from the index it was looked up by, as rp1_pio_close() had to. No functional change. Signed-off-by: Hans Verkuil Assisted-by: Claude-Code:claude-opus-5 --- drivers/misc/rp1-pio.c | 22 +++++++++++++--------- 1 file changed, 13 insertions(+), 9 deletions(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index 1ade2a3abed0f..0a06144ea0684 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -101,6 +101,7 @@ struct dma_info { size_t buf_count; size_t burst_bytes; bool cyclic; + bool tx; unsigned int head_idx; unsigned int tail_idx; struct dma_buf_info bufs[DMA_BOUNCE_BUFFER_COUNT]; @@ -1024,6 +1025,7 @@ static void rp1_pio_sm_dma_free(struct dma_info *dma) dma_release_channel(dma->chan); dma->chan = NULL; dma->cyclic = false; + dma->tx = false; } static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint sm, uint dir, @@ -1037,6 +1039,7 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s struct dma_slave_caps dma_caps; struct dma_info *dma = NULL; bool cyclic = flags & RP1_PIO_SM_CONFIG_XFER_FL_DMA_CYCLE; + bool tx = dir == RP1_PIO_DIR_TO_SM; bool prefer_light_dma = flags & BIT(0); bool force_dma_type = flags & BIT(1); bool reconfigure = false; @@ -1077,7 +1080,7 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s /* Allocate and configure a DMA channel */ /* Careful - each SM FIFO has its own DREQ value */ - chan_name[0] = (dir == RP1_PIO_DIR_TO_SM) ? 't' : 'r'; + chan_name[0] = tx ? 't' : 'r'; chan_name[1] = 'x'; chan_name[2] = '0' + sm; if (prefer_light_dma) @@ -1087,6 +1090,7 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s chan_name[4] = '\0'; dma->cyclic = false; + dma->tx = tx; dma->chan = dma_request_chan(dev, chan_name); if (IS_ERR(dma->chan)) { ret = PTR_ERR(dma->chan); @@ -1143,18 +1147,18 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s fifo_addr = pio->phys_addr; fifo_addr += sm * (RP1_PIO_FIFO_TX1 - RP1_PIO_FIFO_TX0); - fifo_addr += (dir == RP1_PIO_DIR_TO_SM) ? RP1_PIO_FIFO_TX0 : RP1_PIO_FIFO_RX0; + fifo_addr += tx ? RP1_PIO_FIFO_TX0 : RP1_PIO_FIFO_RX0; config.src_addr_width = DMA_SLAVE_BUSWIDTH_4_BYTES; config.dst_addr_width = DMA_SLAVE_BUSWIDTH_4_BYTES; config.src_addr = fifo_addr; config.dst_addr = fifo_addr; - config.direction = (dir == RP1_PIO_DIR_TO_SM) ? DMA_MEM_TO_DEV : DMA_DEV_TO_MEM; + config.direction = tx ? DMA_MEM_TO_DEV : DMA_DEV_TO_MEM; dma_caps.max_burst = 4; dma_get_slave_caps(dma->chan, &dma_caps); if (dma_caps.max_burst > RP1_PIO_FIFO_DEPTH) dma_caps.max_burst = RP1_PIO_FIFO_DEPTH; - if (dir == RP1_PIO_DIR_TO_SM) + if (tx) config.dst_maxburst = dma_caps.max_burst; else config.src_maxburst = dma_caps.max_burst; @@ -1165,12 +1169,12 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s goto err_dma_free; set_dmactrl_args.sm = sm; - set_dmactrl_args.is_tx = (dir == RP1_PIO_DIR_TO_SM); - if (dir == RP1_PIO_DIR_FROM_SM) - set_dmactrl_args.ctrl = RP1_PIO_DMACTRL_DEFAULT | config.src_maxburst; - else + set_dmactrl_args.is_tx = tx; + if (tx) set_dmactrl_args.ctrl = RP1_PIO_DMACTRL_DEFAULT | (RP1_PIO_FIFO_DEPTH - config.dst_maxburst); + else + set_dmactrl_args.ctrl = RP1_PIO_DMACTRL_DEFAULT | config.src_maxburst; ret = rp1_pio_sm_set_dmactrl(client, &set_dmactrl_args); if (ret) @@ -1715,7 +1719,7 @@ void rp1_pio_close(struct rp1_pio_client *client) claimed &= ~mask; /* The SMs have been disabled, so this is safe */ - if ((i & 1) == RP1_PIO_DIR_FROM_SM) + if (!dma->tx) rp1_pio_sm_dma_flush_rx(dma); rp1_pio_sm_dma_free(dma); } From 35f081ca67cc79833864cf1da5f848b9200a72f3 Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Mon, 5 Oct 2026 13:44:20 +0200 Subject: [PATCH 2/8] misc: rp1-pio: guard against unreasonable buffer allocations Refuse total buffer sizes of > 512 MB. This also guards against overflows if userspace passes values like ~0. Also allow buf_count values > 4 for cyclic DMA: the limit of 4 is specific to non-cyclic DMA. Signed-off-by: Hans Verkuil --- drivers/misc/rp1-pio.c | 17 ++++++++++++++++- 1 file changed, 16 insertions(+), 1 deletion(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index 0a06144ea0684..dd2498b27acca 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -79,6 +79,13 @@ #define DMA_BOUNCE_BUFFER_SIZE 0x1000 #define DMA_BOUNCE_BUFFER_COUNT 4 +/* + * Upper bound on the buffer memory a transfer may ask the driver to + * allocate, i.e. on buf_size * buf_count, for cyclic and non-cyclic + * transfers alike. Larger requests are rejected with -EINVAL. + */ +#define RP1_PIO_MAX_TOTAL_BUF_SIZE (512 * 1024 * 1024) + struct dma_xfer_state { struct dma_info *dma; void (*callback)(void *param); @@ -1052,7 +1059,15 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s return -EINVAL; if ((buf_count || buf_size) && (!buf_size || (buf_size & 3) || - !buf_count || buf_count > DMA_BOUNCE_BUFFER_COUNT)) + !buf_count || (!cyclic && buf_count > DMA_BOUNCE_BUFFER_COUNT))) + return -EINVAL; + + /* + * Sanity check for insane amounts of buffer data. + * This also guards against potential overflow issues if either + * buf_size or buf_count are e.g. ~0. + */ + if (buf_count && buf_size > RP1_PIO_MAX_TOTAL_BUF_SIZE / buf_count) return -EINVAL; /* Cyclic DMA is currently only supported for FROM_SM */ if (cyclic && dir == RP1_PIO_DIR_TO_SM) From 9cff240d50a7dc0e0de0c1ffe8468e5c55a04593 Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Tue, 29 Sep 2026 10:58:02 +0200 Subject: [PATCH 3/8] misc: rp1-pio: ensure >= 2 buffers are available for cyclic DMA In the case of cyclic DMA it was possible to accept < 2 buffers or a zero buffer size. Check this and return -EINVAL in that case. Signed-off-by: Hans Verkuil Assisted-by: Claude-Code:claude-opus-5 --- drivers/misc/rp1-pio.c | 11 ++++++++--- 1 file changed, 8 insertions(+), 3 deletions(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index dd2498b27acca..a337569f5899a 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -1069,9 +1069,14 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s */ if (buf_count && buf_size > RP1_PIO_MAX_TOTAL_BUF_SIZE / buf_count) return -EINVAL; - /* Cyclic DMA is currently only supported for FROM_SM */ - if (cyclic && dir == RP1_PIO_DIR_TO_SM) - return -EINVAL; + if (cyclic) { + /* + * Cyclic DMA is currently not supported for TO_SM, and + * needs at least two real buffers to cycle through. + */ + if (tx || !buf_size || buf_count < 2) + return -EINVAL; + } dma_mask = 1 << (sm * 2 + dir); From deb325b131235beafed6ee16ff44a5f80f376e7c Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Tue, 6 Oct 2026 11:27:01 +0200 Subject: [PATCH 4/8] misc: rp1-pio: add RP1_PIO_CYCLIC_MIN_BUF_SIZE Set the minimum DMA buf_size to RP1_PIO_CYCLIC_MIN_BUF_SIZE. This ensures that a single period is larger than the FIFO + burst, which avoids some corner cases. Signed-off-by: Hans Verkuil --- drivers/misc/rp1-pio.c | 9 ++++++++- include/uapi/misc/rp1_pio_if.h | 1 + 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index a337569f5899a..09344f052ad5c 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -1070,11 +1070,18 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s if (buf_count && buf_size > RP1_PIO_MAX_TOTAL_BUF_SIZE / buf_count) return -EINVAL; if (cyclic) { + /* + * RP1_PIO_CYCLIC_MIN_BUF_SIZE is > FIFO + burst + margin, + * which ensures that you can't fit a period of data in just + * the FIFO + burst, which might cause weird behavior. + */ + if (buf_size < RP1_PIO_CYCLIC_MIN_BUF_SIZE) + return -EINVAL; /* * Cyclic DMA is currently not supported for TO_SM, and * needs at least two real buffers to cycle through. */ - if (tx || !buf_size || buf_count < 2) + if (tx || buf_count < 2) return -EINVAL; } diff --git a/include/uapi/misc/rp1_pio_if.h b/include/uapi/misc/rp1_pio_if.h index ee9e7b85b2a03..f6cb13f7eed78 100644 --- a/include/uapi/misc/rp1_pio_if.h +++ b/include/uapi/misc/rp1_pio_if.h @@ -188,6 +188,7 @@ struct rp1_pio_sm_config_xfer32_args { #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_FORCE_HEAVY 2 #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_FORCE_LIGHT 3 +#define RP1_PIO_CYCLIC_MIN_BUF_SIZE 128 #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_CYCLE (1 << 2) struct rp1_pio_sm_config_xfer_v2_args { From c8910fb7730fb2e018a7166ed085f35b6bae7195 Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Tue, 6 Oct 2026 13:28:25 +0200 Subject: [PATCH 5/8] misc: rp1-pio: buf_size was overwritten, which was confusing In rp1_pio_sm_config_xfer_internal() the buf_size argument was overwritten when calculating the aligned buffer sizes. Use a new variable for that rather than changing the argument. That was rather confusing. Signed-off-by: Hans Verkuil --- drivers/misc/rp1-pio.c | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index 09344f052ad5c..e2801fa04ab30 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -1134,12 +1134,12 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s dma->buf_size = buf_size * buf_count; dma->buf_count = 0; /* Round up the allocations */ - buf_size = ROUND_UP(dma->buf_size, PAGE_SIZE); + uint size = ROUND_UP(dma->buf_size, PAGE_SIZE); /* Alloc and map bounce buffer */ struct dma_buf_info *dbi = &dma->bufs[0]; - dbi->buf = dma_alloc_coherent(dma->chan->device->dev, buf_size, + dbi->buf = dma_alloc_coherent(dma->chan->device->dev, size, &dbi->dma_addr, GFP_KERNEL); if (!dbi->buf) { ret = -ENOMEM; @@ -1152,13 +1152,13 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s } else { dma->buf_size = buf_size; /* Round up the allocations */ - buf_size = ROUND_UP(buf_size, PAGE_SIZE); + uint size = ROUND_UP(buf_size, PAGE_SIZE); /* Alloc and map bounce buffers */ for (dma->buf_count = 0; dma->buf_count < buf_count; dma->buf_count++) { struct dma_buf_info *dbi = &dma->bufs[dma->buf_count]; - dbi->buf = dma_alloc_coherent(dma->chan->device->dev, buf_size, + dbi->buf = dma_alloc_coherent(dma->chan->device->dev, size, &dbi->dma_addr, GFP_KERNEL); if (!dbi->buf) { ret = -ENOMEM; From 82164016bdf179d3f902eba6b893886e28285dd8 Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Tue, 6 Oct 2026 13:13:16 +0200 Subject: [PATCH 6/8] misc: rp1-pio: better handling of corner cases Cyclic RX DMA support was missing some corner cases: - More than one period might have been DMAed when the callback is called: check the actual DMA status to detect this. If the dma status reports that it is no longer in progress, then return -EIO. - The head/tail indices wrap around at 2^32, but for a buf_count of e.g. 3 it will get out of sync when using % buf_count to get the actual buffer. When head/tail index >= rounddown(INT_MAX, buf_count) wrap it around. - copy_to_user can take a while, during which the buffer that's being copied might be overwritten. Detect such an overflow and report it. - Replace the dmaengine_terminate_all call with dmaengine_terminate_sync, this ensures all callback are finished. Signed-off-by: Hans Verkuil --- drivers/misc/rp1-pio.c | 114 ++++++++++++++++++++++++++++----- include/uapi/misc/rp1_pio_if.h | 15 ++++- 2 files changed, 111 insertions(+), 18 deletions(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index e2801fa04ab30..b8dc79c31bb3f 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -103,12 +103,18 @@ struct dma_buf_info { struct dma_info { struct semaphore buf_sem; + wait_queue_head_t cyclic_wait; struct dma_chan *chan; size_t buf_size; size_t buf_count; size_t burst_bytes; bool cyclic; bool tx; + dma_cookie_t cyclic_cookie; + spinlock_t cyclic_lock; + bool dma_not_running; + bool cyclic_overflow; + unsigned int wrap_around; unsigned int head_idx; unsigned int tail_idx; struct dma_buf_info bufs[DMA_BOUNCE_BUFFER_COUNT]; @@ -954,9 +960,47 @@ static void rp1_pio_sm_dma_callback(void *param) { struct dma_info *dma = param; - if (dma->cyclic) - WRITE_ONCE(dma->head_idx, dma->head_idx + 1); - up(&dma->buf_sem); + if (!dma->cyclic) { + up(&dma->buf_sem); + return; + } + + struct dma_tx_state state; + + int ret = dmaengine_tx_status(dma->chan, dma->cyclic_cookie, &state); + + if (ret != DMA_IN_PROGRESS && ret != DMA_PAUSED) { + spin_lock(&dma->cyclic_lock); + dma->dma_not_running = true; + spin_unlock(&dma->cyclic_lock); + wake_up_interruptible(&dma->cyclic_wait); + return; + } + + spin_lock(&dma->cyclic_lock); + unsigned int period = dma->buf_size / dma->buf_count; + size_t completed = dma->buf_size - state.residue; + unsigned int cur_slot = (completed / period) % dma->buf_count; + unsigned int head_slot = dma->head_idx % dma->buf_count; + + dma->head_idx += (cur_slot + dma->buf_count - head_slot) % dma->buf_count; + + if (dma->head_idx - dma->tail_idx >= dma->buf_count) { + /* + * The new head is more than buf_count ahead of tail_idx, + * so mark this as an overflow condition and set + * tail_idx to the first available valid buffer. + */ + dma->tail_idx = dma->head_idx - dma->buf_count + 1; + dma->cyclic_overflow = true; + } + if (dma->head_idx > dma->wrap_around && + dma->tail_idx > dma->wrap_around) { + dma->head_idx -= dma->wrap_around; + dma->tail_idx -= dma->wrap_around; + } + spin_unlock(&dma->cyclic_lock); + wake_up_interruptible(&dma->cyclic_wait); } static void rp1_pio_sm_kernel_dma_callback(void *param) @@ -1015,7 +1059,7 @@ static void rp1_pio_sm_dma_free(struct dma_info *dma) /* The buffers were allocated for the DMA controller, so free them there */ struct device *dma_dev = dma->chan->device->dev; - dmaengine_terminate_all(dma->chan); + dmaengine_terminate_sync(dma->chan); if (dma->cyclic) { dma->buf_count = 0; dma_free_coherent(dma_dev, ROUND_UP(dma->buf_size, PAGE_SIZE), @@ -1103,8 +1147,15 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s if (reconfigure) rp1_pio_sm_dma_free(dma); + /* Note: the semaphore is only used by non-cyclic DMA */ sema_init(&dma->buf_sem, 0); + if (!reconfigure) { + /* the waitqueue and spinlock are only used by cyclic DMA */ + init_waitqueue_head(&dma->cyclic_wait); + spin_lock_init(&dma->cyclic_lock); + } + /* Allocate and configure a DMA channel */ /* Careful - each SM FIFO has its own DREQ value */ chan_name[0] = tx ? 't' : 'r'; @@ -1117,6 +1168,8 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s chan_name[4] = '\0'; dma->cyclic = false; + dma->dma_not_running = false; + dma->cyclic_overflow = false; dma->tx = tx; dma->chan = dma_request_chan(dev, chan_name); if (IS_ERR(dma->chan)) { @@ -1149,6 +1202,7 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s sg_dma_address(&dbi->sgl) = dbi->dma_addr; dma->buf_count = buf_count; dma->cyclic = cyclic; + dma->wrap_around = rounddown(INT_MAX, buf_count); } else { dma->buf_size = buf_size; /* Round up the allocations */ @@ -1225,10 +1279,11 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s desc->callback = rp1_pio_sm_dma_callback; desc->callback_param = dma; - /* Submit the buffer - the callback will kick the semaphore */ + /* Submit the buffer - the callback will wake cyclic_wait */ ret = dmaengine_submit(desc); if (ret < 0) goto err_dma_free; + dma->cyclic_cookie = ret; dma_async_issue_pending(dma->chan); } @@ -1399,29 +1454,54 @@ static int rp1_pio_sm_rx_user(struct rp1_pio_device *pio, struct dma_info *dma, if (dma->cyclic) { size_t period = dma->buf_size / dma->buf_count; struct dma_buf_info *dbi = &dma->bufs[0]; + unsigned int tail_slot; if (bytes > period) return -EINVAL; +retry: /* - * The callback posts the semaphore once per period, so consume - * exactly one 'period' worth of data per read. + * Wait until there is new data, or the dma is no longer in + * progress, or there is an overflow. */ - if (down_interruptible(&dma->buf_sem)) - return -ERESTARTSYS; - - /* The DMA has wrapped onto the period we were about to read. */ - if (READ_ONCE(dma->head_idx) - dma->tail_idx > dma->buf_count) - return -EOVERFLOW; + ret = wait_event_interruptible(dma->cyclic_wait, + dma->dma_not_running || + dma->cyclic_overflow || + dma->head_idx > dma->tail_idx); + if (ret) + return ret; + + scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { + if (dma->dma_not_running) + return -EIO; + if (dma->cyclic_overflow) { + dma->cyclic_overflow = false; + return -EOVERFLOW; + } + /* Check that this condition still holds */ + if (dma->head_idx > dma->tail_idx) + tail_slot = dma->tail_idx % dma->buf_count; + else + goto retry; + } /* Pair with the DMAC's writes into the period we are about to copy. */ dma_rmb(); - if (copy_to_user(userbuf, - dbi->buf + (dma->tail_idx % dma->buf_count) * period, - bytes)) + if (copy_to_user(userbuf, dbi->buf + tail_slot * period, bytes)) return -EFAULT; - dma->tail_idx++; + scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { + if (dma->cyclic_overflow) { + /* + * copy_to_user can take so much time that + * the buffer is overwritten, so exit with + * -EOVERFLOW in that case. + */ + dma->cyclic_overflow = false; + return -EOVERFLOW; + } + dma->tail_idx++; + } return 0; } diff --git a/include/uapi/misc/rp1_pio_if.h b/include/uapi/misc/rp1_pio_if.h index f6cb13f7eed78..d967b3c17a7a1 100644 --- a/include/uapi/misc/rp1_pio_if.h +++ b/include/uapi/misc/rp1_pio_if.h @@ -188,8 +188,21 @@ struct rp1_pio_sm_config_xfer32_args { #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_FORCE_HEAVY 2 #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_FORCE_LIGHT 3 -#define RP1_PIO_CYCLIC_MIN_BUF_SIZE 128 +/* + * Use cyclic DMA: buf_size is the size of one period and buf_count is the + * number of periods. Each PIO_IOC_SM_XFER_DATA* call transfers at most one + * period and blocks until that period is available; a zero-byte transfer is + * rejected with -EINVAL. + * + * Currently only cyclic RX DMA is supported. + * + * A read returns -EOVERFLOW if the DMA wrapped onto the period that was about + * to be read. No data is returned in that case. Detection is best effort, but + * the stream resynchronises to the oldest still-valid period so that subsequent + * reads succeed again without needing to be reconfigured. + */ #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_CYCLE (1 << 2) +#define RP1_PIO_CYCLIC_MIN_BUF_SIZE 128 struct rp1_pio_sm_config_xfer_v2_args { uint16_t sm; From f71356f788a32c00965819ad054406038c269df6 Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Thu, 8 Oct 2026 09:30:56 +0200 Subject: [PATCH 7/8] misc: rp1-pio: split off cyclic part of rp1_pio_sm_rx_user The cyclic DMA handling of rp1_pio_sm_rx_user is split off into rp1_pio_sm_rx_user_cyclic, that's easier to read than having both the cyclic and non-cyclic implementations in the same function. Signed-off-by: Hans Verkuil --- drivers/misc/rp1-pio.c | 121 ++++++++++++++++++++++------------------- 1 file changed, 66 insertions(+), 55 deletions(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index b8dc79c31bb3f..4e8c31f1399cb 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -1435,75 +1435,85 @@ static int rp1_pio_sm_rx_submit(struct rp1_pio_device *pio, struct dma_info *dma return 0; } -/* - * A NULL userbuf queues a receive of exactly "bytes" and returns without - * waiting: it puts the app in control of its own read-ahead depth and - * chunk sizes, rather than us assuming every read is a full dma->buf_size - * chunk (reads can be any size up to buf_size, and need not match it). - */ -static int rp1_pio_sm_rx_user(struct rp1_pio_device *pio, struct dma_info *dma, - struct rp1_pio_sm_xfer_data32_args *args) +static int rp1_pio_sm_rx_user_cyclic(struct rp1_pio_device *pio, + struct dma_info *dma, + struct rp1_pio_sm_xfer_data32_args *args) { + size_t period = dma->buf_size / dma->buf_count; + struct dma_buf_info *dbi = &dma->bufs[0]; void __user *userbuf = args->data; size_t bytes = args->data_bytes; + unsigned int tail_slot; int ret; if (!bytes) return -EINVAL; - if (dma->cyclic) { - size_t period = dma->buf_size / dma->buf_count; - struct dma_buf_info *dbi = &dma->bufs[0]; - unsigned int tail_slot; - - if (bytes > period) - return -EINVAL; + if (bytes > period) + return -EINVAL; retry: - /* - * Wait until there is new data, or the dma is no longer in - * progress, or there is an overflow. - */ - ret = wait_event_interruptible(dma->cyclic_wait, - dma->dma_not_running || - dma->cyclic_overflow || - dma->head_idx > dma->tail_idx); - if (ret) - return ret; - - scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { - if (dma->dma_not_running) - return -EIO; - if (dma->cyclic_overflow) { - dma->cyclic_overflow = false; - return -EOVERFLOW; - } - /* Check that this condition still holds */ - if (dma->head_idx > dma->tail_idx) - tail_slot = dma->tail_idx % dma->buf_count; - else - goto retry; + /* + * Wait until there is new data, or the dma is no longer in + * progress, or there is an overflow. + */ + ret = wait_event_interruptible(dma->cyclic_wait, + dma->dma_not_running || + dma->cyclic_overflow || + dma->head_idx > dma->tail_idx); + if (ret) + return ret; + + scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { + if (dma->dma_not_running) + return -EIO; + if (dma->cyclic_overflow) { + dma->cyclic_overflow = false; + return -EOVERFLOW; } + /* Check that this condition still holds */ + if (dma->head_idx > dma->tail_idx) + tail_slot = dma->tail_idx % dma->buf_count; + else + goto retry; + } - /* Pair with the DMAC's writes into the period we are about to copy. */ - dma_rmb(); + /* Pair with the DMAC's writes into the period we are about to copy. */ + dma_rmb(); - if (copy_to_user(userbuf, dbi->buf + tail_slot * period, bytes)) - return -EFAULT; - scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { - if (dma->cyclic_overflow) { - /* - * copy_to_user can take so much time that - * the buffer is overwritten, so exit with - * -EOVERFLOW in that case. - */ - dma->cyclic_overflow = false; - return -EOVERFLOW; - } - dma->tail_idx++; + if (copy_to_user(userbuf, dbi->buf + tail_slot * period, bytes)) + return -EFAULT; + + scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { + if (dma->cyclic_overflow) { + /* + * copy_to_user can take so much time that + * the buffer is overwritten, so exit with + * -EOVERFLOW in that case. + */ + dma->cyclic_overflow = false; + return -EOVERFLOW; } - return 0; + dma->tail_idx++; } + return 0; +} + +/* + * A NULL userbuf queues a receive of exactly "bytes" and returns without + * waiting: it puts the app in control of its own read-ahead depth and + * chunk sizes, rather than us assuming every read is a full dma->buf_size + * chunk (reads can be any size up to buf_size, and need not match it). + */ +static int rp1_pio_sm_rx_user(struct rp1_pio_device *pio, struct dma_info *dma, + struct rp1_pio_sm_xfer_data32_args *args) +{ + void __user *userbuf = args->data; + size_t bytes = args->data_bytes; + int ret; + + if (!bytes) + return -EINVAL; if (!userbuf) { if (dma->head_idx - dma->tail_idx == dma->buf_count) @@ -1567,7 +1577,8 @@ static int rp1_pio_sm_xfer_data32_user(struct rp1_pio_client *client, void *para if (args->dir == RP1_PIO_DIR_TO_SM) return rp1_pio_sm_tx_user(pio, dma, args); else - return rp1_pio_sm_rx_user(pio, dma, args); + return dma->cyclic ? rp1_pio_sm_rx_user_cyclic(pio, dma, args) : + rp1_pio_sm_rx_user(pio, dma, args); } static int rp1_pio_sm_xfer_data_user(struct rp1_pio_client *client, void *param) From 829793114ff6f60761b8e14e2f7e7b5b7d9b5457 Mon Sep 17 00:00:00 2001 From: Hans Verkuil Date: Tue, 6 Oct 2026 15:24:01 +0200 Subject: [PATCH 8/8] misc: rp1-pio: add cyclic TX DMA support Add support for cyclic TX DMA. Signed-off-by: Hans Verkuil --- drivers/misc/rp1-pio.c | 111 ++++++++++++++++++++++++++++++--- include/uapi/misc/rp1_pio_if.h | 20 +++++- 2 files changed, 119 insertions(+), 12 deletions(-) diff --git a/drivers/misc/rp1-pio.c b/drivers/misc/rp1-pio.c index 4e8c31f1399cb..2dada1f540730 100644 --- a/drivers/misc/rp1-pio.c +++ b/drivers/misc/rp1-pio.c @@ -112,8 +112,10 @@ struct dma_info { bool tx; dma_cookie_t cyclic_cookie; spinlock_t cyclic_lock; + bool cyclic_tx_issued; bool dma_not_running; bool cyclic_overflow; + bool cyclic_underflow; unsigned int wrap_around; unsigned int head_idx; unsigned int tail_idx; @@ -985,7 +987,15 @@ static void rp1_pio_sm_dma_callback(void *param) dma->head_idx += (cur_slot + dma->buf_count - head_slot) % dma->buf_count; - if (dma->head_idx - dma->tail_idx >= dma->buf_count) { + if (dma->tx && dma->head_idx >= dma->tail_idx) { + /* + * We started writing from a buffer that has stale data. + * Mark this as underflow. Increment tail_idx to head_idx + 2, + * which gives userspace the chance to write new data. + */ + dma->tail_idx = dma->head_idx + 2; + dma->cyclic_underflow = true; + } else if (!dma->tx && dma->head_idx - dma->tail_idx >= dma->buf_count) { /* * The new head is more than buf_count ahead of tail_idx, * so mark this as an overflow condition and set @@ -1122,10 +1132,10 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s if (buf_size < RP1_PIO_CYCLIC_MIN_BUF_SIZE) return -EINVAL; /* - * Cyclic DMA is currently not supported for TO_SM, and - * needs at least two real buffers to cycle through. + * Cyclic DMA needs at least two (RX) or three (TX) real + * buffers to cycle through. */ - if (tx || buf_count < 2) + if (buf_count < (tx ? 3 : 2)) return -EINVAL; } @@ -1170,6 +1180,8 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s dma->cyclic = false; dma->dma_not_running = false; dma->cyclic_overflow = false; + dma->cyclic_underflow = false; + dma->cyclic_tx_issued = false; dma->tx = tx; dma->chan = dma_request_chan(dev, chan_name); if (IS_ERR(dma->chan)) { @@ -1267,9 +1279,9 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s sg_dma_len(&dbi->sgl) = dma->buf_size; desc = dmaengine_prep_dma_cyclic(dma->chan, dbi->dma_addr, - dma->buf_size, dma->buf_size / dma->buf_count, - DMA_DEV_TO_MEM, - DMA_PREP_INTERRUPT | DMA_CTRL_ACK); + dma->buf_size, dma->buf_size / dma->buf_count, + tx ? DMA_MEM_TO_DEV : DMA_DEV_TO_MEM, + DMA_PREP_INTERRUPT | DMA_CTRL_ACK); if (!desc) { dev_err(dev, "DMA preparation failed\n"); ret = -EIO; @@ -1285,7 +1297,8 @@ static int rp1_pio_sm_config_xfer_internal(struct rp1_pio_client *client, uint s goto err_dma_free; dma->cyclic_cookie = ret; - dma_async_issue_pending(dma->chan); + if (!tx) + dma_async_issue_pending(dma->chan); } return 0; @@ -1340,6 +1353,85 @@ static void rp1_pio_sm_xfer_progress(struct rp1_pio_sm_xfer_data32_args *args, args->data_bytes -= bytes; } +static int rp1_pio_sm_tx_user_cyclic(struct rp1_pio_device *pio, + struct dma_info *dma, + struct rp1_pio_sm_xfer_data32_args *args) +{ + size_t period = dma->buf_size / dma->buf_count; + struct dma_buf_info *dbi = &dma->bufs[0]; + const void __user *userbuf = args->data; + size_t bytes = args->data_bytes; + unsigned int tail_idx; + void *dst; + int ret; + + if (!bytes || bytes > period) + return -EINVAL; + + if (!dma->cyclic_tx_issued) { + unsigned int tail_slot = dma->tail_idx % dma->buf_count; + + dst = dbi->buf + tail_slot * period; + if (copy_from_user(dst, userbuf, bytes)) + return -EFAULT; + memset(dst + bytes, 0, period - bytes); + scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { + dma->tail_idx++; + if (dma->tail_idx == dma->buf_count) { + /* All buffers are filled, start DMA */ + dma->cyclic_tx_issued = true; + dma_async_issue_pending(dma->chan); + } + } + return 0; + } + +retry: + /* + * Wait until a free slot is available, or the dma is no longer + * in progress, or there is an underflow. + */ + ret = wait_event_interruptible(dma->cyclic_wait, + dma->dma_not_running || + dma->cyclic_underflow || + dma->tail_idx - dma->head_idx < dma->buf_count); + if (ret) + return ret; + + scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { + if (dma->dma_not_running) + return -EIO; + if (dma->cyclic_underflow) { + dma->cyclic_underflow = false; + return -EPIPE; + } + /* Check that this condition still holds */ + if (dma->tail_idx - dma->head_idx < dma->buf_count) + tail_idx = dma->tail_idx; + else + goto retry; + } + + dst = dbi->buf + (tail_idx % dma->buf_count) * period; + if (copy_from_user(dst, userbuf, bytes)) + return -EFAULT; + memset(dst + bytes, 0, period - bytes); + + scoped_guard(spinlock_irqsave, &dma->cyclic_lock) { + if (dma->cyclic_underflow) { + /* + * copy_from_user can take so much time that + * the buffer was DMAed before it was fully + * written, so exit with -EPIPE in that case. + */ + dma->cyclic_underflow = false; + return -EPIPE; + } + dma->tail_idx++; + } + return 0; +} + static int rp1_pio_sm_tx_user(struct rp1_pio_device *pio, struct dma_info *dma, struct rp1_pio_sm_xfer_data32_args *args) { @@ -1575,7 +1667,8 @@ static int rp1_pio_sm_xfer_data32_user(struct rp1_pio_client *client, void *para return -EINVAL; if (args->dir == RP1_PIO_DIR_TO_SM) - return rp1_pio_sm_tx_user(pio, dma, args); + return dma->cyclic ? rp1_pio_sm_tx_user_cyclic(pio, dma, args) : + rp1_pio_sm_tx_user(pio, dma, args); else return dma->cyclic ? rp1_pio_sm_rx_user_cyclic(pio, dma, args) : rp1_pio_sm_rx_user(pio, dma, args); diff --git a/include/uapi/misc/rp1_pio_if.h b/include/uapi/misc/rp1_pio_if.h index d967b3c17a7a1..7b7c173d028a6 100644 --- a/include/uapi/misc/rp1_pio_if.h +++ b/include/uapi/misc/rp1_pio_if.h @@ -189,17 +189,31 @@ struct rp1_pio_sm_config_xfer32_args { #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_FORCE_LIGHT 3 /* - * Use cyclic DMA: buf_size is the size of one period and buf_count is the - * number of periods. Each PIO_IOC_SM_XFER_DATA* call transfers at most one + * How to use cyclic DMA: buf_size is the size of one period and buf_count is + * the number of periods. Each PIO_IOC_SM_XFER_DATA* call transfers at most one * period and blocks until that period is available; a zero-byte transfer is * rejected with -EINVAL. * - * Currently only cyclic RX DMA is supported. + * Input accepts any buf_count of two or more. Output requires at least three + * periods. + * + * The minimum value for buf_size is RP1_PIO_CYCLIC_MIN_BUF_SIZE. This ensures + * that the size of each period is more than the PIO FIFO size plus DMA burst + * plus margin. * * A read returns -EOVERFLOW if the DMA wrapped onto the period that was about * to be read. No data is returned in that case. Detection is best effort, but * the stream resynchronises to the oldest still-valid period so that subsequent * reads succeed again without needing to be reconfigured. + * + * For output a write returns -EPIPE if the DMA caught up and resent stale data. + * Detection is best effort, but the stream resynchronises to the first available + * period that has not yet been transmitted so that subsequent writes succeed + * again without needing to be reconfigured. A failed user copy does not + * advance the writer or start DMA. A write shorter than a period zeroes the + * remainder of that period. + * + * All buffers must be filled before the output DMA will start. */ #define RP1_PIO_SM_CONFIG_XFER_FL_DMA_CYCLE (1 << 2) #define RP1_PIO_CYCLIC_MIN_BUF_SIZE 128