fix(gdma): treat burst size 0 and 1 as burst disabled

The GDMA layer used `max_data_burst_size == 0` as the only way to disable the
data burst. That conflicts with the upstream drivers' convention where a zeroed
config struct means "unset", so users had no way to ask for the driver default
burst size.

GDMA now treats both 0 and 1 as "no data burst": a single-beat burst has no
benefit over the non-burst mode. The MSPI alignment constraint under Flash
Encryption / PSRAM ECC still takes precedence and is reported with a warning.

The upstream drivers using GDMA now apply their own default burst size (16
bytes) when the user leaves `dma_burst_size` as 0, following the UHCI driver:

- esp_async_crc (AHB / AXI GDMA backend)
- esp_async_memcpy (AHB / AXI / LP-AHB / DW_GDMA backend)

Callers that really want no burst can now set `dma_burst_size` to 1.
This commit is contained in:
morris
2026-09-08 23:29:14 +08:00
parent c17ec317ef
commit f5afb99e3b
35 changed files with 203 additions and 111 deletions
@@ -52,7 +52,9 @@ typedef bool (*async_crc_isr_cb_t)(async_crc_handle_t crc_hdl, async_crc_event_d
typedef struct {
uint32_t backlog; /*!< Maximum number of pending CRC requests that can be queued per driver instance.
Higher values use more memory but provide better throughput for bursty workloads. */
size_t dma_burst_size; /*!< DMA transfer burst size, in bytes */
size_t dma_burst_size; /*!< DMA transfer burst size, in bytes, must be a power of 2.
Set to 0 to use the driver default.
Set to 1 to disable the data burst. */
uint32_t intr_priority; /*!< DMA interrupt priority. 0 means default low/medium priority. */
} async_crc_config_t;
@@ -52,7 +52,9 @@ typedef bool (*async_memcpy_isr_cb_t)(async_memcpy_handle_t mcp_hdl, async_memcp
typedef struct {
uint32_t backlog; /*!< Maximum number of transactions that can be prepared in the background */
uint32_t weight; /*!< Weight of async memcpy dma channel, higher weight means higher average bandwidth */
size_t dma_burst_size; /*!< DMA transfer burst size, in bytes */
size_t dma_burst_size; /*!< DMA transfer burst size, in bytes, must be a power of 2.
Set to 0 to use the driver default.
Set to 1 to disable the data burst. */
uint32_t flags; /*!< Extra flags to control async memcpy feature */
} async_memcpy_config_t;
@@ -160,7 +162,7 @@ esp_err_t esp_async_memcpy_install_dw_gdma(const async_memcpy_config_t *config,
* - ESP_FAIL: Install async memcpy driver failed because of other error
*/
esp_err_t esp_async_memcpy_install(const async_memcpy_config_t *config, async_memcpy_handle_t *mcp)
__attribute__((deprecated("Select a DMA backend explicitly with esp_async_memcpy_install_* instead")));
__attribute__((deprecated("Select a DMA backend explicitly with esp_async_memcpy_install_* instead")));
/** @endcond */
/**
@@ -201,8 +201,8 @@ esp_err_t gdma_disconnect(gdma_channel_handle_t dma_chan);
*/
typedef struct {
uint32_t max_data_burst_size; /*!< Set the max burst size when DMA read/write the data buffer.
Set to 0 means to disable the data burst.
Other values must be powers of 2 or supported by the selected GDMA bus. */
Set to 0 or 1 means to disable the data burst.
Other value must be power of 2 and supported by the DMA bus interface */
bool access_ext_mem; /*!< Set this if the DMA transfer will access external memory */
} gdma_transfer_config_t;
@@ -22,6 +22,8 @@ ESP_LOG_ATTR_TAG(TAG, "async_crc_gdma");
#define CRC_DMA_DESCRIPTOR_BUFFER_MAX_SIZE 4095
#define CRC_DMA_RX_SINK_BUFFER_SIZE 32
/// Default DMA burst size (in bytes), used when the user leaves `dma_burst_size` as 0
#define CRC_DMA_DEFAULT_BURST_SIZE 16
__attribute__((always_inline))
static inline uint32_t bit_reverse32(uint32_t val)
@@ -155,8 +157,9 @@ esp_err_t esp_async_crc_install_gdma_template(const async_crc_config_t *config,
gdma_apply_strategy(crc_gdma->rx_channel, &rx_strategy_cfg);
// Configure DMA transfer
// Note: 0 means "unset" in the config struct, fall back to the driver default burst size.
gdma_transfer_config_t transfer_cfg = {
.max_data_burst_size = config->dma_burst_size,
.max_data_burst_size = config->dma_burst_size ? config->dma_burst_size : CRC_DMA_DEFAULT_BURST_SIZE,
.access_ext_mem = true, // allow to copy data from external memory
};
ESP_GOTO_ON_ERROR(gdma_config_transfer(crc_gdma->tx_channel, &transfer_cfg), err, TAG, "config TX DMA transfer failed");
@@ -35,6 +35,9 @@ ESP_LOG_ATTR_TAG(TAG, "async_mcp.dw_gdma");
/// @brief Maximum body transfer width (in bits), capped by the AXI data width.
#define MCP_DW_GDMA_MAX_BODY_WIDTH_BITS 64
/// Default DMA burst size (in bytes), used when the user leaves `dma_burst_size` as 0
#define MCP_DW_GDMA_DEFAULT_BURST_SIZE 16
/// @brief Transaction object for async memcpy
typedef struct async_memcpy_transaction_t {
dw_gdma_link_list_handle_t link_list; // DW_GDMA link list for this transaction (body only)
@@ -184,7 +187,8 @@ esp_err_t esp_async_memcpy_install_dw_gdma(const async_memcpy_config_t *config,
portMUX_INITIALIZE(&mcp_dw_gdma->spin_lock);
atomic_init(&mcp_dw_gdma->fsm, MCP_FSM_IDLE);
mcp_dw_gdma->num_trans_objs = trans_queue_len;
mcp_dw_gdma->dma_burst_size = config->dma_burst_size;
// Note: 0 means "unset" in the config struct, fall back to the driver default burst size
mcp_dw_gdma->dma_burst_size = config->dma_burst_size ? config->dma_burst_size : MCP_DW_GDMA_DEFAULT_BURST_SIZE;
mcp_dw_gdma->parent.del = mcp_dw_gdma_del;
mcp_dw_gdma->parent.memcpy = mcp_dw_gdma_memcpy;
@@ -30,6 +30,8 @@
ESP_LOG_ATTR_TAG(TAG, "async_mcp.gdma");
#define MCP_DMA_DESCRIPTOR_BUFFER_MAX_SIZE 4095
/// Default DMA burst size (in bytes), used when the user leaves `dma_burst_size` as 0
#define MCP_GDMA_DEFAULT_BURST_SIZE 16
/// @brief Transaction object for async memcpy
typedef struct async_memcpy_transaction_t {
@@ -138,8 +140,10 @@ static esp_err_t esp_async_memcpy_install_gdma_template(const async_memcpy_confi
ESP_GOTO_ON_ERROR(gdma_set_weight(mcp_gdma->tx_channel, config->weight), err, TAG, "Set GDMA tx channel weight failed");
}
#endif
// Note: 0 means "unset" in the config struct, fall back to the driver default burst size.
// To disable the data burst explicitly, set `dma_burst_size` to 1.
gdma_transfer_config_t transfer_cfg = {
.max_data_burst_size = config->dma_burst_size,
.max_data_burst_size = config->dma_burst_size ? config->dma_burst_size : MCP_GDMA_DEFAULT_BURST_SIZE,
.access_ext_mem = true, // allow to do memory copy from/to external memory
};
ESP_GOTO_ON_ERROR(gdma_config_transfer(mcp_gdma->tx_channel, &transfer_cfg), err, TAG, "config transfer for tx channel failed");
+5 -6
View File
@@ -431,7 +431,8 @@ esp_err_t gdma_config_transfer(gdma_channel_handle_t dma_chan, const gdma_transf
if (config->access_ext_mem) {
#if (SOC_PSRAM_DMA_CAPABLE || SOC_DMA_CAN_ACCESS_FLASH) && SOC_AHB_GDMA_VERSION != 1
// Under Flash Encryption/PSRAM ECC, DMA must use MSPI-aligned bursts.
// Under Flash Encryption/PSRAM ECC, DMA must use MSPI-aligned bursts, so this hardware
// constraint takes precedence over a user requested burst disable.
size_t mspi_alignment = esp_mspi_get_alignment(NULL);
if (mspi_alignment > 1) {
if (max_data_burst_size < mspi_alignment) {
@@ -445,13 +446,11 @@ esp_err_t gdma_config_transfer(gdma_channel_handle_t dma_chan, const gdma_transf
TAG, "max_data_burst_size must not exceed %d when accessing external memory", GDMA_LL_MAX_BURST_SIZE_PSRAM);
#endif
}
if (max_data_burst_size) {
// treat 0 and 1 as "no burst": a single-beat burst has no benefit over the non-burst mode.
bool en_data_burst = max_data_burst_size > 1;
if (en_data_burst) {
ESP_RETURN_ON_FALSE(gdma_hal_check_burst_size(hal, max_data_burst_size), ESP_ERR_INVALID_ARG,
TAG, "invalid max_data_burst_size: %"PRIu32, max_data_burst_size);
}
bool en_data_burst = max_data_burst_size > 0;
if (en_data_burst) {
#if CONFIG_GDMA_ENABLE_WEIGHTED_ARBITRATION
// due to hardware limitation, if weighted arbitration is enabled, the data must be aligned to burst size
int_mem_alignment = MAX(int_mem_alignment, max_data_burst_size);
@@ -162,25 +162,20 @@ static void test_memory_copy_blocking(async_memcpy_handle_t driver)
.align = 16,
};
for (int i = 0; i < sizeof(test_buffer_size) / sizeof(test_buffer_size[0]); i++) {
// Test different align edge
for (int off = 0; off < 4; off++) {
test_context.buffer_size = test_buffer_size[i];
test_context.seed = i;
if (!gdma_test_mspi_strict_alignment_required()) {
test_context.src_offset = off;
test_context.dst_offset = off;
}
async_memcpy_setup_testbench(&test_context);
test_context.buffer_size = test_buffer_size[i];
test_context.seed = i;
async_memcpy_setup_testbench(&test_context);
TEST_ESP_OK(esp_memcpy_blocking(driver, test_context.to_addr, test_context.from_addr, test_context.copy_size, -1));
async_memcpy_verify_and_clear_testbench(test_context.copy_size, test_context.src_buf, test_context.dst_buf,
test_context.from_addr, test_context.to_addr);
}
TEST_ESP_OK(esp_memcpy_blocking(driver, test_context.to_addr, test_context.from_addr, test_context.copy_size, -1));
async_memcpy_verify_and_clear_testbench(test_context.copy_size, test_context.src_buf, test_context.dst_buf,
test_context.from_addr, test_context.to_addr);
}
}
TEST_CASE("memory copy by DMA (blocking)", "[async mcp]")
{
// Aligned copies with the driver default burst.
// Unaligned dest is covered by "memory copy with dest address unaligned" case.
async_memcpy_config_t config = {
.backlog = 1,
.dma_burst_size = 0,
@@ -247,63 +242,83 @@ TEST_CASE("memory copy by DMA (blocking)", "[async mcp]")
}
}
TEST_CASE("memory copy with dest address unaligned", "[async mcp]")
typedef esp_err_t (*test_mcp_install_fn)(const async_memcpy_config_t *config, async_memcpy_handle_t *mcp);
// SRAM can disable burst to cover the unaligned software path on chips whose RX
// burst requires dest alignment. PSRAM cannot: the external-memory block size
// (e.g. ESP32-S3 ext_mem_bk_size) is programmed together with the burst size and
// must match the cache line. Dest is cache-split so the DMA body stays aligned.
[[maybe_unused]] static void test_unaligned_dest_with_backend(const char *name, test_mcp_install_fn install, bool psram_capable)
{
[[maybe_unused]] async_memcpy_config_t driver_config = {
async_memcpy_config_t config = {
.backlog = 4,
.dma_burst_size = 32,
};
[[maybe_unused]] async_memcpy_handle_t driver = NULL;
async_memcpy_handle_t driver = NULL;
#if SOC_GDMA_SUPPORTED && (GDMA_LL_AHB_RX_BURST_NEEDS_ALIGNMENT || CONFIG_GDMA_ENABLE_WEIGHTED_ARBITRATION)
config.dma_burst_size = 1;
#endif
printf("Testing memcpy by %s\r\n", name);
TEST_ESP_OK(install(&config, &driver));
test_memcpy_with_dest_addr_unaligned(driver, false, false);
TEST_ESP_OK(esp_async_memcpy_uninstall(driver));
#if SOC_HAS(SPIRAM)
if (psram_capable) {
config.dma_burst_size = 32;
#if CONFIG_GDMA_ENABLE_WEIGHTED_ARBITRATION
// Weighted arbitration still needs every buffer aligned to the burst
// size, including the TX source body. Keep burst disabled there.
config.dma_burst_size = 1;
#endif
printf("Testing memcpy by %s (PSRAM)\r\n", name);
TEST_ESP_OK(install(&config, &driver));
test_memcpy_with_dest_addr_unaligned(driver, true, true);
TEST_ESP_OK(esp_async_memcpy_uninstall(driver));
}
#else
(void)psram_capable;
#endif // SOC_HAS(SPIRAM)
}
TEST_CASE("memory copy with dest address unaligned", "[async mcp]")
{
if (gdma_test_mspi_strict_alignment_required()) {
TEST_PASS_MESSAGE("MSPI strict alignment required (Flash Encryption / PSRAM ECC), skip this test");
}
#if SOC_CP_DMA_SUPPORTED
printf("Testing memcpy by CP DMA\r\n");
TEST_ESP_OK(esp_async_memcpy_install_cpdma(&driver_config, &driver));
test_memcpy_with_dest_addr_unaligned(driver, false, false);
TEST_ESP_OK(esp_async_memcpy_uninstall(driver));
test_unaligned_dest_with_backend("CP DMA", esp_async_memcpy_install_cpdma, false);
#endif // SOC_CP_DMA_SUPPORTED
#if SOC_HAS(AHB_GDMA) && !GDMA_LL_AHB_RX_BURST_NEEDS_ALIGNMENT && !CONFIG_GDMA_ENABLE_WEIGHTED_ARBITRATION
printf("Testing memcpy by AHB GDMA\r\n");
TEST_ESP_OK(esp_async_memcpy_install_gdma_ahb(&driver_config, &driver));
test_memcpy_with_dest_addr_unaligned(driver, false, false);
#if GDMA_LL_GET(AHB_PSRAM_CAPABLE) && SOC_HAS(SPIRAM)
test_memcpy_with_dest_addr_unaligned(driver, true, true);
#endif // GDMA_LL_GET(AHB_PSRAM_CAPABLE) && SOC_HAS(SPIRAM)
TEST_ESP_OK(esp_async_memcpy_uninstall(driver));
#if SOC_HAS(AHB_GDMA)
#if GDMA_LL_GET(AHB_PSRAM_CAPABLE)
test_unaligned_dest_with_backend("AHB GDMA", esp_async_memcpy_install_gdma_ahb, true);
#else
test_unaligned_dest_with_backend("AHB GDMA", esp_async_memcpy_install_gdma_ahb, false);
#endif
#endif // SOC_HAS(AHB_GDMA)
#if SOC_HAS(AXI_GDMA) && !CONFIG_GDMA_ENABLE_WEIGHTED_ARBITRATION
printf("Testing memcpy by AXI GDMA\r\n");
TEST_ESP_OK(esp_async_memcpy_install_gdma_axi(&driver_config, &driver));
test_memcpy_with_dest_addr_unaligned(driver, false, false);
#if GDMA_LL_GET(AXI_PSRAM_CAPABLE) && SOC_HAS(SPIRAM)
test_memcpy_with_dest_addr_unaligned(driver, true, true);
#endif // GDMA_LL_GET(AXI_PSRAM_CAPABLE) && SOC_HAS(SPIRAM)
TEST_ESP_OK(esp_async_memcpy_uninstall(driver));
#if SOC_HAS(AXI_GDMA)
#if GDMA_LL_GET(AXI_PSRAM_CAPABLE)
test_unaligned_dest_with_backend("AXI GDMA", esp_async_memcpy_install_gdma_axi, true);
#else
test_unaligned_dest_with_backend("AXI GDMA", esp_async_memcpy_install_gdma_axi, false);
#endif
#endif // SOC_HAS(AXI_GDMA)
#if SOC_HAS(LP_AHB_GDMA) && !CONFIG_GDMA_ENABLE_WEIGHTED_ARBITRATION
printf("Testing memcpy by LP AHB GDMA\r\n");
TEST_ESP_OK(esp_async_memcpy_install_gdma_lp_ahb(&driver_config, &driver));
test_memcpy_with_dest_addr_unaligned(driver, false, false);
#if GDMA_LL_GET(LP_AHB_PSRAM_CAPABLE) && SOC_HAS(SPIRAM)
test_memcpy_with_dest_addr_unaligned(driver, true, true);
#endif // GDMA_LL_GET(LP_AHB_PSRAM_CAPABLE) && SOC_HAS(SPIRAM)
TEST_ESP_OK(esp_async_memcpy_uninstall(driver));
#if SOC_HAS(LP_AHB_GDMA)
#if GDMA_LL_GET(LP_AHB_PSRAM_CAPABLE)
test_unaligned_dest_with_backend("LP AHB GDMA", esp_async_memcpy_install_gdma_lp_ahb, true);
#else
test_unaligned_dest_with_backend("LP AHB GDMA", esp_async_memcpy_install_gdma_lp_ahb, false);
#endif
#endif // SOC_HAS(LP_AHB_GDMA)
#if SOC_HAS(DW_GDMA)
printf("Testing memcpy by DW_GDMA\r\n");
TEST_ESP_OK(esp_async_memcpy_install_dw_gdma(&driver_config, &driver));
test_memcpy_with_dest_addr_unaligned(driver, false, false);
#if SOC_HAS(SPIRAM)
test_memcpy_with_dest_addr_unaligned(driver, true, true);
#endif // SOC_HAS(SPIRAM)
TEST_ESP_OK(esp_async_memcpy_uninstall(driver));
test_unaligned_dest_with_backend("DW_GDMA", esp_async_memcpy_install_dw_gdma, true);
#endif // SOC_HAS(DW_GDMA)
}
@@ -941,6 +941,13 @@ static void test_gdma_burst_size_validation(gdma_new_channel_func_t new_channel,
};
TEST_ESP_OK(gdma_config_transfer(tx_chan, &transfer_config));
// 0 and 1 both mean "disable data burst", and must be accepted even on chips
// whose hardware burst size is not programmable to 1 (e.g. S3: 16/32/64 only).
transfer_config.max_data_burst_size = 0;
TEST_ESP_OK(gdma_config_transfer(tx_chan, &transfer_config));
transfer_config.max_data_burst_size = 1;
TEST_ESP_OK(gdma_config_transfer(tx_chan, &transfer_config));
transfer_config.max_data_burst_size = 3;
TEST_ESP_ERR(ESP_ERR_INVALID_ARG, gdma_config_transfer(tx_chan, &transfer_config));
@@ -13,20 +13,22 @@ def get_flash_encryption_marks(target: str) -> tuple[pytest.MarkDecorator, ...]:
return (pytest.mark.flash_encryption,)
@pytest.mark.generic
def get_psram_marks(target: str) -> tuple[pytest.MarkDecorator, ...]:
if target == 'esp32s3':
return (pytest.mark.octal_psram,)
return (pytest.mark.generic,)
@pytest.mark.parametrize(
'config',
'config, target',
[
'release',
pytest.param('release', target, marks=get_psram_marks(target))
for target in soc_filtered_targets('SOC_GDMA_SUPPORTED == 1 or SOC_CP_DMA_SUPPORTED == 1')
],
indirect=True,
)
@idf_parametrize(
'target',
['esp32s2', 'esp32s31', 'esp32c2', 'esp32c3', 'esp32c5', 'esp32c6', 'esp32c61', 'esp32h2', 'esp32h4', 'esp32p4'],
indirect=['target'],
)
def test_dma(dut: Dut) -> None:
def test_gdma(dut: Dut) -> None:
dut.run_all_single_board_cases()
@@ -40,20 +42,7 @@ def test_dma(dut: Dut) -> None:
indirect=True,
)
@idf_parametrize('target', ['esp32p4'], indirect=['target'])
def test_dma_esp32p4_rev1(dut: Dut) -> None:
dut.run_all_single_board_cases()
@pytest.mark.octal_psram
@pytest.mark.parametrize(
'config',
[
'release',
],
indirect=True,
)
@idf_parametrize('target', ['esp32s3'], indirect=['target'])
def test_dma_psram(dut: Dut) -> None:
def test_gdma_esp32p4_rev1(dut: Dut) -> None:
dut.run_all_single_board_cases()
@@ -66,7 +55,7 @@ def test_dma_psram(dut: Dut) -> None:
indirect=True,
)
@idf_parametrize('target', soc_filtered_targets('SOC_GDMA_SUPPORT_WEIGHTED_ARBITRATION == 1'), indirect=['target'])
def test_dma_weighted_arbitration(dut: Dut) -> None:
def test_gdma_weighted_arbitration(dut: Dut) -> None:
dut.run_all_single_board_cases()
@@ -80,5 +69,5 @@ def test_dma_weighted_arbitration(dut: Dut) -> None:
],
indirect=True,
)
def test_dma_flash_encryption(dut: Dut) -> None:
def test_gdma_flash_encryption(dut: Dut) -> None:
dut.run_all_single_board_cases()