feat(sdmmc): Add support for multi-block read/writes

This has the potential of speeding up SD card access significantly.

Reusing the DMA aligned buffer gives the additional option of having to
allocate the transaction buffer only once instead of for every
transaction, while still keeping the improved transaction time.

To give users the maximum amount of control, the Kconfig option for the
transaction buffer size applies only to the temporary buffer, which is
only allocated if the DMA aligned buffer has not been pre-allocated.

Closes https://github.com/espressif/esp-idf/pull/17642

Co-authored-by: Adam Múdry <adam.mudry@espressif.com>
This commit is contained in:
jofrev
2026-03-16 13:33:03 +01:00
committed by Adam Múdry
co-authored by Adam Múdry
parent 642a2d2f54
commit 6c24854436
3 changed files with 106 additions and 36 deletions
+78 -35
View File
@@ -1,16 +1,16 @@
/*
* SPDX-FileCopyrightText: 2015-2025 Espressif Systems (Shanghai) CO LTD
* SPDX-FileCopyrightText: 2015-2026 Espressif Systems (Shanghai) CO LTD
*
* SPDX-License-Identifier: Apache-2.0
*/
#include <inttypes.h>
#include "freertos/FreeRTOS.h"
#include <sys/param.h> // for MIN/MAX
#include "esp_private/sdmmc_common.h"
static const char* TAG = "sdmmc_cmd";
esp_err_t sdmmc_send_cmd(sdmmc_card_t* card, sdmmc_command_t* cmd)
{
if (card->host.command_timeout_ms != 0) {
@@ -463,31 +463,52 @@ esp_err_t sdmmc_write_sectors(sdmmc_card_t* card, const void* src,
err = sdmmc_write_sectors_dma(card, src, start_block, block_count, block_size * block_count);
} else {
// SDMMC peripheral needs DMA-capable buffers. Split the write into
// separate single block writes, if needed, and allocate a temporary
// separate (multi) block writes, if needed, and allocate a temporary
// DMA-capable buffer.
void *tmp_buf = NULL;
size_t actual_size = 0;
// We don't want to force the allocation into SPIRAM, the allocator
// will decide based on the buffer size and memory availability.
tmp_buf = heap_caps_malloc(block_size, MALLOC_CAP_DMA);
if (!tmp_buf) {
ESP_LOGE(TAG, "%s: not enough mem, err=0x%x", __func__, ESP_ERR_NO_MEM);
return ESP_ERR_NO_MEM;
size_t blocks_per_write = MIN(CONFIG_SD_UNALIGNED_MULTI_BLOCK_RW_MAX_CHUNK_SIZE, block_count);
// prefer using DMA aligned buffer if available over allocating local temporary buffer
bool use_dma_aligned_buffer = (card->host.dma_aligned_buffer != NULL);
void* buf = use_dma_aligned_buffer ? card->host.dma_aligned_buffer : NULL;
// only allocate temporary buffer if we can't use the dma_aligned buffer
if (!use_dma_aligned_buffer) {
// We don't want to force the allocation into SPIRAM, the allocator
// will decide based on the buffer size and memory availability.
buf = heap_caps_malloc(block_size * blocks_per_write, MALLOC_CAP_DMA);
if (!buf) {
ESP_LOGE(TAG, "%s: not enough mem, err=0x%x", __func__, ESP_ERR_NO_MEM);
return ESP_ERR_NO_MEM;
}
}
size_t actual_size = heap_caps_get_allocated_size(buf);
blocks_per_write = actual_size / card->csd.sector_size;
// we should still respect the user configured maximum size
blocks_per_write = MIN(CONFIG_SD_UNALIGNED_MULTI_BLOCK_RW_MAX_CHUNK_SIZE, blocks_per_write);
if (blocks_per_write == 0) {
if (!use_dma_aligned_buffer) {
free(buf);
}
ESP_LOGE(TAG, "%s: buffer smaller than sector size: buf=%d, sector=%d", __func__, actual_size, card->csd.sector_size);
return ESP_ERR_INVALID_SIZE;
}
actual_size = heap_caps_get_allocated_size(tmp_buf);
const uint8_t* cur_src = (const uint8_t*) src;
for (size_t i = 0; i < block_count; ++i) {
memcpy(tmp_buf, cur_src, block_size);
cur_src += block_size;
err = sdmmc_write_sectors_dma(card, tmp_buf, start_block + i, 1, actual_size);
for (size_t i = 0; i < block_count; i += blocks_per_write) {
// make sure not to write more than the remaining blocks, i.e. block_count - i
blocks_per_write = MIN(blocks_per_write, (block_count - i));
memcpy(buf, cur_src, block_size * blocks_per_write);
cur_src += block_size * blocks_per_write;
err = sdmmc_write_sectors_dma(card, buf, start_block + i, blocks_per_write, actual_size);
if (err != ESP_OK) {
ESP_LOGD(TAG, "%s: error 0x%x writing block %d+%d",
__func__, err, start_block, i);
ESP_LOGD(TAG, "%s: error 0x%x writing blocks %d+[%d..%d]",
__func__, err, start_block, i, i + blocks_per_write - 1);
break;
}
}
free(tmp_buf);
if (!use_dma_aligned_buffer) {
free(buf);
}
}
return err;
}
@@ -601,33 +622,55 @@ esp_err_t sdmmc_read_sectors(sdmmc_card_t* card, void* dst,
err = sdmmc_read_sectors_dma(card, dst, start_block, block_count, block_size * block_count);
} else {
// SDMMC peripheral needs DMA-capable buffers. Split the read into
// separate single block reads, if needed, and allocate a temporary
// separate (multi) block reads, if needed, and allocate a temporary
// DMA-capable buffer.
void *tmp_buf = NULL;
size_t actual_size = 0;
tmp_buf = heap_caps_malloc(block_size, MALLOC_CAP_DMA);
if (!tmp_buf) {
ESP_LOGE(TAG, "%s: not enough mem, err=0x%x", __func__, ESP_ERR_NO_MEM);
return ESP_ERR_NO_MEM;
size_t blocks_per_read = MIN(CONFIG_SD_UNALIGNED_MULTI_BLOCK_RW_MAX_CHUNK_SIZE, block_count);
// prefer using DMA aligned buffer if available over allocating local temporary buffer
bool use_dma_aligned_buffer = (card->host.dma_aligned_buffer != NULL);
void* buf = use_dma_aligned_buffer ? card->host.dma_aligned_buffer : NULL;
// only allocate temporary buffer if we can't use the dma_aligned buffer
if (!use_dma_aligned_buffer) {
// We don't want to force the allocation into SPIRAM, the allocator
// will decide based on the buffer size and memory availability.
buf = heap_caps_malloc(block_size * blocks_per_read, MALLOC_CAP_DMA);
if (!buf) {
ESP_LOGE(TAG, "%s: not enough mem, err=0x%x", __func__, ESP_ERR_NO_MEM);
return ESP_ERR_NO_MEM;
}
}
size_t actual_size = heap_caps_get_allocated_size(buf);
blocks_per_read = actual_size / card->csd.sector_size;
// we should still respect the user configured maximum size
blocks_per_read = MIN(CONFIG_SD_UNALIGNED_MULTI_BLOCK_RW_MAX_CHUNK_SIZE, blocks_per_read);
if (blocks_per_read == 0) {
if (!use_dma_aligned_buffer) {
free(buf);
}
ESP_LOGE(TAG, "%s: buffer smaller than sector size: buf=%d, sector=%d", __func__, actual_size, card->csd.sector_size);
return ESP_ERR_INVALID_SIZE;
}
actual_size = heap_caps_get_allocated_size(tmp_buf);
uint8_t* cur_dst = (uint8_t*) dst;
for (size_t i = 0; i < block_count; ++i) {
err = sdmmc_read_sectors_dma(card, tmp_buf, start_block + i, 1, actual_size);
for (size_t i = 0; i < block_count; i += blocks_per_read) {
// make sure not to read more than the remaining blocks, i.e. block_count - i
blocks_per_read = MIN(blocks_per_read, (block_count - i));
err = sdmmc_read_sectors_dma(card, buf, start_block + i, blocks_per_read, actual_size);
if (err != ESP_OK) {
ESP_LOGD(TAG, "%s: error 0x%x writing block %d+%d",
__func__, err, start_block, i);
ESP_LOGD(TAG, "%s: error 0x%x reading blocks %d+[%d..%d]",
__func__, err, start_block, i, i + blocks_per_read - 1);
break;
}
memcpy(cur_dst, tmp_buf, block_size);
cur_dst += block_size;
memcpy(cur_dst, buf, block_size * blocks_per_read);
cur_dst += block_size * blocks_per_read;
}
if (!use_dma_aligned_buffer) {
free(buf);
}
free(tmp_buf);
}
return err;
}
esp_err_t sdmmc_read_sectors_dma(sdmmc_card_t* card, void* dst,
size_t start_block, size_t block_count, size_t buffer_len)
{