feat(uhci): uhci receive can be called in isr

(cherry picked from commit fcd5c55194)
This commit is contained in:
C.S.M
2026-08-10 14:50:17 +08:00
parent 6a9c44fe7e
commit 1709f4bbdf
9 changed files with 594 additions and 59 deletions
+121 -56
View File
@@ -16,6 +16,7 @@
#include "esp_attr.h"
#include "esp_log.h"
#include "esp_check.h"
#include "esp_macros.h"
#include "freertos/FreeRTOS.h"
#include "freertos/semphr.h"
#include "freertos/queue.h"
@@ -38,7 +39,7 @@
#include "esp_memory_utils.h"
#include "esp_cache.h"
static const char* TAG = "uhci";
#define TAG "uhci"
typedef struct uhci_platform_t {
_lock_t mutex; // platform level mutex lock.
@@ -114,7 +115,9 @@ static bool uhci_gdma_rx_callback_done(gdma_channel_handle_t dma_chan, gdma_even
return false;
}
if (event_data->flags.abnormal_eof || event_data->flags.normal_eof) {
const bool frame_end = event_data->flags.normal_eof || event_data->flags.abnormal_eof;
const bool rx_terminal = frame_end && !uhci_ctrl->rx_dir.continuous;
if (rx_terminal) {
// An EOF signal does not automatically stop the DMA transfer, so we need to stop it manually.
gdma_stop(uhci_ctrl->rx_dir.dma_chan);
// stop() cannot prevent already prefetched DMA descriptors from being processed.
@@ -122,50 +125,56 @@ static bool uhci_gdma_rx_callback_done(gdma_channel_handle_t dma_chan, gdma_even
gdma_reset(uhci_ctrl->rx_dir.dma_chan);
}
uhci_rx_event_data_t evt_data = {0};
if (!event_data->flags.abnormal_eof) {
const size_t cache_line = uhci_ctrl->rx_dir.cache_line;
size_t rx_size, sync_size;
if (!event_data->flags.normal_eof) {
rx_size = uhci_ctrl->rx_dir.buffer_size_per_desc_node[uhci_ctrl->rx_dir.node_index];
sync_size = rx_size;
} else {
rx_size = gdma_link_count_buffer_size_till_eof(uhci_ctrl->rx_dir.dma_link, uhci_ctrl->rx_dir.node_index);
sync_size = UHCI_ALIGN_UP(rx_size, cache_line); // round up to the next cache line
}
evt_data = (uhci_rx_event_data_t) {
.data = uhci_ctrl->rx_dir.buffer_pointers[uhci_ctrl->rx_dir.node_index],
.recv_size = rx_size,
.flags.totally_received = event_data->flags.normal_eof,
};
if (esp_ptr_external_ram(evt_data.data)) {
esp_psram_mspi_mb();
}
// DMA just finished writing the node's buffer. Because the descriptor link is circular,
// the same buffer region gets overwritten on every loop. On targets where the buffer is
// backed by a cache, the CPU must invalidate the range before reading, otherwise it will
// return stale data from a previous loop.
if (cache_line > 0) {
esp_cache_msync((void *)evt_data.data, sync_size, ESP_CACHE_MSYNC_FLAG_DIR_M2C);
}
const size_t cache_line = uhci_ctrl->rx_dir.cache_line;
size_t rx_size, sync_size;
if (!frame_end) {
rx_size = uhci_ctrl->rx_dir.buffer_size_per_desc_node[uhci_ctrl->rx_dir.node_index];
sync_size = rx_size;
} else {
rx_size = gdma_link_count_buffer_size_till_eof(uhci_ctrl->rx_dir.dma_link, uhci_ctrl->rx_dir.node_index);
// Round the invalidate size up to a full cache line. Each node buffer is itself cache-line
// aligned and a whole multiple of the cache line, so the extra bytes stay inside this same
// node buffer (never a neighbor) and only discard DMA scratch past the frame end.
sync_size = ESP_ALIGN_UP(rx_size, cache_line);
}
if (event_data->flags.abnormal_eof || event_data->flags.normal_eof) {
uhci_ctrl->rx_dir.node_index = 0;
uhci_rx_event_data_t evt_data = {
.data = uhci_ctrl->rx_dir.buffer_pointers[uhci_ctrl->rx_dir.node_index],
.recv_size = rx_size,
.flags.totally_received = frame_end,
};
if (esp_ptr_external_ram(evt_data.data)) {
esp_psram_mspi_mb();
}
// DMA just finished writing the node's buffer. Because the descriptor link is circular,
// the same buffer region gets overwritten on every loop. On targets where the buffer is
// backed by a cache, the CPU must invalidate the range before reading, otherwise it will
// return stale data from a previous loop.
if (cache_line > 0) {
esp_cache_msync((void *)evt_data.data, sync_size, ESP_CACHE_MSYNC_FLAG_DIR_M2C);
}
if (rx_terminal) {
// One-shot completion (or abnormal EOF): return to idle so uhci_receive() can re-arm and
// the controller can be deleted. Atomically claim the RUN->ENABLE transition; only the
// winner releases the PM lock, so a concurrent uhci_stop_receive() (possibly on another
// core) cannot double-release the single pm_lock shared with TX.
uhci_ctrl->rx_dir.node_index = 0;
uhci_rx_fsm_t expected = UHCI_RX_FSM_RUN;
if (atomic_compare_exchange_strong(&uhci_ctrl->rx_dir.rx_fsm, &expected, UHCI_RX_FSM_ENABLE)) {
#if CONFIG_PM_ENABLE
// release power manager lock
if (uhci_ctrl->pm_lock) {
esp_pm_lock_release(uhci_ctrl->pm_lock);
}
// release power manager lock
if (uhci_ctrl->pm_lock) {
esp_pm_lock_release(uhci_ctrl->pm_lock);
}
#endif
atomic_store(&uhci_ctrl->rx_dir.rx_fsm, UHCI_RX_FSM_ENABLE);
}
} else {
// A filled node (any mode) or a completed frame in continuous mode: advance to the next
// node of the circular link and keep the DMA running. In continuous mode the PM lock stays
// held until uhci_stop_receive().
uhci_ctrl->rx_dir.node_index++;
// Go back to 0 as its a circle descriptor link
if (uhci_ctrl->rx_dir.node_index >= uhci_ctrl->rx_dir.rx_num_dma_nodes) {
@@ -236,11 +245,6 @@ static esp_err_t uhci_gdma_initialize(uhci_controller_handle_t uhci_ctrl, const
ESP_RETURN_ON_ERROR(gdma_new_link_list(&dma_link_config, &uhci_ctrl->rx_dir.dma_link), TAG, "DMA rx link list alloc failed");
ESP_LOGD(TAG, "rx_dma node number is %d", uhci_ctrl->rx_dir.rx_num_dma_nodes);
uhci_ctrl->rx_dir.buffer_size_per_desc_node = heap_caps_calloc(uhci_ctrl->rx_dir.rx_num_dma_nodes, sizeof(*uhci_ctrl->rx_dir.buffer_size_per_desc_node), UHCI_MEM_ALLOC_CAPS);
ESP_RETURN_ON_FALSE(uhci_ctrl->rx_dir.buffer_size_per_desc_node, ESP_ERR_NO_MEM, TAG, "no memory for recording buffer size for desc node");
uhci_ctrl->rx_dir.buffer_pointers = heap_caps_calloc(uhci_ctrl->rx_dir.rx_num_dma_nodes, sizeof(*uhci_ctrl->rx_dir.buffer_pointers), UHCI_MEM_ALLOC_CAPS);
ESP_RETURN_ON_FALSE(uhci_ctrl->rx_dir.buffer_pointers, ESP_ERR_NO_MEM, TAG, "no memory for recording buffer pointers for desc node");
// Register callbacks
gdma_tx_event_callbacks_t tx_cbk = {
.on_trans_eof = uhci_gdma_tx_callback_eof,
@@ -309,13 +313,15 @@ static void uhci_do_transmit(uhci_controller_handle_t uhci_ctrl, uhci_transactio
gdma_start(uhci_ctrl->tx_dir.dma_chan, gdma_link_get_head_addr(uhci_ctrl->tx_dir.dma_link));
}
esp_err_t uhci_receive(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer, size_t buffer_size)
static esp_err_t uhci_receive_internal(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer, size_t buffer_size, bool continuous)
{
ESP_RETURN_ON_FALSE(uhci_ctrl, ESP_ERR_INVALID_ARG, TAG, "invalid argument");
ESP_RETURN_ON_FALSE(read_buffer != NULL && buffer_size > 0, ESP_ERR_INVALID_ARG, TAG, "read buffer null or buffer size is 0");
// Use the ISR-safe check variants: uhci_receive() is documented to be callable from the RX-done
// callback (ISR context), where the plain ESP_LOGE-based macros would take the log mutex.
ESP_RETURN_ON_FALSE_ISR(uhci_ctrl, ESP_ERR_INVALID_ARG, TAG, "invalid argument");
ESP_RETURN_ON_FALSE_ISR(read_buffer != NULL && buffer_size > 0, ESP_ERR_INVALID_ARG, TAG, "read buffer null or buffer size is 0");
uhci_rx_fsm_t expected_fsm = UHCI_RX_FSM_ENABLE;
ESP_RETURN_ON_FALSE(atomic_compare_exchange_strong(&uhci_ctrl->rx_dir.rx_fsm, &expected_fsm, UHCI_RX_FSM_RUN_WAIT), ESP_ERR_INVALID_STATE, TAG, "controller not in enable state");
ESP_RETURN_ON_FALSE_ISR(atomic_compare_exchange_strong(&uhci_ctrl->rx_dir.rx_fsm, &expected_fsm, UHCI_RX_FSM_RUN_WAIT), ESP_ERR_INVALID_STATE, TAG, "controller not in enable state");
esp_err_t ret = ESP_OK;
@@ -329,7 +335,7 @@ esp_err_t uhci_receive(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer,
uintptr_t aligned_address = ((uintptr_t)read_buffer + max_alignment_needed - 1) & ~(max_alignment_needed - 1);
size_t offset = aligned_address - (uintptr_t)read_buffer;
ESP_GOTO_ON_FALSE(buffer_size > offset, ESP_ERR_INVALID_ARG, err, TAG, "buffer size too small to align");
ESP_GOTO_ON_FALSE_ISR(buffer_size > offset, ESP_ERR_INVALID_ARG, err, TAG, "buffer size too small to align");
read_buffer = (uint8_t *)aligned_address;
buffer_size -= offset;
@@ -342,7 +348,9 @@ esp_err_t uhci_receive(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer,
size_t remaining_size = usable_size - (base_size * node_count);
{
gdma_buffer_mount_config_t mount_configs[node_count];
// Reuse the pre-allocated scratch array instead of a VLA: this function may run in ISR
// context, where a large node_count on the stack could overflow the small ISR stack.
gdma_buffer_mount_config_t *mount_configs = uhci_ctrl->rx_dir.mount_configs;
memset(mount_configs, 0, node_count * sizeof(gdma_buffer_mount_config_t));
for (size_t i = 0; i < node_count; i++) {
@@ -354,8 +362,8 @@ esp_err_t uhci_receive(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer,
} else {
uhci_ctrl->rx_dir.buffer_size_per_desc_node[i] = base_size;
}
ESP_GOTO_ON_FALSE(uhci_ctrl->rx_dir.buffer_size_per_desc_node[i] != 0 && uhci_ctrl->rx_dir.buffer_size_per_desc_node[i] <= DMA_DESCRIPTOR_BUFFER_MAX_SIZE,
ESP_ERR_INVALID_ARG, err, TAG, "buffer_size is too small or too large");
ESP_GOTO_ON_FALSE_ISR(uhci_ctrl->rx_dir.buffer_size_per_desc_node[i] != 0 && uhci_ctrl->rx_dir.buffer_size_per_desc_node[i] <= DMA_DESCRIPTOR_BUFFER_MAX_SIZE,
ESP_ERR_INVALID_ARG, err, TAG, "buffer_size is too small or too large");
size_t buffer_alignment = esp_ptr_internal(read_buffer) ? uhci_ctrl->rx_dir.int_mem_align : uhci_ctrl->rx_dir.ext_mem_align;
mount_configs[i] = (gdma_buffer_mount_config_t) {
@@ -366,18 +374,18 @@ esp_err_t uhci_receive(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer,
.mark_final = GDMA_FINAL_LINK_TO_DEFAULT,
}
};
ESP_LOGD(TAG, "The DMA node %d has %d byte", i, uhci_ctrl->rx_dir.buffer_size_per_desc_node[i]);
ESP_DRAM_LOGD(TAG, "The DMA node %d has %d byte", i, uhci_ctrl->rx_dir.buffer_size_per_desc_node[i]);
read_buffer += uhci_ctrl->rx_dir.buffer_size_per_desc_node[i];
}
ESP_GOTO_ON_ERROR(gdma_link_mount_buffers(uhci_ctrl->rx_dir.dma_link, 0, mount_configs, node_count, NULL), err, TAG, "DMA link mount buffers failed");
ESP_GOTO_ON_ERROR_ISR(gdma_link_mount_buffers(uhci_ctrl->rx_dir.dma_link, 0, mount_configs, node_count, NULL), err, TAG, "DMA link mount buffers failed");
// Invalidate cache before DMA starts to ensure no dirty cache lines.
// All DMA nodes (mount_configs) share the same contiguous user buffer, so checking mount_configs[0].buffer is sufficient.
bool need_cache_sync = esp_ptr_internal(mount_configs[0].buffer) ? (uhci_ctrl->int_mem_cache_line_size > 0) : (uhci_ctrl->ext_mem_cache_line_size > 0);
if (need_cache_sync) {
ESP_GOTO_ON_ERROR(esp_cache_msync(mount_configs[0].buffer, usable_size, ESP_CACHE_MSYNC_FLAG_DIR_M2C), err, TAG, "cache sync failed");
ESP_GOTO_ON_ERROR_ISR(esp_cache_msync(mount_configs[0].buffer, usable_size, ESP_CACHE_MSYNC_FLAG_DIR_M2C), err, TAG, "cache sync failed");
}
}
@@ -388,6 +396,7 @@ esp_err_t uhci_receive(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer,
}
#endif
uhci_ctrl->rx_dir.continuous = continuous;
atomic_store(&uhci_ctrl->rx_dir.rx_fsm, UHCI_RX_FSM_RUN);
gdma_reset(uhci_ctrl->rx_dir.dma_chan);
@@ -399,6 +408,49 @@ err:
return ret;
}
esp_err_t uhci_receive(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer, size_t buffer_size)
{
return uhci_receive_internal(uhci_ctrl, read_buffer, buffer_size, false);
}
esp_err_t uhci_start_receive_continuous(uhci_controller_handle_t uhci_ctrl, uint8_t *read_buffer, size_t buffer_size)
{
return uhci_receive_internal(uhci_ctrl, read_buffer, buffer_size, true);
}
esp_err_t uhci_stop_receive(uhci_controller_handle_t uhci_ctrl)
{
ESP_RETURN_ON_FALSE(uhci_ctrl, ESP_ERR_INVALID_ARG, TAG, "invalid argument");
// Atomically claim the RUN->ENABLE transition. If the RX EOF ISR already ended the session
// (state is ENABLE) or wins this race, we must not stop the DMA or release the shared PM lock
// again, otherwise the single pm_lock (shared with TX) would be double-released.
uhci_rx_fsm_t expected = UHCI_RX_FSM_RUN;
if (!atomic_compare_exchange_strong(&uhci_ctrl->rx_dir.rx_fsm, &expected, UHCI_RX_FSM_ENABLE)) {
// RUN_WAIT means a receive is concurrently being armed (e.g. re-armed from the RX-done ISR).
// Report it instead of silently returning ESP_OK, which would let that start win the race and
// keep the DMA running after the caller believes it stopped.
if (expected == UHCI_RX_FSM_RUN_WAIT) {
return ESP_ERR_INVALID_STATE;
}
return ESP_OK;
}
gdma_stop(uhci_ctrl->rx_dir.dma_chan);
gdma_reset(uhci_ctrl->rx_dir.dma_chan);
uhci_ctrl->rx_dir.node_index = 0;
uhci_ctrl->rx_dir.continuous = false;
#if CONFIG_PM_ENABLE
// In continuous mode the PM lock is held for the whole session; release it here.
if (uhci_ctrl->pm_lock) {
esp_pm_lock_release(uhci_ctrl->pm_lock);
}
#endif
return ESP_OK;
}
esp_err_t uhci_multi_buffer_transmit(uhci_controller_handle_t uhci_ctrl, const uhci_transmit_buffer_info_t *buffer_info_array, size_t array_size)
{
ESP_RETURN_ON_FALSE(uhci_ctrl, ESP_ERR_INVALID_ARG, TAG, "invalid argument");
@@ -523,6 +575,9 @@ esp_err_t uhci_del_controller(uhci_controller_handle_t uhci_ctrl)
if (uhci_ctrl->rx_dir.buffer_pointers) {
free(uhci_ctrl->rx_dir.buffer_pointers);
}
if (uhci_ctrl->rx_dir.mount_configs) {
heap_caps_free(uhci_ctrl->rx_dir.mount_configs);
}
#if CONFIG_PM_ENABLE
if (uhci_ctrl->pm_lock) {
@@ -624,6 +679,16 @@ esp_err_t uhci_new_controller(const uhci_controller_config_t *config, uhci_contr
ESP_GOTO_ON_ERROR(uhci_gdma_initialize(uhci_ctrl, config), err, TAG, "uhci gdma initialize failed");
// rx_num_dma_nodes is only known after uhci_gdma_initialize() queried the DMA alignment, so the
// per-node RX scratch arrays are allocated here (mirroring how tx_dir.mount_configs is allocated).
uhci_ctrl->rx_dir.buffer_size_per_desc_node = heap_caps_calloc(uhci_ctrl->rx_dir.rx_num_dma_nodes, sizeof(*uhci_ctrl->rx_dir.buffer_size_per_desc_node), UHCI_MEM_ALLOC_CAPS);
ESP_GOTO_ON_FALSE(uhci_ctrl->rx_dir.buffer_size_per_desc_node, ESP_ERR_NO_MEM, err, TAG, "no memory for recording buffer size for desc node");
uhci_ctrl->rx_dir.buffer_pointers = heap_caps_calloc(uhci_ctrl->rx_dir.rx_num_dma_nodes, sizeof(*uhci_ctrl->rx_dir.buffer_pointers), UHCI_MEM_ALLOC_CAPS);
ESP_GOTO_ON_FALSE(uhci_ctrl->rx_dir.buffer_pointers, ESP_ERR_NO_MEM, err, TAG, "no memory for recording buffer pointers for desc node");
// Pre-allocate the mount config scratch array so uhci_receive() never puts a VLA on the (small) ISR stack.
uhci_ctrl->rx_dir.mount_configs = heap_caps_calloc(uhci_ctrl->rx_dir.rx_num_dma_nodes, sizeof(gdma_buffer_mount_config_t), UHCI_MEM_ALLOC_CAPS);
ESP_GOTO_ON_FALSE(uhci_ctrl->rx_dir.mount_configs, ESP_ERR_NO_MEM, err, TAG, "no memory for rx buffer mount config array");
*ret_uhci_ctrl = uhci_ctrl;
return ESP_OK;
err:
@@ -22,7 +22,6 @@ extern "C" {
typedef struct uhci_controller_t uhci_controller_t;
#define UHCI_ALIGN_UP(num, align) (((num) + ((align) - 1)) & ~((align) - 1))
#define UHCI_MAX(a, b) (((a)>(b))?(a):(b))
#define UHCI_PM_LOCK_NAME_LEN_MAX 16
@@ -90,6 +89,8 @@ typedef struct {
size_t int_mem_align; // Alignment for internal memory
size_t ext_mem_align; // Alignment for external memory
size_t rx_num_dma_nodes; // rx dma number nodes
gdma_buffer_mount_config_t *mount_configs; // scratch array (capacity rx_num_dma_nodes) reused by every receive to mount buffer segments; avoids a VLA in ISR context
bool continuous; // continuous mode: keep DMA running across EOFs instead of stopping
} uhci_rx_dir;
struct uhci_controller_t {