Merge branch 'bugfix/ppa_srm_stuck_on_dma_2' into 'master'

fix(ppa): fix for SRM operation potential block if do 90/270 degree rotation

Closes IDFGH-18204 and IDFGH-17870

See merge request espressif/esp-idf!49657
This commit is contained in:
morris
2026-09-09 11:55:27 +08:00
2 changed files with 103 additions and 55 deletions
+50 -15
View File
@@ -115,7 +115,13 @@ bool ppa_srm_transaction_on_picked(uint32_t num_chans, const dma2d_trans_channel
// Configure the block size to be received by the SRM engine, which is passed from the 2D-DMA TX channel (i.e. 2D-DMA dscr-port mode)
uint32_t block_h = 0, block_v = 0;
ppa_ll_srm_get_dma_dscr_port_mode_block_size(platform->hal.dev, ppa_in_color_mode, ppa_ll_srm_get_mb_size(platform->hal.dev), &block_h, &block_v);
ppa_ll_srm_mb_size_t mb_size = ppa_ll_srm_get_mb_size(platform->hal.dev);
#if CONFIG_ESP32P4_SELECTS_REV_LESS_V3
assert(mb_size == PPA_LL_SRM_MB_SIZE_16_16);
#else
assert(mb_size == PPA_LL_SRM_MB_SIZE_32_32);
#endif
ppa_ll_srm_get_dma_dscr_port_mode_block_size(platform->hal.dev, ppa_in_color_mode, mb_size, &block_h, &block_v);
dma2d_dscr_port_mode_config_t dma_dscr_port_mode_config = {
.block_h = block_h,
.block_v = block_v,
@@ -156,18 +162,38 @@ bool ppa_srm_transaction_on_picked(uint32_t num_chans, const dma2d_trans_channel
#if CONFIG_IDF_TARGET_ESP32P4
// Hardware bug workaround (DIG-734)
uint32_t w_out = srm_trans_desc->in.block_w * srm_trans_desc->scale_x_int + srm_trans_desc->in.block_w * srm_trans_desc->scale_x_frag / PPA_LL_SRM_SCALING_FRAG_MAX;
// Leftover block data size less than DMA FIFO depth could cause DMA to miss the counting of such batch,
// so DMA never thinks it has received all the data, and will not raise EOF/DONE interrupt.
// The workaround is to bypass macro block order, so no such small leftover blocks.
uint32_t in_block_w, in_block_h;
uint32_t scale_x_int, scale_x_frag, scale_y_int, scale_y_frag;
if (srm_trans_desc->rotation_angle == PPA_SRM_ROTATION_ANGLE_90 || srm_trans_desc->rotation_angle == PPA_SRM_ROTATION_ANGLE_270) {
in_block_w = srm_trans_desc->in.block_h;
in_block_h = srm_trans_desc->in.block_w;
scale_x_int = srm_trans_desc->scale_y_int;
scale_x_frag = srm_trans_desc->scale_y_frag;
scale_y_int = srm_trans_desc->scale_x_int;
scale_y_frag = srm_trans_desc->scale_x_frag;
} else {
in_block_w = srm_trans_desc->in.block_w;
in_block_h = srm_trans_desc->in.block_h;
scale_x_int = srm_trans_desc->scale_x_int;
scale_x_frag = srm_trans_desc->scale_x_frag;
scale_y_int = srm_trans_desc->scale_y_int;
scale_y_frag = srm_trans_desc->scale_y_frag;
}
uint32_t w_out = in_block_w * scale_x_int + in_block_w * scale_x_frag / PPA_LL_SRM_SCALING_FRAG_MAX;
uint32_t w_divisor = (ppa_out_color_mode == PPA_SRM_COLOR_MODE_ARGB8888 || ppa_out_color_mode == PPA_SRM_COLOR_MODE_RGB888) ? 32 : 64;
uint32_t w_left = w_out % w_divisor;
w_left = (w_left == 0) ? w_divisor : w_left;
uint32_t h_mb = (ppa_ll_srm_get_mb_size(platform->hal.dev) == PPA_LL_SRM_MB_SIZE_16_16) ? 16 : 32;
uint32_t h_in_left = srm_trans_desc->in.block_h % h_mb;
uint32_t h_mb = (mb_size == PPA_LL_SRM_MB_SIZE_16_16) ? 16 : 32;
uint32_t h_in_left = in_block_h % h_mb;
h_in_left = (h_in_left == 0) ? h_mb : h_in_left;
uint32_t h_left = h_in_left * srm_trans_desc->scale_y_int + h_in_left * srm_trans_desc->scale_y_frag / PPA_LL_SRM_SCALING_FRAG_MAX;
uint32_t h_left = h_in_left * scale_y_int + h_in_left * scale_y_frag / PPA_LL_SRM_SCALING_FRAG_MAX;
const uint32_t dma2d_fifo_depth_bits = 12 * 128;
uint32_t out_pixel_depth = color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)ppa_out_color_mode);
bool bypass_mb_order = false;
if (((w_out > w_divisor) || (srm_trans_desc->in.block_h > h_mb)) && // will be cut into more than one trans unit
if (((w_out > w_divisor) || (in_block_h > h_mb)) && // will be cut into more than one trans unit
((w_left * h_left * out_pixel_depth) < dma2d_fifo_depth_bits)
) {
bypass_mb_order = true;
@@ -229,6 +255,18 @@ esp_err_t ppa_do_scale_rotate_mirror(ppa_client_handle_t ppa_client, const ppa_s
uint32_t scale_x_frag = (uint32_t)(config->scale_x * PPA_LL_SRM_SCALING_FRAG_MAX) & (PPA_LL_SRM_SCALING_FRAG_MAX - 1);
uint32_t scale_y_int = (uint32_t)config->scale_y;
uint32_t scale_y_frag = (uint32_t)(config->scale_y * PPA_LL_SRM_SCALING_FRAG_MAX) & (PPA_LL_SRM_SCALING_FRAG_MAX - 1);
// SRM processes in blocks. Block x/(y) (including leftover block) cannot be scaled to odd number when YUV422/YUV420 is the output color mode
// When block size is 16x16, odd number is possible, so needs to make them even
// When block size is 32x32, calculated frag values for full macro blocks are always even
#if CONFIG_ESP32P4_SELECTS_REV_LESS_V3
// macro block size is 16x16, will do sanity check in ppa_srm_transaction_on_picked
if (config->out.srm_cm == PPA_SRM_COLOR_MODE_YUV420) {
scale_x_frag = scale_x_frag & ~1;
scale_y_frag = scale_y_frag & ~1;
} else if (PPA_IS_CM_YUV422(config->out.srm_cm)) {
scale_x_frag = scale_x_frag & ~1;
}
#endif
uint32_t new_block_w = 0;
uint32_t new_block_h = 0;
if (config->rotation_angle == PPA_SRM_ROTATION_ANGLE_0 || config->rotation_angle == PPA_SRM_ROTATION_ANGLE_180) {
@@ -244,6 +282,12 @@ esp_err_t ppa_do_scale_rotate_mirror(ppa_client_handle_t ppa_client, const ppa_s
config->out.block_offset_y < config->out.pic_h &&
new_block_h <= (config->out.pic_h - config->out.block_offset_y),
ESP_ERR_INVALID_ARG, TAG, "scale does not fit in the out pic");
// Check for the leftover block w/(h) to be even in YUV output color mode
if (config->out.srm_cm == PPA_SRM_COLOR_MODE_YUV420) {
ESP_RETURN_ON_FALSE(new_block_w % 2 == 0 && new_block_h % 2 == 0, ESP_ERR_INVALID_ARG, TAG, "YUV420 output does not support scaled block w/h to be odd");
} else if (PPA_IS_CM_YUV422(config->out.srm_cm)) {
ESP_RETURN_ON_FALSE(new_block_w % 2 == 0, ESP_ERR_INVALID_ARG, TAG, "YUV422 output does not support scaled block w to be odd");
}
if (!ppa_check_buffer_alignment(ppa_client, &config->in, true, config->in.block_w) ||
!ppa_check_buffer_alignment(ppa_client, &config->out, false, new_block_w)) {
@@ -299,15 +343,6 @@ esp_err_t ppa_do_scale_rotate_mirror(ppa_client_handle_t ppa_client, const ppa_s
srm_trans_desc->scale_x_frag = scale_x_frag;
srm_trans_desc->scale_y_int = scale_y_int;
srm_trans_desc->scale_y_frag = scale_y_frag;
// SRM processes in blocks. Block x/(y) cannot be scaled to odd number when YUV422/YUV420 is the output color mode
// When block size is 16x16, odd number is possible, so needs to make them even
// When block size is 32x32, calculated frag values are always even
if (config->out.srm_cm == PPA_SRM_COLOR_MODE_YUV420) {
srm_trans_desc->scale_x_frag = srm_trans_desc->scale_x_frag & ~1;
srm_trans_desc->scale_y_frag = srm_trans_desc->scale_y_frag & ~1;
} else if (PPA_IS_CM_YUV422(config->out.srm_cm)) {
srm_trans_desc->scale_x_frag = srm_trans_desc->scale_x_frag & ~1;
}
srm_trans_desc->alpha_value = new_alpha_value;
srm_trans_desc->data_burst_length = ppa_client->data_burst_length;
@@ -1030,12 +1030,20 @@ TEST_CASE("ppa_srm_stress_test", "[PPA]")
const uint32_t h = 200;
const ppa_srm_color_mode_t in_cm = PPA_SRM_COLOR_MODE_RGB565;
const ppa_srm_color_mode_t out_cm = PPA_SRM_COLOR_MODE_RGB565;
const ppa_srm_rotation_angle_t rotation = PPA_SRM_ROTATION_ANGLE_0;
const float scale_x = 1.0;
const float scale_y = 1.0;
const ppa_srm_rotation_angle_t rotations[] = {
PPA_SRM_ROTATION_ANGLE_0,
PPA_SRM_ROTATION_ANGLE_90,
};
const float scale_pairs[][2] = {
{1.0f, 1.0f},
{1.0f, 1.5f},
{1.2f, 1.0f},
};
const uint32_t out_w = 2 * w; // leave a large output buffer, since we test with >1.0 scale
const uint32_t out_h = 2 * h;
uint32_t in_buf_size = w * h * color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)in_cm) / 8;
uint32_t out_buf_size = ESP_ALIGN_UP(w * h * color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)out_cm) / 8, 64);
uint32_t out_buf_size = ESP_ALIGN_UP(out_w * out_h * color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)out_cm) / 8, 64);
uint8_t *out_buf = static_cast<uint8_t *>(heap_caps_aligned_calloc(4, out_buf_size, sizeof(uint8_t), MALLOC_CAP_SPIRAM | MALLOC_CAP_DMA));
TEST_ASSERT_NOT_NULL(out_buf);
uint8_t *in_buf = static_cast<uint8_t *>(heap_caps_aligned_calloc(4, in_buf_size, sizeof(uint8_t), MALLOC_CAP_INTERNAL | MALLOC_CAP_8BIT | MALLOC_CAP_DMA));
@@ -1047,43 +1055,48 @@ TEST_CASE("ppa_srm_stress_test", "[PPA]")
ppa_client_config.max_pending_trans_num = 1;
TEST_ESP_OK(ppa_register_client(&ppa_client_config, &ppa_client_handle));
// Test on different sizes of the block
int test_iterations = 50;
while (test_iterations-- > 0) {
uint32_t block_w_initial = esp_random() % (w - 100);
uint32_t block_h_initial = esp_random() % (h - 100);
block_w_initial = (block_w_initial == 0) ? 1 : block_w_initial;
block_h_initial = (block_h_initial == 0) ? 1 : block_h_initial;
uint32_t block_w = 0;
uint32_t block_h = 0;
for (int i = 0; i < 100; i++) {
block_w = block_w_initial + i;
block_h = block_h_initial + i;
// printf("block_w = %ld, block_h = %ld\n", block_w, block_h);
ppa_srm_oper_config_t oper_config = {};
oper_config.in.buffer = in_buf;
oper_config.in.pic_w = w;
oper_config.in.pic_h = h;
oper_config.in.block_w = block_w;
oper_config.in.block_h = block_h;
oper_config.in.block_offset_x = 0;
oper_config.in.block_offset_y = 0;
oper_config.in.srm_cm = in_cm;
oper_config.out.buffer = out_buf;
oper_config.out.buffer_size = out_buf_size;
oper_config.out.pic_w = block_w;
oper_config.out.pic_h = block_h;
oper_config.out.block_offset_x = 0;
oper_config.out.block_offset_y = 0;
oper_config.out.srm_cm = out_cm;
oper_config.rotation_angle = rotation;
oper_config.scale_x = scale_x;
oper_config.scale_y = scale_y;
oper_config.rgb_swap = 0;
oper_config.byte_swap = 0;
oper_config.mode = PPA_TRANS_MODE_BLOCKING;
for (ppa_srm_rotation_angle_t rotation : rotations) {
for (const auto &scale_pair : scale_pairs) {
const float scale_x = scale_pair[0];
const float scale_y = scale_pair[1];
printf("SRM stress: rot=%d scale_x=%.1f scale_y=%.1f\n", (int)rotation, scale_x, scale_y);
TEST_ESP_OK(ppa_do_scale_rotate_mirror(ppa_client_handle, &oper_config));
// Test on different sizes of the block
int test_iterations = 50;
while (test_iterations-- > 0) {
uint32_t block_w_initial = esp_random() % (w - 100);
uint32_t block_h_initial = esp_random() % (h - 100);
block_w_initial = (block_w_initial == 0) ? 1 : block_w_initial;
block_h_initial = (block_h_initial == 0) ? 1 : block_h_initial;
for (int i = 0; i < 100; i++) {
uint32_t block_w = block_w_initial + i;
uint32_t block_h = block_h_initial + i;
ppa_srm_oper_config_t oper_config = {};
oper_config.in.buffer = in_buf;
oper_config.in.pic_w = w;
oper_config.in.pic_h = h;
oper_config.in.block_w = block_w;
oper_config.in.block_h = block_h;
oper_config.in.block_offset_x = 0;
oper_config.in.block_offset_y = 0;
oper_config.in.srm_cm = in_cm;
oper_config.out.buffer = out_buf;
oper_config.out.buffer_size = out_buf_size;
oper_config.out.pic_w = out_w;
oper_config.out.pic_h = out_h;
oper_config.out.block_offset_x = 0;
oper_config.out.block_offset_y = 0;
oper_config.out.srm_cm = out_cm;
oper_config.rotation_angle = rotation;
oper_config.scale_x = scale_x;
oper_config.scale_y = scale_y;
oper_config.rgb_swap = 0;
oper_config.byte_swap = 0;
oper_config.mode = PPA_TRANS_MODE_BLOCKING;
TEST_ESP_OK(ppa_do_scale_rotate_mirror(ppa_client_handle, &oper_config));
}
}
}
}