diff --git a/components/esp_driver_ppa/src/ppa_srm.c b/components/esp_driver_ppa/src/ppa_srm.c index 1f7b1ff53d1..5e1df703f25 100644 --- a/components/esp_driver_ppa/src/ppa_srm.c +++ b/components/esp_driver_ppa/src/ppa_srm.c @@ -115,7 +115,13 @@ bool ppa_srm_transaction_on_picked(uint32_t num_chans, const dma2d_trans_channel // Configure the block size to be received by the SRM engine, which is passed from the 2D-DMA TX channel (i.e. 2D-DMA dscr-port mode) uint32_t block_h = 0, block_v = 0; - ppa_ll_srm_get_dma_dscr_port_mode_block_size(platform->hal.dev, ppa_in_color_mode, ppa_ll_srm_get_mb_size(platform->hal.dev), &block_h, &block_v); + ppa_ll_srm_mb_size_t mb_size = ppa_ll_srm_get_mb_size(platform->hal.dev); +#if CONFIG_ESP32P4_SELECTS_REV_LESS_V3 + assert(mb_size == PPA_LL_SRM_MB_SIZE_16_16); +#else + assert(mb_size == PPA_LL_SRM_MB_SIZE_32_32); +#endif + ppa_ll_srm_get_dma_dscr_port_mode_block_size(platform->hal.dev, ppa_in_color_mode, mb_size, &block_h, &block_v); dma2d_dscr_port_mode_config_t dma_dscr_port_mode_config = { .block_h = block_h, .block_v = block_v, @@ -156,18 +162,38 @@ bool ppa_srm_transaction_on_picked(uint32_t num_chans, const dma2d_trans_channel #if CONFIG_IDF_TARGET_ESP32P4 // Hardware bug workaround (DIG-734) - uint32_t w_out = srm_trans_desc->in.block_w * srm_trans_desc->scale_x_int + srm_trans_desc->in.block_w * srm_trans_desc->scale_x_frag / PPA_LL_SRM_SCALING_FRAG_MAX; + // Leftover block data size less than DMA FIFO depth could cause DMA to miss the counting of such batch, + // so DMA never thinks it has received all the data, and will not raise EOF/DONE interrupt. + // The workaround is to bypass macro block order, so no such small leftover blocks. + uint32_t in_block_w, in_block_h; + uint32_t scale_x_int, scale_x_frag, scale_y_int, scale_y_frag; + if (srm_trans_desc->rotation_angle == PPA_SRM_ROTATION_ANGLE_90 || srm_trans_desc->rotation_angle == PPA_SRM_ROTATION_ANGLE_270) { + in_block_w = srm_trans_desc->in.block_h; + in_block_h = srm_trans_desc->in.block_w; + scale_x_int = srm_trans_desc->scale_y_int; + scale_x_frag = srm_trans_desc->scale_y_frag; + scale_y_int = srm_trans_desc->scale_x_int; + scale_y_frag = srm_trans_desc->scale_x_frag; + } else { + in_block_w = srm_trans_desc->in.block_w; + in_block_h = srm_trans_desc->in.block_h; + scale_x_int = srm_trans_desc->scale_x_int; + scale_x_frag = srm_trans_desc->scale_x_frag; + scale_y_int = srm_trans_desc->scale_y_int; + scale_y_frag = srm_trans_desc->scale_y_frag; + } + uint32_t w_out = in_block_w * scale_x_int + in_block_w * scale_x_frag / PPA_LL_SRM_SCALING_FRAG_MAX; uint32_t w_divisor = (ppa_out_color_mode == PPA_SRM_COLOR_MODE_ARGB8888 || ppa_out_color_mode == PPA_SRM_COLOR_MODE_RGB888) ? 32 : 64; uint32_t w_left = w_out % w_divisor; w_left = (w_left == 0) ? w_divisor : w_left; - uint32_t h_mb = (ppa_ll_srm_get_mb_size(platform->hal.dev) == PPA_LL_SRM_MB_SIZE_16_16) ? 16 : 32; - uint32_t h_in_left = srm_trans_desc->in.block_h % h_mb; + uint32_t h_mb = (mb_size == PPA_LL_SRM_MB_SIZE_16_16) ? 16 : 32; + uint32_t h_in_left = in_block_h % h_mb; h_in_left = (h_in_left == 0) ? h_mb : h_in_left; - uint32_t h_left = h_in_left * srm_trans_desc->scale_y_int + h_in_left * srm_trans_desc->scale_y_frag / PPA_LL_SRM_SCALING_FRAG_MAX; + uint32_t h_left = h_in_left * scale_y_int + h_in_left * scale_y_frag / PPA_LL_SRM_SCALING_FRAG_MAX; const uint32_t dma2d_fifo_depth_bits = 12 * 128; uint32_t out_pixel_depth = color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)ppa_out_color_mode); bool bypass_mb_order = false; - if (((w_out > w_divisor) || (srm_trans_desc->in.block_h > h_mb)) && // will be cut into more than one trans unit + if (((w_out > w_divisor) || (in_block_h > h_mb)) && // will be cut into more than one trans unit ((w_left * h_left * out_pixel_depth) < dma2d_fifo_depth_bits) ) { bypass_mb_order = true; @@ -229,6 +255,18 @@ esp_err_t ppa_do_scale_rotate_mirror(ppa_client_handle_t ppa_client, const ppa_s uint32_t scale_x_frag = (uint32_t)(config->scale_x * PPA_LL_SRM_SCALING_FRAG_MAX) & (PPA_LL_SRM_SCALING_FRAG_MAX - 1); uint32_t scale_y_int = (uint32_t)config->scale_y; uint32_t scale_y_frag = (uint32_t)(config->scale_y * PPA_LL_SRM_SCALING_FRAG_MAX) & (PPA_LL_SRM_SCALING_FRAG_MAX - 1); + // SRM processes in blocks. Block x/(y) (including leftover block) cannot be scaled to odd number when YUV422/YUV420 is the output color mode + // When block size is 16x16, odd number is possible, so needs to make them even + // When block size is 32x32, calculated frag values for full macro blocks are always even +#if CONFIG_ESP32P4_SELECTS_REV_LESS_V3 + // macro block size is 16x16, will do sanity check in ppa_srm_transaction_on_picked + if (config->out.srm_cm == PPA_SRM_COLOR_MODE_YUV420) { + scale_x_frag = scale_x_frag & ~1; + scale_y_frag = scale_y_frag & ~1; + } else if (PPA_IS_CM_YUV422(config->out.srm_cm)) { + scale_x_frag = scale_x_frag & ~1; + } +#endif uint32_t new_block_w = 0; uint32_t new_block_h = 0; if (config->rotation_angle == PPA_SRM_ROTATION_ANGLE_0 || config->rotation_angle == PPA_SRM_ROTATION_ANGLE_180) { @@ -244,6 +282,12 @@ esp_err_t ppa_do_scale_rotate_mirror(ppa_client_handle_t ppa_client, const ppa_s config->out.block_offset_y < config->out.pic_h && new_block_h <= (config->out.pic_h - config->out.block_offset_y), ESP_ERR_INVALID_ARG, TAG, "scale does not fit in the out pic"); + // Check for the leftover block w/(h) to be even in YUV output color mode + if (config->out.srm_cm == PPA_SRM_COLOR_MODE_YUV420) { + ESP_RETURN_ON_FALSE(new_block_w % 2 == 0 && new_block_h % 2 == 0, ESP_ERR_INVALID_ARG, TAG, "YUV420 output does not support scaled block w/h to be odd"); + } else if (PPA_IS_CM_YUV422(config->out.srm_cm)) { + ESP_RETURN_ON_FALSE(new_block_w % 2 == 0, ESP_ERR_INVALID_ARG, TAG, "YUV422 output does not support scaled block w to be odd"); + } if (!ppa_check_buffer_alignment(ppa_client, &config->in, true, config->in.block_w) || !ppa_check_buffer_alignment(ppa_client, &config->out, false, new_block_w)) { @@ -299,15 +343,6 @@ esp_err_t ppa_do_scale_rotate_mirror(ppa_client_handle_t ppa_client, const ppa_s srm_trans_desc->scale_x_frag = scale_x_frag; srm_trans_desc->scale_y_int = scale_y_int; srm_trans_desc->scale_y_frag = scale_y_frag; - // SRM processes in blocks. Block x/(y) cannot be scaled to odd number when YUV422/YUV420 is the output color mode - // When block size is 16x16, odd number is possible, so needs to make them even - // When block size is 32x32, calculated frag values are always even - if (config->out.srm_cm == PPA_SRM_COLOR_MODE_YUV420) { - srm_trans_desc->scale_x_frag = srm_trans_desc->scale_x_frag & ~1; - srm_trans_desc->scale_y_frag = srm_trans_desc->scale_y_frag & ~1; - } else if (PPA_IS_CM_YUV422(config->out.srm_cm)) { - srm_trans_desc->scale_x_frag = srm_trans_desc->scale_x_frag & ~1; - } srm_trans_desc->alpha_value = new_alpha_value; srm_trans_desc->data_burst_length = ppa_client->data_burst_length; diff --git a/components/esp_driver_ppa/test_apps/main/test_ppa.cpp b/components/esp_driver_ppa/test_apps/main/test_ppa.cpp index 09f7b263d73..731d4131164 100644 --- a/components/esp_driver_ppa/test_apps/main/test_ppa.cpp +++ b/components/esp_driver_ppa/test_apps/main/test_ppa.cpp @@ -1030,12 +1030,20 @@ TEST_CASE("ppa_srm_stress_test", "[PPA]") const uint32_t h = 200; const ppa_srm_color_mode_t in_cm = PPA_SRM_COLOR_MODE_RGB565; const ppa_srm_color_mode_t out_cm = PPA_SRM_COLOR_MODE_RGB565; - const ppa_srm_rotation_angle_t rotation = PPA_SRM_ROTATION_ANGLE_0; - const float scale_x = 1.0; - const float scale_y = 1.0; + const ppa_srm_rotation_angle_t rotations[] = { + PPA_SRM_ROTATION_ANGLE_0, + PPA_SRM_ROTATION_ANGLE_90, + }; + const float scale_pairs[][2] = { + {1.0f, 1.0f}, + {1.0f, 1.5f}, + {1.2f, 1.0f}, + }; + const uint32_t out_w = 2 * w; // leave a large output buffer, since we test with >1.0 scale + const uint32_t out_h = 2 * h; uint32_t in_buf_size = w * h * color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)in_cm) / 8; - uint32_t out_buf_size = ESP_ALIGN_UP(w * h * color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)out_cm) / 8, 64); + uint32_t out_buf_size = ESP_ALIGN_UP(out_w * out_h * color_hal_pixel_format_fourcc_get_bit_depth((esp_color_fourcc_t)out_cm) / 8, 64); uint8_t *out_buf = static_cast(heap_caps_aligned_calloc(4, out_buf_size, sizeof(uint8_t), MALLOC_CAP_SPIRAM | MALLOC_CAP_DMA)); TEST_ASSERT_NOT_NULL(out_buf); uint8_t *in_buf = static_cast(heap_caps_aligned_calloc(4, in_buf_size, sizeof(uint8_t), MALLOC_CAP_INTERNAL | MALLOC_CAP_8BIT | MALLOC_CAP_DMA)); @@ -1047,43 +1055,48 @@ TEST_CASE("ppa_srm_stress_test", "[PPA]") ppa_client_config.max_pending_trans_num = 1; TEST_ESP_OK(ppa_register_client(&ppa_client_config, &ppa_client_handle)); - // Test on different sizes of the block - int test_iterations = 50; - while (test_iterations-- > 0) { - uint32_t block_w_initial = esp_random() % (w - 100); - uint32_t block_h_initial = esp_random() % (h - 100); - block_w_initial = (block_w_initial == 0) ? 1 : block_w_initial; - block_h_initial = (block_h_initial == 0) ? 1 : block_h_initial; - uint32_t block_w = 0; - uint32_t block_h = 0; - for (int i = 0; i < 100; i++) { - block_w = block_w_initial + i; - block_h = block_h_initial + i; - // printf("block_w = %ld, block_h = %ld\n", block_w, block_h); - ppa_srm_oper_config_t oper_config = {}; - oper_config.in.buffer = in_buf; - oper_config.in.pic_w = w; - oper_config.in.pic_h = h; - oper_config.in.block_w = block_w; - oper_config.in.block_h = block_h; - oper_config.in.block_offset_x = 0; - oper_config.in.block_offset_y = 0; - oper_config.in.srm_cm = in_cm; - oper_config.out.buffer = out_buf; - oper_config.out.buffer_size = out_buf_size; - oper_config.out.pic_w = block_w; - oper_config.out.pic_h = block_h; - oper_config.out.block_offset_x = 0; - oper_config.out.block_offset_y = 0; - oper_config.out.srm_cm = out_cm; - oper_config.rotation_angle = rotation; - oper_config.scale_x = scale_x; - oper_config.scale_y = scale_y; - oper_config.rgb_swap = 0; - oper_config.byte_swap = 0; - oper_config.mode = PPA_TRANS_MODE_BLOCKING; + for (ppa_srm_rotation_angle_t rotation : rotations) { + for (const auto &scale_pair : scale_pairs) { + const float scale_x = scale_pair[0]; + const float scale_y = scale_pair[1]; + printf("SRM stress: rot=%d scale_x=%.1f scale_y=%.1f\n", (int)rotation, scale_x, scale_y); - TEST_ESP_OK(ppa_do_scale_rotate_mirror(ppa_client_handle, &oper_config)); + // Test on different sizes of the block + int test_iterations = 50; + while (test_iterations-- > 0) { + uint32_t block_w_initial = esp_random() % (w - 100); + uint32_t block_h_initial = esp_random() % (h - 100); + block_w_initial = (block_w_initial == 0) ? 1 : block_w_initial; + block_h_initial = (block_h_initial == 0) ? 1 : block_h_initial; + for (int i = 0; i < 100; i++) { + uint32_t block_w = block_w_initial + i; + uint32_t block_h = block_h_initial + i; + ppa_srm_oper_config_t oper_config = {}; + oper_config.in.buffer = in_buf; + oper_config.in.pic_w = w; + oper_config.in.pic_h = h; + oper_config.in.block_w = block_w; + oper_config.in.block_h = block_h; + oper_config.in.block_offset_x = 0; + oper_config.in.block_offset_y = 0; + oper_config.in.srm_cm = in_cm; + oper_config.out.buffer = out_buf; + oper_config.out.buffer_size = out_buf_size; + oper_config.out.pic_w = out_w; + oper_config.out.pic_h = out_h; + oper_config.out.block_offset_x = 0; + oper_config.out.block_offset_y = 0; + oper_config.out.srm_cm = out_cm; + oper_config.rotation_angle = rotation; + oper_config.scale_x = scale_x; + oper_config.scale_y = scale_y; + oper_config.rgb_swap = 0; + oper_config.byte_swap = 0; + oper_config.mode = PPA_TRANS_MODE_BLOCKING; + + TEST_ESP_OK(ppa_do_scale_rotate_mirror(ppa_client_handle, &oper_config)); + } + } } }