From 4502948cd85170109303207d28d47fff0475e7ce Mon Sep 17 00:00:00 2001 From: Michael Giacomelli Date: Mon, 5 Oct 2026 14:36:12 -0400 Subject: [PATCH] wmapro: add a two-channel path to the channel transform inverse_channel_transform() ran its general N-channel matrix loop for every sample of a stereo stream, where the matrix is exactly +-1.0. That loop was a quarter to a third of the whole decode, and GCC 9.5.0 compiles it worse than 4.9.4 did, which made the codec 6-9% slower on ARM7TDMI after the toolchain update. Handle a group of two channels separately: add and subtract when the matrix is +-1.0, and a plain four multiply loop otherwise. More than two channels still use the general loop. This reverses the regression from the GCC 9.5.0 update and goes well past it. The loop the newer compiler handled badly is no longer used for stereo, so the two compilers now give the same speed to within 1%, about 25% faster than the codec was with GCC 4.9.4 (estimated with perfsim, e200v1, wmapro_141k: 25.81 MHz with 4.9.4 before this change, 19.5 MHz with either compiler after it). Output is bit-identical: whole-file PCM hashes match before and after for five stereo files at 55-271 kbps, built with GCC 9.5.0 and with 4.9.4, and also with the multiply path forced on. Measured with test_codec, wmapro_141k.wma, MHz for real time: Sansa e200v1 27.99 -> 19.70 Sansa Clip+ 21.78 -> 15.80 Estimated with perfsim for the other files (e200v1 / Clip+): wmapro_55k 25.21 -> 17.06 / 20.02 -> 13.71 wmapro_80k 26.17 -> 18.01 / 20.75 -> 14.44 wmapro_173k 28.52 -> 20.29 / 22.55 -> 16.25 wmapro_271k 30.85 -> 22.52 / 24.34 -> 18.04 Co-Authored-By: Claude Opus 5.5 --- lib/rbcodec/codecs/libwmapro/wmaprodec.c | 69 +++++++++++++++++++----- 1 file changed, 55 insertions(+), 14 deletions(-) diff --git a/lib/rbcodec/codecs/libwmapro/wmaprodec.c b/lib/rbcodec/codecs/libwmapro/wmaprodec.c index 578b8ddb97..09dea96504 100644 --- a/lib/rbcodec/codecs/libwmapro/wmaprodec.c +++ b/lib/rbcodec/codecs/libwmapro/wmaprodec.c @@ -1030,6 +1030,38 @@ static int decode_scale_factors(WMAProDecodeCtx* s) return 0; } +/** + *@brief Apply a 2x2 decorrelation matrix to one band of two channels. + * Two channels are by far the common case; the results are the + * same as the general loop's in inverse_channel_transform(). + *@param ch0 first channel's coefficients + *@param ch1 second channel's coefficients + *@param mat the matrix, as 16.16 fixed point + *@param len number of coefficients + */ +static inline void decorrelate_stereo(int32_t *ch0, int32_t *ch1, + const int32_t *mat, int len) +{ + const int32_t m0 = mat[0], m1 = mat[1], m2 = mat[2], m3 = mat[3]; + + if (m0 == ONE_FRACT16 && m1 == -ONE_FRACT16 && + m2 == ONE_FRACT16 && m3 == ONE_FRACT16) { + /* The matrix of a stereo stream. Multiplying by +-1.0 is exact, + * so add and subtract; unsigned, as the multiply's result wraps. */ + for (; len > 0; len--) { + uint32_t a = *ch0, b = *ch1; + *ch0++ = a - b; + *ch1++ = a + b; + } + } else { + for (; len > 0; len--) { + int32_t a = *ch0, b = *ch1; + *ch0++ = fixmul16(m0, a) + fixmul16(m1, b); + *ch1++ = fixmul16(m2, a) + fixmul16(m3, b); + } + } +} + /** *@brief Reconstruct the individual channel data. *@param s codec context @@ -1052,24 +1084,33 @@ static void inverse_channel_transform(WMAProDecodeCtx *s) sfb < s->cur_sfb_offsets + s->num_bands; sfb++) { int y; if (*tb++ == 1) { - /** multiply values with the decorrelation_matrix */ - for (y = sfb[0]; y < FFMIN(sfb[1], s->subframe_len); y++) { - const int32_t* mat = s->chgroup[i].fixdecorrelation_matrix; - const int32_t* data_end = data + num_channels; - int32_t* data_ptr = data; - int32_t** ch; + if (num_channels == 2) { + decorrelate_stereo(ch_data[0] + sfb[0], + ch_data[1] + sfb[0], + s->chgroup[i].fixdecorrelation_matrix, + FFMIN(sfb[1], s->subframe_len) - sfb[0]); + } else { + /** multiply values with the decorrelation_matrix */ + for (y = sfb[0]; + y < FFMIN(sfb[1], s->subframe_len); y++) { + const int32_t* mat = + s->chgroup[i].fixdecorrelation_matrix; + const int32_t* data_end = data + num_channels; + int32_t* data_ptr = data; + int32_t** ch; - for (ch = ch_data; ch < ch_end; ch++) - *data_ptr++ = (*ch)[y]; + for (ch = ch_data; ch < ch_end; ch++) + *data_ptr++ = (*ch)[y]; - for (ch = ch_data; ch < ch_end; ch++) { - int32_t sum = 0; - data_ptr = data; + for (ch = ch_data; ch < ch_end; ch++) { + int32_t sum = 0; + data_ptr = data; - while (data_ptr < data_end) - sum += fixmul16(*mat++, *data_ptr++); + while (data_ptr < data_end) + sum += fixmul16(*mat++, *data_ptr++); - (*ch)[y] = sum; + (*ch)[y] = sum; + } } } } else if (s->num_channels == 2) {