mirror of
https://github.com/Rockbox/rockbox.git
synced 2026-10-10 08:03:04 -04:00
wmapro: add a two-channel path to the channel transform
inverse_channel_transform() ran its general N-channel matrix loop for every sample of a stereo stream, where the matrix is exactly +-1.0. That loop was a quarter to a third of the whole decode, and GCC 9.5.0 compiles it worse than 4.9.4 did, which made the codec 6-9% slower on ARM7TDMI after the toolchain update. Handle a group of two channels separately: add and subtract when the matrix is +-1.0, and a plain four multiply loop otherwise. More than two channels still use the general loop. This reverses the regression from the GCC 9.5.0 update and goes well past it. The loop the newer compiler handled badly is no longer used for stereo, so the two compilers now give the same speed to within 1%, about 25% faster than the codec was with GCC 4.9.4 (estimated with perfsim, e200v1, wmapro_141k: 25.81 MHz with 4.9.4 before this change, 19.5 MHz with either compiler after it). Output is bit-identical: whole-file PCM hashes match before and after for five stereo files at 55-271 kbps, built with GCC 9.5.0 and with 4.9.4, and also with the multiply path forced on. Measured with test_codec, wmapro_141k.wma, MHz for real time: Sansa e200v1 27.99 -> 19.70 Sansa Clip+ 21.78 -> 15.80 Estimated with perfsim for the other files (e200v1 / Clip+): wmapro_55k 25.21 -> 17.06 / 20.02 -> 13.71 wmapro_80k 26.17 -> 18.01 / 20.75 -> 14.44 wmapro_173k 28.52 -> 20.29 / 22.55 -> 16.25 wmapro_271k 30.85 -> 22.52 / 24.34 -> 18.04 Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
361913fdd2
commit
4502948cd8
1 changed files with 55 additions and 14 deletions
|
|
@ -1030,6 +1030,38 @@ static int decode_scale_factors(WMAProDecodeCtx* s)
|
|||
return 0;
|
||||
}
|
||||
|
||||
/**
|
||||
*@brief Apply a 2x2 decorrelation matrix to one band of two channels.
|
||||
* Two channels are by far the common case; the results are the
|
||||
* same as the general loop's in inverse_channel_transform().
|
||||
*@param ch0 first channel's coefficients
|
||||
*@param ch1 second channel's coefficients
|
||||
*@param mat the matrix, as 16.16 fixed point
|
||||
*@param len number of coefficients
|
||||
*/
|
||||
static inline void decorrelate_stereo(int32_t *ch0, int32_t *ch1,
|
||||
const int32_t *mat, int len)
|
||||
{
|
||||
const int32_t m0 = mat[0], m1 = mat[1], m2 = mat[2], m3 = mat[3];
|
||||
|
||||
if (m0 == ONE_FRACT16 && m1 == -ONE_FRACT16 &&
|
||||
m2 == ONE_FRACT16 && m3 == ONE_FRACT16) {
|
||||
/* The matrix of a stereo stream. Multiplying by +-1.0 is exact,
|
||||
* so add and subtract; unsigned, as the multiply's result wraps. */
|
||||
for (; len > 0; len--) {
|
||||
uint32_t a = *ch0, b = *ch1;
|
||||
*ch0++ = a - b;
|
||||
*ch1++ = a + b;
|
||||
}
|
||||
} else {
|
||||
for (; len > 0; len--) {
|
||||
int32_t a = *ch0, b = *ch1;
|
||||
*ch0++ = fixmul16(m0, a) + fixmul16(m1, b);
|
||||
*ch1++ = fixmul16(m2, a) + fixmul16(m3, b);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
*@brief Reconstruct the individual channel data.
|
||||
*@param s codec context
|
||||
|
|
@ -1052,24 +1084,33 @@ static void inverse_channel_transform(WMAProDecodeCtx *s)
|
|||
sfb < s->cur_sfb_offsets + s->num_bands; sfb++) {
|
||||
int y;
|
||||
if (*tb++ == 1) {
|
||||
/** multiply values with the decorrelation_matrix */
|
||||
for (y = sfb[0]; y < FFMIN(sfb[1], s->subframe_len); y++) {
|
||||
const int32_t* mat = s->chgroup[i].fixdecorrelation_matrix;
|
||||
const int32_t* data_end = data + num_channels;
|
||||
int32_t* data_ptr = data;
|
||||
int32_t** ch;
|
||||
if (num_channels == 2) {
|
||||
decorrelate_stereo(ch_data[0] + sfb[0],
|
||||
ch_data[1] + sfb[0],
|
||||
s->chgroup[i].fixdecorrelation_matrix,
|
||||
FFMIN(sfb[1], s->subframe_len) - sfb[0]);
|
||||
} else {
|
||||
/** multiply values with the decorrelation_matrix */
|
||||
for (y = sfb[0];
|
||||
y < FFMIN(sfb[1], s->subframe_len); y++) {
|
||||
const int32_t* mat =
|
||||
s->chgroup[i].fixdecorrelation_matrix;
|
||||
const int32_t* data_end = data + num_channels;
|
||||
int32_t* data_ptr = data;
|
||||
int32_t** ch;
|
||||
|
||||
for (ch = ch_data; ch < ch_end; ch++)
|
||||
*data_ptr++ = (*ch)[y];
|
||||
for (ch = ch_data; ch < ch_end; ch++)
|
||||
*data_ptr++ = (*ch)[y];
|
||||
|
||||
for (ch = ch_data; ch < ch_end; ch++) {
|
||||
int32_t sum = 0;
|
||||
data_ptr = data;
|
||||
for (ch = ch_data; ch < ch_end; ch++) {
|
||||
int32_t sum = 0;
|
||||
data_ptr = data;
|
||||
|
||||
while (data_ptr < data_end)
|
||||
sum += fixmul16(*mat++, *data_ptr++);
|
||||
while (data_ptr < data_end)
|
||||
sum += fixmul16(*mat++, *data_ptr++);
|
||||
|
||||
(*ch)[y] = sum;
|
||||
(*ch)[y] = sum;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (s->num_channels == 2) {
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue