From 6ffbee954b75d68e6e65471d57e7de72b93c6c49 Mon Sep 17 00:00:00 2001 From: Michael Giacomelli Date: Mon, 5 Oct 2026 20:12:58 -0400 Subject: [PATCH] wmapro: fix the level and accuracy of 16 bit streams A WMA Professional stream of 16 bits per sample played about 48 dB too quiet and with a noise floor near -70 dBFS. ffmpeg scales the transform's output by the stream's sample size. That was dropped when the decoder was converted to fixed point (d884af2b99, 16284ae8ae), and the output is passed to the DSP as if every stream had 24 bits. A stream of fewer bits has a lower quantization step to match, so it came out low by the difference, and it used the bottom of the integer quantization table, where the factors have only a few significant bits. Decode a 16 or 20 bit stream at the level of a 24 bit stream: use that stream's quantization step, and scale each band's factor by the ratio that is left. 24 bit streams are not affected. Checked with perfsim (Sansa e200v1 build) against ffmpeg's decode of a 16 bit, 192 kbps stereo file, the only such file to hand: level SNR vs ffmpeg noise before -48 dB 37 dB -70 dBFS after correct 85 dB -117 dBFS A 24 bit file's output is byte-identical before and after. The 20 bit case follows the same rule but is untested. Co-Authored-By: Claude Opus 5.5 --- lib/rbcodec/codecs/libwmapro/wmaprodec.c | 21 +++++++++++++++++++-- 1 file changed, 19 insertions(+), 2 deletions(-) diff --git a/lib/rbcodec/codecs/libwmapro/wmaprodec.c b/lib/rbcodec/codecs/libwmapro/wmaprodec.c index 09dea96504..75bdf4d8a1 100644 --- a/lib/rbcodec/codecs/libwmapro/wmaprodec.c +++ b/lib/rbcodec/codecs/libwmapro/wmaprodec.c @@ -224,6 +224,8 @@ typedef struct WMAProDecodeCtx { uint8_t len_prefix; ///< frame is prefixed with its length uint8_t dynamic_range_compression; ///< frame contains DRC data uint8_t bits_per_sample; ///< integer audio sample size for the unscaled IMDCT output (used to scale to [-1.0, 1.0]) + uint8_t quant_step_bias; ///< added to the quantization step of a stream of less than 24 bits + int32_t quant_scale; ///< s1.30 factor for its quantization factors, 0 if none uint16_t samples_per_frame; ///< number of samples to output uint16_t log2_frame_size; int8_t num_channels; ///< number of channels in the stream (same as AVCodecContext.num_channels) @@ -332,6 +334,18 @@ int decode_init(asf_waveformatex_t *wfx) s->decode_flags = AV_RL16(edata_ptr+14); channel_mask = AV_RL32(edata_ptr+2); s->bits_per_sample = AV_RL16(edata_ptr); + + /* A stream of fewer bits has a lower quantization step. Decode it at + * the level of a 24 bit stream instead: use that stream's quantization + * step, and scale the factors by what is left, + * 2^(24-bits) / 10^(bias/20). */ + if (s->bits_per_sample == 16) { + s->quant_step_bias = (90 * 24 >> 4) - (90 * 16 >> 4); + s->quant_scale = 1545752065; /* 256 / 10^(45/20) */ + } else if (s->bits_per_sample == 20) { + s->quant_step_bias = (90 * 24 >> 4) - (90 * 20 >> 4); + s->quant_scale = 1216241597; /* 16 / 10^(23/20) */ + } /** dump the extradata */ for (i = 0; i < wfx->datalen; i++) DEBUGF("[%x] ", wfx->data[i]); @@ -1270,7 +1284,7 @@ static int decode_subframe(WMAProDecodeCtx *s) if (transmit_coeffs) { int step; - int quant_step = 90 * s->bits_per_sample >> 4; + int quant_step = (90 * s->bits_per_sample >> 4) + s->quant_step_bias; /** decode number of vector coded coefficients */ if ((s->transmit_num_vec_coeffs = get_bits1(&s->gb))) { @@ -1371,9 +1385,12 @@ static int decode_subframe(WMAProDecodeCtx *s) DEBUGF("in wmaprodec.c : unhandled value for exp (%d), please report sample.\n", exp); return -1; } - const int32_t quant = QUANT(exp); + int32_t quant = QUANT(exp); int start = s->cur_sfb_offsets[b]; + if (s->quant_scale) + quant = (int64_t)quant * s->quant_scale >> 30; + vector_fixmul_scalar(s->tmp+start, s->channel[c].coeffs + start, quant, end-start);