mirror of
https://github.com/Rockbox/rockbox.git
synced 2026-10-09 23:53:28 -04:00
wmapro: fix the level and accuracy of 16 bit streams
A WMA Professional stream of 16 bits per sample played about 48 dB too quiet and with a noise floor near -70 dBFS. ffmpeg scales the transform's output by the stream's sample size. That was dropped when the decoder was converted to fixed point (d884af2b99,16284ae8ae), and the output is passed to the DSP as if every stream had 24 bits. A stream of fewer bits has a lower quantization step to match, so it came out low by the difference, and it used the bottom of the integer quantization table, where the factors have only a few significant bits. Decode a 16 or 20 bit stream at the level of a 24 bit stream: use that stream's quantization step, and scale each band's factor by the ratio that is left. 24 bit streams are not affected. Checked with perfsim (Sansa e200v1 build) against ffmpeg's decode of a 16 bit, 192 kbps stereo file, the only such file to hand: level SNR vs ffmpeg noise before -48 dB 37 dB -70 dBFS after correct 85 dB -117 dBFS A 24 bit file's output is byte-identical before and after. The 20 bit case follows the same rule but is untested. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
4b8ea97fa9
commit
6ffbee954b
1 changed files with 19 additions and 2 deletions
|
|
@ -224,6 +224,8 @@ typedef struct WMAProDecodeCtx {
|
|||
uint8_t len_prefix; ///< frame is prefixed with its length
|
||||
uint8_t dynamic_range_compression; ///< frame contains DRC data
|
||||
uint8_t bits_per_sample; ///< integer audio sample size for the unscaled IMDCT output (used to scale to [-1.0, 1.0])
|
||||
uint8_t quant_step_bias; ///< added to the quantization step of a stream of less than 24 bits
|
||||
int32_t quant_scale; ///< s1.30 factor for its quantization factors, 0 if none
|
||||
uint16_t samples_per_frame; ///< number of samples to output
|
||||
uint16_t log2_frame_size;
|
||||
int8_t num_channels; ///< number of channels in the stream (same as AVCodecContext.num_channels)
|
||||
|
|
@ -332,6 +334,18 @@ int decode_init(asf_waveformatex_t *wfx)
|
|||
s->decode_flags = AV_RL16(edata_ptr+14);
|
||||
channel_mask = AV_RL32(edata_ptr+2);
|
||||
s->bits_per_sample = AV_RL16(edata_ptr);
|
||||
|
||||
/* A stream of fewer bits has a lower quantization step. Decode it at
|
||||
* the level of a 24 bit stream instead: use that stream's quantization
|
||||
* step, and scale the factors by what is left,
|
||||
* 2^(24-bits) / 10^(bias/20). */
|
||||
if (s->bits_per_sample == 16) {
|
||||
s->quant_step_bias = (90 * 24 >> 4) - (90 * 16 >> 4);
|
||||
s->quant_scale = 1545752065; /* 256 / 10^(45/20) */
|
||||
} else if (s->bits_per_sample == 20) {
|
||||
s->quant_step_bias = (90 * 24 >> 4) - (90 * 20 >> 4);
|
||||
s->quant_scale = 1216241597; /* 16 / 10^(23/20) */
|
||||
}
|
||||
/** dump the extradata */
|
||||
for (i = 0; i < wfx->datalen; i++)
|
||||
DEBUGF("[%x] ", wfx->data[i]);
|
||||
|
|
@ -1270,7 +1284,7 @@ static int decode_subframe(WMAProDecodeCtx *s)
|
|||
|
||||
if (transmit_coeffs) {
|
||||
int step;
|
||||
int quant_step = 90 * s->bits_per_sample >> 4;
|
||||
int quant_step = (90 * s->bits_per_sample >> 4) + s->quant_step_bias;
|
||||
|
||||
/** decode number of vector coded coefficients */
|
||||
if ((s->transmit_num_vec_coeffs = get_bits1(&s->gb))) {
|
||||
|
|
@ -1371,9 +1385,12 @@ static int decode_subframe(WMAProDecodeCtx *s)
|
|||
DEBUGF("in wmaprodec.c : unhandled value for exp (%d), please report sample.\n", exp);
|
||||
return -1;
|
||||
}
|
||||
const int32_t quant = QUANT(exp);
|
||||
int32_t quant = QUANT(exp);
|
||||
int start = s->cur_sfb_offsets[b];
|
||||
|
||||
if (s->quant_scale)
|
||||
quant = (int64_t)quant * s->quant_scale >> 30;
|
||||
|
||||
vector_fixmul_scalar(s->tmp+start,
|
||||
s->channel[c].coeffs + start,
|
||||
quant, end-start);
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue