atrac3: output samples with 28 fractional bits

The decoder gave the DSP samples with two fractional bits (sample
depth 17), far fewer than any other codec. The DSP's filters are
only as precise as the samples they are given: with an equalizer
band at a low frequency the output was noise, or held at full
scale.

Scale the last stage of the synthesis filter up by 11 bits and
set the sample depth to 28. On ARMv4 the bits are taken from the
64-bit sums of the dewindowing, which had them. Elsewhere the
input of that stage is scaled, in its matrixing step; that is not
done on ARMv4 because its multiplier takes longer for the larger
operands (51.97 against 48.85 MHz below). The Coldfire path uses
the C matrixing and has not been run.

Seven files decoded under perfsim for the Sansa e200v1 and Clip+:
shifted down again, the e200v1 output is within one step of the
old output, and the Clip+ output no further from the e200v1's
than before. Through a ten band equalizer with bands at 32 and
64 Hz (the filter fix of the DSP included), the noise of
atrac3_lp2_132.oma falls from -64 dB to -101 dB relative to the
signal.

Estimated with perfsim (not measured on a device),
atrac3_lp2_132.oma: e200v1 48.81 -> 48.85 MHz, Clip+ 26.91 ->
26.96 MHz.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Michael Giacomelli 2026-10-06 21:33:31 -04:00
parent 3966b420a3
commit 02d5520c5f
5 changed files with 133 additions and 18 deletions

View file

@ -63,7 +63,7 @@ enum codec_status codec_run(void)
ci->memset(&q,0,sizeof(ATRAC3Context));
ci->configure(DSP_SET_FREQUENCY, ci->id3->frequency);
ci->configure(DSP_SET_SAMPLE_DEPTH, 17); /* Remark: atrac3 uses s15.0 by default, s15.2 was hacked. */
ci->configure(DSP_SET_SAMPLE_DEPTH, ATRAC3_OUTPUT_DEPTH);
const uint8_t channels = 2;
ci->configure(DSP_SET_STEREO_MODE, channels == 1 ?
STEREO_MONO : STEREO_NONINTERLEAVED);

View file

@ -92,7 +92,7 @@ enum codec_status codec_run(void)
init_rm(&rmctx);
ci->configure(DSP_SET_FREQUENCY, ci->id3->frequency);
ci->configure(DSP_SET_SAMPLE_DEPTH, 17); /* Remark: atrac3 uses s15.0 by default, s15.2 was hacked. */
ci->configure(DSP_SET_SAMPLE_DEPTH, ATRAC3_OUTPUT_DEPTH);
ci->configure(DSP_SET_STEREO_MODE, rmctx.nb_channels == 1 ?
STEREO_MONO : STEREO_NONINTERLEAVED);

View file

@ -86,6 +86,13 @@ static int vlcs_initialized = 0;
int32_t *inlo,
int32_t *inhi,
unsigned int nIn);
/* The same, with the results scaled up by ATRAC3_OUT_SHIFT bits
* (ARMv5 and later) */
extern void
atrac3_iqmf_matrixing_out(int32_t *p3,
int32_t *inlo,
int32_t *inhi,
unsigned int nIn);
#else
static inline void
atrac3_iqmf_matrixing(int32_t *p3,
@ -101,6 +108,22 @@ static int vlcs_initialized = 0;
p3[2*i+3] = inlo[i+1] - inhi[i+1];
}
}
/* The same, with the results scaled up by ATRAC3_OUT_SHIFT bits */
static inline void
atrac3_iqmf_matrixing_out(int32_t *p3,
int32_t *inlo,
int32_t *inhi,
unsigned int nIn)
{
uint32_t i;
for(i=0; i<nIn; i+=2){
p3[2*i+0] = (inlo[i ] + inhi[i ]) * (1 << ATRAC3_OUT_SHIFT);
p3[2*i+1] = (inlo[i ] - inhi[i ]) * (1 << ATRAC3_OUT_SHIFT);
p3[2*i+2] = (inlo[i+1] + inhi[i+1]) * (1 << ATRAC3_OUT_SHIFT);
p3[2*i+3] = (inlo[i+1] - inhi[i+1]) * (1 << ATRAC3_OUT_SHIFT);
}
}
#endif
@ -150,6 +173,15 @@ static int vlcs_initialized = 0;
int32_t *in,
int32_t *win,
unsigned int nIn);
/* The same, with the results scaled up by ATRAC3_OUT_SHIFT bits. The
* ARMv4 multiplier takes longer for large operands, so there the last
* stage is scaled here and not in its matrixing. */
#define ATRAC3_SCALE_IN_DEWINDOWING
extern void
atrac3_iqmf_dewindowing_out(int32_t *out,
int32_t *in,
int32_t *win,
unsigned int nIn);
#elif defined (CPU_COLDFIRE)
#define MULTIPLY_ADD_BLOCK \
@ -274,19 +306,35 @@ atrac3_imdct_windowing(int32_t *buffer,
* @param pOut out buffer
* @param delayBuf delayBuf buffer
* @param temp temp buffer
* @param last true for the stage that makes the output, which is
* scaled up by ATRAC3_OUT_SHIFT bits
*/
static void iqmf (int32_t *inlo, int32_t *inhi, unsigned int nIn, int32_t *pOut, int32_t *delayBuf, int32_t *temp)
static void iqmf (int32_t *inlo, int32_t *inhi, unsigned int nIn, int32_t *pOut, int32_t *delayBuf, int32_t *temp, bool last)
{
/* Restore the delay buffer */
memcpy(temp, delayBuf, 46*sizeof(int32_t));
#ifdef ATRAC3_SCALE_IN_DEWINDOWING
/* loop1: matrixing */
atrac3_iqmf_matrixing(temp + 46, inlo, inhi, nIn);
/* loop2: dewindowing */
if (last)
atrac3_iqmf_dewindowing_out(pOut, temp, qmf_window, nIn);
else
atrac3_iqmf_dewindowing(pOut, temp, qmf_window, nIn);
#else
/* loop1: matrixing */
if (last)
atrac3_iqmf_matrixing_out(temp + 46, inlo, inhi, nIn);
else
atrac3_iqmf_matrixing(temp + 46, inlo, inhi, nIn);
/* loop2: dewindowing */
atrac3_iqmf_dewindowing(pOut, temp, qmf_window, nIn);
#endif
/* Save the delay buffer */
memcpy(delayBuf, temp + (nIn << 1), 46*sizeof(int32_t));
@ -1105,9 +1153,9 @@ static int decodeFrame(ATRAC3Context *q, const uint8_t* databuf, int off)
p2= p1+256;
p3= p2+256;
p4= p3+256;
iqmf (p1, p2, 256, p1, q->pUnits[i].delayBuf1, q->tempBuf);
iqmf (p4, p3, 256, p3, q->pUnits[i].delayBuf2, q->tempBuf);
iqmf (p1, p3, 512, p1, q->pUnits[i].delayBuf3, q->tempBuf);
iqmf (p1, p2, 256, p1, q->pUnits[i].delayBuf1, q->tempBuf, false);
iqmf (p4, p3, 256, p3, q->pUnits[i].delayBuf2, q->tempBuf, false);
iqmf (p1, p3, 512, p1, q->pUnits[i].delayBuf3, q->tempBuf, true);
p1 +=1024;
}

View file

@ -35,6 +35,13 @@
#define ICONST_ATTR_LARGE_IRAM
#endif
/* The decoder works on samples with two fractional bits, s15.2. The last
* stage of the synthesis filter is scaled up by this many more bits, so the
* output has 28 bits below the sign like that of most codecs and the
* precision of that stage's multiplications is not thrown away. */
#define ATRAC3_OUT_SHIFT 11
#define ATRAC3_OUTPUT_DEPTH (17 + ATRAC3_OUT_SHIFT)
/* These structures are needed to store the parsed gain control data. */
typedef struct {
int num_gain_data;

View file

@ -71,6 +71,58 @@ atrac3_iqmf_matrixing:
.atrac3_iqmf_matrixing_end:
.size atrac3_iqmf_matrixing,.atrac3_iqmf_matrixing_end-atrac3_iqmf_matrixing
/* ATRAC3_OUT_SHIFT of atrac3.h */
#define OUT_SHIFT 11
/****************************************************************************
* void atrac3_iqmf_matrixing_out(int32_t *dest,
* int32_t *inlo,
* int32_t *inhi,
* unsigned int count);
*
* The matrixing step of the last iqmf stage: as atrac3_iqmf_matrixing, with
* the results scaled up by OUT_SHIFT bits, which then is the scale of the
* decoder's output. Not used on ARMv4, where the multiplications of the
* dewindowing would take longer for it: see atrac3_iqmf_dewindowing_out.
****************************************************************************/
#if ARM_ARCH >= 5
.align 2
.global atrac3_iqmf_matrixing_out
.type atrac3_iqmf_matrixing_out, %function
atrac3_iqmf_matrixing_out:
/* r0 = dest */
/* r1 = inlo */
/* r2 = inhi */
/* r3 = counter */
stmfd sp!, {r4-r9, lr} /* save non-scratch registers */
.iqmf_matrixing_out_loop:
ldmia r1!, { r4, r6, r8, r12} /* load inlo[0...3] */
ldmia r2!, { r5, r7, r9, lr } /* load inhi[0...3] */
add r4, r4, r5 /* r4 = inlo[0] + inhi[0] */
mov r4, r4, asl #OUT_SHIFT
sub r5, r4, r5, asl #OUT_SHIFT+1 /* r5 = inlo[0] - inhi[0] */
add r6, r6, r7 /* r6 = inlo[1] + inhi[1] */
mov r6, r6, asl #OUT_SHIFT
sub r7, r6, r7, asl #OUT_SHIFT+1 /* r7 = inlo[1] - inhi[1] */
add r8, r8, r9 /* r8 = inlo[2] + inhi[2] */
mov r8, r8, asl #OUT_SHIFT
sub r9, r8, r9, asl #OUT_SHIFT+1 /* r9 = inlo[2] - inhi[2] */
add r12, r12, lr /* r12 = inlo[3] + inhi[3] */
mov r12, r12, asl #OUT_SHIFT
sub lr , r12, lr, asl #OUT_SHIFT+1 /* lr = inlo[3] - inhi[3] */
stmia r0!, {r4-r9, r12, lr} /* store results to dest */
subs r3, r3, #4 /* counter -= 4 */
bgt .iqmf_matrixing_out_loop
ldmpc regs=r4-r9 /* restore registers */
.atrac3_iqmf_matrixing_out_end:
.size atrac3_iqmf_matrixing_out,.atrac3_iqmf_matrixing_out_end-atrac3_iqmf_matrixing_out
#endif /* ARM_ARCH >= 5 */
/****************************************************************************
* atrac3_iqmf_dewindowing(int32_t *out,
@ -122,11 +174,16 @@ atrac3_iqmf_matrixing:
smlal lr , r9, r7, r10; /* s1 += win[2k+3] * in[44-2k] */ \
smlal r12, r8, r6, r11; /* s2 += win[2k+2] * in[45-2k] */
/* atrac3_iqmf_dewindowing is the function above. atrac3_iqmf_dewindowing_out
* is the same for the last iqmf stage, with the result scaled up by OUT_SHIFT
* bits, which then is the scale of the decoder's output: the bits are taken
* from the 64-bit sums, that have them. */
.macro DEWINDOWING name, shift
.align 2
.global atrac3_iqmf_dewindowing
.type atrac3_iqmf_dewindowing, %function
atrac3_iqmf_dewindowing:
.global \name
.type \name, %function
\name:
/* r0 = dest */
/* r1 = input samples */
/* r2 = window coefficients */
@ -136,7 +193,7 @@ atrac3_iqmf_dewindowing:
stmfd sp!, {r3}
add r3, r1, #192 /* r3 = end of the 48 inputs */
.iqmf_dewindow_outer_loop: /* outer loop 0...counter-1 */
1: /* outer loop 0...counter-1 */
DEWIN_8_SAMPLES_ASM(smull, smull) /* in[ 0.. 3], in[44..47] */
DEWIN_8_SAMPLES_ASM(smlal, smlal) /* in[ 4.. 7], in[40..43] */
@ -145,10 +202,10 @@ atrac3_iqmf_dewindowing:
DEWIN_8_SAMPLES_ASM(smlal, smlal) /* in[16..19], in[28..31] */
DEWIN_8_SAMPLES_ASM(smlal, smlal) /* in[20..23], in[24..27] */
mov lr , lr , lsr #31
orr r9, lr , r9, lsl #1 /* s1 = low>>31 || hi<<1 */
mov r12, r12, lsr #31
orr r8, r12, r8, lsl #1 /* s2 = low>>31 || hi<<1 */
mov lr , lr , lsr #31-\shift
orr r9, lr , r9, lsl #1+\shift /* s1 = low>>31 || hi<<1, << shift */
mov r12, r12, lsr #31-\shift
orr r8, r12, r8, lsl #1+\shift /* s2 = low>>31 || hi<<1, << shift */
stmia r0!, {r8, r9} /* store result out[0]=s2, out[1]=s1 */
ldr r4, [sp] /* end of dest */
@ -157,10 +214,13 @@ atrac3_iqmf_dewindowing:
sub r2, r2, #96 /* roll back 24 entries = win[0] */
cmp r0, r4
bne .iqmf_dewindow_outer_loop
bne 1b
add sp, sp, #4
ldmpc regs=r4-r11 /* restore registers */
.atrac3_iqmf_dewindowing_end:
.size atrac3_iqmf_dewindowing,.atrac3_iqmf_dewindowing_end-atrac3_iqmf_dewindowing
.size \name, .-\name
.endm
DEWINDOWING atrac3_iqmf_dewindowing, 0
DEWINDOWING atrac3_iqmf_dewindowing_out, OUT_SHIFT