mirror of
https://github.com/Rockbox/rockbox.git
synced 2026-10-09 23:53:28 -04:00
atrac3: output samples with 28 fractional bits
The decoder gave the DSP samples with two fractional bits (sample depth 17), far fewer than any other codec. The DSP's filters are only as precise as the samples they are given: with an equalizer band at a low frequency the output was noise, or held at full scale. Scale the last stage of the synthesis filter up by 11 bits and set the sample depth to 28. On ARMv4 the bits are taken from the 64-bit sums of the dewindowing, which had them. Elsewhere the input of that stage is scaled, in its matrixing step; that is not done on ARMv4 because its multiplier takes longer for the larger operands (51.97 against 48.85 MHz below). The Coldfire path uses the C matrixing and has not been run. Seven files decoded under perfsim for the Sansa e200v1 and Clip+: shifted down again, the e200v1 output is within one step of the old output, and the Clip+ output no further from the e200v1's than before. Through a ten band equalizer with bands at 32 and 64 Hz (the filter fix of the DSP included), the noise of atrac3_lp2_132.oma falls from -64 dB to -101 dB relative to the signal. Estimated with perfsim (not measured on a device), atrac3_lp2_132.oma: e200v1 48.81 -> 48.85 MHz, Clip+ 26.91 -> 26.96 MHz. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
parent
3966b420a3
commit
02d5520c5f
5 changed files with 133 additions and 18 deletions
|
|
@ -63,7 +63,7 @@ enum codec_status codec_run(void)
|
|||
ci->memset(&q,0,sizeof(ATRAC3Context));
|
||||
|
||||
ci->configure(DSP_SET_FREQUENCY, ci->id3->frequency);
|
||||
ci->configure(DSP_SET_SAMPLE_DEPTH, 17); /* Remark: atrac3 uses s15.0 by default, s15.2 was hacked. */
|
||||
ci->configure(DSP_SET_SAMPLE_DEPTH, ATRAC3_OUTPUT_DEPTH);
|
||||
const uint8_t channels = 2;
|
||||
ci->configure(DSP_SET_STEREO_MODE, channels == 1 ?
|
||||
STEREO_MONO : STEREO_NONINTERLEAVED);
|
||||
|
|
|
|||
|
|
@ -92,7 +92,7 @@ enum codec_status codec_run(void)
|
|||
init_rm(&rmctx);
|
||||
|
||||
ci->configure(DSP_SET_FREQUENCY, ci->id3->frequency);
|
||||
ci->configure(DSP_SET_SAMPLE_DEPTH, 17); /* Remark: atrac3 uses s15.0 by default, s15.2 was hacked. */
|
||||
ci->configure(DSP_SET_SAMPLE_DEPTH, ATRAC3_OUTPUT_DEPTH);
|
||||
ci->configure(DSP_SET_STEREO_MODE, rmctx.nb_channels == 1 ?
|
||||
STEREO_MONO : STEREO_NONINTERLEAVED);
|
||||
|
||||
|
|
|
|||
|
|
@ -86,6 +86,13 @@ static int vlcs_initialized = 0;
|
|||
int32_t *inlo,
|
||||
int32_t *inhi,
|
||||
unsigned int nIn);
|
||||
/* The same, with the results scaled up by ATRAC3_OUT_SHIFT bits
|
||||
* (ARMv5 and later) */
|
||||
extern void
|
||||
atrac3_iqmf_matrixing_out(int32_t *p3,
|
||||
int32_t *inlo,
|
||||
int32_t *inhi,
|
||||
unsigned int nIn);
|
||||
#else
|
||||
static inline void
|
||||
atrac3_iqmf_matrixing(int32_t *p3,
|
||||
|
|
@ -101,6 +108,22 @@ static int vlcs_initialized = 0;
|
|||
p3[2*i+3] = inlo[i+1] - inhi[i+1];
|
||||
}
|
||||
}
|
||||
|
||||
/* The same, with the results scaled up by ATRAC3_OUT_SHIFT bits */
|
||||
static inline void
|
||||
atrac3_iqmf_matrixing_out(int32_t *p3,
|
||||
int32_t *inlo,
|
||||
int32_t *inhi,
|
||||
unsigned int nIn)
|
||||
{
|
||||
uint32_t i;
|
||||
for(i=0; i<nIn; i+=2){
|
||||
p3[2*i+0] = (inlo[i ] + inhi[i ]) * (1 << ATRAC3_OUT_SHIFT);
|
||||
p3[2*i+1] = (inlo[i ] - inhi[i ]) * (1 << ATRAC3_OUT_SHIFT);
|
||||
p3[2*i+2] = (inlo[i+1] + inhi[i+1]) * (1 << ATRAC3_OUT_SHIFT);
|
||||
p3[2*i+3] = (inlo[i+1] - inhi[i+1]) * (1 << ATRAC3_OUT_SHIFT);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
|
|
@ -150,6 +173,15 @@ static int vlcs_initialized = 0;
|
|||
int32_t *in,
|
||||
int32_t *win,
|
||||
unsigned int nIn);
|
||||
/* The same, with the results scaled up by ATRAC3_OUT_SHIFT bits. The
|
||||
* ARMv4 multiplier takes longer for large operands, so there the last
|
||||
* stage is scaled here and not in its matrixing. */
|
||||
#define ATRAC3_SCALE_IN_DEWINDOWING
|
||||
extern void
|
||||
atrac3_iqmf_dewindowing_out(int32_t *out,
|
||||
int32_t *in,
|
||||
int32_t *win,
|
||||
unsigned int nIn);
|
||||
|
||||
#elif defined (CPU_COLDFIRE)
|
||||
#define MULTIPLY_ADD_BLOCK \
|
||||
|
|
@ -274,19 +306,35 @@ atrac3_imdct_windowing(int32_t *buffer,
|
|||
* @param pOut out buffer
|
||||
* @param delayBuf delayBuf buffer
|
||||
* @param temp temp buffer
|
||||
* @param last true for the stage that makes the output, which is
|
||||
* scaled up by ATRAC3_OUT_SHIFT bits
|
||||
*/
|
||||
|
||||
static void iqmf (int32_t *inlo, int32_t *inhi, unsigned int nIn, int32_t *pOut, int32_t *delayBuf, int32_t *temp)
|
||||
static void iqmf (int32_t *inlo, int32_t *inhi, unsigned int nIn, int32_t *pOut, int32_t *delayBuf, int32_t *temp, bool last)
|
||||
{
|
||||
|
||||
/* Restore the delay buffer */
|
||||
memcpy(temp, delayBuf, 46*sizeof(int32_t));
|
||||
|
||||
#ifdef ATRAC3_SCALE_IN_DEWINDOWING
|
||||
/* loop1: matrixing */
|
||||
atrac3_iqmf_matrixing(temp + 46, inlo, inhi, nIn);
|
||||
|
||||
/* loop2: dewindowing */
|
||||
if (last)
|
||||
atrac3_iqmf_dewindowing_out(pOut, temp, qmf_window, nIn);
|
||||
else
|
||||
atrac3_iqmf_dewindowing(pOut, temp, qmf_window, nIn);
|
||||
#else
|
||||
/* loop1: matrixing */
|
||||
if (last)
|
||||
atrac3_iqmf_matrixing_out(temp + 46, inlo, inhi, nIn);
|
||||
else
|
||||
atrac3_iqmf_matrixing(temp + 46, inlo, inhi, nIn);
|
||||
|
||||
/* loop2: dewindowing */
|
||||
atrac3_iqmf_dewindowing(pOut, temp, qmf_window, nIn);
|
||||
#endif
|
||||
|
||||
/* Save the delay buffer */
|
||||
memcpy(delayBuf, temp + (nIn << 1), 46*sizeof(int32_t));
|
||||
|
|
@ -1105,9 +1153,9 @@ static int decodeFrame(ATRAC3Context *q, const uint8_t* databuf, int off)
|
|||
p2= p1+256;
|
||||
p3= p2+256;
|
||||
p4= p3+256;
|
||||
iqmf (p1, p2, 256, p1, q->pUnits[i].delayBuf1, q->tempBuf);
|
||||
iqmf (p4, p3, 256, p3, q->pUnits[i].delayBuf2, q->tempBuf);
|
||||
iqmf (p1, p3, 512, p1, q->pUnits[i].delayBuf3, q->tempBuf);
|
||||
iqmf (p1, p2, 256, p1, q->pUnits[i].delayBuf1, q->tempBuf, false);
|
||||
iqmf (p4, p3, 256, p3, q->pUnits[i].delayBuf2, q->tempBuf, false);
|
||||
iqmf (p1, p3, 512, p1, q->pUnits[i].delayBuf3, q->tempBuf, true);
|
||||
p1 +=1024;
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -35,6 +35,13 @@
|
|||
#define ICONST_ATTR_LARGE_IRAM
|
||||
#endif
|
||||
|
||||
/* The decoder works on samples with two fractional bits, s15.2. The last
|
||||
* stage of the synthesis filter is scaled up by this many more bits, so the
|
||||
* output has 28 bits below the sign like that of most codecs and the
|
||||
* precision of that stage's multiplications is not thrown away. */
|
||||
#define ATRAC3_OUT_SHIFT 11
|
||||
#define ATRAC3_OUTPUT_DEPTH (17 + ATRAC3_OUT_SHIFT)
|
||||
|
||||
/* These structures are needed to store the parsed gain control data. */
|
||||
typedef struct {
|
||||
int num_gain_data;
|
||||
|
|
|
|||
|
|
@ -71,6 +71,58 @@ atrac3_iqmf_matrixing:
|
|||
.atrac3_iqmf_matrixing_end:
|
||||
.size atrac3_iqmf_matrixing,.atrac3_iqmf_matrixing_end-atrac3_iqmf_matrixing
|
||||
|
||||
|
||||
/* ATRAC3_OUT_SHIFT of atrac3.h */
|
||||
#define OUT_SHIFT 11
|
||||
|
||||
/****************************************************************************
|
||||
* void atrac3_iqmf_matrixing_out(int32_t *dest,
|
||||
* int32_t *inlo,
|
||||
* int32_t *inhi,
|
||||
* unsigned int count);
|
||||
*
|
||||
* The matrixing step of the last iqmf stage: as atrac3_iqmf_matrixing, with
|
||||
* the results scaled up by OUT_SHIFT bits, which then is the scale of the
|
||||
* decoder's output. Not used on ARMv4, where the multiplications of the
|
||||
* dewindowing would take longer for it: see atrac3_iqmf_dewindowing_out.
|
||||
****************************************************************************/
|
||||
#if ARM_ARCH >= 5
|
||||
.align 2
|
||||
.global atrac3_iqmf_matrixing_out
|
||||
.type atrac3_iqmf_matrixing_out, %function
|
||||
|
||||
atrac3_iqmf_matrixing_out:
|
||||
/* r0 = dest */
|
||||
/* r1 = inlo */
|
||||
/* r2 = inhi */
|
||||
/* r3 = counter */
|
||||
stmfd sp!, {r4-r9, lr} /* save non-scratch registers */
|
||||
|
||||
.iqmf_matrixing_out_loop:
|
||||
ldmia r1!, { r4, r6, r8, r12} /* load inlo[0...3] */
|
||||
ldmia r2!, { r5, r7, r9, lr } /* load inhi[0...3] */
|
||||
add r4, r4, r5 /* r4 = inlo[0] + inhi[0] */
|
||||
mov r4, r4, asl #OUT_SHIFT
|
||||
sub r5, r4, r5, asl #OUT_SHIFT+1 /* r5 = inlo[0] - inhi[0] */
|
||||
add r6, r6, r7 /* r6 = inlo[1] + inhi[1] */
|
||||
mov r6, r6, asl #OUT_SHIFT
|
||||
sub r7, r6, r7, asl #OUT_SHIFT+1 /* r7 = inlo[1] - inhi[1] */
|
||||
add r8, r8, r9 /* r8 = inlo[2] + inhi[2] */
|
||||
mov r8, r8, asl #OUT_SHIFT
|
||||
sub r9, r8, r9, asl #OUT_SHIFT+1 /* r9 = inlo[2] - inhi[2] */
|
||||
add r12, r12, lr /* r12 = inlo[3] + inhi[3] */
|
||||
mov r12, r12, asl #OUT_SHIFT
|
||||
sub lr , r12, lr, asl #OUT_SHIFT+1 /* lr = inlo[3] - inhi[3] */
|
||||
stmia r0!, {r4-r9, r12, lr} /* store results to dest */
|
||||
subs r3, r3, #4 /* counter -= 4 */
|
||||
bgt .iqmf_matrixing_out_loop
|
||||
|
||||
ldmpc regs=r4-r9 /* restore registers */
|
||||
|
||||
.atrac3_iqmf_matrixing_out_end:
|
||||
.size atrac3_iqmf_matrixing_out,.atrac3_iqmf_matrixing_out_end-atrac3_iqmf_matrixing_out
|
||||
#endif /* ARM_ARCH >= 5 */
|
||||
|
||||
|
||||
/****************************************************************************
|
||||
* atrac3_iqmf_dewindowing(int32_t *out,
|
||||
|
|
@ -122,11 +174,16 @@ atrac3_iqmf_matrixing:
|
|||
smlal lr , r9, r7, r10; /* s1 += win[2k+3] * in[44-2k] */ \
|
||||
smlal r12, r8, r6, r11; /* s2 += win[2k+2] * in[45-2k] */
|
||||
|
||||
/* atrac3_iqmf_dewindowing is the function above. atrac3_iqmf_dewindowing_out
|
||||
* is the same for the last iqmf stage, with the result scaled up by OUT_SHIFT
|
||||
* bits, which then is the scale of the decoder's output: the bits are taken
|
||||
* from the 64-bit sums, that have them. */
|
||||
.macro DEWINDOWING name, shift
|
||||
.align 2
|
||||
.global atrac3_iqmf_dewindowing
|
||||
.type atrac3_iqmf_dewindowing, %function
|
||||
|
||||
atrac3_iqmf_dewindowing:
|
||||
.global \name
|
||||
.type \name, %function
|
||||
|
||||
\name:
|
||||
/* r0 = dest */
|
||||
/* r1 = input samples */
|
||||
/* r2 = window coefficients */
|
||||
|
|
@ -136,7 +193,7 @@ atrac3_iqmf_dewindowing:
|
|||
stmfd sp!, {r3}
|
||||
add r3, r1, #192 /* r3 = end of the 48 inputs */
|
||||
|
||||
.iqmf_dewindow_outer_loop: /* outer loop 0...counter-1 */
|
||||
1: /* outer loop 0...counter-1 */
|
||||
|
||||
DEWIN_8_SAMPLES_ASM(smull, smull) /* in[ 0.. 3], in[44..47] */
|
||||
DEWIN_8_SAMPLES_ASM(smlal, smlal) /* in[ 4.. 7], in[40..43] */
|
||||
|
|
@ -145,10 +202,10 @@ atrac3_iqmf_dewindowing:
|
|||
DEWIN_8_SAMPLES_ASM(smlal, smlal) /* in[16..19], in[28..31] */
|
||||
DEWIN_8_SAMPLES_ASM(smlal, smlal) /* in[20..23], in[24..27] */
|
||||
|
||||
mov lr , lr , lsr #31
|
||||
orr r9, lr , r9, lsl #1 /* s1 = low>>31 || hi<<1 */
|
||||
mov r12, r12, lsr #31
|
||||
orr r8, r12, r8, lsl #1 /* s2 = low>>31 || hi<<1 */
|
||||
mov lr , lr , lsr #31-\shift
|
||||
orr r9, lr , r9, lsl #1+\shift /* s1 = low>>31 || hi<<1, << shift */
|
||||
mov r12, r12, lsr #31-\shift
|
||||
orr r8, r12, r8, lsl #1+\shift /* s2 = low>>31 || hi<<1, << shift */
|
||||
|
||||
stmia r0!, {r8, r9} /* store result out[0]=s2, out[1]=s1 */
|
||||
ldr r4, [sp] /* end of dest */
|
||||
|
|
@ -157,10 +214,13 @@ atrac3_iqmf_dewindowing:
|
|||
sub r2, r2, #96 /* roll back 24 entries = win[0] */
|
||||
|
||||
cmp r0, r4
|
||||
bne .iqmf_dewindow_outer_loop
|
||||
bne 1b
|
||||
|
||||
add sp, sp, #4
|
||||
ldmpc regs=r4-r11 /* restore registers */
|
||||
|
||||
.atrac3_iqmf_dewindowing_end:
|
||||
.size atrac3_iqmf_dewindowing,.atrac3_iqmf_dewindowing_end-atrac3_iqmf_dewindowing
|
||||
.size \name, .-\name
|
||||
.endm
|
||||
|
||||
DEWINDOWING atrac3_iqmf_dewindowing, 0
|
||||
DEWINDOWING atrac3_iqmf_dewindowing_out, OUT_SHIFT
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue