mirror of
https://github.com/Rockbox/rockbox.git
synced 2026-10-10 08:03:04 -04:00
opus: ARM stereo de-emphasis kernel
deemphasis_stereo_simple on both cores, with the filter state kept unshifted and shifted inside the add that consumes it. Modelled: -0.53% ARMv4, -0.96% ARMv5E. Measured with the three preceding commits: e200v1 49.42 -> 48.96 MHz, Clip+ 32.83 -> 30.89 MHz. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Change-Id: Id09297974f3b66a5a193b524d68fd8844f71cc02
This commit is contained in:
parent
bdbbb753c5
commit
08d3332edf
5 changed files with 215 additions and 0 deletions
|
|
@ -13,12 +13,14 @@ celt/arm/kiss_fft_armv4_asm.S
|
|||
celt/arm/mdct_armv4_asm.S
|
||||
celt/arm/comb_filter_armv4_asm.S
|
||||
celt/arm/denorm_armv4_asm.S
|
||||
celt/arm/deemph_armv4_asm.S
|
||||
#elif defined(CPU_ARM) && (ARM_ARCH == 5)
|
||||
celt/arm/comb_filter_armv5e_asm.S
|
||||
celt/arm/denorm_armv5e_asm.S
|
||||
celt/arm/exp_rotation1_armv5e_asm.S
|
||||
celt/arm/haar1_armv5e_asm.S
|
||||
celt/arm/normres_armv5e_asm.S
|
||||
celt/arm/deemph_armv5e_asm.S
|
||||
#endif
|
||||
celt/laplace.c
|
||||
celt/mathops.c
|
||||
|
|
|
|||
|
|
@ -54,6 +54,13 @@ void comb_filter_const_armv4(opus_val32 *y, opus_val32 *x, int T, int N,
|
|||
void celt_sat_armv4(celt_sig *x, int n);
|
||||
# define celt_sat(x, n) celt_sat_armv4((x), (n))
|
||||
|
||||
/* deemphasis_stereo_simple, the last stage before PCM. */
|
||||
# define OVERRIDE_DEEMPH_STEREO
|
||||
void deemph_stereo_armv4(celt_sig *x0, celt_sig *x1, opus_val16 *pcm, int N,
|
||||
int coef0, celt_sig *mem);
|
||||
# define deemphasis_stereo_simple(in, pcm, N, coef0, mem) \
|
||||
deemph_stereo_armv4((in)[0], (in)[1], (pcm), (N), (coef0), (mem))
|
||||
|
||||
# elif defined(OPUS_ARM_INLINE_EDSP) && (ARM_ARCH == 5)
|
||||
|
||||
# define OVERRIDE_COMB_FILTER_CONST
|
||||
|
|
@ -67,6 +74,12 @@ void comb_filter_const_armv5e(opus_val32 *y, opus_val32 *x, int T, int N,
|
|||
void celt_sat_armv5e(celt_sig *x, int n);
|
||||
# define celt_sat(x, n) celt_sat_armv5e((x), (n))
|
||||
|
||||
# define OVERRIDE_DEEMPH_STEREO
|
||||
void deemph_stereo_armv5e(celt_sig *x0, celt_sig *x1, opus_val16 *pcm, int N,
|
||||
int coef0, celt_sig *mem);
|
||||
# define deemphasis_stereo_simple(in, pcm, N, coef0, mem) \
|
||||
deemph_stereo_armv5e((in)[0], (in)[1], (pcm), (N), (coef0), (mem))
|
||||
|
||||
# endif
|
||||
|
||||
#endif
|
||||
|
|
|
|||
101
lib/rbcodec/codecs/libopus/celt/arm/deemph_armv4_asm.S
Normal file
101
lib/rbcodec/codecs/libopus/celt/arm/deemph_armv4_asm.S
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
/* ARMv4 stereo de-emphasis for the CELT decoder output.
|
||||
*
|
||||
* Copyright (c) 2007-2008 CSIRO
|
||||
* Copyright (c) 2007-2010 Xiph.Org Foundation
|
||||
* Copyright (c) 2008 Gregory Maxwell
|
||||
* Copyright (c) 2026 Michael Giacomelli
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the conditions stated in
|
||||
* celt/celt_decoder.c are met.
|
||||
*
|
||||
* Why this exists
|
||||
* ---------------
|
||||
* deemphasis_stereo_simple is the decoder's last stage: a one-pole filter
|
||||
* per channel and the conversion to 16-bit PCM. On the ARM7TDMI the two
|
||||
* smull dominate and cannot be avoided, but the compiled loop also shifts
|
||||
* both filter states left once per sample and clamps to 16 bits with four
|
||||
* instructions and a register holding -32768.
|
||||
*
|
||||
* Here the state is kept as the high word of the smull and shifted inside
|
||||
* the add that consumes it, "add r8, r8, r4, lsl #1", and the clamp is the
|
||||
* three-instruction asr / teq / eorne form. The incoming state from mem[]
|
||||
* is a full value, so the first sample adds it unshifted. See
|
||||
* deemph_armv5e_asm.S for the clamp.
|
||||
*
|
||||
* Bit-exact: the high word of smull(tmp, coef0<<16) shifted left one is
|
||||
* exactly MULT16_32_Q15_armv4, low bit dropped included.
|
||||
*/
|
||||
|
||||
#if defined(__thumb__) || defined(__thumb2__)
|
||||
#error "deemph_armv4_asm.S must be assembled in ARM mode"
|
||||
#endif
|
||||
|
||||
/* These kernels are 2,892 bytes that carry 43% of the cycles an ARMv4 decode
|
||||
* spends, which makes them the densest thing in the codec to put in IRAM.
|
||||
* config.h sets OPUS_ARM_ICODE on the targets whose codec IRAM window has
|
||||
* room; everywhere else this stays in .text. The firmware config.h it reads
|
||||
* that from is assembly-safe, and libopus/config.h guards the rest of its
|
||||
* includes with __ASSEMBLER__ so this include costs nothing here. */
|
||||
#ifdef HAVE_CONFIG_H
|
||||
#include "config.h"
|
||||
#endif
|
||||
|
||||
.text
|
||||
.align 2
|
||||
|
||||
.macro PCM16 src, tmp
|
||||
add \tmp, \src, #0x800 @ PSHR32(x, 12)
|
||||
mov \tmp, \tmp, asr #12
|
||||
mov lr, \tmp, asr #15
|
||||
teq lr, \tmp, asr #31
|
||||
eorne \tmp, r7, \tmp, asr #31
|
||||
strh \tmp, [r2], #2
|
||||
.endm
|
||||
|
||||
/* ------------------------------------------------------------------------
|
||||
* void deemph_stereo_armv4(celt_sig *x0, celt_sig *x1, opus_val16 *pcm,
|
||||
* int N, int coef0, celt_sig *mem)
|
||||
*
|
||||
* r0 x0, r1 x1, r2 pcm, r6 the count, r4 and r5 the two filter states,
|
||||
* r7 32767, r8 and r9 the two channel sums, r10 coef0<<16, r11 and lr
|
||||
* scratch, ip mem. smull needs RdLo, RdHi and Rm distinct: lr, r4/r5, r8/r9.
|
||||
* ------------------------------------------------------------------------ */
|
||||
.global deemph_stereo_armv4
|
||||
.type deemph_stereo_armv4, %function
|
||||
deemph_stereo_armv4:
|
||||
push {r4-r11, lr}
|
||||
ldr r10, [sp, #36] @ coef0
|
||||
ldr ip, [sp, #40] @ mem
|
||||
mov r10, r10, lsl #16 @ into the top half of Rs, as the C does
|
||||
mov r6, r3
|
||||
ldmia ip, {r4, r5} @ m0, m1, full values
|
||||
mov r7, #0x7F00
|
||||
orr r7, r7, #0xFF
|
||||
cmp r6, #0
|
||||
ble .Lde4_store
|
||||
ldr r8, [r0], #4
|
||||
ldr r9, [r1], #4
|
||||
add r8, r8, r4 @ first sample: the state as stored
|
||||
add r9, r9, r5
|
||||
b .Lde4_body
|
||||
.Lde4_loop:
|
||||
ldr r8, [r0], #4
|
||||
ldr r9, [r1], #4
|
||||
add r8, r8, r4, lsl #1 @ tmp0 = x0[j] + m0
|
||||
add r9, r9, r5, lsl #1 @ tmp1 = x1[j] + m1
|
||||
.Lde4_body:
|
||||
smull lr, r4, r8, r10 @ m0 = r4 << 1
|
||||
smull lr, r5, r9, r10 @ m1 = r5 << 1
|
||||
PCM16 r8, r11 @ pcm[2j]
|
||||
PCM16 r9, r11 @ pcm[2j+1]
|
||||
subs r6, r6, #1
|
||||
bne .Lde4_loop
|
||||
mov r4, r4, lsl #1
|
||||
mov r5, r5, lsl #1
|
||||
.Lde4_store:
|
||||
stmia ip, {r4, r5} @ mem[0], mem[1]
|
||||
pop {r4-r11, pc}
|
||||
.size deemph_stereo_armv4, .-deemph_stereo_armv4
|
||||
|
||||
.section .note.GNU-stack,"",%progbits
|
||||
97
lib/rbcodec/codecs/libopus/celt/arm/deemph_armv5e_asm.S
Normal file
97
lib/rbcodec/codecs/libopus/celt/arm/deemph_armv5e_asm.S
Normal file
|
|
@ -0,0 +1,97 @@
|
|||
/* ARMv5E stereo de-emphasis for the CELT decoder output.
|
||||
*
|
||||
* Copyright (c) 2007-2008 CSIRO
|
||||
* Copyright (c) 2007-2010 Xiph.Org Foundation
|
||||
* Copyright (c) 2008 Gregory Maxwell
|
||||
* Copyright (c) 2026 Michael Giacomelli
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the conditions stated in
|
||||
* celt/celt_decoder.c are met.
|
||||
*
|
||||
* Why this exists
|
||||
* ---------------
|
||||
* deemphasis_stereo_simple is the decoder's last stage: a one-pole filter
|
||||
* per channel and the conversion to 16-bit PCM. The compiled loop feeds
|
||||
* each channel's ldr straight into an add, a stall apiece, shifts both
|
||||
* filter states left once per sample, and clamps to 16 bits with four
|
||||
* instructions and a register holding -32768.
|
||||
*
|
||||
* Here the state is kept as the smulwb result and shifted inside the add
|
||||
* that consumes it, "add r8, r8, r4, lsl #1", which removes two shifts per
|
||||
* sample. The incoming state from mem[] is a full value, so the first
|
||||
* sample adds it unshifted and joins the loop after the adds; nothing is
|
||||
* assumed about its low bit. The clamp is the three-instruction form
|
||||
*
|
||||
* mov lr, y, asr #15
|
||||
* teq lr, y, asr #31 @ in range iff bits 31..15 agree
|
||||
* eorne y, r7, y, asr #31 @ else 0x7FFF, or 0xFFFF8000 for negatives
|
||||
*
|
||||
* whose low halfword is exactly the clamped value strh stores.
|
||||
*
|
||||
* Bit-exact: smulwb followed by a shift is MULT16_32_Q15 on this
|
||||
* architecture, the rounding shift is PSHR32(x, SIG_SHIFT), and additions
|
||||
* wrap exactly as the C does.
|
||||
*/
|
||||
|
||||
#if defined(__thumb__) || defined(__thumb2__)
|
||||
#error "deemph_armv5e_asm.S must be assembled in ARM mode"
|
||||
#endif
|
||||
|
||||
.text
|
||||
.align 2
|
||||
|
||||
.macro PCM16 src, tmp
|
||||
add \tmp, \src, #0x800 @ PSHR32(x, 12)
|
||||
mov \tmp, \tmp, asr #12
|
||||
mov lr, \tmp, asr #15
|
||||
teq lr, \tmp, asr #31
|
||||
eorne \tmp, r7, \tmp, asr #31
|
||||
strh \tmp, [r2], #2
|
||||
.endm
|
||||
|
||||
/* ------------------------------------------------------------------------
|
||||
* void deemph_stereo_armv5e(celt_sig *x0, celt_sig *x1, opus_val16 *pcm,
|
||||
* int N, int coef0, celt_sig *mem)
|
||||
*
|
||||
* r0 x0, r1 x1, r2 pcm, r3 and r6 the count, r4 and r5 the two filter
|
||||
* states, r7 32767, r8 and r9 the two channel sums, r10 coef0, r11 and lr
|
||||
* scratch, ip mem.
|
||||
* ------------------------------------------------------------------------ */
|
||||
.global deemph_stereo_armv5e
|
||||
.type deemph_stereo_armv5e, %function
|
||||
deemph_stereo_armv5e:
|
||||
push {r4-r11, lr}
|
||||
ldr r10, [sp, #36] @ coef0; smulwb reads its low halfword
|
||||
ldr ip, [sp, #40] @ mem
|
||||
mov r6, r3
|
||||
ldmia ip, {r4, r5} @ m0, m1, full values
|
||||
mov r7, #0x7F00
|
||||
orr r7, r7, #0xFF
|
||||
cmp r6, #0
|
||||
ble .Lde5_store
|
||||
ldr r8, [r0], #4
|
||||
ldr r9, [r1], #4
|
||||
add r8, r8, r4 @ first sample: the state as stored
|
||||
add r9, r9, r5
|
||||
b .Lde5_body
|
||||
.Lde5_loop:
|
||||
ldr r8, [r0], #4
|
||||
ldr r9, [r1], #4
|
||||
add r8, r8, r4, lsl #1 @ tmp0 = x0[j] + m0
|
||||
add r9, r9, r5, lsl #1 @ tmp1 = x1[j] + m1
|
||||
.Lde5_body:
|
||||
smulwb r4, r8, r10 @ m0 = this << 1
|
||||
smulwb r5, r9, r10 @ m1 = this << 1
|
||||
PCM16 r8, r11 @ pcm[2j]
|
||||
PCM16 r9, r11 @ pcm[2j+1]
|
||||
subs r6, r6, #1
|
||||
bne .Lde5_loop
|
||||
mov r4, r4, lsl #1
|
||||
mov r5, r5, lsl #1
|
||||
.Lde5_store:
|
||||
stmia ip, {r4, r5} @ mem[0], mem[1]
|
||||
pop {r4-r11, pc}
|
||||
.size deemph_stereo_armv5e, .-deemph_stereo_armv5e
|
||||
|
||||
.section .note.GNU-stack,"",%progbits
|
||||
|
|
@ -227,6 +227,7 @@ void opus_custom_decoder_destroy(CELTDecoder *st)
|
|||
/* Special case for stereo with no downsampling and no accumulation. This is
|
||||
quite common and we can make it faster by processing both channels in the
|
||||
same loop, reducing overhead due to the dependency loop in the IIR filter. */
|
||||
#ifndef OVERRIDE_DEEMPH_STEREO
|
||||
static void deemphasis_stereo_simple(celt_sig *in[], opus_val16 *pcm, int N, const opus_val16 coef0,
|
||||
celt_sig *mem)
|
||||
{
|
||||
|
|
@ -253,6 +254,7 @@ static void deemphasis_stereo_simple(celt_sig *in[], opus_val16 *pcm, int N, con
|
|||
mem[1] = m1;
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifndef RESYNTH
|
||||
static
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue