opus: ARM stereo de-emphasis kernel

deemphasis_stereo_simple on both cores, with the filter state kept
unshifted and shifted inside the add that consumes it.

Modelled: -0.53% ARMv4, -0.96% ARMv5E.
Measured with the three preceding commits: e200v1 49.42 -> 48.96 MHz,
Clip+ 32.83 -> 30.89 MHz.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Change-Id: Id09297974f3b66a5a193b524d68fd8844f71cc02
This commit is contained in:
Michael Giacomelli 2026-09-18 11:59:24 -04:00 • committed by Solomon Peachy
parent bdbbb753c5
commit 08d3332edf
5 changed files with 215 additions and 0 deletions

View file

@ -13,12 +13,14 @@ celt/arm/kiss_fft_armv4_asm.S
celt/arm/mdct_armv4_asm.S
celt/arm/comb_filter_armv4_asm.S
celt/arm/denorm_armv4_asm.S
celt/arm/deemph_armv4_asm.S
#elif defined(CPU_ARM) && (ARM_ARCH == 5)
celt/arm/comb_filter_armv5e_asm.S
celt/arm/denorm_armv5e_asm.S
celt/arm/exp_rotation1_armv5e_asm.S
celt/arm/haar1_armv5e_asm.S
celt/arm/normres_armv5e_asm.S
celt/arm/deemph_armv5e_asm.S
#endif
celt/laplace.c
celt/mathops.c

View file

@ -54,6 +54,13 @@ void comb_filter_const_armv4(opus_val32 *y, opus_val32 *x, int T, int N,
void celt_sat_armv4(celt_sig *x, int n);
# define celt_sat(x, n) celt_sat_armv4((x), (n))
/* deemphasis_stereo_simple, the last stage before PCM. */
# define OVERRIDE_DEEMPH_STEREO
void deemph_stereo_armv4(celt_sig *x0, celt_sig *x1, opus_val16 *pcm, int N,
int coef0, celt_sig *mem);
# define deemphasis_stereo_simple(in, pcm, N, coef0, mem) \
deemph_stereo_armv4((in)[0], (in)[1], (pcm), (N), (coef0), (mem))
# elif defined(OPUS_ARM_INLINE_EDSP) && (ARM_ARCH == 5)
# define OVERRIDE_COMB_FILTER_CONST
@ -67,6 +74,12 @@ void comb_filter_const_armv5e(opus_val32 *y, opus_val32 *x, int T, int N,
void celt_sat_armv5e(celt_sig *x, int n);
# define celt_sat(x, n) celt_sat_armv5e((x), (n))
# define OVERRIDE_DEEMPH_STEREO
void deemph_stereo_armv5e(celt_sig *x0, celt_sig *x1, opus_val16 *pcm, int N,
int coef0, celt_sig *mem);
# define deemphasis_stereo_simple(in, pcm, N, coef0, mem) \
deemph_stereo_armv5e((in)[0], (in)[1], (pcm), (N), (coef0), (mem))
# endif
#endif

View file

@ -0,0 +1,101 @@
/* ARMv4 stereo de-emphasis for the CELT decoder output.
*
* Copyright (c) 2007-2008 CSIRO
* Copyright (c) 2007-2010 Xiph.Org Foundation
* Copyright (c) 2008 Gregory Maxwell
* Copyright (c) 2026 Michael Giacomelli
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the conditions stated in
* celt/celt_decoder.c are met.
*
* Why this exists
* ---------------
* deemphasis_stereo_simple is the decoder's last stage: a one-pole filter
* per channel and the conversion to 16-bit PCM. On the ARM7TDMI the two
* smull dominate and cannot be avoided, but the compiled loop also shifts
* both filter states left once per sample and clamps to 16 bits with four
* instructions and a register holding -32768.
*
* Here the state is kept as the high word of the smull and shifted inside
* the add that consumes it, "add r8, r8, r4, lsl #1", and the clamp is the
* three-instruction asr / teq / eorne form. The incoming state from mem[]
* is a full value, so the first sample adds it unshifted. See
* deemph_armv5e_asm.S for the clamp.
*
* Bit-exact: the high word of smull(tmp, coef0<<16) shifted left one is
* exactly MULT16_32_Q15_armv4, low bit dropped included.
*/
#if defined(__thumb__) || defined(__thumb2__)
#error "deemph_armv4_asm.S must be assembled in ARM mode"
#endif
/* These kernels are 2,892 bytes that carry 43% of the cycles an ARMv4 decode
* spends, which makes them the densest thing in the codec to put in IRAM.
* config.h sets OPUS_ARM_ICODE on the targets whose codec IRAM window has
* room; everywhere else this stays in .text. The firmware config.h it reads
* that from is assembly-safe, and libopus/config.h guards the rest of its
* includes with __ASSEMBLER__ so this include costs nothing here. */
#ifdef HAVE_CONFIG_H
#include "config.h"
#endif
.text
.align 2
.macro PCM16 src, tmp
add \tmp, \src, #0x800 @ PSHR32(x, 12)
mov \tmp, \tmp, asr #12
mov lr, \tmp, asr #15
teq lr, \tmp, asr #31
eorne \tmp, r7, \tmp, asr #31
strh \tmp, [r2], #2
.endm
/* ------------------------------------------------------------------------
* void deemph_stereo_armv4(celt_sig *x0, celt_sig *x1, opus_val16 *pcm,
* int N, int coef0, celt_sig *mem)
*
* r0 x0, r1 x1, r2 pcm, r6 the count, r4 and r5 the two filter states,
* r7 32767, r8 and r9 the two channel sums, r10 coef0<<16, r11 and lr
* scratch, ip mem. smull needs RdLo, RdHi and Rm distinct: lr, r4/r5, r8/r9.
* ------------------------------------------------------------------------ */
.global deemph_stereo_armv4
.type deemph_stereo_armv4, %function
deemph_stereo_armv4:
push {r4-r11, lr}
ldr r10, [sp, #36] @ coef0
ldr ip, [sp, #40] @ mem
mov r10, r10, lsl #16 @ into the top half of Rs, as the C does
mov r6, r3
ldmia ip, {r4, r5} @ m0, m1, full values
mov r7, #0x7F00
orr r7, r7, #0xFF
cmp r6, #0
ble .Lde4_store
ldr r8, [r0], #4
ldr r9, [r1], #4
add r8, r8, r4 @ first sample: the state as stored
add r9, r9, r5
b .Lde4_body
.Lde4_loop:
ldr r8, [r0], #4
ldr r9, [r1], #4
add r8, r8, r4, lsl #1 @ tmp0 = x0[j] + m0
add r9, r9, r5, lsl #1 @ tmp1 = x1[j] + m1
.Lde4_body:
smull lr, r4, r8, r10 @ m0 = r4 << 1
smull lr, r5, r9, r10 @ m1 = r5 << 1
PCM16 r8, r11 @ pcm[2j]
PCM16 r9, r11 @ pcm[2j+1]
subs r6, r6, #1
bne .Lde4_loop
mov r4, r4, lsl #1
mov r5, r5, lsl #1
.Lde4_store:
stmia ip, {r4, r5} @ mem[0], mem[1]
pop {r4-r11, pc}
.size deemph_stereo_armv4, .-deemph_stereo_armv4
.section .note.GNU-stack,"",%progbits

View file

@ -0,0 +1,97 @@
/* ARMv5E stereo de-emphasis for the CELT decoder output.
*
* Copyright (c) 2007-2008 CSIRO
* Copyright (c) 2007-2010 Xiph.Org Foundation
* Copyright (c) 2008 Gregory Maxwell
* Copyright (c) 2026 Michael Giacomelli
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the conditions stated in
* celt/celt_decoder.c are met.
*
* Why this exists
* ---------------
* deemphasis_stereo_simple is the decoder's last stage: a one-pole filter
* per channel and the conversion to 16-bit PCM. The compiled loop feeds
* each channel's ldr straight into an add, a stall apiece, shifts both
* filter states left once per sample, and clamps to 16 bits with four
* instructions and a register holding -32768.
*
* Here the state is kept as the smulwb result and shifted inside the add
* that consumes it, "add r8, r8, r4, lsl #1", which removes two shifts per
* sample. The incoming state from mem[] is a full value, so the first
* sample adds it unshifted and joins the loop after the adds; nothing is
* assumed about its low bit. The clamp is the three-instruction form
*
* mov lr, y, asr #15
* teq lr, y, asr #31 @ in range iff bits 31..15 agree
* eorne y, r7, y, asr #31 @ else 0x7FFF, or 0xFFFF8000 for negatives
*
* whose low halfword is exactly the clamped value strh stores.
*
* Bit-exact: smulwb followed by a shift is MULT16_32_Q15 on this
* architecture, the rounding shift is PSHR32(x, SIG_SHIFT), and additions
* wrap exactly as the C does.
*/
#if defined(__thumb__) || defined(__thumb2__)
#error "deemph_armv5e_asm.S must be assembled in ARM mode"
#endif
.text
.align 2
.macro PCM16 src, tmp
add \tmp, \src, #0x800 @ PSHR32(x, 12)
mov \tmp, \tmp, asr #12
mov lr, \tmp, asr #15
teq lr, \tmp, asr #31
eorne \tmp, r7, \tmp, asr #31
strh \tmp, [r2], #2
.endm
/* ------------------------------------------------------------------------
* void deemph_stereo_armv5e(celt_sig *x0, celt_sig *x1, opus_val16 *pcm,
* int N, int coef0, celt_sig *mem)
*
* r0 x0, r1 x1, r2 pcm, r3 and r6 the count, r4 and r5 the two filter
* states, r7 32767, r8 and r9 the two channel sums, r10 coef0, r11 and lr
* scratch, ip mem.
* ------------------------------------------------------------------------ */
.global deemph_stereo_armv5e
.type deemph_stereo_armv5e, %function
deemph_stereo_armv5e:
push {r4-r11, lr}
ldr r10, [sp, #36] @ coef0; smulwb reads its low halfword
ldr ip, [sp, #40] @ mem
mov r6, r3
ldmia ip, {r4, r5} @ m0, m1, full values
mov r7, #0x7F00
orr r7, r7, #0xFF
cmp r6, #0
ble .Lde5_store
ldr r8, [r0], #4
ldr r9, [r1], #4
add r8, r8, r4 @ first sample: the state as stored
add r9, r9, r5
b .Lde5_body
.Lde5_loop:
ldr r8, [r0], #4
ldr r9, [r1], #4
add r8, r8, r4, lsl #1 @ tmp0 = x0[j] + m0
add r9, r9, r5, lsl #1 @ tmp1 = x1[j] + m1
.Lde5_body:
smulwb r4, r8, r10 @ m0 = this << 1
smulwb r5, r9, r10 @ m1 = this << 1
PCM16 r8, r11 @ pcm[2j]
PCM16 r9, r11 @ pcm[2j+1]
subs r6, r6, #1
bne .Lde5_loop
mov r4, r4, lsl #1
mov r5, r5, lsl #1
.Lde5_store:
stmia ip, {r4, r5} @ mem[0], mem[1]
pop {r4-r11, pc}
.size deemph_stereo_armv5e, .-deemph_stereo_armv5e
.section .note.GNU-stack,"",%progbits

View file

@ -227,6 +227,7 @@ void opus_custom_decoder_destroy(CELTDecoder *st)
/* Special case for stereo with no downsampling and no accumulation. This is
quite common and we can make it faster by processing both channels in the
same loop, reducing overhead due to the dependency loop in the IIR filter. */
#ifndef OVERRIDE_DEEMPH_STEREO
static void deemphasis_stereo_simple(celt_sig *in[], opus_val16 *pcm, int N, const opus_val16 coef0,
celt_sig *mem)
{
@ -253,6 +254,7 @@ static void deemphasis_stereo_simple(celt_sig *in[], opus_val16 *pcm, int N, con
mem[1] = m1;
}
#endif
#endif
#ifndef RESYNTH
static