mirror of
https://github.com/Rockbox/rockbox.git
synced 2026-10-10 08:03:04 -04:00
opus: dual_inner_prod on ARMv5E
One ldr fetches two celt_norm coefficients and smlabb/smlatt take the halves apart; scalar fallback when the three pointers disagree on alignment. Modelled: -0.88% ARMv5E. Measured: Clip+ 28.13 -> 28.08 MHz, the same build with and without it. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Change-Id: I79719132ccecfb7664616bca019e3856ee3479da
This commit is contained in:
parent
841007dfa1
commit
7b4d1a7f75
4 changed files with 140 additions and 0 deletions
|
|
@ -17,6 +17,7 @@ celt/arm/deemph_armv4_asm.S
|
|||
#elif defined(CPU_ARM) && (ARM_ARCH >= 5) && (ARCH_PROFILE == ARM_PROFILE_CLASSIC)
|
||||
celt/arm/kiss_fft_armv5e_asm.S
|
||||
celt/arm/mdct_armv5e_asm.S
|
||||
celt/arm/pitch_armv5e_asm.S
|
||||
celt/arm/comb_filter_armv5e_asm.S
|
||||
celt/arm/denorm_armv5e_asm.S
|
||||
celt/arm/exp_rotation1_armv5e_asm.S
|
||||
|
|
|
|||
|
|
@ -66,6 +66,23 @@ extern void (*const DUAL_INNER_PROD_IMPL[OPUS_ARCHMASK+1])(const opus_val16 *x,
|
|||
# endif
|
||||
# endif
|
||||
|
||||
/* The decoder reaches dual_inner_prod from one place, stereo_merge. The C
|
||||
reads three signed halfwords per element and stalls on each of them, since
|
||||
an LDRSH result lands two cycles late on this core; one LDR fetches two
|
||||
coefficients instead and smlabb/smlatt take the halves apart with no sign
|
||||
extension of their own. Measured at 12.1 cycles per element before and
|
||||
5.5 after. ARMv4 is deliberately left alone: without the packed
|
||||
multiplies the extraction would cost exactly what the saved loads buy.
|
||||
Escape hatch OPUS_ARM_NO_PITCH_ASM. */
|
||||
# if defined(OPUS_ARM_INLINE_EDSP) && defined(FIXED_POINT) \
|
||||
&& !defined(OVERRIDE_DUAL_INNER_PROD) && !defined(OPUS_ARM_NO_PITCH_ASM)
|
||||
void dual_inner_prod_armv5e(const opus_val16 *x, const opus_val16 *y01,
|
||||
const opus_val16 *y02, int N, opus_val32 *xy1, opus_val32 *xy2);
|
||||
# define OVERRIDE_DUAL_INNER_PROD (1)
|
||||
# define dual_inner_prod(x, y01, y02, N, xy1, xy2, arch) \
|
||||
((void)(arch), dual_inner_prod_armv5e(x, y01, y02, N, xy1, xy2))
|
||||
# endif
|
||||
|
||||
# if defined(FIXED_POINT)
|
||||
|
||||
# if defined(OPUS_ARM_MAY_HAVE_NEON)
|
||||
|
|
|
|||
121
lib/rbcodec/codecs/libopus/celt/arm/pitch_armv5e_asm.S
Normal file
121
lib/rbcodec/codecs/libopus/celt/arm/pitch_armv5e_asm.S
Normal file
|
|
@ -0,0 +1,121 @@
|
|||
/* ARMv5E dual_inner_prod for the CELT fixed-point decoder.
|
||||
*
|
||||
* Copyright (c) 2007-2008 CSIRO
|
||||
* Copyright (c) 2007-2009 Xiph.Org Foundation
|
||||
* Copyright (c) 2026 Michael Giacomelli
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the conditions stated in
|
||||
* celt/pitch.h are met.
|
||||
*
|
||||
* Why this exists
|
||||
* ---------------
|
||||
* The decoder reaches dual_inner_prod from one place, stereo_merge, which
|
||||
* calls it as dual_inner_prod(Y, X, Y, N, &xp, &side). Over eight packets of
|
||||
* stereo decode that is 6,400 elements costing 12.1 cycles each, against two
|
||||
* multiplies of real work.
|
||||
*
|
||||
* The cost is the loads. The C reads three signed halfwords per element, and
|
||||
* on ARM926EJ-S an LDRSH result is two cycles late rather than one, so the
|
||||
* loop stalls on every one of them. One LDR fetches two coefficients
|
||||
* instead, and smlabb and smlatt take the halves apart without a separate
|
||||
* sign extension -- which is also why this is worth doing here and not on
|
||||
* ARMv4, where the extraction would cost exactly what the extra loads save.
|
||||
*
|
||||
* Alignment. A packed load needs a word-aligned address, and celt_norm is a
|
||||
* halfword, so a band can start on an odd element. The three pointers can
|
||||
* only be packed together if they share their alignment, so the kernel checks
|
||||
* that first and falls back to the scalar loop if they disagree; when they
|
||||
* agree it steps one element at a time until the address is word-aligned and
|
||||
* then runs packed. Both fallbacks compute exactly what the fast path does,
|
||||
* in the same order.
|
||||
*
|
||||
* Bit-exactness. MAC16_16 is smlabb and the sum is a plain 32-bit
|
||||
* accumulation, so the order of accumulation is the only thing that could
|
||||
* differ, and this keeps it: element 0 then 1 within each word, low
|
||||
* addresses first. Output is bit-identical to the C, which the test harness
|
||||
* checks sample by sample.
|
||||
*/
|
||||
|
||||
#if defined(__thumb__) || defined(__thumb2__)
|
||||
#error "pitch_armv5e_asm.S must be assembled in ARM mode"
|
||||
#endif
|
||||
|
||||
.text
|
||||
.align 2
|
||||
|
||||
/* ------------------------------------------------------------------------
|
||||
* void dual_inner_prod_armv5e(const opus_val16 *x, const opus_val16 *y01,
|
||||
* const opus_val16 *y02, int N,
|
||||
* opus_val32 *xy1, opus_val32 *xy2)
|
||||
*
|
||||
* r0 = x r1 = y01 r2 = y02 r3 = N [sp+0] = xy1 [sp+4] = xy2
|
||||
* ------------------------------------------------------------------------ */
|
||||
.global dual_inner_prod_armv5e
|
||||
.type dual_inner_prod_armv5e, %function
|
||||
dual_inner_prod_armv5e:
|
||||
push {r4-r8, lr}
|
||||
mov r7, #0 @ xy01
|
||||
mov r8, #0 @ xy02
|
||||
cmp r3, #0
|
||||
ble .Ldip_done
|
||||
|
||||
/* The three streams can only be read in pairs if they agree on their
|
||||
alignment; two bits of exclusive-or settle it. */
|
||||
eor ip, r0, r1
|
||||
eor lr, r0, r2
|
||||
orr ip, ip, lr
|
||||
tst ip, #3
|
||||
bne .Ldip_scalar
|
||||
|
||||
.Ldip_align: @ step singly until x is word-aligned
|
||||
tst r0, #3
|
||||
beq .Ldip_pairs
|
||||
ldrsh r4, [r0], #2
|
||||
ldrsh r5, [r1], #2
|
||||
ldrsh r6, [r2], #2
|
||||
smlabb r7, r4, r5, r7
|
||||
smlabb r8, r4, r6, r8
|
||||
subs r3, r3, #1
|
||||
bne .Ldip_align
|
||||
b .Ldip_done
|
||||
|
||||
.Ldip_pairs:
|
||||
subs r3, r3, #2
|
||||
bmi .Ldip_tail
|
||||
.Ldip_loop:
|
||||
ldr r4, [r0], #4 @ x[i], x[i+1]
|
||||
ldr r5, [r1], #4 @ y01[i], y01[i+1]
|
||||
ldr r6, [r2], #4 @ y02[i], y02[i+1]
|
||||
smlabb r7, r4, r5, r7
|
||||
smlabb r8, r4, r6, r8
|
||||
smlatt r7, r4, r5, r7
|
||||
smlatt r8, r4, r6, r8
|
||||
subs r3, r3, #2
|
||||
bpl .Ldip_loop
|
||||
.Ldip_tail:
|
||||
tst r3, #1 @ one element left when N was odd
|
||||
beq .Ldip_done
|
||||
ldrsh r4, [r0]
|
||||
ldrsh r5, [r1]
|
||||
ldrsh r6, [r2]
|
||||
smlabb r7, r4, r5, r7
|
||||
smlabb r8, r4, r6, r8
|
||||
b .Ldip_done
|
||||
|
||||
.Ldip_scalar:
|
||||
ldrsh r4, [r0], #2
|
||||
ldrsh r5, [r1], #2
|
||||
ldrsh r6, [r2], #2
|
||||
smlabb r7, r4, r5, r7
|
||||
smlabb r8, r4, r6, r8
|
||||
subs r3, r3, #1
|
||||
bne .Ldip_scalar
|
||||
|
||||
.Ldip_done:
|
||||
ldr ip, [sp, #24] @ xy1
|
||||
ldr lr, [sp, #28] @ xy2
|
||||
str r7, [ip]
|
||||
str r8, [lr]
|
||||
pop {r4-r8, pc}
|
||||
.size dual_inner_prod_armv5e, .-dual_inner_prod_armv5e
|
||||
|
|
@ -64,6 +64,7 @@
|
|||
#if (ARCH_PROFILE != ARM_PROFILE_CLASSIC)
|
||||
#define OPUS_ARM_NO_FFT_ASM
|
||||
#define OPUS_ARM_NO_MDCT_ASM
|
||||
#define OPUS_ARM_NO_PITCH_ASM
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue