opus: ARMv4 SILK resampler interpolation

On ARMv4 the fixed-phase FIR cycles at 8, 12 and 16 kHz run as adds of
shifted samples rather than multiplies: every coefficient is a constant,
and a shifted add is one cycle where mla plus loading the coefficient is
five or six.  Each sample is loaded once per cycle and added into the
two or three outputs it feeds, sharing partial products such as 31x
between them, about 27 adds per output.  The kernels are generated by
silk/arm/gen_resampler_armv4.py.  Bit-exact; OPUS_ARM_NO_SILK_ASM
disables them.

Measured on the e200v1, silk_5.opus (WB SILK), with the previous commit:
36.31 -> 28.85 MHz, -20.5%.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Change-Id: Ic1fd5777e91d182248f43954e4b6ae977309dc5d
This commit is contained in:
Michael Giacomelli 2026-09-28 10:14:58 -04:00 • committed by Solomon Peachy
parent 89e9ade855
commit 15eec82d8f
5 changed files with 673 additions and 0 deletions

View file

@ -21,6 +21,9 @@ Celt:
Silk:
* added fixed-phase interpolation paths for the 8, 12 and 16 kHz to 48 kHz
steps in resampler_private_IIR_FIR.c (disable with OPUS_NO_SILK_FIXED_PHASE)
* resampler_private_IIR_FIR.c calls an ARMv4 kernel for those three steps,
silk/arm/resampler_armv4_asm.S, generated by gen_resampler_armv4.py beside
it (disable with OPUS_ARM_NO_SILK_ASM)
Opus-tools:
* copied src/opus_header.h and src/opus_header.c to lib/rbcodec/codecs/libopus

View file

@ -16,6 +16,7 @@ celt/arm/comb_filter_armv4_asm.S
celt/arm/denorm_armv4_asm.S
celt/arm/deemph_armv4_asm.S
celt/arm/exp_rotation1_armv4_asm.S
silk/arm/resampler_armv4_asm.S
#elif defined(CPU_ARM) && (ARM_ARCH >= 5) && (ARCH_PROFILE != ARM_PROFILE_MICRO)
celt/arm/kiss_fft_armv5e_asm.S
celt/arm/mdct_armv5e_asm.S

View file

@ -0,0 +1,274 @@
#!/usr/bin/env python3
"""Generate silk/arm/resampler_armv4_asm.S.
The FIR half of silk_resampler_private_IIR_FIR at 8, 12 and 16 kHz uses two
or three fixed phase filters in a fixed cycle (see the C), so every
coefficient is a compile-time constant. On ARM7TDMI a constant multiply is
cheaper as shifts and adds than as mla: each add with a shifted operand is
one cycle, where mla is 3-4 cycles plus a load of the coefficient.
The kernel is sample-major: each input sample is loaded once per cycle and
added into the two or three accumulators it feeds, so a partial product
such as x*(1+2^a) can be shared between the coefficients one sample meets.
The search below finds, per sample, the cheapest set of up to two such
temporaries and the fewest shifted terms per coefficient.
python3 gen_resampler_armv4.py > resampler_armv4_asm.S
"""
import itertools
import random
import sys
from functools import lru_cache
T = [[189, -600, 617, 30567], [117, -159, -1070, 29704],
[52, 221, -2392, 28276], [-4, 529, -3350, 26341],
[-48, 758, -3956, 23973], [-80, 905, -4235, 21254],
[-99, 972, -4222, 18278], [-107, 967, -3957, 15143],
[-103, 896, -3487, 11950], [-91, 773, -2865, 8798],
[-71, 611, -2143, 5784], [-46, 425, -1375, 2996]]
# The eight taps of phase p, in sample order, as the C applies them.
P = {p: T[p] + T[11 - p][::-1] for p in range(12)}
LIM = 1 << 17
def csd(c):
"""Canonical signed digits of c as (sign, shift)."""
d, e = [], 0
while c:
if c & 1:
r = 2 - (c & 3)
d.append((r, e))
c -= r
c >>= 1
e += 1
return d
@lru_cache(None)
def singles(V):
s = {}
for v in V:
k = 0
while (v << k) < LIM:
for sg in (1, -1):
s.setdefault(sg * (v << k), (sg, v, k))
k += 1
return s
@lru_cache(None)
def doubles(V):
s1 = singles(V)
d = {}
for a, b in itertools.product(s1.items(), repeat=2):
d.setdefault(a[0] + b[0], (a[1], b[1]))
return d
@lru_cache(None)
def terms(c, V):
"""Fewest (sign, v, shift) terms, v in V, summing to c (None if > 4)."""
s1, s2 = singles(V), doubles(V)
if c in s1:
return [s1[c]]
if c in s2:
return list(s2[c])
for a, ta in s1.items():
if c - a in s2:
return [ta] + list(s2[c - a])
for a, ta in s2.items():
if c - a in s2:
return list(ta) + list(s2[c - a])
return None
def plain(c):
return [(sg, 1, k) for sg, k in csd(c)]
TEMPS = sorted({(1 << a) + 1 for a in range(1, 15)} |
{(1 << a) - 1 for a in range(2, 15)})
def program(cs):
"""(temps, [terms per coefficient]) with the fewest instructions."""
best = (sum(len(csd(c)) for c in cs), (), [plain(c) for c in cs])
for n in (1, 2):
for ts in itertools.combinations(TEMPS, n):
V = (1,) + ts
tl = []
cost = n
for c in cs:
t = terms(c, V)
if t is None or len(t) > len(csd(c)):
t = plain(c)
tl.append(t)
cost += len(t)
if cost >= best[0]:
break
else:
best = (cost, ts, tl)
return best
# Rates: step, phases in cycle order, window offset of each output, samples
# advanced per cycle.
RATES = [
("16k", 43691, [0, 8, 4], [0, 0, 1], 2),
("8k", 21846, [0, 4, 8], [0, 0, 0], 1),
("12k", 32768, [0, 6], [0, 0], 1),
]
ACC = ["r3", "r4", "r5"]
XR = ["r6", "r7"]
TR = ["r8", "r9"]
def gen_rate(name, step, phases, offs, adv, out):
nsamp = max(offs) + 8
# coefficients each sample meets, per accumulator
need = []
for j in range(nsamp):
cs = []
for o, (p, off) in enumerate(zip(phases, offs)):
i = j - off
if 0 <= i < 8:
cs.append((o, P[p][i]))
need.append(cs)
# Process x1 .. x(n-1), then x0, whose load post-increments the pointer.
order = list(range(1, nsamp)) + [0]
progs = {}
total = 0
for j in order:
cost, ts, tl = program(tuple(c for _, c in need[j]))
progs[j] = (ts, tl)
total += cost
# Check every program arithmetically before emitting it.
for j in order:
ts, tl = progs[j]
for x in (1, -1, 32767, -32768, random.randint(-32768, 32767)):
val = {1: x}
for t in ts:
val[t] = x * t
for (o, c), tt in zip(need[j], tl):
assert sum(sg * (val[v] << k) for sg, v, k in tt) == c * x, \
(name, j, c, tt)
w = out.write
fn = "silk_IIR_FIR_%s_armv4" % name
w("\n/* %s: phases %s, %d samples a cycle, advancing %d. %d shifted"
" adds a\n cycle, %.1f an output. */\n"
% (name, "/".join(map(str, phases)), nsamp, adv, total,
total / len(phases)))
w(" .global %s\n .type %s, %%function\n%s:\n" % (fn, fn, fn))
w(" push {r4-r11, lr}\n")
w(" mov r10, #0x7f00\n orr r10, r10, #0xff @ 0x7fff, for the clamp\n")
w(" mov r11, #0x4000 @ the rounding bias\n")
w(" ldrsh %s, [r1, #2] @ x1\n" % XR[0])
w(".L%s_loop:\n" % name)
started = set()
for n, j in enumerate(order):
xr = XR[n % 2]
ts, tl = progs[j]
body = []
treg = {1: xr}
for t, r in zip(ts, TR):
treg[t] = r
if (t - 1) & (t - 2) == 0: # 2^a + 1
a = (t - 1).bit_length() - 1
body.append("add %s, %s, %s, lsl #%d" % (r, xr, xr, a))
else: # 2^a - 1
a = (t + 1).bit_length() - 1
assert (1 << a) - 1 == t
body.append("rsb %s, %s, %s, lsl #%d" % (r, xr, xr, a))
for (o, c), tt in zip(need[j], tl):
acc = ACC[o]
first = len(body)
for sg, v, k in tt:
sh = (", lsl #%d" % k) if k else ""
base = acc if o in started else "r11"
started.add(o)
body.append("%s %s, %s, %s%s" % ("add" if sg > 0 else "sub",
acc, base, treg[v], sh))
body[first] += " " * max(1, 34 - len(body[first])) + \
"@ x%d * %d" % (j, c)
# The next sample's load goes after this program's first instruction,
# into the register the sample before this one has finished with.
if n + 1 < len(order):
nj = order[n + 1]
if nj == 0:
ld = "ldrsh %s, [r1], #%d" % (XR[(n + 1) % 2], 2 * adv)
ld += " " * max(1, 34 - len(ld)) + "@ x0, and advance"
else:
ld = "ldrsh %s, [r1, #%d]" % (XR[(n + 1) % 2], 2 * nj)
ld += " " * max(1, 34 - len(ld)) + "@ x%d" % nj
body.insert(1, ld)
for b in body:
w(" " + b + "\n")
# Clamp and store; the next cycle's x1 loads under the first clamp.
for o in range(len(phases)):
acc, r = ACC[o], TR[o % 2]
w(" teq %s, %s, lsl #1 @ out of 16 bits?\n" % (acc, acc))
if o == 0:
w(" ldrsh %s, [r1, #2] @ next x1\n" % XR[0])
w(" mov %s, %s, asr #15\n" % (r, acc))
w(" eormi %s, r10, %s, asr #31\n" % (r, acc))
w(" strh %s, [r0], #2\n" % r)
w(" subs r2, r2, #1\n")
w(" bne .L%s_loop\n" % name)
w(" pop {r4-r11, pc}\n")
w(" .size %s, .-%s\n" % (fn, fn))
return total
HEAD = """/* Copyright (c) 2026 Michael Giacomelli
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the conditions stated in
* silk/resampler_private_IIR_FIR.c are met.
*/
/* GENERATED by silk/arm/gen_resampler_armv4.py -- edit that, not this.
*
* ARMv4 FIR interpolation for silk_resampler_private_IIR_FIR at the three
* SILK rates. Each step visits two or three fixed phase filters in a fixed
* cycle, so every coefficient is a constant, and on ARM7TDMI a constant
* multiply is cheaper as adds of shifted operands (one cycle each) than as
* mla (three or four, plus loading the coefficient).
*
* Sample-major: each input sample is loaded once a cycle and added into the
* two or three accumulators it feeds, sharing a partial product x*(2^a+-1)
* between them where that saves an add. The rounding bias rides in on
* each accumulator's first add. Every sum is exact in 32 bits: the largest
* phase has sum |c| = 50,045, so |sum| < 2^31 for 16-bit input.
*
* opus_int16 *silk_IIR_FIR_<rate>_armv4(opus_int16 *out,
* const opus_int16 *buf, int cycles)
*
* Runs `cycles` whole cycles (cycles > 0) and returns the advanced out.
* r0 out, r1 buf, r2 cycles, r3-r5 accumulators, r6/r7 samples, r8/r9
* partial products, r10 0x7fff, r11 the rounding bias.
*/
#if defined(__thumb__) || defined(__thumb2__)
#error "resampler_armv4_asm.S must be assembled in ARM mode"
#endif
.text
.align 2
"""
def main():
random.seed(1)
out = sys.stdout
out.write(HEAD)
tot = {}
for r in RATES:
tot[r[0]] = gen_rate(*r, out=out)
sys.stderr.write("adds per cycle: %s\n" % tot)
if __name__ == "__main__":
main()

View file

@ -0,0 +1,356 @@
/* Copyright (c) 2026 Michael Giacomelli
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted provided that the conditions stated in
* silk/resampler_private_IIR_FIR.c are met.
*/
/* GENERATED by silk/arm/gen_resampler_armv4.py -- edit that, not this.
*
* ARMv4 FIR interpolation for silk_resampler_private_IIR_FIR at the three
* SILK rates. Each step visits two or three fixed phase filters in a fixed
* cycle, so every coefficient is a constant, and on ARM7TDMI a constant
* multiply is cheaper as adds of shifted operands (one cycle each) than as
* mla (three or four, plus loading the coefficient).
*
* Sample-major: each input sample is loaded once a cycle and added into the
* two or three accumulators it feeds, sharing a partial product x*(2^a+-1)
* between them where that saves an add. The rounding bias rides in on
* each accumulator's first add. Every sum is exact in 32 bits: the largest
* phase has sum |c| = 50,045, so |sum| < 2^31 for 16-bit input.
*
* opus_int16 *silk_IIR_FIR_<rate>_armv4(opus_int16 *out,
* const opus_int16 *buf, int cycles)
*
* Runs `cycles` whole cycles (cycles > 0) and returns the advanced out.
* r0 out, r1 buf, r2 cycles, r3-r5 accumulators, r6/r7 samples, r8/r9
* partial products, r10 0x7fff, r11 the rounding bias.
*/
#if defined(__thumb__) || defined(__thumb2__)
#error "resampler_armv4_asm.S must be assembled in ARM mode"
#endif
.text
.align 2
/* 16k: phases 0/8/4, 9 samples a cycle, advancing 2. 82 shifted adds a
cycle, 27.3 an output. */
.global silk_IIR_FIR_16k_armv4
.type silk_IIR_FIR_16k_armv4, %function
silk_IIR_FIR_16k_armv4:
push {r4-r11, lr}
mov r10, #0x7f00
orr r10, r10, #0xff @ 0x7fff, for the clamp
mov r11, #0x4000 @ the rounding bias
ldrsh r6, [r1, #2] @ x1
.L16k_loop:
add r8, r6, r6, lsl #1
ldrsh r7, [r1, #4] @ x2
add r3, r11, r6, lsl #3 @ x1 * -600
sub r3, r3, r6, lsl #9
sub r3, r3, r8, lsl #5
add r4, r11, r6, lsl #7 @ x1 * 896
add r4, r4, r8, lsl #8
sub r5, r11, r8, lsl #4 @ x1 * -48
rsb r8, r7, r7, lsl #5
ldrsh r6, [r1, #6] @ x3
add r3, r3, r7 @ x2 * 617
sub r3, r3, r7, lsl #2
add r3, r3, r8, lsl #2
add r3, r3, r8, lsl #4
add r4, r4, r7, lsl #9 @ x2 * -3487
sub r4, r4, r8
sub r4, r4, r8, lsl #7
sub r5, r5, r7, lsl #1 @ x2 * 758
add r5, r5, r7, lsl #9
add r5, r5, r8, lsl #3
rsb r8, r6, r6, lsl #5
ldrsh r7, [r1, #8] @ x4
add r3, r3, r6, lsl #15 @ x3 * 30567
add r3, r3, r8
sub r3, r3, r8, lsl #3
sub r3, r3, r8, lsl #6
sub r4, r4, r6, lsl #4 @ x3 * 11950
add r4, r4, r8, lsl #1
add r4, r4, r8, lsl #7
add r4, r4, r8, lsl #8
add r5, r5, r6, lsl #2 @ x3 * -3956
add r5, r5, r6, lsl #3
sub r5, r5, r8, lsl #7
add r8, r7, r7, lsl #1
ldrsh r6, [r1, #10] @ x5
rsb r9, r7, r7, lsl #6
sub r3, r3, r7, lsl #6 @ x4 * 2996
sub r3, r3, r8, lsl #2
add r3, r3, r8, lsl #10
add r4, r4, r7 @ x4 * 26341
add r4, r4, r8, lsl #13
sub r4, r4, r9, lsl #2
add r4, r4, r9, lsl #5
sub r5, r5, r8 @ x4 * 23973
sub r5, r5, r8, lsl #5
add r5, r5, r8, lsl #13
sub r5, r5, r9, lsl #3
add r8, r6, r6, lsl #1
ldrsh r7, [r1, #12] @ x6
add r9, r6, r6, lsl #8
sub r3, r3, r8, lsl #5 @ x5 * -1375
sub r3, r3, r8, lsl #9
add r3, r3, r9
sub r4, r4, r8, lsl #1 @ x5 * -3350
add r4, r4, r8, lsl #8
sub r4, r4, r9, lsl #4
sub r5, r5, r6, lsl #10 @ x5 * 15143
sub r5, r5, r8, lsl #3
sub r5, r5, r9
add r5, r5, r9, lsl #6
add r8, r7, r7, lsl #4
ldrsh r6, [r1, #14] @ x7
add r3, r3, r8 @ x6 * 425
add r3, r3, r8, lsl #3
add r3, r3, r8, lsl #4
add r4, r4, r7, lsl #9 @ x6 * 529
add r4, r4, r8
add r5, r5, r7 @ x6 * -3957
add r5, r5, r7, lsl #1
sub r5, r5, r7, lsl #12
add r5, r5, r8, lsl #3
rsb r8, r6, r6, lsl #3
ldrsh r7, [r1, #16] @ x8
sub r3, r3, r6, lsl #5 @ x7 * -46
sub r3, r3, r8, lsl #1
sub r4, r4, r6, lsl #2 @ x7 * -4
sub r5, r5, r6 @ x7 * 967
add r5, r5, r6, lsl #10
sub r5, r5, r8, lsl #3
add r5, r5, r7 @ x8 * -107
ldrsh r6, [r1], #4 @ x0, and advance
add r5, r5, r7, lsl #2
add r5, r5, r7, lsl #4
sub r5, r5, r7, lsl #7
add r8, r6, r6, lsl #1
sub r3, r3, r8 @ x0 * 189
add r3, r3, r8, lsl #6
add r4, r4, r6 @ x0 * -103
sub r4, r4, r6, lsl #3
sub r4, r4, r8, lsl #5
teq r3, r3, lsl #1 @ out of 16 bits?
ldrsh r6, [r1, #2] @ next x1
mov r8, r3, asr #15
eormi r8, r10, r3, asr #31
strh r8, [r0], #2
teq r4, r4, lsl #1 @ out of 16 bits?
mov r9, r4, asr #15
eormi r9, r10, r4, asr #31
strh r9, [r0], #2
teq r5, r5, lsl #1 @ out of 16 bits?
mov r8, r5, asr #15
eormi r8, r10, r5, asr #31
strh r8, [r0], #2
subs r2, r2, #1
bne .L16k_loop
pop {r4-r11, pc}
.size silk_IIR_FIR_16k_armv4, .-silk_IIR_FIR_16k_armv4
/* 8k: phases 0/4/8, 8 samples a cycle, advancing 1. 79 shifted adds a
cycle, 26.3 an output. */
.global silk_IIR_FIR_8k_armv4
.type silk_IIR_FIR_8k_armv4, %function
silk_IIR_FIR_8k_armv4:
push {r4-r11, lr}
mov r10, #0x7f00
orr r10, r10, #0xff @ 0x7fff, for the clamp
mov r11, #0x4000 @ the rounding bias
ldrsh r6, [r1, #2] @ x1
.L8k_loop:
add r8, r6, r6, lsl #2
ldrsh r7, [r1, #4] @ x2
add r3, r11, r8, lsl #3 @ x1 * -600
sub r3, r3, r8, lsl #7
add r4, r11, r6, lsl #7 @ x1 * 758
sub r4, r4, r8, lsl #1
add r4, r4, r8, lsl #7
sub r5, r11, r6, lsl #7 @ x1 * 896
add r5, r5, r6, lsl #10
add r8, r7, r7, lsl #1
ldrsh r6, [r1, #6] @ x3
rsb r9, r7, r7, lsl #5
sub r3, r3, r8 @ x2 * 617
add r3, r3, r9, lsl #2
add r3, r3, r9, lsl #4
add r4, r4, r8, lsl #2 @ x2 * -3956
sub r4, r4, r9, lsl #7
add r5, r5, r7, lsl #9 @ x2 * -3487
sub r5, r5, r9
sub r5, r5, r9, lsl #7
rsb r8, r6, r6, lsl #4
ldrsh r7, [r1, #8] @ x4
add r9, r6, r6, lsl #5
sub r3, r3, r8, lsl #3 @ x3 * 30567
add r3, r3, r8, lsl #11
sub r3, r3, r9
add r4, r4, r8, lsl #10 @ x3 * 23973
add r4, r4, r9
add r4, r4, r9, lsl #2
add r4, r4, r9, lsl #8
add r5, r5, r6, lsl #4 @ x3 * 11950
add r5, r5, r8, lsl #1
add r5, r5, r8, lsl #9
add r5, r5, r9, lsl #7
add r8, r7, r7, lsl #1
ldrsh r6, [r1, #10] @ x5
rsb r9, r7, r7, lsl #3
sub r3, r3, r7, lsl #6 @ x4 * 2996
sub r3, r3, r8, lsl #2
add r3, r3, r8, lsl #10
add r4, r4, r7, lsl #5 @ x4 * 15143
add r4, r4, r8, lsl #8
add r4, r4, r9
add r4, r4, r9, lsl #11
add r5, r5, r7 @ x4 * 26341
add r5, r5, r8, lsl #13
sub r5, r5, r9, lsl #2
add r5, r5, r9, lsl #8
rsb r8, r6, r6, lsl #3
ldrsh r7, [r1, #12] @ x6
rsb r9, r6, r6, lsl #5
add r3, r3, r8, lsl #6 @ x5 * -1375
sub r3, r3, r8, lsl #8
sub r3, r3, r9
add r4, r4, r6, lsl #2 @ x5 * -3957
add r4, r4, r8
sub r4, r4, r9, lsl #7
sub r5, r5, r8, lsl #1 @ x5 * -3350
sub r5, r5, r8, lsl #9
add r5, r5, r9, lsl #3
rsb r8, r7, r7, lsl #3
ldrsh r6, [r1, #14] @ x7
sub r3, r3, r7, lsl #4 @ x6 * 425
sub r3, r3, r8
add r3, r3, r8, lsl #6
sub r4, r4, r7 @ x6 * 967
add r4, r4, r7, lsl #10
sub r4, r4, r8, lsl #3
add r5, r5, r7 @ x6 * 529
add r5, r5, r7, lsl #4
add r5, r5, r7, lsl #9
add r8, r6, r6, lsl #1
ldrsh r7, [r1], #2 @ x0, and advance
add r3, r3, r6, lsl #1 @ x7 * -46
sub r3, r3, r8, lsl #4
add r4, r4, r6 @ x7 * -107
sub r4, r4, r8, lsl #2
sub r4, r4, r8, lsl #5
sub r5, r5, r6, lsl #2 @ x7 * -4
add r8, r7, r7, lsl #1
sub r3, r3, r8 @ x0 * 189
add r3, r3, r8, lsl #6
sub r4, r4, r8, lsl #4 @ x0 * -48
add r5, r5, r7 @ x0 * -103
sub r5, r5, r7, lsl #3
sub r5, r5, r8, lsl #5
teq r3, r3, lsl #1 @ out of 16 bits?
ldrsh r6, [r1, #2] @ next x1
mov r8, r3, asr #15
eormi r8, r10, r3, asr #31
strh r8, [r0], #2
teq r4, r4, lsl #1 @ out of 16 bits?
mov r9, r4, asr #15
eormi r9, r10, r4, asr #31
strh r9, [r0], #2
teq r5, r5, lsl #1 @ out of 16 bits?
mov r8, r5, asr #15
eormi r8, r10, r5, asr #31
strh r8, [r0], #2
subs r2, r2, #1
bne .L8k_loop
pop {r4-r11, pc}
.size silk_IIR_FIR_8k_armv4, .-silk_IIR_FIR_8k_armv4
/* 12k: phases 0/6, 8 samples a cycle, advancing 1. 55 shifted adds a
cycle, 27.5 an output. */
.global silk_IIR_FIR_12k_armv4
.type silk_IIR_FIR_12k_armv4, %function
silk_IIR_FIR_12k_armv4:
push {r4-r11, lr}
mov r10, #0x7f00
orr r10, r10, #0xff @ 0x7fff, for the clamp
mov r11, #0x4000 @ the rounding bias
ldrsh r6, [r1, #2] @ x1
.L12k_loop:
add r8, r6, r6, lsl #2
ldrsh r7, [r1, #4] @ x2
add r3, r11, r8, lsl #3 @ x1 * -600
sub r3, r3, r8, lsl #7
sub r4, r11, r6, lsl #5 @ x1 * 972
add r4, r4, r6, lsl #10
sub r4, r4, r8, lsl #2
rsb r8, r7, r7, lsl #3
ldrsh r6, [r1, #6] @ x3
add r3, r3, r7, lsl #9 @ x2 * 617
sub r3, r3, r8
add r3, r3, r8, lsl #4
add r4, r4, r7, lsl #1 @ x2 * -4222
sub r4, r4, r7, lsl #7
sub r4, r4, r7, lsl #12
add r8, r6, r6, lsl #3
ldrsh r7, [r1, #8] @ x4
sub r3, r3, r6, lsl #11 @ x3 * 30567
add r3, r3, r6, lsl #15
sub r3, r3, r8
sub r3, r3, r8, lsl #4
sub r4, r4, r6 @ x3 * 18278
sub r4, r4, r8
sub r4, r4, r8, lsl #4
add r4, r4, r8, lsl #11
add r8, r7, r7, lsl #1
ldrsh r6, [r1, #10] @ x5
sub r3, r3, r7, lsl #6 @ x4 * 2996
sub r3, r3, r8, lsl #2
add r3, r3, r8, lsl #10
sub r4, r4, r7, lsl #8 @ x4 * 21254
add r4, r4, r8, lsl #1
sub r4, r4, r8, lsl #10
add r4, r4, r8, lsl #13
add r8, r6, r6, lsl #1
ldrsh r7, [r1, #12] @ x6
add r9, r6, r6, lsl #5
add r3, r3, r6, lsl #7 @ x5 * -1375
sub r3, r3, r8, lsl #9
add r3, r3, r9
add r4, r4, r6 @ x5 * -4235
sub r4, r4, r8, lsl #2
sub r4, r4, r9, lsl #7
rsb r8, r7, r7, lsl #3
ldrsh r6, [r1, #14] @ x7
sub r3, r3, r7, lsl #4 @ x6 * 425
sub r3, r3, r8
add r3, r3, r8, lsl #6
add r4, r4, r7 @ x6 * 905
add r4, r4, r7, lsl #3
add r4, r4, r8, lsl #7
add r3, r3, r6, lsl #1 @ x7 * -46
ldrsh r7, [r1], #2 @ x0, and advance
add r3, r3, r6, lsl #4
sub r3, r3, r6, lsl #6
sub r4, r4, r6, lsl #4 @ x7 * -80
sub r4, r4, r6, lsl #6
add r8, r7, r7, lsl #1
sub r3, r3, r8 @ x0 * 189
add r3, r3, r8, lsl #6
sub r4, r4, r8 @ x0 * -99
sub r4, r4, r8, lsl #5
teq r3, r3, lsl #1 @ out of 16 bits?
ldrsh r6, [r1, #2] @ next x1
mov r8, r3, asr #15
eormi r8, r10, r3, asr #31
strh r8, [r0], #2
teq r4, r4, lsl #1 @ out of 16 bits?
mov r9, r4, asr #15
eormi r9, r10, r4, asr #31
strh r9, [r0], #2
subs r2, r2, #1
bne .L12k_loop
pop {r4-r11, pc}
.size silk_IIR_FIR_12k_armv4, .-silk_IIR_FIR_12k_armv4

View file

@ -108,6 +108,36 @@ static OPUS_INLINE opus_int16 *silk_IIR_FIR_cycle2(
}
return out;
}
#if defined(OPUS_ARM_ASM_ARMV4_ONLY) && !defined(OPUS_ARM_NO_SILK_ASM)
/* The same cycles as shifts and adds of constants, in
silk/arm/resampler_armv4_asm.S; each runs whole cycles only. */
#define SILK_IIR_FIR_ASM
#define SILK_IIR_FIR_16K silk_IIR_FIR_16k_armv4, 43691, 3, 1, 2
#define SILK_IIR_FIR_8K silk_IIR_FIR_8k_armv4, 21846, 3, 1, 1
#define SILK_IIR_FIR_12K silk_IIR_FIR_12k_armv4, 32768, 2, 1, 1
#endif
#ifdef SILK_IIR_FIR_ASM
opus_int16 *silk_IIR_FIR_16k_armv4( opus_int16 *out, const opus_int16 *buf, opus_int32 iters );
opus_int16 *silk_IIR_FIR_8k_armv4( opus_int16 *out, const opus_int16 *buf, opus_int32 iters );
opus_int16 *silk_IIR_FIR_12k_armv4( opus_int16 *out, const opus_int16 *buf, opus_int32 iters );
/* Run the kernel over every whole iteration of cyc cycles of n outputs at
step inc, each advancing adv samples, and leave the C the rest. Cycle j
is whole when its last output's index, (j*n + n-1)*inc, is below
max_index_Q16. */
#define SILK_IIR_FIR_RUN( args ) SILK_IIR_FIR_RUN_( args )
#define SILK_IIR_FIR_RUN_( fn, inc, n, cyc, adv ) \
do { \
opus_int32 iters = ( max_index_Q16 + (inc) - 1 ) / ( (n) * (inc) ) / (cyc); \
if( iters > 0 ) { \
out = fn( out, buf, iters ); \
buf += (adv) * iters; \
max_index_Q16 -= iters * (cyc) * (n) * (inc); \
} \
} while( 0 )
#endif
#endif
static OPUS_INLINE opus_int16 *silk_resampler_private_IIR_FIR_INTERPOL(
@ -124,10 +154,19 @@ static OPUS_INLINE opus_int16 *silk_resampler_private_IIR_FIR_INTERPOL(
#ifndef OPUS_NO_SILK_FIXED_PHASE
switch( index_increment_Q16 ) {
case 43691: /* 16 kHz */
#ifdef SILK_IIR_FIR_ASM
SILK_IIR_FIR_RUN( SILK_IIR_FIR_16K );
#endif
return silk_IIR_FIR_cycle3( out, buf, max_index_Q16, 43691, 0, 8, 4, 1, 2 );
case 21846: /* 8 kHz */
#ifdef SILK_IIR_FIR_ASM
SILK_IIR_FIR_RUN( SILK_IIR_FIR_8K );
#endif
return silk_IIR_FIR_cycle3( out, buf, max_index_Q16, 21846, 0, 4, 8, 0, 1 );
case 32768: /* 12 kHz */
#ifdef SILK_IIR_FIR_ASM
SILK_IIR_FIR_RUN( SILK_IIR_FIR_12K );
#endif
return silk_IIR_FIR_cycle2( out, buf, max_index_Q16 );
}
#endif