From 15eec82d8f9fb7c782a4371664bdadd474cb98a5 Mon Sep 17 00:00:00 2001 From: Michael Giacomelli Date: Mon, 28 Sep 2026 10:14:58 -0400 Subject: [PATCH] opus: ARMv4 SILK resampler interpolation On ARMv4 the fixed-phase FIR cycles at 8, 12 and 16 kHz run as adds of shifted samples rather than multiplies: every coefficient is a constant, and a shifted add is one cycle where mla plus loading the coefficient is five or six. Each sample is loaded once per cycle and added into the two or three outputs it feeds, sharing partial products such as 31x between them, about 27 adds per output. The kernels are generated by silk/arm/gen_resampler_armv4.py. Bit-exact; OPUS_ARM_NO_SILK_ASM disables them. Measured on the e200v1, silk_5.opus (WB SILK), with the previous commit: 36.31 -> 28.85 MHz, -20.5%. Co-Authored-By: Claude Opus 5.5 Change-Id: Ic1fd5777e91d182248f43954e4b6ae977309dc5d --- lib/rbcodec/codecs/libopus/README.rockbox | 3 + lib/rbcodec/codecs/libopus/SOURCES | 1 + .../libopus/silk/arm/gen_resampler_armv4.py | 274 ++++++++++++++ .../libopus/silk/arm/resampler_armv4_asm.S | 356 ++++++++++++++++++ .../libopus/silk/resampler_private_IIR_FIR.c | 39 ++ 5 files changed, 673 insertions(+) create mode 100644 lib/rbcodec/codecs/libopus/silk/arm/gen_resampler_armv4.py create mode 100644 lib/rbcodec/codecs/libopus/silk/arm/resampler_armv4_asm.S diff --git a/lib/rbcodec/codecs/libopus/README.rockbox b/lib/rbcodec/codecs/libopus/README.rockbox index 41d38363f2..9c6eefe4a9 100644 --- a/lib/rbcodec/codecs/libopus/README.rockbox +++ b/lib/rbcodec/codecs/libopus/README.rockbox @@ -21,6 +21,9 @@ Celt: Silk: * added fixed-phase interpolation paths for the 8, 12 and 16 kHz to 48 kHz steps in resampler_private_IIR_FIR.c (disable with OPUS_NO_SILK_FIXED_PHASE) +* resampler_private_IIR_FIR.c calls an ARMv4 kernel for those three steps, + silk/arm/resampler_armv4_asm.S, generated by gen_resampler_armv4.py beside + it (disable with OPUS_ARM_NO_SILK_ASM) Opus-tools: * copied src/opus_header.h and src/opus_header.c to lib/rbcodec/codecs/libopus diff --git a/lib/rbcodec/codecs/libopus/SOURCES b/lib/rbcodec/codecs/libopus/SOURCES index baf2751702..bacf07dd86 100644 --- a/lib/rbcodec/codecs/libopus/SOURCES +++ b/lib/rbcodec/codecs/libopus/SOURCES @@ -16,6 +16,7 @@ celt/arm/comb_filter_armv4_asm.S celt/arm/denorm_armv4_asm.S celt/arm/deemph_armv4_asm.S celt/arm/exp_rotation1_armv4_asm.S +silk/arm/resampler_armv4_asm.S #elif defined(CPU_ARM) && (ARM_ARCH >= 5) && (ARCH_PROFILE != ARM_PROFILE_MICRO) celt/arm/kiss_fft_armv5e_asm.S celt/arm/mdct_armv5e_asm.S diff --git a/lib/rbcodec/codecs/libopus/silk/arm/gen_resampler_armv4.py b/lib/rbcodec/codecs/libopus/silk/arm/gen_resampler_armv4.py new file mode 100644 index 0000000000..a39d5d0ed0 --- /dev/null +++ b/lib/rbcodec/codecs/libopus/silk/arm/gen_resampler_armv4.py @@ -0,0 +1,274 @@ +#!/usr/bin/env python3 +"""Generate silk/arm/resampler_armv4_asm.S. + +The FIR half of silk_resampler_private_IIR_FIR at 8, 12 and 16 kHz uses two +or three fixed phase filters in a fixed cycle (see the C), so every +coefficient is a compile-time constant. On ARM7TDMI a constant multiply is +cheaper as shifts and adds than as mla: each add with a shifted operand is +one cycle, where mla is 3-4 cycles plus a load of the coefficient. + +The kernel is sample-major: each input sample is loaded once per cycle and +added into the two or three accumulators it feeds, so a partial product +such as x*(1+2^a) can be shared between the coefficients one sample meets. +The search below finds, per sample, the cheapest set of up to two such +temporaries and the fewest shifted terms per coefficient. + + python3 gen_resampler_armv4.py > resampler_armv4_asm.S +""" +import itertools +import random +import sys +from functools import lru_cache + +T = [[189, -600, 617, 30567], [117, -159, -1070, 29704], + [52, 221, -2392, 28276], [-4, 529, -3350, 26341], + [-48, 758, -3956, 23973], [-80, 905, -4235, 21254], + [-99, 972, -4222, 18278], [-107, 967, -3957, 15143], + [-103, 896, -3487, 11950], [-91, 773, -2865, 8798], + [-71, 611, -2143, 5784], [-46, 425, -1375, 2996]] +# The eight taps of phase p, in sample order, as the C applies them. +P = {p: T[p] + T[11 - p][::-1] for p in range(12)} + +LIM = 1 << 17 + + +def csd(c): + """Canonical signed digits of c as (sign, shift).""" + d, e = [], 0 + while c: + if c & 1: + r = 2 - (c & 3) + d.append((r, e)) + c -= r + c >>= 1 + e += 1 + return d + + +@lru_cache(None) +def singles(V): + s = {} + for v in V: + k = 0 + while (v << k) < LIM: + for sg in (1, -1): + s.setdefault(sg * (v << k), (sg, v, k)) + k += 1 + return s + + +@lru_cache(None) +def doubles(V): + s1 = singles(V) + d = {} + for a, b in itertools.product(s1.items(), repeat=2): + d.setdefault(a[0] + b[0], (a[1], b[1])) + return d + + +@lru_cache(None) +def terms(c, V): + """Fewest (sign, v, shift) terms, v in V, summing to c (None if > 4).""" + s1, s2 = singles(V), doubles(V) + if c in s1: + return [s1[c]] + if c in s2: + return list(s2[c]) + for a, ta in s1.items(): + if c - a in s2: + return [ta] + list(s2[c - a]) + for a, ta in s2.items(): + if c - a in s2: + return list(ta) + list(s2[c - a]) + return None + + +def plain(c): + return [(sg, 1, k) for sg, k in csd(c)] + + +TEMPS = sorted({(1 << a) + 1 for a in range(1, 15)} | + {(1 << a) - 1 for a in range(2, 15)}) + + +def program(cs): + """(temps, [terms per coefficient]) with the fewest instructions.""" + best = (sum(len(csd(c)) for c in cs), (), [plain(c) for c in cs]) + for n in (1, 2): + for ts in itertools.combinations(TEMPS, n): + V = (1,) + ts + tl = [] + cost = n + for c in cs: + t = terms(c, V) + if t is None or len(t) > len(csd(c)): + t = plain(c) + tl.append(t) + cost += len(t) + if cost >= best[0]: + break + else: + best = (cost, ts, tl) + return best + + +# Rates: step, phases in cycle order, window offset of each output, samples +# advanced per cycle. +RATES = [ + ("16k", 43691, [0, 8, 4], [0, 0, 1], 2), + ("8k", 21846, [0, 4, 8], [0, 0, 0], 1), + ("12k", 32768, [0, 6], [0, 0], 1), +] + +ACC = ["r3", "r4", "r5"] +XR = ["r6", "r7"] +TR = ["r8", "r9"] + + +def gen_rate(name, step, phases, offs, adv, out): + nsamp = max(offs) + 8 + # coefficients each sample meets, per accumulator + need = [] + for j in range(nsamp): + cs = [] + for o, (p, off) in enumerate(zip(phases, offs)): + i = j - off + if 0 <= i < 8: + cs.append((o, P[p][i])) + need.append(cs) + # Process x1 .. x(n-1), then x0, whose load post-increments the pointer. + order = list(range(1, nsamp)) + [0] + progs = {} + total = 0 + for j in order: + cost, ts, tl = program(tuple(c for _, c in need[j])) + progs[j] = (ts, tl) + total += cost + # Check every program arithmetically before emitting it. + for j in order: + ts, tl = progs[j] + for x in (1, -1, 32767, -32768, random.randint(-32768, 32767)): + val = {1: x} + for t in ts: + val[t] = x * t + for (o, c), tt in zip(need[j], tl): + assert sum(sg * (val[v] << k) for sg, v, k in tt) == c * x, \ + (name, j, c, tt) + + w = out.write + fn = "silk_IIR_FIR_%s_armv4" % name + w("\n/* %s: phases %s, %d samples a cycle, advancing %d. %d shifted" + " adds a\n cycle, %.1f an output. */\n" + % (name, "/".join(map(str, phases)), nsamp, adv, total, + total / len(phases))) + w(" .global %s\n .type %s, %%function\n%s:\n" % (fn, fn, fn)) + w(" push {r4-r11, lr}\n") + w(" mov r10, #0x7f00\n orr r10, r10, #0xff @ 0x7fff, for the clamp\n") + w(" mov r11, #0x4000 @ the rounding bias\n") + w(" ldrsh %s, [r1, #2] @ x1\n" % XR[0]) + w(".L%s_loop:\n" % name) + started = set() + for n, j in enumerate(order): + xr = XR[n % 2] + ts, tl = progs[j] + body = [] + treg = {1: xr} + for t, r in zip(ts, TR): + treg[t] = r + if (t - 1) & (t - 2) == 0: # 2^a + 1 + a = (t - 1).bit_length() - 1 + body.append("add %s, %s, %s, lsl #%d" % (r, xr, xr, a)) + else: # 2^a - 1 + a = (t + 1).bit_length() - 1 + assert (1 << a) - 1 == t + body.append("rsb %s, %s, %s, lsl #%d" % (r, xr, xr, a)) + for (o, c), tt in zip(need[j], tl): + acc = ACC[o] + first = len(body) + for sg, v, k in tt: + sh = (", lsl #%d" % k) if k else "" + base = acc if o in started else "r11" + started.add(o) + body.append("%s %s, %s, %s%s" % ("add" if sg > 0 else "sub", + acc, base, treg[v], sh)) + body[first] += " " * max(1, 34 - len(body[first])) + \ + "@ x%d * %d" % (j, c) + # The next sample's load goes after this program's first instruction, + # into the register the sample before this one has finished with. + if n + 1 < len(order): + nj = order[n + 1] + if nj == 0: + ld = "ldrsh %s, [r1], #%d" % (XR[(n + 1) % 2], 2 * adv) + ld += " " * max(1, 34 - len(ld)) + "@ x0, and advance" + else: + ld = "ldrsh %s, [r1, #%d]" % (XR[(n + 1) % 2], 2 * nj) + ld += " " * max(1, 34 - len(ld)) + "@ x%d" % nj + body.insert(1, ld) + for b in body: + w(" " + b + "\n") + # Clamp and store; the next cycle's x1 loads under the first clamp. + for o in range(len(phases)): + acc, r = ACC[o], TR[o % 2] + w(" teq %s, %s, lsl #1 @ out of 16 bits?\n" % (acc, acc)) + if o == 0: + w(" ldrsh %s, [r1, #2] @ next x1\n" % XR[0]) + w(" mov %s, %s, asr #15\n" % (r, acc)) + w(" eormi %s, r10, %s, asr #31\n" % (r, acc)) + w(" strh %s, [r0], #2\n" % r) + w(" subs r2, r2, #1\n") + w(" bne .L%s_loop\n" % name) + w(" pop {r4-r11, pc}\n") + w(" .size %s, .-%s\n" % (fn, fn)) + return total + + +HEAD = """/* Copyright (c) 2026 Michael Giacomelli + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the conditions stated in + * silk/resampler_private_IIR_FIR.c are met. + */ + +/* GENERATED by silk/arm/gen_resampler_armv4.py -- edit that, not this. + * + * ARMv4 FIR interpolation for silk_resampler_private_IIR_FIR at the three + * SILK rates. Each step visits two or three fixed phase filters in a fixed + * cycle, so every coefficient is a constant, and on ARM7TDMI a constant + * multiply is cheaper as adds of shifted operands (one cycle each) than as + * mla (three or four, plus loading the coefficient). + * + * Sample-major: each input sample is loaded once a cycle and added into the + * two or three accumulators it feeds, sharing a partial product x*(2^a+-1) + * between them where that saves an add. The rounding bias rides in on + * each accumulator's first add. Every sum is exact in 32 bits: the largest + * phase has sum |c| = 50,045, so |sum| < 2^31 for 16-bit input. + * + * opus_int16 *silk_IIR_FIR__armv4(opus_int16 *out, + * const opus_int16 *buf, int cycles) + * + * Runs `cycles` whole cycles (cycles > 0) and returns the advanced out. + * r0 out, r1 buf, r2 cycles, r3-r5 accumulators, r6/r7 samples, r8/r9 + * partial products, r10 0x7fff, r11 the rounding bias. + */ + +#if defined(__thumb__) || defined(__thumb2__) +#error "resampler_armv4_asm.S must be assembled in ARM mode" +#endif + + .text + .align 2 +""" + + +def main(): + random.seed(1) + out = sys.stdout + out.write(HEAD) + tot = {} + for r in RATES: + tot[r[0]] = gen_rate(*r, out=out) + sys.stderr.write("adds per cycle: %s\n" % tot) + + +if __name__ == "__main__": + main() diff --git a/lib/rbcodec/codecs/libopus/silk/arm/resampler_armv4_asm.S b/lib/rbcodec/codecs/libopus/silk/arm/resampler_armv4_asm.S new file mode 100644 index 0000000000..df7983eeb6 --- /dev/null +++ b/lib/rbcodec/codecs/libopus/silk/arm/resampler_armv4_asm.S @@ -0,0 +1,356 @@ +/* Copyright (c) 2026 Michael Giacomelli + * + * Redistribution and use in source and binary forms, with or without + * modification, are permitted provided that the conditions stated in + * silk/resampler_private_IIR_FIR.c are met. + */ + +/* GENERATED by silk/arm/gen_resampler_armv4.py -- edit that, not this. + * + * ARMv4 FIR interpolation for silk_resampler_private_IIR_FIR at the three + * SILK rates. Each step visits two or three fixed phase filters in a fixed + * cycle, so every coefficient is a constant, and on ARM7TDMI a constant + * multiply is cheaper as adds of shifted operands (one cycle each) than as + * mla (three or four, plus loading the coefficient). + * + * Sample-major: each input sample is loaded once a cycle and added into the + * two or three accumulators it feeds, sharing a partial product x*(2^a+-1) + * between them where that saves an add. The rounding bias rides in on + * each accumulator's first add. Every sum is exact in 32 bits: the largest + * phase has sum |c| = 50,045, so |sum| < 2^31 for 16-bit input. + * + * opus_int16 *silk_IIR_FIR__armv4(opus_int16 *out, + * const opus_int16 *buf, int cycles) + * + * Runs `cycles` whole cycles (cycles > 0) and returns the advanced out. + * r0 out, r1 buf, r2 cycles, r3-r5 accumulators, r6/r7 samples, r8/r9 + * partial products, r10 0x7fff, r11 the rounding bias. + */ + +#if defined(__thumb__) || defined(__thumb2__) +#error "resampler_armv4_asm.S must be assembled in ARM mode" +#endif + + .text + .align 2 + +/* 16k: phases 0/8/4, 9 samples a cycle, advancing 2. 82 shifted adds a + cycle, 27.3 an output. */ + .global silk_IIR_FIR_16k_armv4 + .type silk_IIR_FIR_16k_armv4, %function +silk_IIR_FIR_16k_armv4: + push {r4-r11, lr} + mov r10, #0x7f00 + orr r10, r10, #0xff @ 0x7fff, for the clamp + mov r11, #0x4000 @ the rounding bias + ldrsh r6, [r1, #2] @ x1 +.L16k_loop: + add r8, r6, r6, lsl #1 + ldrsh r7, [r1, #4] @ x2 + add r3, r11, r6, lsl #3 @ x1 * -600 + sub r3, r3, r6, lsl #9 + sub r3, r3, r8, lsl #5 + add r4, r11, r6, lsl #7 @ x1 * 896 + add r4, r4, r8, lsl #8 + sub r5, r11, r8, lsl #4 @ x1 * -48 + rsb r8, r7, r7, lsl #5 + ldrsh r6, [r1, #6] @ x3 + add r3, r3, r7 @ x2 * 617 + sub r3, r3, r7, lsl #2 + add r3, r3, r8, lsl #2 + add r3, r3, r8, lsl #4 + add r4, r4, r7, lsl #9 @ x2 * -3487 + sub r4, r4, r8 + sub r4, r4, r8, lsl #7 + sub r5, r5, r7, lsl #1 @ x2 * 758 + add r5, r5, r7, lsl #9 + add r5, r5, r8, lsl #3 + rsb r8, r6, r6, lsl #5 + ldrsh r7, [r1, #8] @ x4 + add r3, r3, r6, lsl #15 @ x3 * 30567 + add r3, r3, r8 + sub r3, r3, r8, lsl #3 + sub r3, r3, r8, lsl #6 + sub r4, r4, r6, lsl #4 @ x3 * 11950 + add r4, r4, r8, lsl #1 + add r4, r4, r8, lsl #7 + add r4, r4, r8, lsl #8 + add r5, r5, r6, lsl #2 @ x3 * -3956 + add r5, r5, r6, lsl #3 + sub r5, r5, r8, lsl #7 + add r8, r7, r7, lsl #1 + ldrsh r6, [r1, #10] @ x5 + rsb r9, r7, r7, lsl #6 + sub r3, r3, r7, lsl #6 @ x4 * 2996 + sub r3, r3, r8, lsl #2 + add r3, r3, r8, lsl #10 + add r4, r4, r7 @ x4 * 26341 + add r4, r4, r8, lsl #13 + sub r4, r4, r9, lsl #2 + add r4, r4, r9, lsl #5 + sub r5, r5, r8 @ x4 * 23973 + sub r5, r5, r8, lsl #5 + add r5, r5, r8, lsl #13 + sub r5, r5, r9, lsl #3 + add r8, r6, r6, lsl #1 + ldrsh r7, [r1, #12] @ x6 + add r9, r6, r6, lsl #8 + sub r3, r3, r8, lsl #5 @ x5 * -1375 + sub r3, r3, r8, lsl #9 + add r3, r3, r9 + sub r4, r4, r8, lsl #1 @ x5 * -3350 + add r4, r4, r8, lsl #8 + sub r4, r4, r9, lsl #4 + sub r5, r5, r6, lsl #10 @ x5 * 15143 + sub r5, r5, r8, lsl #3 + sub r5, r5, r9 + add r5, r5, r9, lsl #6 + add r8, r7, r7, lsl #4 + ldrsh r6, [r1, #14] @ x7 + add r3, r3, r8 @ x6 * 425 + add r3, r3, r8, lsl #3 + add r3, r3, r8, lsl #4 + add r4, r4, r7, lsl #9 @ x6 * 529 + add r4, r4, r8 + add r5, r5, r7 @ x6 * -3957 + add r5, r5, r7, lsl #1 + sub r5, r5, r7, lsl #12 + add r5, r5, r8, lsl #3 + rsb r8, r6, r6, lsl #3 + ldrsh r7, [r1, #16] @ x8 + sub r3, r3, r6, lsl #5 @ x7 * -46 + sub r3, r3, r8, lsl #1 + sub r4, r4, r6, lsl #2 @ x7 * -4 + sub r5, r5, r6 @ x7 * 967 + add r5, r5, r6, lsl #10 + sub r5, r5, r8, lsl #3 + add r5, r5, r7 @ x8 * -107 + ldrsh r6, [r1], #4 @ x0, and advance + add r5, r5, r7, lsl #2 + add r5, r5, r7, lsl #4 + sub r5, r5, r7, lsl #7 + add r8, r6, r6, lsl #1 + sub r3, r3, r8 @ x0 * 189 + add r3, r3, r8, lsl #6 + add r4, r4, r6 @ x0 * -103 + sub r4, r4, r6, lsl #3 + sub r4, r4, r8, lsl #5 + teq r3, r3, lsl #1 @ out of 16 bits? + ldrsh r6, [r1, #2] @ next x1 + mov r8, r3, asr #15 + eormi r8, r10, r3, asr #31 + strh r8, [r0], #2 + teq r4, r4, lsl #1 @ out of 16 bits? + mov r9, r4, asr #15 + eormi r9, r10, r4, asr #31 + strh r9, [r0], #2 + teq r5, r5, lsl #1 @ out of 16 bits? + mov r8, r5, asr #15 + eormi r8, r10, r5, asr #31 + strh r8, [r0], #2 + subs r2, r2, #1 + bne .L16k_loop + pop {r4-r11, pc} + .size silk_IIR_FIR_16k_armv4, .-silk_IIR_FIR_16k_armv4 + +/* 8k: phases 0/4/8, 8 samples a cycle, advancing 1. 79 shifted adds a + cycle, 26.3 an output. */ + .global silk_IIR_FIR_8k_armv4 + .type silk_IIR_FIR_8k_armv4, %function +silk_IIR_FIR_8k_armv4: + push {r4-r11, lr} + mov r10, #0x7f00 + orr r10, r10, #0xff @ 0x7fff, for the clamp + mov r11, #0x4000 @ the rounding bias + ldrsh r6, [r1, #2] @ x1 +.L8k_loop: + add r8, r6, r6, lsl #2 + ldrsh r7, [r1, #4] @ x2 + add r3, r11, r8, lsl #3 @ x1 * -600 + sub r3, r3, r8, lsl #7 + add r4, r11, r6, lsl #7 @ x1 * 758 + sub r4, r4, r8, lsl #1 + add r4, r4, r8, lsl #7 + sub r5, r11, r6, lsl #7 @ x1 * 896 + add r5, r5, r6, lsl #10 + add r8, r7, r7, lsl #1 + ldrsh r6, [r1, #6] @ x3 + rsb r9, r7, r7, lsl #5 + sub r3, r3, r8 @ x2 * 617 + add r3, r3, r9, lsl #2 + add r3, r3, r9, lsl #4 + add r4, r4, r8, lsl #2 @ x2 * -3956 + sub r4, r4, r9, lsl #7 + add r5, r5, r7, lsl #9 @ x2 * -3487 + sub r5, r5, r9 + sub r5, r5, r9, lsl #7 + rsb r8, r6, r6, lsl #4 + ldrsh r7, [r1, #8] @ x4 + add r9, r6, r6, lsl #5 + sub r3, r3, r8, lsl #3 @ x3 * 30567 + add r3, r3, r8, lsl #11 + sub r3, r3, r9 + add r4, r4, r8, lsl #10 @ x3 * 23973 + add r4, r4, r9 + add r4, r4, r9, lsl #2 + add r4, r4, r9, lsl #8 + add r5, r5, r6, lsl #4 @ x3 * 11950 + add r5, r5, r8, lsl #1 + add r5, r5, r8, lsl #9 + add r5, r5, r9, lsl #7 + add r8, r7, r7, lsl #1 + ldrsh r6, [r1, #10] @ x5 + rsb r9, r7, r7, lsl #3 + sub r3, r3, r7, lsl #6 @ x4 * 2996 + sub r3, r3, r8, lsl #2 + add r3, r3, r8, lsl #10 + add r4, r4, r7, lsl #5 @ x4 * 15143 + add r4, r4, r8, lsl #8 + add r4, r4, r9 + add r4, r4, r9, lsl #11 + add r5, r5, r7 @ x4 * 26341 + add r5, r5, r8, lsl #13 + sub r5, r5, r9, lsl #2 + add r5, r5, r9, lsl #8 + rsb r8, r6, r6, lsl #3 + ldrsh r7, [r1, #12] @ x6 + rsb r9, r6, r6, lsl #5 + add r3, r3, r8, lsl #6 @ x5 * -1375 + sub r3, r3, r8, lsl #8 + sub r3, r3, r9 + add r4, r4, r6, lsl #2 @ x5 * -3957 + add r4, r4, r8 + sub r4, r4, r9, lsl #7 + sub r5, r5, r8, lsl #1 @ x5 * -3350 + sub r5, r5, r8, lsl #9 + add r5, r5, r9, lsl #3 + rsb r8, r7, r7, lsl #3 + ldrsh r6, [r1, #14] @ x7 + sub r3, r3, r7, lsl #4 @ x6 * 425 + sub r3, r3, r8 + add r3, r3, r8, lsl #6 + sub r4, r4, r7 @ x6 * 967 + add r4, r4, r7, lsl #10 + sub r4, r4, r8, lsl #3 + add r5, r5, r7 @ x6 * 529 + add r5, r5, r7, lsl #4 + add r5, r5, r7, lsl #9 + add r8, r6, r6, lsl #1 + ldrsh r7, [r1], #2 @ x0, and advance + add r3, r3, r6, lsl #1 @ x7 * -46 + sub r3, r3, r8, lsl #4 + add r4, r4, r6 @ x7 * -107 + sub r4, r4, r8, lsl #2 + sub r4, r4, r8, lsl #5 + sub r5, r5, r6, lsl #2 @ x7 * -4 + add r8, r7, r7, lsl #1 + sub r3, r3, r8 @ x0 * 189 + add r3, r3, r8, lsl #6 + sub r4, r4, r8, lsl #4 @ x0 * -48 + add r5, r5, r7 @ x0 * -103 + sub r5, r5, r7, lsl #3 + sub r5, r5, r8, lsl #5 + teq r3, r3, lsl #1 @ out of 16 bits? + ldrsh r6, [r1, #2] @ next x1 + mov r8, r3, asr #15 + eormi r8, r10, r3, asr #31 + strh r8, [r0], #2 + teq r4, r4, lsl #1 @ out of 16 bits? + mov r9, r4, asr #15 + eormi r9, r10, r4, asr #31 + strh r9, [r0], #2 + teq r5, r5, lsl #1 @ out of 16 bits? + mov r8, r5, asr #15 + eormi r8, r10, r5, asr #31 + strh r8, [r0], #2 + subs r2, r2, #1 + bne .L8k_loop + pop {r4-r11, pc} + .size silk_IIR_FIR_8k_armv4, .-silk_IIR_FIR_8k_armv4 + +/* 12k: phases 0/6, 8 samples a cycle, advancing 1. 55 shifted adds a + cycle, 27.5 an output. */ + .global silk_IIR_FIR_12k_armv4 + .type silk_IIR_FIR_12k_armv4, %function +silk_IIR_FIR_12k_armv4: + push {r4-r11, lr} + mov r10, #0x7f00 + orr r10, r10, #0xff @ 0x7fff, for the clamp + mov r11, #0x4000 @ the rounding bias + ldrsh r6, [r1, #2] @ x1 +.L12k_loop: + add r8, r6, r6, lsl #2 + ldrsh r7, [r1, #4] @ x2 + add r3, r11, r8, lsl #3 @ x1 * -600 + sub r3, r3, r8, lsl #7 + sub r4, r11, r6, lsl #5 @ x1 * 972 + add r4, r4, r6, lsl #10 + sub r4, r4, r8, lsl #2 + rsb r8, r7, r7, lsl #3 + ldrsh r6, [r1, #6] @ x3 + add r3, r3, r7, lsl #9 @ x2 * 617 + sub r3, r3, r8 + add r3, r3, r8, lsl #4 + add r4, r4, r7, lsl #1 @ x2 * -4222 + sub r4, r4, r7, lsl #7 + sub r4, r4, r7, lsl #12 + add r8, r6, r6, lsl #3 + ldrsh r7, [r1, #8] @ x4 + sub r3, r3, r6, lsl #11 @ x3 * 30567 + add r3, r3, r6, lsl #15 + sub r3, r3, r8 + sub r3, r3, r8, lsl #4 + sub r4, r4, r6 @ x3 * 18278 + sub r4, r4, r8 + sub r4, r4, r8, lsl #4 + add r4, r4, r8, lsl #11 + add r8, r7, r7, lsl #1 + ldrsh r6, [r1, #10] @ x5 + sub r3, r3, r7, lsl #6 @ x4 * 2996 + sub r3, r3, r8, lsl #2 + add r3, r3, r8, lsl #10 + sub r4, r4, r7, lsl #8 @ x4 * 21254 + add r4, r4, r8, lsl #1 + sub r4, r4, r8, lsl #10 + add r4, r4, r8, lsl #13 + add r8, r6, r6, lsl #1 + ldrsh r7, [r1, #12] @ x6 + add r9, r6, r6, lsl #5 + add r3, r3, r6, lsl #7 @ x5 * -1375 + sub r3, r3, r8, lsl #9 + add r3, r3, r9 + add r4, r4, r6 @ x5 * -4235 + sub r4, r4, r8, lsl #2 + sub r4, r4, r9, lsl #7 + rsb r8, r7, r7, lsl #3 + ldrsh r6, [r1, #14] @ x7 + sub r3, r3, r7, lsl #4 @ x6 * 425 + sub r3, r3, r8 + add r3, r3, r8, lsl #6 + add r4, r4, r7 @ x6 * 905 + add r4, r4, r7, lsl #3 + add r4, r4, r8, lsl #7 + add r3, r3, r6, lsl #1 @ x7 * -46 + ldrsh r7, [r1], #2 @ x0, and advance + add r3, r3, r6, lsl #4 + sub r3, r3, r6, lsl #6 + sub r4, r4, r6, lsl #4 @ x7 * -80 + sub r4, r4, r6, lsl #6 + add r8, r7, r7, lsl #1 + sub r3, r3, r8 @ x0 * 189 + add r3, r3, r8, lsl #6 + sub r4, r4, r8 @ x0 * -99 + sub r4, r4, r8, lsl #5 + teq r3, r3, lsl #1 @ out of 16 bits? + ldrsh r6, [r1, #2] @ next x1 + mov r8, r3, asr #15 + eormi r8, r10, r3, asr #31 + strh r8, [r0], #2 + teq r4, r4, lsl #1 @ out of 16 bits? + mov r9, r4, asr #15 + eormi r9, r10, r4, asr #31 + strh r9, [r0], #2 + subs r2, r2, #1 + bne .L12k_loop + pop {r4-r11, pc} + .size silk_IIR_FIR_12k_armv4, .-silk_IIR_FIR_12k_armv4 diff --git a/lib/rbcodec/codecs/libopus/silk/resampler_private_IIR_FIR.c b/lib/rbcodec/codecs/libopus/silk/resampler_private_IIR_FIR.c index c09551d3a0..670280185f 100644 --- a/lib/rbcodec/codecs/libopus/silk/resampler_private_IIR_FIR.c +++ b/lib/rbcodec/codecs/libopus/silk/resampler_private_IIR_FIR.c @@ -108,6 +108,36 @@ static OPUS_INLINE opus_int16 *silk_IIR_FIR_cycle2( } return out; } + +#if defined(OPUS_ARM_ASM_ARMV4_ONLY) && !defined(OPUS_ARM_NO_SILK_ASM) +/* The same cycles as shifts and adds of constants, in + silk/arm/resampler_armv4_asm.S; each runs whole cycles only. */ +#define SILK_IIR_FIR_ASM +#define SILK_IIR_FIR_16K silk_IIR_FIR_16k_armv4, 43691, 3, 1, 2 +#define SILK_IIR_FIR_8K silk_IIR_FIR_8k_armv4, 21846, 3, 1, 1 +#define SILK_IIR_FIR_12K silk_IIR_FIR_12k_armv4, 32768, 2, 1, 1 +#endif + +#ifdef SILK_IIR_FIR_ASM +opus_int16 *silk_IIR_FIR_16k_armv4( opus_int16 *out, const opus_int16 *buf, opus_int32 iters ); +opus_int16 *silk_IIR_FIR_8k_armv4( opus_int16 *out, const opus_int16 *buf, opus_int32 iters ); +opus_int16 *silk_IIR_FIR_12k_armv4( opus_int16 *out, const opus_int16 *buf, opus_int32 iters ); + +/* Run the kernel over every whole iteration of cyc cycles of n outputs at + step inc, each advancing adv samples, and leave the C the rest. Cycle j + is whole when its last output's index, (j*n + n-1)*inc, is below + max_index_Q16. */ +#define SILK_IIR_FIR_RUN( args ) SILK_IIR_FIR_RUN_( args ) +#define SILK_IIR_FIR_RUN_( fn, inc, n, cyc, adv ) \ + do { \ + opus_int32 iters = ( max_index_Q16 + (inc) - 1 ) / ( (n) * (inc) ) / (cyc); \ + if( iters > 0 ) { \ + out = fn( out, buf, iters ); \ + buf += (adv) * iters; \ + max_index_Q16 -= iters * (cyc) * (n) * (inc); \ + } \ + } while( 0 ) +#endif #endif static OPUS_INLINE opus_int16 *silk_resampler_private_IIR_FIR_INTERPOL( @@ -124,10 +154,19 @@ static OPUS_INLINE opus_int16 *silk_resampler_private_IIR_FIR_INTERPOL( #ifndef OPUS_NO_SILK_FIXED_PHASE switch( index_increment_Q16 ) { case 43691: /* 16 kHz */ +#ifdef SILK_IIR_FIR_ASM + SILK_IIR_FIR_RUN( SILK_IIR_FIR_16K ); +#endif return silk_IIR_FIR_cycle3( out, buf, max_index_Q16, 43691, 0, 8, 4, 1, 2 ); case 21846: /* 8 kHz */ +#ifdef SILK_IIR_FIR_ASM + SILK_IIR_FIR_RUN( SILK_IIR_FIR_8K ); +#endif return silk_IIR_FIR_cycle3( out, buf, max_index_Q16, 21846, 0, 4, 8, 0, 1 ); case 32768: /* 12 kHz */ +#ifdef SILK_IIR_FIR_ASM + SILK_IIR_FIR_RUN( SILK_IIR_FIR_12K ); +#endif return silk_IIR_FIR_cycle2( out, buf, max_index_Q16 ); } #endif