mirror of
https://github.com/Rockbox/rockbox.git
synced 2026-10-09 23:53:28 -04:00
flac: ARM assembly for the 64-bit LPC filter
Replaces flac_lpc_32_c, the wide predictor path every 24-bit stream takes, and 16-bit streams encoded at high coefficient precision. The coefficients are invariant for the whole call, but the C loop reloaded all of them for every output sample and spilled its loop bound to the stack on top of that. Orders 1-8 now keep every coefficient in a register; orders 10 and 12, which is what -8 emits, get their own unrolled loops instead of the generic chunked one. An ARMv5E kernel using the packed 16-bit multiplies is included behind FLAC_LPC32_NARROW_ASM, off by default: it needs bps <= 16, and only about half the subframes of such a stream can use it. Measured with test_codec on 24-bit/96kHz streams. At predictor order 12: 36.24 MHz on e200 (ARMv4) against 62.29 MHz before. The same file improves from 37.65 MHz to 27.3 MHz on Clip+ (ARMv5). Bit-exact against flac -d over four complete streams on both targets, with and without the assembly, and the kernels are checked against an int64_t reference across every order 1-32, qlevel 0-15 and coefficient precision 1-15. The rarely-executed orders stay in DRAM: the codec's IRAM window on PP502x is nearly full and demoting them measured no cost, leaving 96 bytes free. FLAC_LPC32_NO_IRAM demotes the whole filter. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Change-Id: Icd8bd298973b9f5c26c103af8adf47b03489d69f
This commit is contained in:
parent
1896c2128b
commit
374f7c0334
3 changed files with 569 additions and 1 deletions
|
|
@ -270,3 +270,541 @@ lpc_decode_arm:
|
|||
.exit:
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
|
||||
/* ---------------------------------------------------------------------------
|
||||
The 64-bit LPC filter.
|
||||
|
||||
void flac_lpc_32_arm(int32_t *decoded, int coeffs[], int pred_order,
|
||||
int qlevel, int len);
|
||||
|
||||
Same contract as flac_lpc_32_c in decoder.c, which decode_subframe_lpc uses
|
||||
whenever bps + coeff_prec + av_log2(pred_order) exceeds 32 -- every 24-bit
|
||||
stream, and any 16-bit stream encoded at high coefficient precision:
|
||||
|
||||
for (i = pred_order; i < len; i++, decoded++) {
|
||||
int64_t sum = 0;
|
||||
for (j = 0; j < pred_order; j++)
|
||||
sum += (int64_t)coeffs[pred_order-j-1] * decoded[j];
|
||||
decoded[j] += sum >> qlevel;
|
||||
}
|
||||
|
||||
The oldest history sample multiplies the highest-numbered coefficient, the
|
||||
same convention lpc_decode_arm above uses.
|
||||
|
||||
Orders 1..8 get their own loop with every coefficient held in a register:
|
||||
the C loop reloads all of them for every output sample although they are
|
||||
invariant for the whole call, and spills the loop bound to the stack on top
|
||||
of that. Orders above 8 go through a chunked loop; the reference encoder's
|
||||
default maximum order is 8 and no test stream here exceeds it.
|
||||
|
||||
Registers, all orders:
|
||||
|
||||
r0 history pointer, walks on one sample per output
|
||||
r1:r2 64-bit accumulator (r1 low); on entry coeffs and pred_order
|
||||
r3, lr history samples, and scratch in the tail
|
||||
r12 ((outputs remaining - 1) << 5) | qlevel
|
||||
r4... coefficients, ascending: r4 = coeffs[0]
|
||||
...r11 further coefficients, then spare history samples
|
||||
|
||||
Packing the output counter and qlevel into one register is what makes order
|
||||
8 fit: eight coefficients, a 64-bit accumulator, a pointer and the counter
|
||||
are already thirteen of the fourteen usable registers. Unpacking costs one
|
||||
AND per output sample and buys a second sample register, which is what keeps
|
||||
every load two instructions clear of the multiply that consumes it -- worth
|
||||
a cycle per tap on both cores.
|
||||
|
||||
Multiply operand order is deliberate. On ARM7TDMI the multiplier walks Rs
|
||||
eight bits at a time and stops once the rest of Rs is all sign, so the
|
||||
coefficient -- fifteen bits or fewer by construction, since coeff_prec is
|
||||
capped at 15 -- goes in Rs and the sample in Rm. That is two passes rather
|
||||
than four whatever the sample width, which is why a 24-bit stream pays no
|
||||
more per multiply here than a 16-bit one.
|
||||
*/
|
||||
|
||||
/* Placement. The hot half of this filter wants IRAM on PP502x, but the
|
||||
codec's 80 KB window there is nearly full before it arrives: FLAC already
|
||||
keeps four 18 KB sample buffers in it. So the orders that are a fraction
|
||||
of a percent of any real stream, and the generic loop that covers the rest,
|
||||
are left in DRAM -- a dense loop like this is cache-resident after its first
|
||||
pass anyway. Define FLAC_LPC32_NO_IRAM to demote the whole thing on a
|
||||
target that needs the space back. */
|
||||
#if defined(USE_IRAM) && !defined(FLAC_LPC32_NO_IRAM)
|
||||
#define LPC32_HOT .section .icode,"ax",%progbits
|
||||
#else
|
||||
#define LPC32_HOT .text
|
||||
#endif
|
||||
#define LPC32_COLD .text
|
||||
|
||||
/* Shift the accumulator down by qlevel, add the residual waiting at
|
||||
p[order], store it back, and step the window on by one sample. The
|
||||
loads above leave r0 at p[order], so the store's post-index rewinds it.
|
||||
A register-specified shift by 32 yields zero on ARM, which is exactly
|
||||
what the qlevel == 0 case needs from the high word. */
|
||||
.macro LPCTAIL order
|
||||
ldr r3, [r0] @ residual
|
||||
and lr, r12, #31 @ qlevel
|
||||
mov r1, r1, lsr lr
|
||||
rsb lr, lr, #32
|
||||
orr r1, r1, r2, lsl lr
|
||||
add r1, r3, r1
|
||||
str r1, [r0], #-(4 * \order - 4)
|
||||
subs r12, r12, #32 @ one output done; the borrow out of the
|
||||
.endm @ count field is the loop's exit test
|
||||
|
||||
/* The high-order tail. Orders above 8 cannot hold their coefficients, so
|
||||
those loops keep the accumulator in r10:r11 and have r2 free for a
|
||||
fourth history sample -- the order is a constant there, so nothing
|
||||
needs to remember it. */
|
||||
.macro HTAIL order
|
||||
ldr r2, [r0] @ residual
|
||||
and lr, r12, #31 @ qlevel
|
||||
mov r10, r10, lsr lr
|
||||
rsb lr, lr, #32
|
||||
orr r10, r10, r11, lsl lr
|
||||
add r10, r2, r10
|
||||
str r10, [r0], #-(4 * \order - 4)
|
||||
subs r12, r12, #32
|
||||
.endm
|
||||
|
||||
/* One pass of four taps, for the orders that cannot keep their
|
||||
coefficients in registers. Both operand streams burst: coefficients
|
||||
descend from the top, history ascends. r2 is the fourth sample
|
||||
register, which is what lets the samples come in one ldm of four
|
||||
rather than three plus a single load. */
|
||||
.macro HCHUNK first
|
||||
ldmdb r3!, { r4-r7 }
|
||||
ldmia r0!, { r2, r8, r9, lr }
|
||||
.ifnb \first
|
||||
smull r10, r11, r2, r7 @ opens the accumulator, so it never
|
||||
.else @ has to be zeroed
|
||||
smlal r10, r11, r2, r7
|
||||
.endif
|
||||
smlal r10, r11, r8, r6
|
||||
smlal r10, r11, r9, r5
|
||||
smlal r10, r11, lr, r4
|
||||
.endm
|
||||
|
||||
LPC32_HOT
|
||||
.align 2
|
||||
.global flac_lpc_32_arm
|
||||
flac_lpc_32_arm:
|
||||
stmfd sp!, { r4-r11, lr }
|
||||
ldr r12, [sp, #36] @ len
|
||||
subs r12, r12, r2 @ outputs = len - pred_order
|
||||
ble .Lx_ret
|
||||
sub r12, r12, #1
|
||||
orr r12, r3, r12, lsl #5
|
||||
.Lx_entry: @ internal: r0, r1, r2 and r12 set up
|
||||
cmp r2, #12
|
||||
addls pc, pc, r2, lsl #2
|
||||
ldr pc, .Lx_pgen @ 13..32: the generic loop, in DRAM
|
||||
@ jumptable:
|
||||
b .Lx_ret @ order 0 cannot occur
|
||||
ldr pc, .Lx_p1 @ 1..3 are a fraction of a percent of
|
||||
ldr pc, .Lx_p2 @ any stream and the tail dominates them
|
||||
ldr pc, .Lx_p3 @ anyway, so they sit in DRAM
|
||||
b .Lx_order4
|
||||
b .Lx_order5
|
||||
b .Lx_order6
|
||||
b .Lx_order7
|
||||
b .Lx_order8
|
||||
ldr pc, .Lx_pgen @ 9 -- generic loop
|
||||
b .Lx_order10
|
||||
ldr pc, .Lx_pgen @ 11 -- generic loop
|
||||
@ order 12 is the last table slot, so it falls through. 10 and 12 get their
|
||||
@ own loops because they are what the reference encoder's -8 actually emits;
|
||||
@ 9 and 11 would too if there were IRAM to put them in.
|
||||
.Lx_order12:
|
||||
add r3, r1, #48 @ coefficient pointer, descending
|
||||
.Lx_loop12:
|
||||
HCHUNK first
|
||||
HCHUNK
|
||||
HCHUNK
|
||||
HTAIL 12
|
||||
bpl .Lx_order12
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
.Lx_order10:
|
||||
add r3, r1, #40
|
||||
.Lx_loop10:
|
||||
ldmdb r3!, { r4, r5 } @ coeffs[8..9], the two odd taps
|
||||
ldmia r0!, { r2, r8 } @ p[0..1]
|
||||
smull r10, r11, r2, r5
|
||||
smlal r10, r11, r8, r4
|
||||
HCHUNK
|
||||
HCHUNK
|
||||
HTAIL 10
|
||||
bpl .Lx_order10
|
||||
ldmpc regs=r4-r11
|
||||
@ order 8 is reached from the table above
|
||||
.Lx_order8:
|
||||
ldmia r1, { r4-r11 } @ coeffs[0..7], resident for the call
|
||||
.Lx_loop8:
|
||||
ldmia r0!, { r3, lr } @ p[0..1]. Eight resident coefficients
|
||||
smull r1, r2, r3, r11 @ leave room for two samples, and a pair
|
||||
smlal r1, r2, lr, r10 @ is still worth bursting: an ldm of n is
|
||||
ldmia r0!, { r3, lr } @ n+1 cycles against 2n for separate
|
||||
smlal r1, r2, r3, r9 @ loads, so two words cost 3 rather than
|
||||
smlal r1, r2, lr, r8 @ 4. The transfer writes r3 before lr,
|
||||
ldmia r0!, { r3, lr } @ so consuming r3 first both avoids the
|
||||
smlal r1, r2, r3, r7 @ stall and earns back the final I cycle.
|
||||
smlal r1, r2, lr, r6
|
||||
ldmia r0!, { r3, lr } @ p[6..7]
|
||||
smlal r1, r2, r3, r5
|
||||
smlal r1, r2, lr, r4
|
||||
LPCTAIL 8
|
||||
bpl .Lx_loop8
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
.Lx_order7:
|
||||
ldmia r1, { r4-r10 } @ coeffs[0..6]
|
||||
.Lx_loop7:
|
||||
ldmia r0!, { r3, r11, lr } @ p[0..2]: 3+2 cycles against 3x2 for
|
||||
smull r1, r2, r3, r10 @ single loads on ARM7TDMI
|
||||
smlal r1, r2, r11, r9
|
||||
smlal r1, r2, lr, r8
|
||||
ldmia r0!, { r3, r11, lr } @ p[3..5]
|
||||
smlal r1, r2, r3, r7 @ r3 lands first, lr last, so reading r3
|
||||
smlal r1, r2, r11, r6 @ straight after the ldm cannot stall
|
||||
ldr r3, [r0], #4 @ p[6]
|
||||
smlal r1, r2, lr, r5
|
||||
smlal r1, r2, r3, r4
|
||||
LPCTAIL 7
|
||||
bpl .Lx_loop7
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
.Lx_order6:
|
||||
ldmia r1, { r4-r9 } @ coeffs[0..5]
|
||||
.Lx_loop6:
|
||||
ldmia r0!, { r3, r10, r11, lr } @ p[0..3]: 4+2 against 4x2
|
||||
smull r1, r2, r3, r9
|
||||
smlal r1, r2, r10, r8
|
||||
smlal r1, r2, r11, r7
|
||||
ldmia r0!, { r3, r10 } @ p[4..5]
|
||||
smlal r1, r2, lr, r6 @ lr is still the first ldm's
|
||||
smlal r1, r2, r3, r5
|
||||
smlal r1, r2, r10, r4
|
||||
LPCTAIL 6
|
||||
bpl .Lx_loop6
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
.Lx_order5:
|
||||
ldmia r1, { r4-r8 } @ coeffs[0..4]
|
||||
.Lx_loop5:
|
||||
ldmia r0!, { r3, r9, r10, r11, lr } @ the whole window in 5+2 cycles
|
||||
smull r1, r2, r3, r8
|
||||
smlal r1, r2, r9, r7
|
||||
smlal r1, r2, r10, r6
|
||||
smlal r1, r2, r11, r5
|
||||
smlal r1, r2, lr, r4
|
||||
LPCTAIL 5
|
||||
bpl .Lx_loop5
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
.Lx_order4:
|
||||
ldmia r1, { r4-r7 } @ coeffs[0..3]
|
||||
.Lx_loop4:
|
||||
ldmia r0!, { r3, r8, r9, lr }
|
||||
smull r1, r2, r3, r7
|
||||
smlal r1, r2, r8, r6
|
||||
smlal r1, r2, r9, r5
|
||||
smlal r1, r2, lr, r4
|
||||
LPCTAIL 4
|
||||
bpl .Lx_loop4
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
LPC32_COLD
|
||||
.align 2
|
||||
.Lx_order3:
|
||||
ldmia r1, { r4-r6 } @ coeffs[0..2]
|
||||
.Lx_loop3:
|
||||
ldmia r0!, { r3, r8, lr }
|
||||
smull r1, r2, r3, r6
|
||||
smlal r1, r2, r8, r5
|
||||
smlal r1, r2, lr, r4
|
||||
LPCTAIL 3
|
||||
bpl .Lx_loop3
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
.Lx_order2:
|
||||
ldmia r1, { r4-r5 } @ coeffs[0..1]
|
||||
.Lx_loop2:
|
||||
ldmia r0!, { r3, lr }
|
||||
smull r1, r2, r3, r5
|
||||
smlal r1, r2, lr, r4
|
||||
LPCTAIL 2
|
||||
bpl .Lx_loop2
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
/* Order 1 carries the sample it just reconstructed into the next
|
||||
iteration in r3: it is the only history the next output needs, so the
|
||||
loop holds one load and one multiply and never stalls. */
|
||||
.Lx_order1:
|
||||
ldr r4, [r1] @ the one coefficient
|
||||
ldr r3, [r0], #4 @ p[0]
|
||||
.Lx_loop1:
|
||||
smull r1, r2, r3, r4
|
||||
ldr r3, [r0] @ residual
|
||||
and lr, r12, #31
|
||||
mov r1, r1, lsr lr
|
||||
rsb lr, lr, #32
|
||||
orr r1, r1, r2, lsl lr
|
||||
add r3, r3, r1 @ and this is the next p[0]
|
||||
str r3, [r0], #4
|
||||
subs r12, r12, #32
|
||||
bpl .Lx_loop1
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
/* Orders 9..32. Four taps a pass, both operand streams bursted: the
|
||||
coefficients descend from coeffs + order and the history ascends, so the
|
||||
inner loop ends when the coefficient pointer reaches coeffs and needs no
|
||||
counter of its own. order & 3 leftover taps go first, one at a time. */
|
||||
.Lx_highgen: @ orders 9, 11 and 13..32
|
||||
.Lx_dnext:
|
||||
add r3, r1, r2, lsl #2 @ coefficient pointer, descending
|
||||
mov r10, #0 @ 64-bit accumulator (r10 low)
|
||||
mov r11, #0
|
||||
ands lr, r2, #3
|
||||
beq .Lx_dchunk
|
||||
.Lx_drem:
|
||||
ldr r8, [r0], #4
|
||||
ldr r9, [r3, #-4]!
|
||||
subs lr, lr, #1
|
||||
smlal r10, r11, r8, r9
|
||||
bne .Lx_drem
|
||||
.Lx_dchunk:
|
||||
ldmdb r3!, { r4-r7 } @ four coefficients, descending
|
||||
ldmia r0!, { r8, r9, lr } @ three samples, and the fourth below,
|
||||
smlal r10, r11, r8, r7 @ because r8 is free again by then
|
||||
ldr r8, [r0], #4
|
||||
smlal r10, r11, r9, r6
|
||||
smlal r10, r11, lr, r5
|
||||
smlal r10, r11, r8, r4
|
||||
cmp r3, r1
|
||||
bhi .Lx_dchunk
|
||||
ldr r3, [r0] @ residual, r0 is at p[order]
|
||||
and lr, r12, #31
|
||||
mov r10, r10, lsr lr
|
||||
rsb lr, lr, #32
|
||||
orr r10, r10, r11, lsl lr
|
||||
add r10, r3, r10
|
||||
str r10, [r0]
|
||||
sub r0, r0, r2, lsl #2 @ back to p[0]...
|
||||
add r0, r0, #4 @ ...and on one sample
|
||||
subs r12, r12, #32
|
||||
bpl .Lx_dnext
|
||||
ldmpc regs=r4-r11
|
||||
|
||||
LPC32_HOT
|
||||
.Lx_ret:
|
||||
ldmpc regs=r4-r11
|
||||
.Lx_p1: .word .Lx_order1
|
||||
.Lx_p2: .word .Lx_order2
|
||||
.Lx_p3: .word .Lx_order3
|
||||
.Lx_pgen: .word .Lx_highgen
|
||||
|
||||
#if ARM_ARCH >= 5
|
||||
/* ---------------------------------------------------------------------------
|
||||
The same filter for ARMv5E, using the packed 16-bit multiplies.
|
||||
|
||||
void flac_lpc_32_arm_narrow(int32_t *decoded, int coeffs[], int pred_order,
|
||||
int qlevel, int len);
|
||||
|
||||
Only valid where every history sample fits a signed halfword, which is what
|
||||
bps <= 16 guarantees on entry: the warm-up samples come from get_sbits(bps),
|
||||
and every sample this loop reconstructs is checked before the next output
|
||||
reads it. The caller selects it on bps; it falls back to the 32-bit kernel
|
||||
above by itself whenever a precondition fails, so it is never the reason an
|
||||
answer differs.
|
||||
|
||||
Why this needs its own kernel, and why ARMv4 gets none:
|
||||
|
||||
smlal is 32x32 -> 64 and costs three cycles on ARM926. smlabb is
|
||||
16x16 + 32 -> 32 and costs one. Two things have to hold before the cheap
|
||||
one is exact. The sample must fit sixteen bits -- true only for bps <= 16,
|
||||
which is why a 24-bit stream cannot use this and stays on smlal. And the
|
||||
sum must fit 32 bits, which holds iff sum|coeffs| * 32768 < 2^31, i.e.
|
||||
sum|coeffs| <= 65535. That is a property of the subframe's coefficients,
|
||||
so it is tested once per call, not per sample. Measured over the test
|
||||
corpus it holds for 94% of the 16-bit high-precision subframes.
|
||||
|
||||
Splitting a wider sample into halves was measured and rejected: two
|
||||
smlalbb plus the extraction is five cycles against smlal's three, so for
|
||||
bps > 16 the 32-bit multiply is not merely adequate, it is optimal.
|
||||
|
||||
Registers:
|
||||
|
||||
r0 history pointer r1 coeffs, kept for the fallback
|
||||
r2 32-bit accumulator r3 scratch
|
||||
r4..r7 coefficients, packed two to a register, low half first
|
||||
r8..r11, lr history samples r12 packed count and qlevel
|
||||
*/
|
||||
|
||||
/* Pack coeffs[2i] and coeffs[2i+1] into one register, low index in the
|
||||
low half, so that smlabb selects the even coefficient and smlabt the
|
||||
odd one. Three instructions once per call, not per sample. */
|
||||
.macro PACK2 dst, even, odd
|
||||
mov \dst, \even, lsl #16
|
||||
mov \dst, \dst, lsr #16
|
||||
orr \dst, \dst, \odd, lsl #16
|
||||
.endm
|
||||
|
||||
/* Accumulator down by qlevel, plus the residual. One register-specified
|
||||
shift folded into the add, where the 64-bit kernel needs a shift, a
|
||||
reverse subtract and an or. */
|
||||
.macro NTAIL order
|
||||
ldr r3, [r0] @ residual
|
||||
and lr, r12, #31 @ qlevel
|
||||
add r2, r3, r2, asr lr
|
||||
str r2, [r0], #-(4 * \order - 4)
|
||||
.endm
|
||||
|
||||
/* Does the sample just reconstructed still fit a signed halfword? If not,
|
||||
the remaining outputs go through the 32-bit kernel. Three cycles per
|
||||
output sample, against the eight per output that the cheap multiplies
|
||||
save at order 8. */
|
||||
.macro NGUARD order
|
||||
add r3, r2, #0x8000
|
||||
cmp r3, #0x10000
|
||||
bcs .Ln_bail\order
|
||||
.endm
|
||||
|
||||
.macro NBAIL order
|
||||
.Ln_bail\order:
|
||||
subs r12, r12, #32 @ that output was finished
|
||||
ldmpc cond=mi, regs=r4-r11 @ and it was the last one
|
||||
mov r2, #\order
|
||||
b .Lx_entry
|
||||
.endm
|
||||
|
||||
.align 2
|
||||
.global flac_lpc_32_arm_narrow
|
||||
flac_lpc_32_arm_narrow:
|
||||
stmfd sp!, { r4-r11, lr }
|
||||
ldr r12, [sp, #36] @ len
|
||||
subs r12, r12, r2 @ outputs = len - pred_order
|
||||
ble .Ln_ret
|
||||
sub r12, r12, #1
|
||||
orr r12, r3, r12, lsl #5
|
||||
@ sum|coeffs| <= 65535, or the 32-bit accumulator could overflow
|
||||
mov r3, #0
|
||||
sub lr, r2, #1
|
||||
.Ln_sum:
|
||||
ldr r4, [r1, lr, lsl #2] @ walks coeffs[order-1] down to coeffs[0]
|
||||
cmp r4, #0
|
||||
rsblt r4, r4, #0
|
||||
add r3, r3, r4
|
||||
subs lr, lr, #1
|
||||
bge .Ln_sum
|
||||
cmp r3, #0x10000
|
||||
bcs .Lx_entry @ too large: use the 32-bit kernel
|
||||
cmp r2, #8
|
||||
addls pc, pc, r2, lsl #2
|
||||
b .Lx_entry @ orders 9 and up: 32-bit kernel
|
||||
@ jumptable:
|
||||
b .Ln_ret @ order 0 cannot occur
|
||||
b .Lx_entry @ orders 1..4 are left to the 32-bit
|
||||
b .Lx_entry @ kernel: they are absent from every
|
||||
b .Lx_entry @ stream that reaches this path, and the
|
||||
b .Lx_entry @ tail dominates them anyway
|
||||
b .Ln_order5
|
||||
b .Ln_order6
|
||||
b .Ln_order7
|
||||
@ order 8 falls through
|
||||
.Ln_order8:
|
||||
ldmia r1, { r8, r9, r10, r11 } @ coeffs[0..3]
|
||||
PACK2 r4, r8, r9
|
||||
PACK2 r5, r10, r11
|
||||
add lr, r1, #16
|
||||
ldmia lr, { r8, r9, r10, r11 } @ coeffs[4..7]
|
||||
PACK2 r6, r8, r9
|
||||
PACK2 r7, r10, r11
|
||||
.Ln_loop8:
|
||||
ldmia r0!, { r8, r9, r10, r11 } @ p[0..3]
|
||||
smulbt r2, r8, r7 @ p[0] * coeffs[7]: one cycle, and it
|
||||
smlabb r2, r9, r7, r2 @ needs no accumulator to be zeroed
|
||||
smlabt r2, r10, r6, r2
|
||||
smlabb r2, r11, r6, r2
|
||||
ldmia r0!, { r8, r9, r10, r11 } @ p[4..7]
|
||||
smlabt r2, r8, r5, r2
|
||||
smlabb r2, r9, r5, r2
|
||||
smlabt r2, r10, r4, r2
|
||||
smlabb r2, r11, r4, r2
|
||||
NTAIL 8
|
||||
NGUARD 8
|
||||
subs r12, r12, #32
|
||||
bpl .Ln_loop8
|
||||
ldmpc regs=r4-r11
|
||||
NBAIL 8
|
||||
|
||||
.Ln_order7:
|
||||
ldmia r1, { r8, r9, r10, r11 } @ coeffs[0..3]
|
||||
PACK2 r4, r8, r9
|
||||
PACK2 r5, r10, r11
|
||||
add lr, r1, #16
|
||||
ldmia lr, { r8, r9, r10 } @ coeffs[4..6]
|
||||
PACK2 r6, r8, r9
|
||||
mov r7, r10 @ coeffs[6] on its own, low half
|
||||
.Ln_loop7:
|
||||
ldmia r0!, { r8, r9, r10, r11 } @ p[0..3]
|
||||
smulbb r2, r8, r7 @ p[0] * coeffs[6]
|
||||
smlabt r2, r9, r6, r2
|
||||
smlabb r2, r10, r6, r2
|
||||
smlabt r2, r11, r5, r2
|
||||
ldmia r0!, { r8, r9, r10 } @ p[4..6]
|
||||
smlabb r2, r8, r5, r2
|
||||
smlabt r2, r9, r4, r2
|
||||
smlabb r2, r10, r4, r2
|
||||
NTAIL 7
|
||||
NGUARD 7
|
||||
subs r12, r12, #32
|
||||
bpl .Ln_loop7
|
||||
ldmpc regs=r4-r11
|
||||
NBAIL 7
|
||||
|
||||
.Ln_order6:
|
||||
ldmia r1, { r8, r9, r10, r11 } @ coeffs[0..3]
|
||||
PACK2 r4, r8, r9
|
||||
PACK2 r5, r10, r11
|
||||
add lr, r1, #16
|
||||
ldmia lr, { r8, r9 } @ coeffs[4..5]
|
||||
PACK2 r6, r8, r9
|
||||
.Ln_loop6:
|
||||
ldmia r0!, { r8, r9, r10, r11 } @ p[0..3]
|
||||
smulbt r2, r8, r6 @ p[0] * coeffs[5]
|
||||
smlabb r2, r9, r6, r2
|
||||
smlabt r2, r10, r5, r2
|
||||
smlabb r2, r11, r5, r2
|
||||
ldmia r0!, { r8, r9 } @ p[4..5]
|
||||
smlabt r2, r8, r4, r2
|
||||
smlabb r2, r9, r4, r2
|
||||
NTAIL 6
|
||||
NGUARD 6
|
||||
subs r12, r12, #32
|
||||
bpl .Ln_loop6
|
||||
ldmpc regs=r4-r11
|
||||
NBAIL 6
|
||||
|
||||
.Ln_order5:
|
||||
ldmia r1, { r8, r9, r10, r11 } @ coeffs[0..3]
|
||||
PACK2 r4, r8, r9
|
||||
PACK2 r5, r10, r11
|
||||
ldr r6, [r1, #16] @ coeffs[4] on its own, low half
|
||||
.Ln_loop5:
|
||||
ldmia r0!, { r8, r9, r10, r11 } @ p[0..3]
|
||||
ldr lr, [r0], #4 @ p[4], loaded well clear of its multiply
|
||||
smulbb r2, r8, r6 @ p[0] * coeffs[4]
|
||||
smlabt r2, r9, r5, r2
|
||||
smlabb r2, r10, r5, r2
|
||||
smlabt r2, r11, r4, r2
|
||||
smlabb r2, lr, r4, r2
|
||||
NTAIL 5
|
||||
NGUARD 5
|
||||
subs r12, r12, #32
|
||||
bpl .Ln_loop5
|
||||
ldmpc regs=r4-r11
|
||||
NBAIL 5
|
||||
|
||||
.Ln_ret:
|
||||
ldmpc regs=r4-r11
|
||||
#endif /* ARM_ARCH >= 5 */
|
||||
|
|
|
|||
|
|
@ -5,4 +5,13 @@
|
|||
|
||||
void lpc_decode_arm(int blocksize, int qlevel, int pred_order, int32_t* data, int* coeffs);
|
||||
|
||||
/* The 64-bit LPC filter, the wide path of decode_subframe_lpc. Both have the
|
||||
same contract as flac_lpc_32_c in decoder.c. The _narrow form uses the
|
||||
ARMv5E packed 16-bit multiplies and is only valid when every history sample
|
||||
fits a signed halfword, which bps <= 16 guarantees. */
|
||||
void flac_lpc_32_arm(int32_t *decoded, int coeffs[], int pred_order,
|
||||
int qlevel, int len);
|
||||
void flac_lpc_32_arm_narrow(int32_t *decoded, int coeffs[], int pred_order,
|
||||
int qlevel, int len);
|
||||
|
||||
#endif
|
||||
|
|
|
|||
|
|
@ -238,8 +238,16 @@ int decode_subframe_fixed(FLACContext *s, int32_t* decoded, int pred_order, int
|
|||
}
|
||||
|
||||
#if !defined(CPU_COLDFIRE)
|
||||
/* The assembler kernels in arm.S implement flac_lpc_32_c exactly; keep the C
|
||||
version compiled anyway when they are in use, so that a build can A/B the
|
||||
two by defining FLAC_NO_LPC32_ASM. */
|
||||
#if defined(CPU_ARM_CLASSIC) && !defined(FLAC_NO_LPC32_ASM)
|
||||
#define FLAC_LPC32_ASM
|
||||
#endif
|
||||
|
||||
static void flac_lpc_32_c(int32_t *decoded, int coeffs[],
|
||||
int pred_order, int qlevel, int len) ICODE_ATTR_FLAC;
|
||||
int pred_order, int qlevel, int len)
|
||||
ICODE_ATTR_FLAC UNUSED_ATTR;
|
||||
static void flac_lpc_32_c(int32_t *decoded, int coeffs[],
|
||||
int pred_order, int qlevel, int len)
|
||||
{
|
||||
|
|
@ -340,7 +348,20 @@ static int decode_subframe_lpc(FLACContext *s, int32_t* decoded, int pred_order,
|
|||
lpc_decode_emac_wide(s->blocksize - pred_order, qlevel, pred_order,
|
||||
decoded + pred_order, coeffs);
|
||||
#else
|
||||
#if defined(FLAC_LPC32_ASM)
|
||||
/* The 16-bit multiply kernel needs every history sample to fit a
|
||||
signed halfword, which bps <= 16 guarantees, and checks its other
|
||||
precondition itself. */
|
||||
#if (ARM_ARCH >= 5) && defined(FLAC_LPC32_NARROW_ASM)
|
||||
if (bps <= 16)
|
||||
flac_lpc_32_arm_narrow(decoded, coeffs, pred_order, qlevel,
|
||||
s->blocksize);
|
||||
else
|
||||
#endif
|
||||
flac_lpc_32_arm(decoded, coeffs, pred_order, qlevel, s->blocksize);
|
||||
#else
|
||||
flac_lpc_32_c(decoded, coeffs, pred_order, qlevel, s->blocksize);
|
||||
#endif
|
||||
|
||||
if (bps <= 16)
|
||||
lpc_analyze_remodulate(decoded, coeffs, pred_order, qlevel, s->blocksize, bps);
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue