diff --git a/apps/plugins/SOURCES b/apps/plugins/SOURCES index 226903f35f..8e11baa23b 100644 --- a/apps/plugins/SOURCES +++ b/apps/plugins/SOURCES @@ -218,7 +218,8 @@ test_codec.c test_core_jpeg.c #endif #if (CONFIG_PLATFORM & PLATFORM_NATIVE) && defined(CPU_ARM) \ - && !defined(CPU_ARM_MICRO) + && !defined(CPU_ARM_MICRO) \ + && PLUGIN_BUFFER_SIZE > 0x20000 /* about 140 KB on ARMv5 */ test_cyc.c #endif test_disk.c diff --git a/apps/plugins/test_cyc.c b/apps/plugins/test_cyc.c index 2a34e7795d..6264b8b3a8 100644 --- a/apps/plugins/test_cyc.c +++ b/apps/plugins/test_cyc.c @@ -38,6 +38,20 @@ * independent loads, a word, a 32-byte line and a four-word ldm at a time, * give what a miss costs when it is free to overlap the loads around it. * + * Instructions whose condition fails are timed too (the loop leaves the + * carry set, so "cc" never executes), and so are stores to the stack, on + * this thread and again on the codec thread, whose stack a codec uses and + * which may not be in the memory the main thread's is. + * + * Where plugins have the room, a store from IRAM code is also timed at + * addresses 4 KB apart across IRAM, since that extra cycle depends on + * which part of IRAM the data is in. + * + * Three more cache costs are timed: a store to a line that is not cached, + * loads that miss far apart rather than next to each other, and an + * instruction fetch miss, from a chain of branches through more DRAM code + * than the cache holds. + * * ARMv5 adds the DSP multiplies, clz, qadd, ldrd/strd and pld; ARMv6 the * top-word and dual 16-bit multiplies, umaal, ssat, rev, the extends, pkhbt, * the SIMD adds and multiply result latency. Each set is behind #if @@ -52,6 +66,33 @@ #include "plugin.h" +/* Rows that have been run and told nothing the rows left in do not, or + that were a step to a finding since made another way. They are kept, + each group with a line on what it showed, and 1 here runs them again: + + - Instruction rows that came out as the sum of their parts on an + ARM926: a return through lr and bx, a conditional return, conditional + instructions that execute, flag-setting, a register shift in an add, + ldm of 3, the third register of an ldm used next, an ldm feeding a + multiply, writeback, stm then ldm, push and pop of ten, mul then use + (as mla then use), a load feeding a register shift. + - Multiplies with 24-bit and negative Rm, and mul, smull and smlal by + operand size: the same whatever the operands, as the rows kept show. + - A miss and then the rest of its line, back to back, which could not + tell a word arriving from the whole line arriving; "miss, then" can. + - Dirty evictions and a last-word miss with 16 instructions after: + the longer gaps kept say the same. Misses 64 and 256 instructions + apart, and an ldm of 2 after a miss: as their neighbours. + - The pair rewrite, a replay of FLAC's decorrelation in DRAM. It never + behaved like the codec, because on the device the codec is not in + DRAM. + - The filter by where its state and code sit, and frames of adds: + they led to "adds in a loop", which measures the cause directly. + - Lines kept after a stream of misses, and after new lines in a set: + too coarse and too noisy to tell how the victim is chosen; "first + line kept" does. */ +#define TEST_CYC_ARCHIVE 0 + #define ITERS 200000ul /* first try at loop iterations per run */ #define UNROLL 16 @@ -61,6 +102,14 @@ static unsigned long iram_buf[64] IBSS_ATTR __attribute__((aligned(32))); static unsigned long dram_buf[64] __attribute__((aligned(32))); +/* On PP5022/PP5024 plugins have 80 KB of IRAM, enough to map what a store + costs across most of it: the extra cycle a store pays when its code and + data share a memory turns out to depend on where in IRAM the data is. */ +#if (CONFIG_CPU == PP5022 || CONFIG_CPU == PP5024) && defined(USE_IRAM) +#define IRAM_MAP_BYTES (64 * 1024) +static unsigned long iram_map[IRAM_MAP_BYTES / 4] IBSS_ATTR; +#endif + typedef void (*benchfn)(unsigned long *p, unsigned long n); @@ -76,13 +125,19 @@ name##_i(unsigned long *p, unsigned long n) \ : [n] "+r" (n) : [p] "r" (p) \ : "r4", "r5", "r6", "r7", "r8", "r9", "cc", "memory"); \ } \ +BENCH_D(name, setup, body) + +/* The DRAM copy alone, for bodies that call the helpers below, which a + branch from IRAM would not reach. lr is free for the body too. */ +#define BENCH_D(name, setup, body) \ static void name##_d(unsigned long *p, unsigned long n) \ { \ asm volatile (setup "1:\n" X16(body) \ " subs %[n], %[n], #1\n" \ " bne 1b\n" \ : [n] "+r" (n) : [p] "r" (p) \ - : "r4", "r5", "r6", "r7", "r8", "r9", "cc", "memory"); \ + : "r4", "r5", "r6", "r7", "r8", "r9", "lr", "cc", \ + "memory"); \ } #define NOSETUP "" @@ -134,6 +189,21 @@ BENCH(b_ldrhreg, OFFSETUP, " ldrh r4, [%[p], r6]\n" " ldrh r5, [%[p], r6]\n") BENCH(b_ldrreguse, OFFSETUP, " ldr r4, [%[p], r6]\n" " add r5, r4, #1\n") +/* The loop's subs leaves the carry set while n has not run out, and this + sets it for the first pass, so a "cc" instruction never executes. */ +#define CCSETUP " cmp r4, r4\n" +BENCH(b_addcc, CCSETUP, " addcc r4, r5, #1\n") +BENCH(b_ldrcc, CCSETUP, " ldrcc r4, [%[p]]\n") +BENCH(b_strcc, CCSETUP, " strcc r4, [%[p]]\n") +BENCH(b_ldmcc, CCSETUP, " ldmccia %[p], { r4-r7 }\n") +BENCH(b_mlacc, MULSETUP CCSETUP, " mlacc r4, r5, r6, r4\n") +BENCH(b_bcc, CCSETUP, " bcc .+4\n") +/* Stack traffic, just below sp so nothing live is touched. */ +BENCH(b_spstr, NOSETUP, " str r4, [sp, #-4]\n") +BENCH(b_spstm, NOSETUP, " stmdb sp, { r4-r7 }\n") +BENCH(b_spldr, NOSETUP, " ldr r4, [sp, #-4]\n") +BENCH(b_sppush, NOSETUP, " stmfd sp!, { r4-r7 }\n" + " ldmfd sp!, { r4-r7 }\n") BENCH(b_mla, MULSETUP, " mla r4, r5, r6, r4\n") BENCH(b_mla32, MUL32SETUP, " mla r4, r5, r6, r4\n") BENCH(b_smlal32, MUL32SETUP, " smlal r4, r5, r8, r7\n") @@ -169,6 +239,104 @@ BENCH(b_qadd16, NOSETUP, " qadd16 r4, r5, r6\n") #endif BENCH(b_smlal, MULSETUP, " smlal r4, r5, r8, r7\n") +/* Calls and returns, results used by the next instruction, and conditional + instructions that execute: what compiled code does all the time and the + rows above do not. The helpers are called from DRAM code only. */ +void tc_leaf(void); +void tc_pop2(void); +void tc_poplr2(void); +void tc_pop10(void); +void tc_popne10(void); +__asm__(".text\n.align 2\n" + ".type tc_leaf, %function\n" + "tc_leaf:\n" + " bx lr\n" + ".type tc_pop2, %function\n" + "tc_pop2:\n" + " stmfd sp!, { r4, lr }\n" + " ldmfd sp!, { r4, pc }\n" + ".type tc_poplr2, %function\n" + "tc_poplr2:\n" + " stmfd sp!, { r4, lr }\n" + " ldmfd sp!, { r4, lr }\n" + " bx lr\n" + ".type tc_pop10, %function\n" + "tc_pop10:\n" + " stmfd sp!, { r4-r11, ip, lr }\n" + " ldmfd sp!, { r4-r11, ip, pc }\n" + ".type tc_popne10, %function\n" + "tc_popne10:\n" + " stmfd sp!, { r4-r11, ip, lr }\n" + " cmp sp, #0\n" + " ldmnefd sp!, { r4-r11, ip, pc }\n"); +/* The loop's subs leaves Z clear while n has not run out, and this clears + it for the first pass, so an "ne" instruction always executes. */ +#define NESETUP " cmp sp, #0\n" +BENCH_D(b_call, NOSETUP, " bl tc_leaf\n") +BENCH_D(b_callpop, NOSETUP, " bl tc_pop2\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_callpoplr, NOSETUP, " bl tc_poplr2\n") +#endif +BENCH_D(b_callpop10, NOSETUP, " bl tc_pop10\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_callpopne, NOSETUP, " bl tc_popne10\n") +#endif +BENCH_D(b_bx, NOSETUP, " adr r4, 2f\n bx r4\n2:\n") +BENCH_D(b_ldrpc, NOSETUP, " adr r4, 2f\n str r4, [%[p]]\n" + " ldr pc, [%[p]]\n2:\n") +BENCH_D(b_bne, NESETUP, " bne .+4\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_addne, NESETUP, " addne r4, r5, #1\n") +BENCH_D(b_ldmne, NESETUP, " ldmneia %[p], { r4-r7 }\n") +BENCH_D(b_stmne, NESETUP, " stmneia %[p], { r4-r7 }\n") +BENCH_D(b_ands, NOSETUP, " ands r4, r5, #63\n") +#endif +BENCH_D(b_orrasr, NOSETUP, " orr r4, r5, r6, asr #30\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_addasr, NOSETUP, " add r4, r5, r6, asr r7\n") +BENCH_D(b_ldm3, NOSETUP, " ldmia %[p], { r4, r5, lr }\n") +#endif +BENCH_D(b_ldm4first, NOSETUP, " ldmia %[p], { r4-r7 }\n" + " add r8, r4, #1\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_ldm4third, NOSETUP, " ldmia %[p], { r4-r7 }\n" + " add r8, r6, #1\n") +#endif +BENCH_D(b_ldm4last, NOSETUP, " ldmia %[p], { r4-r7 }\n" + " add r8, r7, #1\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_ldm4mla, NOSETUP, " ldmia %[p], { r4-r7 }\n" + " mla r8, r4, r5, r8\n") +BENCH_D(b_ldmwb, " mov r8, %[p]\n", + " ldmia r8!, { r4-r7 }\n sub r8, r8, #16\n") +BENCH_D(b_stmwb, " mov r8, %[p]\n", + " stmia r8!, { r4-r7 }\n sub r8, r8, #16\n") +BENCH_D(b_stmldm, " add r8, %[p], #32\n", + " stmia %[p], { r4-r7 }\n ldmia r8, { r4-r7 }\n") +#endif +BENCH_D(b_strldrsame, NOSETUP, " str r4, [%[p]]\n" + " ldr r5, [%[p]]\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_push10, NOSETUP, " stmfd sp!, { r4-r11, ip, lr }\n" + " ldmfd sp!, { r4-r11, ip, lr }\n") +#endif +BENCH_D(b_mlause, MULSETUP, " mla r4, r5, r6, r7\n" + " add r8, r4, #1\n") +BENCH_D(b_mlamul, MULSETUP, " mla r4, r5, r6, r7\n" + " mla r8, r4, r6, r7\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_muladd, MULSETUP, " mul r4, r5, r6\n" + " add r8, r4, #1\n") +#endif +BENCH_D(b_smullacc, MULSETUP, " smull r4, r5, r8, r6\n" + " add r7, r5, #1\n") +BENCH_D(b_ldrmla, MULSETUP, " ldr r4, [%[p]]\n" + " mla r8, r4, r6, r8\n") +#if TEST_CYC_ARCHIVE +BENCH_D(b_ldrshift, NOSETUP, " ldr r6, [%[p]]\n" + " add r4, r5, r4, asr r6\n") +#endif + /* A pointer chase: each word holds the address of the next one, stride bytes on, wrapping at the end. Sixteen dependent loads per iteration, so the loop's two instructions are a small share and every load waits for the @@ -208,6 +376,263 @@ static void name(unsigned long *unused, unsigned long n) \ STREAM(stream_word, " ldr r4, [%[p]], #4\n") STREAM(stream_line, " ldr r4, [%[p]], #32\n") STREAM(stream_ldm4, " ldmia %[p]!, { r4-r7 }\n") +/* The same with stores, to price a store whose line is not cached. */ +STREAM(wstream_word, " str r4, [%[p]], #4\n") +STREAM(wstream_line, " str r4, [%[p]], #32\n") +STREAM(wstream_stm4, " stmia %[p]!, { r4-r7 }\n") + +/* A load per line with work between that touches no memory, to see whether + a miss costs less when the next access is not waiting on it. */ +#define X8(i) i i i i i i i i +#define WORK8 X8(" add r5, r5, #1\n") +STREAM(stream_work8, " ldr r4, [%[p]], #32\n" WORK8) +STREAM(stream_work16, " ldr r4, [%[p]], #32\n" WORK8 WORK8) +STREAM(stream_work32, " ldr r4, [%[p]], #32\n" WORK8 WORK8 WORK8 WORK8) + +/* A miss on the last word of each line, and on its fourth: if the fill + brings the word asked for first these cost what a miss on the first word + does, and if it runs in order from the start of the line they cost + more. */ +STREAM(stream_last, " ldr r4, [%[p], #28]\n add %[p], %[p], #32\n") +STREAM(stream_mid, " ldr r4, [%[p], #12]\n add %[p], %[p], #32\n") +STREAM(stream_first, " ldr r4, [%[p]]\n add %[p], %[p], #32\n") +#if TEST_CYC_ARCHIVE +/* The same misses with the rest of the line read straight after, a word + at a time, in order from the start. */ +#define REST_OF_LINE \ + " ldr r5, [%[p]]\n ldr r5, [%[p], #4]\n" \ + " ldr r5, [%[p], #8]\n ldr r5, [%[p], #16]\n" \ + " ldr r5, [%[p], #20]\n ldr r5, [%[p], #24]\n" \ + " add %[p], %[p], #32\n" +STREAM(stream_last_rest, " ldr r4, [%[p], #28]\n" REST_OF_LINE) +STREAM(stream_mid_rest, " ldr r4, [%[p], #12]\n" REST_OF_LINE) +#endif + +/* A miss whose line is then written, so that every line a later miss + evicts is dirty and has to go back to memory first: one store dirties + half the line on the ARM926, two the whole of it. */ +STREAM(stream_dirty4, " ldr r4, [%[p]]\n str r4, [%[p]]\n" + " add %[p], %[p], #32\n") +STREAM(stream_dirty8, " ldr r4, [%[p]]\n str r4, [%[p]]\n" + " str r4, [%[p], #16]\n" + " add %[p], %[p], #32\n") + +/* The dirty eviction and the miss on a last word again, with work after + that touches no memory: whether the write-back, and the fill from the + last word, hold the core or only the next miss. */ +#define DIRTY8 " ldr r4, [%[p]]\n str r4, [%[p]]\n" \ + " str r4, [%[p], #16]\n add %[p], %[p], #32\n" +#define WORK16 WORK8 WORK8 +#define WORK32 WORK16 WORK16 +#if TEST_CYC_ARCHIVE +STREAM(stream_dirty8_w16, DIRTY8 WORK16) +#endif +STREAM(stream_dirty8_w32, DIRTY8 WORK32) +STREAM(stream_dirty8_w64, DIRTY8 WORK32 WORK32) +STREAM(stream_dirty8_w96, DIRTY8 WORK32 WORK32 WORK32) +#define LAST " ldr r4, [%[p], #28]\n add %[p], %[p], #32\n" +#if TEST_CYC_ARCHIVE +STREAM(stream_last_w16, LAST WORK16) +#endif +STREAM(stream_last_w32, LAST WORK32) + +/* A miss with a long way to the next one, and what else is read from its + line straight after: whether the stall is all there is when nothing + follows, and what a second word of the line, or the rest of an ldm, + waits for. */ +#define WORK64 WORK32 WORK32 +#define LINE0 " ldr r4, [%[p]]\n" +#define NEXTLINE " add %[p], %[p], #32\n" +STREAM(gap_32, LINE0 NEXTLINE WORK32) +#if TEST_CYC_ARCHIVE +STREAM(gap_64, LINE0 NEXTLINE WORK64) +#endif +STREAM(gap_128, LINE0 NEXTLINE WORK64 WORK64) +#if TEST_CYC_ARCHIVE +STREAM(gap_256, LINE0 NEXTLINE WORK64 WORK64 WORK64 WORK64) +#endif +STREAM(gap_w1, LINE0 " ldr r5, [%[p], #4]\n" NEXTLINE WORK32) +STREAM(gap_w7, LINE0 " ldr r5, [%[p], #28]\n" NEXTLINE WORK32) +#if TEST_CYC_ARCHIVE +STREAM(gap_ldm2, " ldmia %[p], { r4, r5 }\n" NEXTLINE WORK32) +#endif +STREAM(gap_ldm4, " ldmia %[p], { r4-r7 }\n" NEXTLINE WORK32) +STREAM(gap_ldm4mid, " add r5, %[p], #8\n" + " ldmia r5, { r4-r7 }\n" NEXTLINE WORK32) +STREAM(gap_str, LINE0 " str r4, [%[p]]\n" NEXTLINE WORK32) + +/* A copy, four words at a time, to 128 KB further on: line fills and + buffered writes both wanting the bus, as in a memcpy between buffers + that are not in the cache. */ +STREAM(copy_4, " ldmia %[p], { r4-r7 }\n" + " add r4, %[p], #0x20000\n" + " stmia r4, { r4-r7 }\n" + " add %[p], %[p], #16\n") + +/* The chase with work after each load, for a miss far from the last. */ +static void chase_work32(unsigned long *start, unsigned long n) +{ + asm volatile ( + " mov r4, %[s]\n" + "1:\n" + X16(" ldr r4, [r4]\n" WORK32) + " subs %[n], %[n], #1\n" + " bne 1b\n" + : [n] "+r" (n) : [s] "r" (start) : "r4", "r5", "cc", "memory"); +} + +#if TEST_CYC_ARCHIVE +/* What a codec does with a block of two channels: each buffer is filtered + in place, one after the other (pair_filter), then both are rewritten + together, four words of each per pass (pair_pass, the loop of the FLAC + decorrelation). pair_frame is all three and pair_filters the first two, + so their difference is the passes alone, as they find the cache. */ +void pair_filter(unsigned long *p, unsigned long passes); +void pair_pass(unsigned long *a, unsigned long *b, unsigned long passes); +__asm__(".text\n.align 2\n.type pair_filter, %function\n" + "pair_filter:\n" + " stmfd sp!, { r4-r7 }\n" + "1:\n" + " ldmia r0, { r4-r7 }\n" + " stmia r0!, { r4-r7 }\n" + " subs r1, r1, #1\n" + " bne 1b\n" + " ldmfd sp!, { r4-r7 }\n" + " bx lr\n" + ".type pair_pass, %function\n" + "pair_pass:\n" + " stmfd sp!, { r4-r11 }\n" + " mov r3, #0\n" + "1:\n" + " ldmia r0, { r4-r7 }\n" + " ldmia r1, { r8-r11 }\n" + " mov r4, r4, lsl r3\n" + " mov r8, r8, lsl r3\n" + " mov r5, r5, lsl r3\n" + " mov r9, r9, lsl r3\n" + " mov r6, r6, lsl r3\n" + " mov r10, r10, lsl r3\n" + " mov r7, r7, lsl r3\n" + " mov r11, r11, lsl r3\n" + " stmia r0!, { r4-r7 }\n" + " stmia r1!, { r8-r11 }\n" + " subs r2, r2, #1\n" + " bne 1b\n" + " ldmfd sp!, { r4-r11 }\n" + " bx lr\n"); + +static unsigned long *pair_a, *pair_b; +static unsigned long pair_passes; /* 16 bytes of each buffer a pass */ + +static void pair_filters(unsigned long *unused, unsigned long n) +{ + (void)unused; + while (n--) + { + pair_filter(pair_a, pair_passes); + pair_filter(pair_b, pair_passes); + } +} + +static void pair_frame(unsigned long *unused, unsigned long n) +{ + (void)unused; + while (n--) + { + pair_filter(pair_a, pair_passes); + pair_filter(pair_b, pair_passes); + pair_pass(pair_a, pair_b, pair_passes); + } +} +#endif + +/* Taken branches, one per cache line of DRAM code, each to the next, so + over more code than the cache holds every one is a fetch miss. Lines + are 16 bytes on the PP502x and 32 on the ARM9 targets. */ +#if ARM_ARCH >= 5 +#define ICHAIN_PAD "28" +#else +#define ICHAIN_PAD "12" +#endif +#define ICHAIN(name, lines) \ +static void name(unsigned long *unused, unsigned long n) \ +{ \ + (void)unused; \ + asm volatile ("1:\n" \ + " .rept " #lines "\n" \ + " b 2f\n" \ + " .space " ICHAIN_PAD "\n" \ + "2:\n" \ + " .endr\n" \ + " subs %[n], %[n], #1\n" \ + " bne 1b\n" \ + : [n] "+r" (n) : : "cc"); \ +} +#define ICHAIN_SMALL 32 /* 512 bytes, or 1 KB */ +#define ICHAIN_BIG 2048 /* 32 KB, or 64 KB */ +ICHAIN(ichain_small, 32) +ICHAIN(ichain_big, 2048) +ICHAIN(ichain_1, 1) +ICHAIN(ichain_2, 2) +ICHAIN(ichain_3, 3) +ICHAIN(ichain_4, 4) +ICHAIN(ichain_5, 5) +ICHAIN(ichain_6, 6) +ICHAIN(ichain_7, 7) +ICHAIN(ichain_8, 8) +ICHAIN(ichain_9, 9) +ICHAIN(ichain_10, 10) +ICHAIN(ichain_12, 12) +ICHAIN(ichain_16, 16) +ICHAIN(ichain_64, 64) +ICHAIN(ichain_128, 128) +ICHAIN(ichain_256, 256) + +/* Adds in a row and nothing else, the loop starting on a cache line: how + much code a loop can run through before an instruction costs more than + its cycle. */ +#define SEQ(name, count) \ +static void name(unsigned long *unused, unsigned long n) \ +{ \ + (void)unused; \ + asm volatile (" b 1f\n" \ + " .balign 32\n" \ + "1:\n" \ + " .rept " #count "\n" \ + " add r4, r4, #1\n" \ + " .endr\n" \ + " subs %[n], %[n], #1\n" \ + " bne 1b\n" \ + : [n] "+r" (n) : : "r4", "cc"); \ +} +SEQ(seq_6, 6) +SEQ(seq_14, 14) +SEQ(seq_22, 22) +SEQ(seq_30, 30) +SEQ(seq_38, 38) +SEQ(seq_46, 46) +SEQ(seq_54, 54) +SEQ(seq_62, 62) +SEQ(seq_70, 70) +SEQ(seq_78, 78) +SEQ(seq_94, 94) +SEQ(seq_126, 126) +SEQ(seq_254, 254) +SEQ(seq_510, 510) +SEQ(seq_1022, 1022) +SEQ(seq_2046, 2046) + +/* The same chain in scattered order: a full-period linear congruence over + the nodes (bytes / stride, a power of two), so consecutive loads are far + apart instead of next to each other. */ +static void make_chain_scattered(unsigned char *buf, unsigned long bytes, + unsigned long stride) +{ + unsigned long i, nodes = bytes / stride; + for (i = 0; i < nodes; i++) + *(unsigned long *)(buf + i * stride) = (unsigned long) + (buf + ((i * 1664525ul + 1013904223ul) & (nodes - 1)) * stride); +} static void make_chain(unsigned char *buf, unsigned long bytes, unsigned long stride) @@ -246,6 +671,22 @@ static void make_chain(unsigned char *buf, unsigned long bytes, #define NOW() (*(volatile unsigned long *)0x8001c0c0) /* DIGCTL usec */ #define PER_SEC 1000000ll #define TIMER_NAME "i.MX233 usec timer" +#elif CONFIG_CPU == AS3525v2 +/* No free-running counter, but the kernel tick's timer counts down from + 15000 at 1.5 MHz, so the tick and that together are one. */ +static unsigned long as3525_now(void) +{ + long t; + unsigned long v; + do { + t = *rb->current_tick; + v = *(volatile unsigned long *)0xC8040024; /* TIMER2_VALUE */ + } while (t != *rb->current_tick); + return (unsigned long)t * 15000 + (15000 - v); +} +#define NOW() as3525_now() +#define PER_SEC 1500000ll +#define TIMER_NAME "AS3525v2 tick timer, 2/3 usec" #else #define NOW() (*rb->current_tick) #define PER_SEC ((long long)HZ) @@ -261,10 +702,10 @@ static void make_chain(unsigned char *buf, unsigned long bytes, /* Hundredths of a cycle per loop iteration. The run doubles in length until the clock has moved far enough for the figure to be good to a percent. */ -static long long cpi(benchfn fn, unsigned long *p) +static long long cpi_from(benchfn fn, unsigned long *p, unsigned long n) { - unsigned long n = ITERS, t0, t1; - fn(p, 64); /* warm the caches */ + unsigned long t0, t1; + fn(p, n < 64 ? n : 64); /* warm the caches */ for (;;) { t0 = NOW(); @@ -277,6 +718,11 @@ static long long cpi(benchfn fn, unsigned long *p) return (long long)(t1 - t0) * *rb->cpu_frequency * 100 / PER_SEC / n; } +static long long cpi(benchfn fn, unsigned long *p) +{ + return cpi_from(fn, p, ITERS); +} + static int fd = -1; /* The screen keeps as many of the latest lines as fit, older ones scrolling @@ -322,12 +768,151 @@ static int per_insn(long long cyc, long long empty) return (int)((cyc - empty) / UNROLL); } -static const struct { +/* Multiplies by the size of each operand: Rm in r5 and Rs in r6, each + 14 bits, 24 bits, 31 bits or a small negative number. */ +#define RM14 " mov r5, #0x3f00\n orr r5, r5, #0xff\n" +#define RM24 " mov r5, #0x7f0000\n orr r5, r5, #0xff00\n" +#define RM31 " mvn r5, #0x80000000\n" +#define RMNEG " mvn r5, #0x3f00\n" +#define RS14 " mov r6, #0x3f00\n orr r6, r6, #0xff\n" +#define RS24 " mov r6, #0x7f0000\n orr r6, r6, #0xff00\n" +#define RS31 " mvn r6, #0x80000000\n" +#define RSNEG " mvn r6, #0x3f00\n" +#define MLA " mla r4, r5, r6, r4\n" +#define MUL " mul r4, r5, r6\n" +#define SMULL " smull r8, r9, r5, r6\n" +#define SMLAL " smlal r8, r9, r5, r6\n" +BENCH_D(b_mla_14_14, RM14 RS14, MLA) +#if TEST_CYC_ARCHIVE +BENCH_D(b_mla_24_14, RM24 RS14, MLA) +#endif +BENCH_D(b_mla_31_14, RM31 RS14, MLA) +#if TEST_CYC_ARCHIVE +BENCH_D(b_mla_n_14, RMNEG RS14, MLA) +BENCH_D(b_mla_14_24, RM14 RS24, MLA) +#endif +BENCH_D(b_mla_14_31, RM14 RS31, MLA) +BENCH_D(b_mla_14_n, RM14 RSNEG, MLA) +#if TEST_CYC_ARCHIVE +BENCH_D(b_mla_24_24, RM24 RS24, MLA) +#endif +BENCH_D(b_mla_31_31, RM31 RS31, MLA) +#if TEST_CYC_ARCHIVE +BENCH_D(b_mul_14_14, RM14 RS14, MUL) +BENCH_D(b_mul_31_14, RM31 RS14, MUL) +BENCH_D(b_mul_14_31, RM14 RS31, MUL) +BENCH_D(b_mul_31_31, RM31 RS31, MUL) +#endif +BENCH_D(b_smull_14_14, RM14 RS14, SMULL) +#if TEST_CYC_ARCHIVE +BENCH_D(b_smull_31_14, RM31 RS14, SMULL) +BENCH_D(b_smull_14_31, RM14 RS31, SMULL) +#endif +BENCH_D(b_smull_31_31, RM31 RS31, SMULL) +#if TEST_CYC_ARCHIVE +BENCH_D(b_smlal_14_14, RM14 RS14, SMLAL) +BENCH_D(b_smlal_31_31, RM31 RS31, SMLAL) +#endif + +struct row { const char *name; benchfn i, d; int per; /* instructions per X16 copy */ int alu; /* ALU cycles per copy, subtracted */ -} B[] = { +}; + +/* DRAM code only; a call row is the whole call, there and back, and the + pairs are both instructions together. */ +#define D(fn) fn##_d, fn##_d +static const struct row B2[] = { + { "bl, bx lr ", D(b_call), 1, 0 }, + { "bl,pop2 pc", D(b_callpop), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "bl,pop2 lr", D(b_callpoplr), 1, 0 }, +#endif + { "bl,pop10pc", D(b_callpop10), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "bl,popne10", D(b_callpopne), 1, 1 }, +#endif + { "bx reg ", D(b_bx), 1, 1 }, + { "ldr pc ", D(b_ldrpc), 1, 2 }, + { "bne taken ", D(b_bne), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "addne ", D(b_addne), 1, 0 }, + { "ldmne 4 ", D(b_ldmne), 1, 0 }, + { "stmne 4 ", D(b_stmne), 1, 0 }, + { "ands ", D(b_ands), 1, 0 }, +#endif + { "orr asr # ", D(b_orrasr), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "add asr Rs", D(b_addasr), 1, 0 }, + { "ldm 3 ", D(b_ldm3), 1, 0 }, +#endif + { "ldm4+use 1", D(b_ldm4first), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "ldm4+use 3", D(b_ldm4third), 1, 0 }, +#endif + { "ldm4+use 4", D(b_ldm4last), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "ldm4+mla 1", D(b_ldm4mla), 1, 0 }, + { "ldm 4 wb ", D(b_ldmwb), 1, 1 }, + { "stm 4 wb ", D(b_stmwb), 1, 1 }, + { "stm4+ldm4 ", D(b_stmldm), 1, 0 }, +#endif + { "str+ldr = ", D(b_strldrsame), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "push+pop10", D(b_push10), 1, 0 }, +#endif + { "mla+use ", D(b_mlause), 1, 0 }, + { "mla+mla Rm", D(b_mlamul), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "mul+use ", D(b_muladd), 1, 0 }, +#endif + { "smull+use ", D(b_smullacc), 1, 0 }, + { "ldr+mla ", D(b_ldrmla), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "ldr+asr Rs", D(b_ldrshift), 1, 0 }, +#endif +}; +#define NB2 ((int)(sizeof(B2) / sizeof(B2[0]))) + +/* Multiplies, named by the bits of Rm and of Rs; n is a small negative. */ +static const struct row B3[] = { + { "mla 14x14 ", D(b_mla_14_14), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "mla 24x14 ", D(b_mla_24_14), 1, 0 }, +#endif + { "mla 31x14 ", D(b_mla_31_14), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "mla nx14 ", D(b_mla_n_14), 1, 0 }, + { "mla 14x24 ", D(b_mla_14_24), 1, 0 }, +#endif + { "mla 14x31 ", D(b_mla_14_31), 1, 0 }, + { "mla 14xn ", D(b_mla_14_n), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "mla 24x24 ", D(b_mla_24_24), 1, 0 }, +#endif + { "mla 31x31 ", D(b_mla_31_31), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "mul 14x14 ", D(b_mul_14_14), 1, 0 }, + { "mul 31x14 ", D(b_mul_31_14), 1, 0 }, + { "mul 14x31 ", D(b_mul_14_31), 1, 0 }, + { "mul 31x31 ", D(b_mul_31_31), 1, 0 }, +#endif + { "smull14x14", D(b_smull_14_14), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "smull31x14", D(b_smull_31_14), 1, 0 }, + { "smull14x31", D(b_smull_14_31), 1, 0 }, +#endif + { "smull31x31", D(b_smull_31_31), 1, 0 }, +#if TEST_CYC_ARCHIVE + { "smlal14x14", D(b_smlal_14_14), 1, 0 }, + { "smlal31x31", D(b_smlal_31_31), 1, 0 }, +#endif +}; +#define NB3 ((int)(sizeof(B3) / sizeof(B3[0]))) + +static const struct row B[] = { { "add ", b_alu_i, b_alu_d, 1, 0 }, { "mov lsl Rs", b_regshift_i, b_regshift_d, 1, 0 }, { "b taken ", b_branch_i, b_branch_d, 1, 0 }, @@ -355,6 +940,12 @@ static const struct { { "ldrb Rm ", b_ldrbreg_i, b_ldrbreg_d, 2, 0 }, { "ldrh Rm ", b_ldrhreg_i, b_ldrhreg_d, 2, 0 }, { "ldr Rm+use", b_ldrreguse_i, b_ldrreguse_d, 1, 1 }, + { "addcc fail", b_addcc_i, b_addcc_d, 1, 0 }, + { "ldrcc fail", b_ldrcc_i, b_ldrcc_d, 1, 0 }, + { "strcc fail", b_strcc_i, b_strcc_d, 1, 0 }, + { "ldmcc fail", b_ldmcc_i, b_ldmcc_d, 1, 0 }, + { "mlacc fail", b_mlacc_i, b_mlacc_d, 1, 0 }, + { "bcc fail ", b_bcc_i, b_bcc_d, 1, 0 }, { "mla 14b Rs", b_mla_i, b_mla_d, 1, 0 }, { "smlal 14b ", b_smlal_i, b_smlal_d, 1, 0 }, { "mla 32b Rs", b_mla32_i, b_mla32_d, 1, 0 }, @@ -396,13 +987,975 @@ static void fmt(char *out, int v) rb->snprintf(out, 8, "%3d.%02d", v / 100, v < 0 ? -v % 100 : v % 100); } +/* Stack rows: where the stack is depends on the thread, so these are run + on the caller's thread, with code in IRAM and in DRAM. */ +static const struct { + const char *name; + benchfn i, d; +} SP[] = { + { "str [sp] ", b_spstr_i, b_spstr_d }, + { "stm 4 [sp]", b_spstm_i, b_spstm_d }, + { "ldr [sp] ", b_spldr_i, b_spldr_d }, + { "push+pop 4", b_sppush_i, b_sppush_d }, +}; +#define NSP ((int)(sizeof(SP) / sizeof(SP[0]))) + +#ifdef USE_IRAM +static long long sp_empty_i; +#endif +static long long sp_empty_d; +static int sp_res[NSP][2]; +static unsigned long sp_addr; +static volatile bool sp_done; + +static void stack_rows(void) +{ + int k; + asm volatile ("mov %0, sp" : "=r" (sp_addr)); + for (k = 0; k < NSP; k++) + { +#ifdef USE_IRAM + sp_res[k][0] = per_insn(cpi(SP[k].i, dram_buf), sp_empty_i); +#endif + sp_res[k][1] = per_insn(cpi(SP[k].d, dram_buf), sp_empty_d); + } + sp_done = true; +} + +static void say_stack_rows(const char *thread) +{ + int k; + say("stack rows, %s thread, sp %08lx", thread, sp_addr); +#ifdef USE_IRAM + say("cyc/insn code:IRAM DRAM"); +#endif + for (k = 0; k < NSP; k++) + { + char a[8], b[8]; + fmt(b, sp_res[k][1]); +#ifdef USE_IRAM + fmt(a, sp_res[k][0]); + say("%s %s %s", SP[k].name, a, b); +#else + (void)a; + say("%s %s", SP[k].name, b); +#endif + } +} + +/* A real kernel, a stage at a time: libtta's hybrid_filter (filter_arm.S), + its error > 0 path, cut off after each stage in turn, to find where a + whole function costs more than its instructions do one by one. Each + stage is the ones before it and: + 0 the frame: push and pop of ten registers, through the pc + 1 the three-word load of the state, five adds, the branches + 2 ldm 4, ldm 4, four adds, stm 4: the first weights + 3 ldm 4 and four mla + 4 stage 2 again for the next four weights + 5 stage 3 again + 6 ldr, ldr, add with a register shift, str, ldr, add, ands, stm 2 + 7 mov, four orr with a shift, three shifts, three subs + 8 the two conditional stm 4 and the conditional return: the whole + function, shifting its delay lines every sixteenth call */ +#define TCF_ADD4 " add r5, r5, r9\n add r6, r6, r10\n" \ + " add r7, r7, r11\n add r8, r8, r12\n" +#define TCF_MLA4 " mla lr, r5, r9, lr\n mla lr, r6, r10, lr\n" \ + " mla lr, r7, r11, lr\n mla lr, r8, r12, lr\n" +__asm__(".text\n.align 2\n" + ".macro TCF name, stage, sub=0, step=4\n" + ".type \\name, %function\n" + "\\name:\n" + " stmdb sp!, { r4-r12, lr }\n" + ".if \\stage >= 1\n" + " ldmia r0, { r5, r6, lr }\n" + " add r2, r0, #148\n" + " add r3, r0, #52\n" + " add r4, r0, #20\n" + " add r2, r2, r5\n" + " add r3, r3, r5\n" + " cmp r6, #0\n" + " bmi 9f\n" + " bne 1f\n" + " b 9f\n" + "1:\n" + ".endif\n" + ".if \\stage >= 2\n" + " ldmia r4, { r5, r6, r7, r8 }\n" + " ldmia r3!, { r9, r10, r11, r12 }\n" + TCF_ADD4 + " stmia r4!, { r5, r6, r7, r8 }\n" + ".endif\n" + ".if \\stage >= 3\n" + " ldmia r2!, { r9, r10, r11, r12 }\n" + TCF_MLA4 + ".endif\n" + ".if \\stage >= 4\n" + " ldmia r4, { r5, r6, r7, r8 }\n" + " ldmia r3!, { r9, r10, r11, r12 }\n" + TCF_ADD4 + " stmia r4!, { r5, r6, r7, r8 }\n" + ".endif\n" + ".if \\stage >= 5\n" + " ldmia r2!, { r9, r10, r11, r12 }\n" + TCF_MLA4 + ".endif\n" + ".irp k, 1, 2, 3, 4, 5, 6, 7, 8\n" + ".if (\\stage >= 6) | (\\sub >= \\k)\n" + ".if \\k == 1\n ldr r5, [r1]\n.endif\n" + ".if \\k == 2\n ldr r6, [r0, #12]\n.endif\n" + ".if \\k == 3\n add lr, r5, lr, asr r6\n.endif\n" + ".if \\k == 4\n str lr, [r1]\n.endif\n" + ".if \\k == 5\n ldr r1, [r0]\n.endif\n" + ".if \\k == 6\n add r1, r1, #\\step\n.endif\n" + ".if \\k == 7\n ands r1, r1, #63\n.endif\n" + ".if \\k == 8\n stmia r0, { r1, r5 }\n.endif\n" + ".endif\n" + ".endr\n" + ".if \\stage >= 7\n" + " mov r4, #1\n" + " orr r5, r4, r9, asr #30\n" + " orr r6, r4, r10, asr #30\n" + " orr r7, r4, r11, asr #30\n" + " orr r8, r4, r12, asr #30\n" + " mov r6, r6, lsl #1\n" + " mov r7, r7, lsl #1\n" + " mov r8, r8, lsl #2\n" + " sub r12, lr, r12\n" + " sub r11, r12, r11\n" + " sub r10, r11, r10\n" + ".endif\n" + ".if \\stage >= 8\n" + " stmneda r2, { r10, r11, r12, lr }\n" + " stmneda r3, { r5, r6, r7, r8 }\n" + " ldmnefd sp!, { r4-r12, pc }\n" + " add r2, r0, #212\n" + " ldmia r2, { r1, r3, r4 }\n" + " sub r2, r2, #64\n" + " stmia r2, { r1, r3, r4, r9-r12, lr }\n" + " add r9, r0, #116\n" + " ldmia r9, { r1, r2, r3, r4 }\n" + " sub r9, r9, #64\n" + " stmia r9, { r1-r8 }\n" + ".endif\n" + "9:\n" + " ldmfd sp!, { r4-r12, pc }\n" + ".endm\n" + "TCF tcf_0, 0\nTCF tcf_1, 1\nTCF tcf_2, 2\nTCF tcf_3, 3\n" + "TCF tcf_4, 4\nTCF tcf_5, 5\nTCF tcf_6, 6\nTCF tcf_7, 7\n" + "TCF tcf_8, 8\n" + "TCF tcf_61, 5, 1\nTCF tcf_62, 5, 2\nTCF tcf_63, 5, 3\n" + "TCF tcf_64, 5, 4\nTCF tcf_65, 5, 5\nTCF tcf_66, 5, 6\n" + "TCF tcf_67, 5, 7\nTCF tcf_6s, 6, 0, 0\nTCF tcf_7s, 7, 0, 0\n" + /* Stage 5 again at each word of a cache line, and frames holding + only adds, as long as the filter's stages are. */ + ".irp k, 0, 1, 2, 3, 4, 5, 6, 7\n" + ".balign 32\n" + ".if \\k\n" + ".space 4 * \\k\n" + ".endif\n" + "TCF tcf_5c\\k, 5\n" + ".endr\n" + ".macro TCP name, n\n" + ".type \\name, %function\n" + "\\name:\n" + " stmdb sp!, { r4-r12, lr }\n" + ".rept \\n\n" + " add r4, r4, #1\n" + ".endr\n" + " ldmfd sp!, { r4-r12, pc }\n" + ".endm\n" + "TCP tcp_40, 40\nTCP tcp_60, 60\nTCP tcp_70, 70\n" + "TCP tcp_80, 80\nTCP tcp_100, 100\nTCP tcp_120, 120\n" + ".type tcf_ret, %function\n" + "tcf_ret:\n" + " bx lr\n"); +int tcf_5c0(int *, int *); int tcf_5c1(int *, int *); +int tcf_5c2(int *, int *); int tcf_5c3(int *, int *); +int tcf_5c4(int *, int *); int tcf_5c5(int *, int *); +int tcf_5c6(int *, int *); int tcf_5c7(int *, int *); +int tcp_40(int *, int *); int tcp_60(int *, int *); int tcp_70(int *, int *); +int tcp_80(int *, int *); int tcp_100(int *, int *); +int tcp_120(int *, int *); +typedef int (*tcf_fn)(int *fs, int *in); +int tcf_0(int *, int *); int tcf_1(int *, int *); int tcf_2(int *, int *); +int tcf_3(int *, int *); int tcf_4(int *, int *); int tcf_5(int *, int *); +int tcf_6(int *, int *); int tcf_7(int *, int *); int tcf_8(int *, int *); +int tcf_61(int *, int *); int tcf_62(int *, int *); int tcf_63(int *, int *); +int tcf_64(int *, int *); int tcf_65(int *, int *); int tcf_66(int *, int *); +int tcf_67(int *, int *); int tcf_6s(int *, int *); int tcf_7s(int *, int *); +int tcf_ret(int *, int *); + +/* The filter's state, as fltst lays it out: index, error, round, shift, + a spare word, then qm[8], dx[24] and dl[24]. */ +static int tcf_buf[61 + 8] __attribute__((aligned(32))); +static int *tcf_fs = tcf_buf; /* the state, somewhere in tcf_buf */ +static int tcf_in[64]; +static tcf_fn tcf_now; + +static void tcf_loop(unsigned long *unused, unsigned long n) +{ + tcf_fn fn = tcf_now; + int *fs = tcf_fs; + unsigned long i; + int v; + (void)unused; + for (i = 0; i < n; i++) + { + v = tcf_in[i & 63]; + fn(fs, &v); + } +} + +/* Hundredths of a cycle a call, the function's own return included. */ +static long long tcf_cycles(tcf_fn fn) +{ + int i; + rb->memset(tcf_buf, 0, sizeof(tcf_buf)); + tcf_fs[1] = 1; /* error > 0 */ + tcf_fs[2] = 512; /* round */ + tcf_fs[3] = 10; /* shift */ + for (i = 0; i < 64; i++) + tcf_in[i] = 500 + i * 37 % 1000; + tcf_now = fn; + return cpi(tcf_loop, NULL); +} + +static void say_filter_rows(void) +{ + /* By the rows above: the frame 24, then 13, 16, 12, 16, 12, 12 and 11, + and 9.75 for the last, the delay lines' shift included. On an + ARM926 the later stages come out a tenth over, which is the cycle a + line that "adds in a loop" measures: with its caller the function + is more than eight lines of code. The archive has stage 6 an + instruction at a time (6.1 to 6.7, 6 being all eight), and stages + 6 and 7 with the index left where it is (6s, 7s). */ + static const struct { const char *name; tcf_fn fn; int want; } T[] = { + { "0 ", tcf_0, 2400 }, { "1 ", tcf_1, 3700 }, + { "2 ", tcf_2, 5300 }, { "3 ", tcf_3, 6500 }, + { "4 ", tcf_4, 8100 }, { "5 ", tcf_5, 9300 }, +#if TEST_CYC_ARCHIVE + { "6.1", tcf_61, 9400 }, { "6.2", tcf_62, 9500 }, + { "6.3", tcf_63, 9800 }, { "6.4", tcf_64, 9900 }, + { "6.5", tcf_65, 10000 }, { "6.6", tcf_66, 10200 }, + { "6.7", tcf_67, 10300 }, +#endif + { "6 ", tcf_6, 10500 }, + { "7 ", tcf_7, 11600 }, { "8 ", tcf_8, 12575 }, +#if TEST_CYC_ARCHIVE + { "6s ", tcf_6s, 10500 }, { "7s ", tcf_7s, 11600 }, +#endif + }; + long long base = tcf_cycles(tcf_ret); + unsigned k; + int last = 0; + say("tta filter, state at %08lx", (unsigned long)tcf_fs); + say(" stage cyc step by rows"); + for (k = 0; k < sizeof(T) / sizeof(T[0]); k++) + { + char a[8], b[8], c[8]; + int v = (int)(tcf_cycles(T[k].fn) - base) + 300; + fmt(a, v); + fmt(b, v - last); + fmt(c, T[k].want); + say(" %s %s %s %s", T[k].name, a, b, c); + last = v; + } +} + +#if TEST_CYC_ARCHIVE +/* Where the filter's cost depends on something other than its + instructions: stage 5 with its state at each word of a cache line, the + same code at each word of a line, and frames of nothing but adds. */ +static void say_filter_layout(void) +{ + static const tcf_fn code[] = { tcf_5c0, tcf_5c1, tcf_5c2, tcf_5c3, + tcf_5c4, tcf_5c5, tcf_5c6, tcf_5c7 }; + static const struct { tcf_fn fn; int n; } pad[] = { + { tcp_40, 40 }, { tcp_60, 60 }, { tcp_70, 70 }, { tcp_80, 80 }, + { tcp_100, 100 }, { tcp_120, 120 } }; + long long base = tcf_cycles(tcf_ret); + unsigned k; + char a[8], b[8]; + say("stage 5 (93) and 4 (81), state at word"); + for (k = 0; k < 8; k++) + { + tcf_fs = tcf_buf + k; + fmt(a, (int)(tcf_cycles(tcf_5) - base) + 300); + fmt(b, (int)(tcf_cycles(tcf_4) - base) + 300); + say(" %u %s %s", k, a, b); + } + tcf_fs = tcf_buf; + say("stage 5 (93), code at word"); + for (k = 0; k < 8; k++) + { + fmt(a, (int)(tcf_cycles(code[k]) - base) + 300); + say(" %u %s at %08lx", k, a, (unsigned long)code[k]); + } + say("frame of adds cyc by rows"); + for (k = 0; k < sizeof(pad) / sizeof(pad[0]); k++) + { + fmt(a, (int)(tcf_cycles(pad[k].fn) - base) + 300); + fmt(b, (24 + pad[k].n) * 100); + say(" %3d adds %s %s", pad[k].n, a, b); + } +} +#endif + +/* The empty loop, by where its code and its data are. */ +static long long e_d_d; +#ifdef USE_IRAM +static long long e_i_i, e_i_d, e_d_i; +#endif + +#ifndef TICK_TIMED +#if TEST_CYC_ARCHIVE +/* Which lines a miss evicts. The 16 KB at the start of the buffer is + read until the data cache holds it, then a stream of misses is run + through memory elsewhere, a load a line, and the first 16 KB is read + once more and timed: what share of its lines are still there. The + stream's loop is padded with 0 to 5 adds, since a victim chosen by a + free-running counter would depend on the loop's length where a random + one would not: at random, a quarter of what is left goes with every 128 + lines, leaving 75%, 32%, and after 2048 lines nothing. */ +#define EVICT(name, pad) \ +static void name(unsigned long *p, unsigned long n) \ +{ \ + asm volatile ("1:\n" \ + " ldr r4, [%[p]], #32\n" \ + pad \ + " subs %[n], %[n], #1\n" \ + " bne 1b\n" \ + : [n] "+&r" (n), [p] "+&r" (p) : : "r4", "r5", "cc"); \ +} +#define PAD1 " add r5, r5, #1\n" +EVICT(evict_0, "") +EVICT(evict_1, PAD1) +EVICT(evict_2, PAD1 PAD1) +EVICT(evict_3, PAD1 PAD1 PAD1) +EVICT(evict_5, PAD1 PAD1 PAD1 PAD1 PAD1) + +/* Timer counts for one read of the first 16 KB, after `lines` misses made + by fn elsewhere; fn NULL leaves it as it is. */ +static unsigned long evict_pass(unsigned char *big, benchfn fn, + unsigned long lines) +{ + unsigned long t; + int i; + for (i = 0; i < 12; i++) + evict_0((unsigned long *)big, 512); + if (fn) + fn((unsigned long *)(big + 65536), lines); + t = NOW(); + evict_0((unsigned long *)big, 512); + return NOW() - t; +} + +/* Which way a miss takes, miss after miss. Four lines are loaded into + one set until they fill it; then some new lines are loaded into that + set, with misses in other sets in between; then the four are read again + and timed, each followed by enough adds for a miss to cost its stall + alone. A victim picked at random each time keeps 75%, 56%, 42% and 32% + of the four after 1 to 4 new lines whatever comes between; one that + takes the ways in turn keeps 75%, 50%, 25% and none. */ +void evict_reload(unsigned char *line0); +__asm__(".text\n.align 2\n.type evict_reload, %function\n" + "evict_reload:\n" + " mov r2, #4\n" + "1:\n" + " ldr r1, [r0]\n" + " add r0, r0, #4096\n" + " .rept 32\n" + " add r3, r3, #1\n" + " .endr\n" + " subs r2, r2, #1\n" + " bne 1b\n" + " bx lr\n"); + +static unsigned long evict_set_trial(unsigned char *big, int news, int gap) +{ + unsigned long total = 0, t, fill = 0; + int set, k, round, i; + for (set = 0; set < 128; set++) + { + unsigned char *line = big + set * 32; + for (round = 0; round < 8; round++) + for (k = 0; k < 4; k++) + (void)*(volatile unsigned long *)(line + k * 4096); + for (k = 0; k < news; k++) + { + (void)*(volatile unsigned long *)(line + (4 + k) * 4096); + for (i = 0; i < gap; i++, fill++) + (void)*(volatile unsigned long *)(big + 131072 + + ((set + 1 + fill % 126) & 127) * 32 + + (fill / 126 % 16) * 4096); + } + t = NOW(); + evict_reload(line); + total += NOW() - t; + } + return total; +} + +static void say_evict_set_rows(void) +{ + static const unsigned char gaps[] = { 0, 1, 2, 3, 7, 127 }; + size_t bufsize; + unsigned char *big = rb->plugin_get_buffer(&bufsize); + unsigned long hit = 0, miss = 0; + unsigned g, n, r; + big = (unsigned char *)(((unsigned long)big + 31) & ~31ul); + if (bufsize < 262144) + return; + for (r = 0; r < 8; r++) + { + hit += evict_set_trial(big, 0, 0); + miss += evict_set_trial(big, 20, 0); + } + say("4 lines of a set: kept %lu, gone %lu counts", hit / 8, miss / 8); + say("kept, %%, after new lines in the set"); + say(" between 1 2 3 4 8"); + for (g = 0; g < sizeof(gaps); g++) + { + static const unsigned char news[] = { 1, 2, 3, 4, 8 }; + int kept[5]; + for (n = 0; n < 5; n++) + { + unsigned long t = 0; + for (r = 0; r < 8; r++) + t += evict_set_trial(big, news[n], gaps[g]); + kept[n] = miss > hit + ? (int)(100 - 100 * ((long long)t - hit) / (miss - hit)) : 0; + } + say(" %3d %4d %4d %4d %4d %4d", gaps[g], kept[0], kept[1], + kept[2], kept[3], kept[4]); + } +} +#endif + +/* The same question asked more sharply: a line is loaded, then one to + five more into its set, each a miss with the bus idle before it (32 + adds, plus 0 to 7 more to move the misses in time), and the first is + read again. Chosen at random it survives 75%, 56%, 42%, 32% and 24% of + the time, whatever the spacing. With the ways taken in turn it always + survives three more and never four. With the victim read off a + free-running counter it depends on the spacing, a row at a time. */ +#define EVICTN(name, pad) \ +void name(unsigned char *line, unsigned long n); +#define EVICTN_ASM(name, pad) \ + ".type " #name ", %function\n" \ + #name ":\n" \ + "1:\n" \ + " ldr r2, [r0]\n" \ + " add r0, r0, #4096\n" \ + " .rept 32 + " #pad "\n" \ + " add r3, r3, #1\n" \ + " .endr\n" \ + " subs r1, r1, #1\n" \ + " bne 1b\n" \ + " bx lr\n" +EVICTN(evictn_0, 0) EVICTN(evictn_1, 1) EVICTN(evictn_2, 2) +EVICTN(evictn_3, 3) EVICTN(evictn_4, 4) EVICTN(evictn_5, 5) +EVICTN(evictn_6, 6) EVICTN(evictn_7, 7) +__asm__(".text\n.align 2\n" + EVICTN_ASM(evictn_0, 0) EVICTN_ASM(evictn_1, 1) + EVICTN_ASM(evictn_2, 2) EVICTN_ASM(evictn_3, 3) + EVICTN_ASM(evictn_4, 4) EVICTN_ASM(evictn_5, 5) + EVICTN_ASM(evictn_6, 6) EVICTN_ASM(evictn_7, 7)); + +/* Timer counts to read back the first of 1 + more lines loaded into each + set in turn. The lines come from 56 that share the set, taken round, + so that nearly every load is a miss. */ +static unsigned long evict_lag_trial(unsigned char *big, + void (*fn)(unsigned char *, + unsigned long), + int more) +{ + static unsigned tag; + unsigned long total = 0, t; + int set; + for (set = 0; set < 128; set++) + { + unsigned char *line; + if (tag + 1 + more > 56) + tag = 0; + line = big + set * 32 + tag * 4096; + fn(line, 1 + more); + t = NOW(); + evictn_0(line, 1); + total += NOW() - t; + tag += 1 + more; + } + return total; +} + +static void say_evict_lag_rows(void) +{ + static void (*const fn[])(unsigned char *, unsigned long) = { + evictn_0, evictn_1, evictn_2, evictn_3, + evictn_4, evictn_5, evictn_6, evictn_7 }; + size_t bufsize; + unsigned char *big = rb->plugin_get_buffer(&bufsize); + unsigned long hit = 0, miss = 0; + unsigned d, k, r; + big = (unsigned char *)(((unsigned long)big + 31) & ~31ul); + if (bufsize < 56 * 4096 + 8192) + return; + for (r = 0; r < 16; r++) + { + hit += evict_lag_trial(big, evictn_0, 0); + miss += evict_lag_trial(big, evictn_0, 16); + } + say("first line back: kept %lu, gone %lu counts", hit / 16, miss / 16); + say("first line kept, %%, after more misses in its set"); + say(" pad 1 2 3 4 5"); + for (d = 0; d < 8; d++) + { + int kept[5]; + for (k = 0; k < 5; k++) + { + unsigned long t = 0; + for (r = 0; r < 16; r++) + t += evict_lag_trial(big, fn[d], k + 1); + kept[k] = miss > hit + ? (int)(100 - 100 * ((long long)t - hit) / (miss - hit)) : 0; + } + say(" %u %4d %4d %4d %4d %4d", d, kept[0], kept[1], kept[2], + kept[3], kept[4]); + } +} + +#if TEST_CYC_ARCHIVE +static void say_evict_rows(void) +{ + static const struct { benchfn fn; int pad; } E[] = { + { evict_0, 0 }, { evict_1, 1 }, { evict_2, 2 }, { evict_3, 3 }, + { evict_5, 5 } }; + static const unsigned short lines[] = { 128, 512, 2048 }; + size_t bufsize; + unsigned char *big = rb->plugin_get_buffer(&bufsize); + unsigned long hit = 0, miss = 0, t; + unsigned k, j, r; + big = (unsigned char *)(((unsigned long)big + 31) & ~31ul); + if (bufsize < 65536 + 2048 * 32 + 64) + return; + for (r = 0; r < 16; r++) + { + hit += evict_pass(big, NULL, 0); + miss += evict_pass(big, evict_0, 4096); + } + say("16 KB read: cached %lu, evicted %lu counts", hit / 16, miss / 16); + say("lines kept, %%, after misses elsewhere"); + say(" pad 128 512 2048 lines"); + for (k = 0; k < sizeof(E) / sizeof(E[0]); k++) + { + int kept[3]; + for (j = 0; j < 3; j++) + { + t = 0; + for (r = 0; r < 16; r++) + t += evict_pass(big, E[k].fn, lines[j]); + kept[j] = miss > hit + ? (int)(100 - 100 * ((long long)t - hit) / (miss - hit)) : 0; + } + say(" %d %4d %4d %4d", E[k].pad, kept[0], kept[1], kept[2]); + } +} +#endif +#endif + +/* Misses with room after them, and back to back, in the memory at big: + the plugin buffer, which is DRAM, or on an AS3525v2 the megabyte of RAM + inside the SoC, where codecs are loaded with all their data. */ +static void say_miss_gap_rows(const char *where, unsigned char *big, + size_t bufsize) +{ + static const struct { const char *name; benchfn fn; } G[] = { + { "+32 alu ", gap_32 }, +#if TEST_CYC_ARCHIVE + { "+64 alu ", gap_64 }, +#endif + { "+128 alu ", gap_128 }, +#if TEST_CYC_ARCHIVE + { "+256 alu ", gap_256 }, +#endif + { "ldr w1, +32 ", gap_w1 }, + { "ldr w7, +32 ", gap_w7 }, +#if TEST_CYC_ARCHIVE + { "ldm 2, +32 ", gap_ldm2 }, +#endif + { "ldm 4, +32 ", gap_ldm4 }, + { "ldm w2-5,+32", gap_ldm4mid }, + { "str w0, +32 ", gap_str }, + }; + static const struct { const char *name; benchfn fn; } P[] = { + { "line, no gap", stream_line }, + { "ldm 4 stream", stream_ldm4 }, + { "dirty 4 ", stream_dirty4 }, + { "dirty 8 ", stream_dirty8 }, + { "str stream ", wstream_word }, + { "str a line ", wstream_line }, + { "stm 4 stream", wstream_stm4 }, + { "copy 4 words", copy_4 }, + }; + unsigned long large; + unsigned k; + char a[8], b[8], c[8]; + int hit, miss; + large = (bufsize > 262144 ? 262144 : bufsize) & ~511ul; + say("%s at %08lx", where, (unsigned long)big); + say("miss, then cached %luKB extra", large / 1024); + for (k = 0; k < sizeof(G) / sizeof(G[0]); k++) + { + sbuf = (unsigned long *)big; + sbytes = 1024; + hit = per_insn(cpi_from(G[k].fn, NULL, 2048), e_d_d); + sbytes = large; + miss = per_insn(cpi_from(G[k].fn, NULL, 2048), e_d_d); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" %s %s %s %s", G[k].name, a, b, c); + } + if (bufsize >= 262144) + { + make_chain(big, 1024, 32); + hit = per_insn(cpi_from(chase_work32, (unsigned long *)big, 2048), + e_d_d); + make_chain(big, 262144, 32); + miss = per_insn(cpi_from(chase_work32, (unsigned long *)big, 2048), + e_d_d); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" chase, +32 %s %s %s", a, b, c); + make_chain_scattered(big, 262144, 32); + miss = per_insn(cpi_from(chase_work32, (unsigned long *)big, 2048), + e_d_d); + fmt(b, miss); + fmt(c, miss - hit); + say(" scattered %s %s %s", a, b, c); + } + /* Back to back. The first kilobyte is read before each row, so the + store rows' first column is of stores to lines that are cached. */ + for (k = 0; k < sizeof(P) / sizeof(P[0]); k++) + { + sbuf = (unsigned long *)big; + sbytes = 1024; + stream_word(NULL, 64); + hit = per_insn(cpi_from(P[k].fn, NULL, 2048), e_d_d); + /* The copy reads the first half of the buffer and writes the + second. */ + sbytes = P[k].fn == copy_4 ? 131072 : large; + miss = per_insn(cpi_from(P[k].fn, NULL, 2048), e_d_d); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" %s %s %s %s", P[k].name, a, b, c); + } +} + +static void say_miss_rows_everywhere(void) +{ + size_t bufsize; + unsigned char *big = rb->plugin_get_buffer(&bufsize); + big = (unsigned char *)(((unsigned long)big + 31) & ~31ul); + say_miss_gap_rows("DRAM", big, bufsize - 32); +#if CONFIG_CPU == AS3525v2 + /* The SoC's own RAM is mapped after the DRAM; its first 32 KB is the + firmware's and the rest is where a codec goes, free while nothing + plays. */ + rb->audio_stop(); + say_miss_gap_rows("SoC RAM", (unsigned char *)0x30000000 + + MEMORYSIZE * 0x100000 + 0x40000, 0x80000); +#endif +} + +#if ARM_ARCH >= 5 +/* Fetch misses in any memory: the code is written where it is to run. A + loop of one taken branch a line, or of lines of eight adds, over 4 KB, + which the cache holds, and over 64 KB, where every line is a miss. */ +static void make_code(unsigned long *code, int lines, bool adds) +{ + unsigned long *w = code; + int k, j; + for (k = 0; k < lines; k++, w += 8) + { + for (j = 0; j < 8; j++) + w[j] = 0xe2800001; /* add r0, r0, #1 */ + if (!adds) + w[0] = 0xea000006; /* b the next line */ + } + w[0] = 0xe2511001; /* subs r1, r1, #1 */ + w[1] = 0x1a000000 /* bne the first line */ + | ((unsigned long)(code - (w + 1) - 2) & 0xfffffful); + w[2] = 0xe12fff1e; /* bx lr */ + rb->commit_dcache(); + rb->commit_discard_idcache(); +} + +static void say_code_miss_rows(const char *where, unsigned long *code) +{ + static const char *const name[] = { "a branch a line", "8 adds a line " }; + int v; + char a[8], b[8], c[8]; + say("%s at %08lx, code", where, (unsigned long)code); + say("cycles a line 4KB 64KB extra"); + for (v = 0; v < 2; v++) + { + int hit, miss; + make_code(code, 128, v); + hit = (int)(cpi_from((benchfn)code, NULL, 512) / 128); + make_code(code, 2048, v); + miss = (int)(cpi_from((benchfn)code, NULL, 64) / 2048); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" %s %s %s %s", name[v], a, b, c); + } +} + +#if CONFIG_CPU == AS3525v2 +/* The copy of copy_4, from one memory to the other: whether a fill in one + waits for the writes still to go out to the other. */ +static unsigned long sdelta; +static void copy_across(unsigned long *unused, unsigned long n) +{ + unsigned long *p = sbuf, *end = sbuf + sbytes / 4; + (void)unused; + asm volatile ("1:\n" + X16(" ldmia %[p], { r4-r7 }\n" + " add r8, %[p], %[d]\n" + " stmia r8, { r4-r7 }\n" + " add %[p], %[p], #16\n") + " cmp %[p], %[end]\n" + " movhs %[p], %[base]\n" + " subs %[n], %[n], #1\n" + " bne 1b\n" + : [n] "+&r" (n), [p] "+&r" (p) + : [end] "r" (end), [base] "r" (sbuf), [d] "r" (sdelta) + : "r4", "r5", "r6", "r7", "r8", "cc", "memory"); +} + +static void say_copy_across(const char *name, unsigned long *from, + unsigned long *to) +{ + char a[8], b[8], c[8]; + int hit, miss; + sbuf = from; + sdelta = (unsigned long)to - (unsigned long)from; + sbytes = 1024; + stream_word(NULL, 64); + hit = per_insn(cpi_from(copy_across, NULL, 2048), e_d_d); + sbytes = 131072; + miss = per_insn(cpi_from(copy_across, NULL, 2048), e_d_d); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" %s %s %s %s", name, a, b, c); +} +#endif + +/* Tremor's window loop, which reads two buffers, one upwards and one + downwards, and stores a word to each of two more, none of them cached: + args is the buffer, a quarter of its size and a small table. And the + two streams of stores alone. Each pass is 16 times the body. */ +#define VBODY_STORES \ + " str r5, [r9, #-4]!\n" \ + " str r5, [r6], #4\n" +#define VBODY_WINDOW \ + " ldr r0, [r7], #4\n" \ + " ldr r1, [ip, #-4]!\n" \ + " ldr r2, [r8]\n" \ + " ldr r3, [r8, #4]\n" \ + " smull r4, r5, r0, r2\n" \ + " smlal r4, r5, r1, r3\n" \ + " rsb r2, r2, #0\n" \ + " mov r5, r5, lsl #1\n" \ + " str r5, [r9, #-4]!\n" \ + " smull r4, r5, r0, r3\n" \ + " smlal r4, r5, r1, r2\n" \ + " mov r5, r5, lsl #1\n" \ + " str r5, [r6], #4\n" +#define VLOOP(name, body, step) \ +void name(unsigned long *args, unsigned long n); \ +asm (" .pushsection .text." #name ", \"ax\", %progbits\n" \ + " .align 2\n" \ + " .type " #name ", %function\n" \ + #name ":\n" \ + " stmfd sp!, { r4-r11, lr }\n" \ + " mov r11, r0\n" \ + " mov r10, r1\n" \ + " ldr r8, [r11, #8]\n" \ + "0:\n" \ + " ldr r7, [r11]\n" \ + " ldr r3, [r11, #4]\n" \ + " add ip, r7, r3, lsl #1\n" \ + " add r6, ip, r3\n" \ + " mov r9, r6\n" \ + " add lr, r7, r3\n" \ + "1:\n" \ + X16(body) \ + step \ + " subs r10, r10, #1\n" \ + " beq 2f\n" \ + " cmp r7, lr\n" \ + " blo 1b\n" \ + " b 0b\n" \ + "2:\n" \ + " ldmfd sp!, { r4-r11, pc }\n" \ + " .popsection\n"); +VLOOP(vloop_stores, VBODY_STORES, " add r7, r7, #64\n") +VLOOP(vloop_window, VBODY_WINDOW, "") + +static void say_window_rows(const char *where, unsigned long *big, + size_t bufsize) +{ + static const struct { const char *name; benchfn fn; } V[] = { + { "2 str streams", vloop_stores }, + { "window loop ", vloop_window }, + }; + static unsigned long table[2] = { 0x3fffffff, 0x12345678 }; + unsigned long args[3]; + unsigned long large = (bufsize > 262144 ? 262144 : bufsize) & ~1023ul; + unsigned k; + char a[8], b[8], c[8]; + say("%s at %08lx", where, (unsigned long)big); + say("two streams out cached %luKB extra", large / 1024); + args[0] = (unsigned long)big; + args[2] = (unsigned long)table; + for (k = 0; k < sizeof(V) / sizeof(V[0]); k++) + { + int hit, miss; + sbuf = big; + sbytes = 1024; + stream_word(NULL, 64); + args[1] = 256; + hit = per_insn(cpi_from(V[k].fn, args, 2048), e_d_d); + args[1] = large / 4; + miss = per_insn(cpi_from(V[k].fn, args, 2048), e_d_d); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" %s %s %s %s", V[k].name, a, b, c); + } +} + +static void say_window_rows_everywhere(void) +{ + size_t bufsize; + unsigned char *big = rb->plugin_get_buffer(&bufsize); + big = (unsigned char *)(((unsigned long)big + 31) & ~31ul); + say_window_rows("DRAM", (unsigned long *)big, bufsize - 32); +#if CONFIG_CPU == AS3525v2 + rb->audio_stop(); + say_window_rows("SoC RAM", (unsigned long *)(0x30000000 + + MEMORYSIZE * 0x100000 + 0x40000), 0x80000); +#endif +} + +static void say_code_miss_rows_everywhere(void) +{ + size_t bufsize; + unsigned char *big = rb->plugin_get_buffer(&bufsize); + big = (unsigned char *)(((unsigned long)big + 31) & ~31ul); + if (bufsize >= 0x10100) + say_code_miss_rows("DRAM", (unsigned long *)big); +#if CONFIG_CPU == AS3525v2 + rb->audio_stop(); + say_code_miss_rows("SoC RAM", (unsigned long *)(0x30000000 + + MEMORYSIZE * 0x100000 + 0x40000)); + if (bufsize >= 0x20100) + { + unsigned long *soc = (unsigned long *)(0x30000000 + + MEMORYSIZE * 0x100000 + 0x40000); + say("copy 4 words cached 128KB extra"); + say_copy_across("DRAM to SoC ", (unsigned long *)big, soc); + say_copy_across("SoC to DRAM ", soc, (unsigned long *)big); + } +#endif +} +#endif + +/* How much code a loop holds against what its instructions cost: loops of + adds filling one cache line to 256 of them, and of one taken branch a + line. */ +static void say_code_size_rows(void) +{ + static const struct { benchfn fn; int n; } S[] = { + { seq_6, 6 }, { seq_14, 14 }, { seq_22, 22 }, { seq_30, 30 }, + { seq_38, 38 }, { seq_46, 46 }, { seq_54, 54 }, { seq_62, 62 }, + { seq_70, 70 }, { seq_78, 78 }, { seq_94, 94 }, + { seq_126, 126 }, { seq_254, 254 }, { seq_510, 510 }, + { seq_1022, 1022 }, { seq_2046, 2046 } }; + static const struct { benchfn fn; int n; } C[] = { + { ichain_1, 1 }, { ichain_2, 2 }, { ichain_3, 3 }, { ichain_4, 4 }, + { ichain_5, 5 }, { ichain_6, 6 }, { ichain_7, 7 }, { ichain_8, 8 }, + { ichain_9, 9 }, { ichain_10, 10 }, { ichain_12, 12 }, + { ichain_16, 16 }, { ichain_small, 32 }, + { ichain_64, 64 }, { ichain_128, 128 }, { ichain_256, 256 } }; + unsigned k; + char a[8], b[8]; + say("adds in a loop lines cyc/add over"); + for (k = 0; k < sizeof(S) / sizeof(S[0]); k++) + { + long long c = cpi_from(S[k].fn, NULL, 3200000ul / S[k].n) - e_d_d; + fmt(a, (int)(c / S[k].n)); + fmt(b, (int)(c - S[k].n * 100)); + say(" %4d adds %3d %s %s", S[k].n, (S[k].n + 2) / 8, a, b); + } + say("a branch a line cyc/branch"); + for (k = 0; k < sizeof(C) / sizeof(C[0]); k++) + { + long long c = cpi_from(C[k].fn, NULL, 1600000ul / C[k].n) - e_d_d; + fmt(a, (int)(c / C[k].n)); + say(" %3d lines %s", C[k].n, a); + } + /* The data side: the pointer chase round a few lines, all cached. */ + { + static const unsigned char lines[] = { 1, 2, 3, 4, 5, 6, 7, 8, 9, + 10, 11, 12, 14, 16, 24, 32 }; + size_t bufsize; + unsigned char *big = rb->plugin_get_buffer(&bufsize); + big = (unsigned char *)(((unsigned long)big + 31) & ~31ul); + say("chase, cached cyc/load"); + for (k = 0; k < sizeof(lines); k++) + { + make_chain(big, lines[k] * 32, 32); + fmt(a, per_insn(cpi(chase, (unsigned long *)big), e_d_d)); + say(" %3d lines %s", lines[k], a); + } + } +} + +static void say_rows(const struct row *r, int n) +{ + int k; + for (k = 0; k < n; k++) + { + char d[8]; + int div = r[k].per, sub = r[k].alu * 100; + fmt(d, (per_insn(cpi(r[k].d, dram_buf), e_d_d) - sub) / div); +#ifdef USE_IRAM + { + char a[8], b[8], c[8]; + fmt(a, (per_insn(cpi(r[k].i, iram_buf), e_i_i) - sub) / div); + fmt(b, (per_insn(cpi(r[k].i, dram_buf), e_i_d) - sub) / div); + fmt(c, (per_insn(cpi(r[k].d, iram_buf), e_d_i) - sub) / div); + say("%s %s %s %s %s", r[k].name, a, b, c, d); + } +#else + say("%s %s", r[k].name, d); +#endif + } +} + +/* Define to run only the newest rows, for a quick turn on a device + whose other rows are already known. */ +/* #define TEST_CYC_QUICK */ + enum plugin_status plugin_start(const void* parameter) { - long long e_d_d, loop_cyc; -#ifdef USE_IRAM - long long e_i_i, e_i_d, e_d_i; -#endif - int k; + long long loop_cyc; (void)parameter; rb->lcd_setfont(FONT_SYSFIXED); @@ -438,23 +1991,75 @@ enum plugin_status plugin_start(const void* parameter) say("cyc/insn"); #endif - for (k = 0; k < NB; k++) +#if ARM_ARCH >= 5 && !defined(CPU_ARM_MICRO) { - char d[8]; - int div = B[k].per, sub = B[k].alu * 100; - fmt(d, (per_insn(cpi(B[k].d, dram_buf), e_d_d) - sub) / div); -#ifdef USE_IRAM - { - char a[8], b[8], c[8]; - fmt(a, (per_insn(cpi(B[k].i, iram_buf), e_i_i) - sub) / div); - fmt(b, (per_insn(cpi(B[k].i, dram_buf), e_i_d) - sub) / div); - fmt(c, (per_insn(cpi(B[k].d, iram_buf), e_d_i) - sub) / div); - say("%s %s %s %s %s", B[k].name, a, b, c, d); - } -#else - say("%s %s", B[k].name, d); -#endif + /* The control register, whose bit 14 picks round-robin cache + replacement over random, and the cache type register. */ + unsigned long ctrl, ctype; + asm volatile ("mrc p15, 0, %0, c1, c0, 0" : "=r" (ctrl)); + asm volatile ("mrc p15, 0, %0, c0, c0, 1" : "=r" (ctype)); + say("cp15 control %08lx (RR bit %lu), cache type %08lx", ctrl, + (ctrl >> 14) & 1, ctype); } +#endif + +#ifdef TEST_CYC_QUICK + say("quick run: the newest rows only"); +#if ARM_ARCH >= 5 + say_window_rows_everywhere(); +#else + say_miss_rows_everywhere(); +#endif +#else + say_rows(B, NB); + say_rows(B2, NB2); + say_rows(B3, NB3); + say_filter_rows(); +#if TEST_CYC_ARCHIVE + say_filter_layout(); +#endif + say_code_size_rows(); +#ifndef TICK_TIMED +#if TEST_CYC_ARCHIVE + say_evict_rows(); + say_evict_set_rows(); +#endif + say_evict_lag_rows(); +#endif + say_miss_rows_everywhere(); +#if ARM_ARCH >= 5 + say_code_miss_rows_everywhere(); + say_window_rows_everywhere(); +#endif + + /* Stack traffic, here and on the codec thread. */ +#ifdef USE_IRAM + sp_empty_i = e_i_d; +#endif + sp_empty_d = e_d_d; + stack_rows(); + say_stack_rows("main"); + sp_done = false; + rb->codec_thread_do_callback(stack_rows, NULL); + while (!sp_done) + rb->sleep(HZ / 10); + rb->codec_thread_do_callback(NULL, NULL); + say_stack_rows("codec"); + +#ifdef IRAM_MAP_BYTES + /* What a store costs from IRAM code, by where in IRAM it lands. */ + { + unsigned long off; + say("iram store map, code at %08lx", (unsigned long)b_str_i); + for (off = 0; off < IRAM_MAP_BYTES; off += 4096) + { + char a[8]; + unsigned long *q = &iram_map[off / 4]; + fmt(a, per_insn(cpi(b_str_i, q), e_i_i)); + say(" %08lx %s", (unsigned long)q, a); + } + } +#endif /* Misses: a pointer chase over working sets either side of the cache, in DRAM code, from the plugin buffer. The loop's own cost is the @@ -514,8 +2119,168 @@ enum plugin_status plugin_start(const void* parameter) say(" %s %s %s %s", S[k2].name, a, b, c); } } + + /* Stores: to 1 KB that was read first, so its lines are cached; + to 1 KB after 32 KB elsewhere was read, so they are not, unless + a store brings its line in; and over the large buffer. */ + { + static const struct { const char *name; benchfn fn; } W[] = { + { "word ", wstream_word }, + { "line 32", wstream_line }, + { "stm 4 ", wstream_stm4 }, + }; + unsigned long large = bufsize > 262144 ? 262144 : bufsize; + large &= ~511ul; + say("store stream 1K read 1K cold %luKB", large / 1024); + for (k2 = 0; k2 < sizeof(W) / sizeof(W[0]); k2++) + { + char a[8], b[8], c[8]; + sbuf = (unsigned long *)big; + sbytes = 1024; + stream_word(NULL, 64); + fmt(a, per_insn(cpi(W[k2].fn, NULL), e_d_d)); + if (bufsize >= 131072) + { + sbuf = (unsigned long *)(big + 65536); + sbytes = 32768; + stream_line(NULL, 256); + } + sbuf = (unsigned long *)big; + sbytes = 1024; + fmt(b, per_insn(cpi_from(W[k2].fn, NULL, ITERS), e_d_d)); + sbytes = large; + fmt(c, per_insn(cpi(W[k2].fn, NULL), e_d_d)); + say(" %s %s %s %s", W[k2].name, a, b, c); + } + } + + /* A miss with work after it. */ + { + static const struct { const char *name; benchfn fn; } K[] = { + { "+0 alu ", stream_line }, + { "+8 alu ", stream_work8 }, + { "+16 alu", stream_work16 }, + { "+32 alu", stream_work32 }, + }; + unsigned long large = bufsize > 262144 ? 262144 : bufsize; + large &= ~511ul; + say("load a line cached %luKB extra", large / 1024); + for (k2 = 0; k2 < sizeof(K) / sizeof(K[0]); k2++) + { + char a[8], b[8], c[8]; + int hit, miss; + sbuf = (unsigned long *)big; + sbytes = 1024; + hit = per_insn(cpi(K[k2].fn, NULL), e_d_d); + sbytes = large; + miss = per_insn(cpi(K[k2].fn, NULL), e_d_d); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" %s %s %s %s", K[k2].name, a, b, c); + } + } + + /* Which word of the line the miss asks for, and what it costs + when the lines it evicts are dirty. */ + { + static const struct { const char *name; benchfn fn; } O[] = { + { "word 0 ", stream_first }, + { "word 3 ", stream_mid }, + { "word 7 ", stream_last }, +#if TEST_CYC_ARCHIVE + { "word 3+rest", stream_mid_rest }, + { "word 7+rest", stream_last_rest }, + { "w0, dirty 4", stream_dirty4 }, + { "w0, dirty 8", stream_dirty8 }, + { "dirty 8 +16", stream_dirty8_w16 }, +#endif + { "dirty 8 +32", stream_dirty8_w32 }, + { "dirty 8 +64", stream_dirty8_w64 }, + { "dirty 8 +96", stream_dirty8_w96 }, +#if TEST_CYC_ARCHIVE + { "word 7 +16 ", stream_last_w16 }, +#endif + { "word 7 +32 ", stream_last_w32 }, + }; + unsigned long large = bufsize > 262144 ? 262144 : bufsize; + large &= ~511ul; + say("miss on cached %luKB extra", large / 1024); + for (k2 = 0; k2 < sizeof(O) / sizeof(O[0]); k2++) + { + char a[8], b[8], c[8]; + int hit, miss; + sbuf = (unsigned long *)big; + sbytes = 1024; + hit = per_insn(cpi(O[k2].fn, NULL), e_d_d); + sbytes = large; + miss = per_insn(cpi(O[k2].fn, NULL), e_d_d); + fmt(a, hit); + fmt(b, miss); + fmt(c, miss - hit); + say(" %s %s %s %s", O[k2].name, a, b, c); + } + } + +#if TEST_CYC_ARCHIVE + /* Two buffers filtered in turn and then rewritten together: cycles + a pass of the rewrite, by buffer size and how far apart they + are. 512 bytes each stay cached; 16 KB each do not both fit. */ + if (bufsize >= 131072) + { + static const struct { unsigned bytes, apart; } P[] = { + { 512, 32768 }, { 4096, 32768 }, { 8192, 32768 }, + { 16384, 32768 }, { 16384, 34816 }, { 16384, 16384 }, + { 32768, 32768 }, + }; + say("pair rewrite cyc/pass, 2 x 16 bytes"); + for (k2 = 0; k2 < sizeof(P) / sizeof(P[0]); k2++) + { + char a[8]; + long long both, filt; + pair_a = (unsigned long *)big; + pair_b = (unsigned long *)(big + P[k2].apart); + pair_passes = P[k2].bytes / 16; + filt = cpi_from(pair_filters, NULL, 4); + both = cpi_from(pair_frame, NULL, 4); + fmt(a, (int)((both - filt) / pair_passes)); + say(" %2uK, %2uK apart %s", P[k2].bytes / 1024, + P[k2].apart / 1024, a); + } + } +#endif + + /* Loads far apart: the chase again, in scattered order. */ + { + static const unsigned long sizes[] = { 65536, 262144, 1048576 }; + say("chase scattered stride 32"); + for (k2 = 0; k2 < sizeof(sizes) / sizeof(sizes[0]); k2++) + { + char a[8]; + if (sizes[k2] > bufsize) + break; + make_chain_scattered(big, sizes[k2], 32); + fmt(a, per_insn(cpi(chase, (unsigned long *)big), e_d_d)); + say(" %4lu KB %s", sizes[k2] / 1024, a); + } + } } + /* Instruction fetch misses: a taken branch per line of DRAM code, over + 32 lines of it and over 2048. */ + { + char a[8], b[8], c[8]; + int small = (int)((cpi_from(ichain_small, NULL, 4096) - e_d_d) + / ICHAIN_SMALL); + int big2 = (int)((cpi_from(ichain_big, NULL, 64) - e_d_d) + / ICHAIN_BIG); + fmt(a, small); + fmt(b, big2); + fmt(c, big2 - small); + say("code chain cyc/branch small %s big %s extra %s", a, b, c); + } +#endif /* TEST_CYC_QUICK */ + #ifdef HAVE_ADJUSTABLE_CPU_FREQ rb->cpu_boost(false); #endif