From 01a2b165a063c4e97fe27543aa6c1e9adbb2d24f Mon Sep 17 00:00:00 2001 From: Samuel Price Date: Fri, 26 Jun 2026 19:30:46 -0400 Subject: [PATCH 1/6] contrib/plugins: add mb_cycles MicroBlaze cycle-counting plugin MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Throughput model (stalls=off, default): counts 1 cycle per issued instruction with opcode-class overrides from UG984 §5 (idiv=34, fpu=6, mts/mfs=2). Uses QEMU_PLUGIN_INLINE_ADD_U64 — minimal overhead. RAW-stall model (stalls=on): additionally inserts pipeline bubble cycles when an instruction reads a register whose forwarded value is not yet ready. Pipeline model: C_AREA_OPTIMIZED=0 with forwarding: ALU→ALU is 0 stalls; load-use (lw/lh/lb → use) is 1 stall; blocking instructions (idiv, fpu) incur no extra stall because their base throughput cost already covers the latency. Uses full per-instruction callbacks in stall mode; throughput mode keeps the INLINE_ADD_U64 path for benchmarking. Options: verbose=on prints per-opcode-class breakdown on exit. stalls=on enables the RAW-stall pipeline model. Co-Authored-By: Claude Sonnet 4.6 --- contrib/plugins/Makefile | 1 + contrib/plugins/mb_cycles.c | 465 ++++++++++++++++++++++++++++++++++++ 2 files changed, 466 insertions(+) create mode 100644 contrib/plugins/mb_cycles.c diff --git a/contrib/plugins/Makefile b/contrib/plugins/Makefile index 0b64d2c1e3a..e5706d3ccd6 100644 --- a/contrib/plugins/Makefile +++ b/contrib/plugins/Makefile @@ -27,6 +27,7 @@ endif NAMES += hwprofile NAMES += cache NAMES += drcov +NAMES += mb_cycles ifeq ($(CONFIG_WIN32),y) SO_SUFFIX := .dll diff --git a/contrib/plugins/mb_cycles.c b/contrib/plugins/mb_cycles.c new file mode 100644 index 00000000000..b2cbb23b51b --- /dev/null +++ b/contrib/plugins/mb_cycles.c @@ -0,0 +1,465 @@ +/* + * mb_cycles.c — QEMU plugin: MicroBlaze per-instruction cycle counting + * + * Two operation modes (select with plugin argument): + * + * stalls=off (default) + * Throughput model: every instruction costs its base cycle count from the + * table below. Uses QEMU_PLUGIN_INLINE_ADD_U64 — minimal overhead. + * Measures differences in instruction COUNT between compilers/options. + * + * stalls=on + * RAW-stall model: additionally inserts pipeline bubble cycles whenever + * an instruction reads a register whose forwarded value is not yet ready. + * Uses full per-instruction callbacks (3-5× slower; run on a quiet + * machine with a single benchmark process). + * + * Pipeline model used by stalls=on (C_AREA_OPTIMIZED=0, forwarding enabled): + * + * Producer Latency Stall on immediate consumer + * --------------- ------- -------------------------- + * ALU 1 cycle 0 (result forwarded EX→EX) + * Load (lw/lh/lb) 2 cycles 1 (result from MEM, one cycle late) + * idiv/idivu 34 cycles 0 extra (blocking; base cost covers it) + * fpu 6 cycles 0 extra (same) + * mts / mfs 2 cycles 0 extra (base cost = 2) + * Branch (link) 1 cycle 0 (PC+8 forwarded from IF stage) + * Store / branch — no GPR written + * + * Stall counts are conservative (may overcount on rare edge cases). + * + * Usage: + * qemu-system-microblazeel ... \ + * -plugin /path/to/libmb_cycles.so[,verbose=on][,stalls=on] + * + * Works with both microblaze (BE) and microblazeel (LE) targets. + */ + +#include +#include +#include +#include +#include +#include + +QEMU_PLUGIN_EXPORT int qemu_plugin_version = QEMU_PLUGIN_VERSION; + +/* + * Base throughput cost per opcode[31:26]. + * Source: UG984 v2026.1 §5 latency tables, C_AREA_OPTIMIZED=0. + */ +static const uint8_t mb_cycle_table[64] = { + /* 0x00 */ 1, /* add */ + /* 0x01 */ 1, /* rsub */ + /* 0x02 */ 1, /* addc */ + /* 0x03 */ 1, /* rsubc */ + /* 0x04 */ 1, /* addk */ + /* 0x05 */ 1, /* rsubk / cmp / cmpu */ + /* 0x06 */ 1, /* addkc */ + /* 0x07 */ 1, /* rsubkc */ + /* 0x08 */ 1, /* addi */ + /* 0x09 */ 1, /* rsubi */ + /* 0x0A */ 1, /* addic */ + /* 0x0B */ 1, /* rsubic */ + /* 0x0C */ 1, /* addik */ + /* 0x0D */ 1, /* rsubik */ + /* 0x0E */ 1, /* addikc */ + /* 0x0F */ 1, /* rsubikc */ + /* 0x10 */ 1, /* mul / mulh / mulhu / mulhsu (pipelined) */ + /* 0x11 */ 1, /* bsrl / bsra / bsll */ + /* 0x12 */34, /* idiv / idivu (UG984: 34 cycles, blocking) */ + /* 0x13 */ 1, /* getd / putd (FSL) */ + /* 0x14 */ 1, /* (reserved) */ + /* 0x15 */ 1, /* (reserved) */ + /* 0x16 */ 6, /* fadd/frsub/fmul/fdiv/flt/fint/fcmp (6 cyc) */ + /* 0x17 */ 1, /* (reserved) */ + /* 0x18 */ 1, /* muli */ + /* 0x19 */ 1, /* bsrli / bsrai / bslli / bsefi / bsifi */ + /* 0x1A */ 1, /* (reserved) */ + /* 0x1B */ 1, /* get / put (FSL) */ + /* 0x1C */ 1, /* (reserved) */ + /* 0x1D */ 1, /* (reserved) */ + /* 0x1E */ 1, /* (reserved) */ + /* 0x1F */ 1, /* (reserved) */ + /* 0x20 */ 1, /* or / pcmpbf */ + /* 0x21 */ 1, /* and */ + /* 0x22 */ 1, /* xor / pcmpeq */ + /* 0x23 */ 1, /* andn / pcmpne */ + /* 0x24 */ 1, /* sra/src/srl/sext8/sext16/clz/swapb/swaph */ + /* 0x25 */ 2, /* mts / mfs / msrset / msrclr (UG984: 2 cyc) */ + /* 0x26 */ 1, /* br/bra/brd/brad/brld/brald/brk */ + /* 0x27 */ 1, /* beq-bge and delay-slot variants */ + /* 0x28 */ 1, /* ori */ + /* 0x29 */ 1, /* andi */ + /* 0x2A */ 1, /* xori */ + /* 0x2B */ 1, /* andni */ + /* 0x2C */ 1, /* imm */ + /* 0x2D */ 1, /* rtsd / rtid / rtbd / rted */ + /* 0x2E */ 1, /* bri/brai/brid/braid/brlid/bralid/brki/mbar */ + /* 0x2F */ 1, /* beqi-bgei and delay-slot variants */ + /* 0x30 */ 1, /* lbu / lbur / lbuea */ + /* 0x31 */ 1, /* lhu / lhur / lhuea */ + /* 0x32 */ 1, /* lw / lwr / lwea / lwx */ + /* 0x33 */ 1, /* (reserved) */ + /* 0x34 */ 1, /* sb / sbr / sbea */ + /* 0x35 */ 1, /* sh / shr / shea */ + /* 0x36 */ 1, /* sw / swr / swea / swx */ + /* 0x37 */ 1, /* (reserved) */ + /* 0x38 */ 1, /* lbui */ + /* 0x39 */ 1, /* lhui */ + /* 0x3A */ 1, /* lwi */ + /* 0x3B */ 1, /* (reserved) */ + /* 0x3C */ 1, /* sbi */ + /* 0x3D */ 1, /* shi */ + /* 0x3E */ 1, /* swi */ + /* 0x3F */ 1, /* (reserved) */ +}; + +/* + * Forwarding latency table for the stall model. + * + * For each opcode: number of cycles after the instruction ISSUES before + * its GPR result is available for forwarding to the next instruction's + * execute stage. 0 means the instruction writes no GPR. + * + * For blocking instructions (idiv=34, fpu=6, mts/mfs=2): the latency equals + * the base throughput cost, so the next instruction (which can only start + * after the blocking instruction completes) finds the result already ready — + * no additional stall is incurred beyond what is already in mb_cycle_table. + */ +static const uint8_t mb_result_latency[64] = { + /* 0x00 add */ 1, /* 0x01 rsub */ 1, /* 0x02 addc */ 1, /* 0x03 rsubc */ 1, + /* 0x04 addk */ 1, /* 0x05 rsubk */ 1, /* 0x06 addkc */ 1, /* 0x07 rsubkc */ 1, + /* 0x08 addi */ 1, /* 0x09 rsubi */ 1, /* 0x0A addic */ 1, /* 0x0B rsubic */ 1, + /* 0x0C addik */ 1, /* 0x0D rsubik */ 1, /* 0x0E addikc */ 1, /* 0x0F rsubikc*/ 1, + /* 0x10 mul */ 1, /* 0x11 bsrl */ 1, /* 0x12 idiv */34, /* 0x13 getd */ 1, + /* 0x14 rsv */ 0, /* 0x15 rsv */ 0, /* 0x16 fpu */ 6, /* 0x17 rsv */ 0, + /* 0x18 muli */ 1, /* 0x19 bsrli */ 1, /* 0x1A rsv */ 0, /* 0x1B get */ 1, + /* 0x1C rsv */ 0, /* 0x1D rsv */ 0, /* 0x1E rsv */ 0, /* 0x1F rsv */ 0, + /* 0x20 or */ 1, /* 0x21 and */ 1, /* 0x22 xor */ 1, /* 0x23 andn */ 1, + /* 0x24 sra */ 1, /* 0x25 mfs */ 2, /* 0x26 br */ 1, /* 0x27 beq */ 0, + /* 0x28 ori */ 1, /* 0x29 andi */ 1, /* 0x2A xori */ 1, /* 0x2B andni */ 1, + /* 0x2C imm */ 0, /* 0x2D rtsd */ 0, /* 0x2E bri */ 1, /* 0x2F beqi */ 0, + /* 0x30 lbu */ 2, /* 0x31 lhu */ 2, /* 0x32 lw */ 2, /* 0x33 rsv */ 0, + /* 0x34 sb */ 0, /* 0x35 sh */ 0, /* 0x36 sw */ 0, /* 0x37 rsv */ 0, + /* 0x38 lbui */ 2, /* 0x39 lhui */ 2, /* 0x3A lwi */ 2, /* 0x3B rsv */ 0, + /* 0x3C sbi */ 0, /* 0x3D shi */ 0, /* 0x3E swi */ 0, /* 0x3F rsv */ 0, +}; + +/* + * For these opcodes, bits[25:21] encode a SOURCE register rather than (or in + * addition to) the usual Rd destination field: + * + * 0x27 beq-bge : bits[25:21] = Ra (condition register) + * 0x2F beqi-bgei : bits[25:21] = Ra (condition register) + * 0x34 sb : bits[25:21] = rD (data to store) + * 0x35 sh : bits[25:21] = rD (data to store) + * 0x36 sw : bits[25:21] = rD (data to store) + * 0x3C sbi : bits[25:21] = rD (data to store, imm addressing) + * 0x3D shi : bits[25:21] = rD (data to store, imm addressing) + * 0x3E swi : bits[25:21] = rD (data to store, imm addressing) + * + * The stall model checks reg_ready for these bits[25:21] as an extra source. + */ +static const bool mb_rd_is_src[64] = { + [0x27] = true, [0x2F] = true, + [0x34] = true, [0x35] = true, [0x36] = true, + [0x3C] = true, [0x3D] = true, [0x3E] = true, +}; + +/* Human-readable class label for each opcode slot (verbose output). */ +static const char *const mb_class_name[64] = { + "add", "rsub", "addc", "rsubc", + "addk", "rsubk/cmp","addkc", "rsubkc", + "addi", "rsubi", "addic", "rsubic", + "addik", "rsubik", "addikc", "rsubikc", + "mul", "bsrl/a/l", "idiv", "getd/putd", + "rsv", "rsv", "fpu", "rsv", + "muli", "bsrli/ai", "rsv", "get/put", + "rsv", "rsv", "rsv", "rsv", + "or", "and", "xor", "andn", + "sra/ext", "mts/mfs", "br/brld", "beq-bge", + "ori", "andi", "xori", "andni", + "imm", "rtsd", "bri/braid","beqi-bgei", + "lbu", "lhu", "lw", "rsv", + "sb", "sh", "sw", "rsv", + "lbui", "lhui", "lwi", "rsv", + "sbi", "shi", "swi", "rsv", +}; + +/* ---- Counters (single-core MicroBlaze; not thread-safe for SMP) ---- */ +static uint64_t total_cycles; +static uint64_t total_insns; +static uint64_t total_stall_cycles; +static uint64_t opcode_cycles[64]; +static uint64_t opcode_count[64]; +static uint64_t trans_count; + +static bool verbose; +static bool model_stalls; +static bool big_endian_target; + +/* ---- Stall model state ---- */ +static uint64_t reg_ready[32]; /* cycle when each GPR result is forwardable */ +static uint64_t pipeline_cycle; /* logical pipeline cycle counter */ + +/* + * Extract the 6-bit opcode from the MSB byte of a MicroBlaze instruction. + * Works for both endian variants (see insn_bytes() comment). + */ +static inline uint32_t insn_opcode(const uint8_t *data) +{ + return (big_endian_target ? data[0] : data[3]) >> 2; +} + +/* + * Reconstruct the full 32-bit instruction word from guest memory bytes. + * MicroBlaze bit-31 = MSB. In little-endian ELF (microblazeel): byte[3] is MSB. + */ +static inline uint32_t insn_word32(const uint8_t *data) +{ + if (big_endian_target) + return ((uint32_t)data[0] << 24) | ((uint32_t)data[1] << 16) + | ((uint32_t)data[2] << 8) | (uint32_t)data[3]; + else + return ((uint32_t)data[3] << 24) | ((uint32_t)data[2] << 16) + | ((uint32_t)data[1] << 8) | (uint32_t)data[0]; +} + +/* + * Return host-virtual pointer to the 4 bytes of a guest instruction. + * qemu_plugin_insn_haddr() is always valid for RAM-backed pages. + */ +static inline const uint8_t *insn_bytes(const struct qemu_plugin_insn *insn) +{ + if (qemu_plugin_insn_size(insn) >= 4) + return (const uint8_t *)qemu_plugin_insn_data(insn); + return (const uint8_t *)qemu_plugin_insn_haddr(insn); +} + +/* Return the number of stall cycles register R would impose right now. */ +static inline uint64_t stall_for(uint8_t r, uint64_t now) +{ + return (r && reg_ready[r] > now) ? reg_ready[r] - now : 0; +} + +/* + * Full callback for the RAW-stall model (stalls=on). + * + * Userdata is a packed 25-bit value: + * bits [ 5: 0] op (6-bit opcode) + * bits [10: 6] rd (bits[25:21] of instruction word) + * bits [15:11] ra (bits[20:16]) + * bits [20:16] rb (bits[15:11], meaningful only if !type_b) + * bit [21] type_b (1 = Type-B instruction, no Rb register field) + * bit [22] rd_is_src (1 = also check rd as a source, per mb_rd_is_src) + * bit [23] valid (1 = instruction bytes were available at trans time) + */ +static void vcpu_insn_exec_stall(unsigned int vcpu_idx, void *userdata) +{ + uintptr_t meta = (uintptr_t)userdata; + uint8_t op = (meta >> 0) & 0x3f; + uint8_t rd = (meta >> 6) & 0x1f; + uint8_t ra = (meta >> 11) & 0x1f; + uint8_t rb = (meta >> 16) & 0x1f; + bool type_b = (meta >> 21) & 1; + bool rd_is_src = (meta >> 22) & 1; + bool valid = (meta >> 23) & 1; + + uint64_t base = mb_cycle_table[op]; + uint64_t stall = 0; + + if (valid) { + uint64_t now = pipeline_cycle; + uint64_t s; + + s = stall_for(ra, now); + if (s > stall) stall = s; + + if (!type_b) { + s = stall_for(rb, now); + if (s > stall) stall = s; + } + + if (rd_is_src) { + s = stall_for(rd, now); + if (s > stall) stall = s; + } + } + + uint64_t issue = pipeline_cycle + stall; + pipeline_cycle = issue + base; + + /* Record when this instruction's GPR result will be forwardable. */ + uint8_t lat = mb_result_latency[op]; + if (lat && rd) + reg_ready[rd] = issue + lat; + + uint64_t cost = base + stall; + total_cycles += cost; + total_stall_cycles += stall; + total_insns++; + opcode_count[op]++; + opcode_cycles[op] += cost; +} + +static void vcpu_tb_trans(qemu_plugin_id_t id, struct qemu_plugin_tb *tb) +{ + trans_count++; + size_t n = qemu_plugin_tb_n_insns(tb); + + for (size_t i = 0; i < n; i++) { + struct qemu_plugin_insn *insn = qemu_plugin_tb_get_insn(tb, i); + const uint8_t *data = insn_bytes(insn); + + if (model_stalls) { + /* --- RAW-stall mode: full callback with packed instruction metadata --- */ + uint8_t op = 63, rd = 0, ra = 0, rb = 0; + bool type_b = false, rd_is_src = false, valid = false; + + if (data) { + uint32_t w = insn_word32(data); + op = (w >> 26) & 0x3f; + rd = (w >> 21) & 0x1f; + ra = (w >> 16) & 0x1f; + rb = (w >> 11) & 0x1f; + type_b = (op & 0x08) != 0; + rd_is_src = mb_rd_is_src[op]; + valid = true; + } + + uintptr_t meta = (uintptr_t)op + | ((uintptr_t)rd << 6) + | ((uintptr_t)ra << 11) + | ((uintptr_t)rb << 16) + | ((uintptr_t)type_b << 21) + | ((uintptr_t)rd_is_src << 22) + | ((uintptr_t)valid << 23); + + qemu_plugin_register_vcpu_insn_exec_cb( + insn, vcpu_insn_exec_stall, + QEMU_PLUGIN_CB_NO_REGS, (void *)meta); + + } else { + /* --- Throughput mode: four inline atomic counters per instruction --- */ + uint8_t op; + uint64_t cycles; + if (data) { + op = insn_opcode(data); + cycles = mb_cycle_table[op]; + } else { + op = 63; + cycles = 1; + } + + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total_insns, 1); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total_cycles, cycles); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &opcode_count[op], 1); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &opcode_cycles[op],cycles); + } + } +} + +static void plugin_exit(qemu_plugin_id_t id, void *userdata) +{ + char buf[320]; + const char *mode = model_stalls + ? "(C_AREA_OPTIMIZED=0, branch-predicted, RAW stalls modeled)" + : "(C_AREA_OPTIMIZED=0, branch-predicted, throughput only)"; + + if (model_stalls) { + snprintf(buf, sizeof(buf), + "MicroBlaze cycle model %s\n" + "Total instructions : %" PRIu64 "\n" + "Total cycles : %" PRIu64 "\n" + "Stall cycles : %" PRIu64 "\n" + "CPI : %.3f\n" + "Stall CPI : %.3f\n" + "TB translations : %" PRIu64 "\n", + mode, + total_insns, total_cycles, total_stall_cycles, + total_insns ? (double)total_cycles / total_insns : 0.0, + total_insns ? (double)total_stall_cycles / total_insns : 0.0, + trans_count); + } else { + snprintf(buf, sizeof(buf), + "MicroBlaze cycle model %s\n" + "Total instructions : %" PRIu64 "\n" + "Total cycles : %" PRIu64 "\n" + "CPI : %.3f\n" + "TB translations : %" PRIu64 "\n", + mode, + total_insns, total_cycles, + total_insns ? (double)total_cycles / total_insns : 0.0, + trans_count); + } + qemu_plugin_outs(buf); + + if (!verbose) + return; + + qemu_plugin_outs("Opcode class breakdown:\n" + " Op Class Insns Cycles CPI\n" + " ---- --------------- ------------ -------- ----\n"); + + for (int op = 0; op < 64; op++) { + if (!opcode_count[op]) + continue; + snprintf(buf, sizeof(buf), + " 0x%02x %-15s %12" PRIu64 " %8" PRIu64 " %.2f\n", + op, mb_class_name[op], + opcode_count[op], opcode_cycles[op], + (double)opcode_cycles[op] / opcode_count[op]); + qemu_plugin_outs(buf); + } +} + +QEMU_PLUGIN_EXPORT +int qemu_plugin_install(qemu_plugin_id_t id, const qemu_info_t *info, + int argc, char **argv) +{ + for (int i = 0; i < argc; i++) { + char *opt = argv[i]; + g_auto(GStrv) tokens = g_strsplit(opt, "=", 2); + const char *key = tokens[0]; + const char *val = tokens[1] ? tokens[1] : ""; + + if (!g_strcmp0(key, "verbose")) { + if (!g_strcmp0(val,"on")||!g_strcmp0(val,"yes")|| + !g_strcmp0(val,"true")||!g_strcmp0(val,"1")) + verbose = true; + else if (!g_strcmp0(val,"off")||!g_strcmp0(val,"no")|| + !g_strcmp0(val,"false")||!g_strcmp0(val,"0")) + verbose = false; + else { + fprintf(stderr, "mb_cycles: bad value for verbose: %s\n", opt); + return -1; + } + } else if (!g_strcmp0(key, "stalls")) { + if (!g_strcmp0(val,"on")||!g_strcmp0(val,"yes")|| + !g_strcmp0(val,"true")||!g_strcmp0(val,"1")) + model_stalls = true; + else if (!g_strcmp0(val,"off")||!g_strcmp0(val,"no")|| + !g_strcmp0(val,"false")||!g_strcmp0(val,"0")) + model_stalls = false; + else { + fprintf(stderr, "mb_cycles: bad value for stalls: %s\n", opt); + return -1; + } + } else { + fprintf(stderr, "mb_cycles: unknown option '%s'\n", opt); + return -1; + } + } + + big_endian_target = info->target_name && + strcmp(info->target_name, "microblaze") == 0; + + qemu_plugin_register_vcpu_tb_trans_cb(id, vcpu_tb_trans); + qemu_plugin_register_atexit_cb(id, plugin_exit, NULL); + return 0; +} From 0b0a8247f9aeaa5770937941f2f2cf8a53a3ba8d Mon Sep 17 00:00:00 2001 From: Samuel Price Date: Sat, 27 Jun 2026 19:56:12 -0400 Subject: [PATCH 2/6] hw/microblaze + plugins: exit device and accurate FPU cycle counting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit petalogix_s3adsp1800_mmu: add memory-mapped exit/counter device at 0xFF000000. - Write 0 → clean QEMU shutdown; nonzero → panic exit with code. - Reads at +4/+8 return lo/hi of QEMU virtual clock (ns), which with -icount 0 equals the instruction count. Used by bare-metal benchmarks. mb_cycles: per-instruction FPU latency dispatch and configurable area_opt. - opcode 0x16 (FPU) now dispatches on instruction bits[9:7] to give per-instruction latencies: fadd/frsub=4, fmul=4, fdiv=30, fcmp=4, flt=5, fint=4, fsqrt=27 (C_AREA_OPTIMIZED=0). - Full 3×8 latency table covers all three C_AREA_OPTIMIZED levels; select at runtime via area_opt=0|1|2 plugin argument. - WIC (opcode 0x24, func bit[3]=1) correctly counted as 2 cycles. - stall model: base_cyc packed into meta[31:24] at translation time, eliminating the table lookup from the hot callback path. Co-Authored-By: Claude Sonnet 4.6 --- contrib/plugins/mb_cycles.c | 882 ++++++++++++++--------- hw/microblaze/petalogix_s3adsp1800_mmu.c | 49 ++ 2 files changed, 593 insertions(+), 338 deletions(-) diff --git a/contrib/plugins/mb_cycles.c b/contrib/plugins/mb_cycles.c index b2cbb23b51b..23322c0b9f5 100644 --- a/contrib/plugins/mb_cycles.c +++ b/contrib/plugins/mb_cycles.c @@ -1,38 +1,76 @@ /* - * mb_cycles.c — QEMU plugin: MicroBlaze per-instruction cycle counting + * mb_cycles.c — QEMU plugin: MicroBlaze deterministic-cycle regression oracle * - * Two operation modes (select with plugin argument): + * PURPOSE + * ------- + * Provide a deterministic, machine-readable cycle model for LLVM/GCC codegen + * regression testing on MicroBlaze. The goal is NOT hardware-cycle accuracy. + * The goal is a reproducible measurement so that compiler changes produce clear + * before/after diffs. * - * stalls=off (default) - * Throughput model: every instruction costs its base cycle count from the - * table below. Uses QEMU_PLUGIN_INLINE_ADD_U64 — minimal overhead. - * Measures differences in instruction COUNT between compilers/options. + * TWO MEASUREMENT MODES + * --------------------- + * weighted (stalls=off, default): + * Counts weighted instruction cycles from a fixed UG984 §5 table. No stalls, + * no memory effects, no branch penalties. Use as the primary regression baseline. * - * stalls=on - * RAW-stall model: additionally inserts pipeline bubble cycles whenever - * an instruction reads a register whose forwarded value is not yet ready. - * Uses full per-instruction callbacks (3-5× slower; run on a quiet - * machine with a single benchmark process). + * raw-stalls (stalls=on): + * Adds in-order RAW forwarding stalls on top of weighted cycles. Tracks + * register readiness for all 32 GPRs. Loads add 1 stall on a direct consumer + * (load result latency = 2 cycles, forwarded from MEM stage). Blocking ops + * (idiv=34, FPU=per-func, mts/mfs=2) stall the pipeline for their base cost; + * no extra RAW stall after them. Ignores WAW/WAR hazards, memory hierarchy, + * and branch penalties. * - * Pipeline model used by stalls=on (C_AREA_OPTIMIZED=0, forwarding enabled): + * BENCHMARK CONTROL (magic NOP instructions) + * ------------------------------------------ + * Benchmarks delimit the measured hot region with "ori r0, r0, 0x4DXX" words. + * Writing to r0 is always a hardware NOP on MicroBlaze. * - * Producer Latency Stall on immediate consumer - * --------------- ------- -------------------------- - * ALU 1 cycle 0 (result forwarded EX→EX) - * Load (lw/lh/lb) 2 cycles 1 (result from MEM, one cycle late) - * idiv/idivu 34 cycles 0 extra (blocking; base cost covers it) - * fpu 6 cycles 0 extra (same) - * mts / mfs 2 cycles 0 extra (base cost = 2) - * Branch (link) 1 cycle 0 (PC+8 forwarded from IF stage) - * Store / branch — no GPR written + * .word 0xA0004D01 BENCH_RESTART — snapshot counters, begin measurement + * .word 0xA0004D02 BENCH_STOP — compute delta, end measurement + * .word 0xA0004D03 BENCH_EXIT_PASS — emit JSON, status=pass + * .word 0xA0004D04 BENCH_EXIT_FAIL — emit JSON, status=fail * - * Stall counts are conservative (may overcount on rare edge cases). + * Only cycles/instructions between BENCH_RESTART and BENCH_STOP are reported. + * Magic instructions themselves are not counted as code instructions. * - * Usage: - * qemu-system-microblazeel ... \ - * -plugin /path/to/libmb_cycles.so[,verbose=on][,stalls=on] + * COUNTERS EMITTED PER BENCHMARK REGION + * -------------------------------------- + * instructions total instruction count + * weighted_cycles sum of UG984 §5 instruction weights + * raw_stall_cycles additional RAW stall cycles (stalls=on only) + * raw_cycles weighted_cycles + raw_stall_cycles + * cpi_weighted weighted_cycles / instructions + * cpi_raw raw_cycles / instructions + * alu_count, load_count, store_count, branch_count + * nop_count, div_count, mul_count, fpu_count, mts_mfs_count + * delayed_branch_count branches with a delay slot (D-form) + * delay_slot_nop_count wasted delay slots (branch + canonical NOP) + * delay_slot_filled_count useful delay slots (branch + non-NOP) + * delay_slot_fill_rate delay_slot_filled / delayed_branch * - * Works with both microblaze (BE) and microblazeel (LE) targets. + * PLUGIN ARGUMENTS + * ---------------- + * bench=NAME benchmark name (default: "unknown") + * compiler=NAME compiler name (default: "unknown") + * opt=FLAGS optimisation flags (default: "") + * target=CONFIG target configuration (default: "") + * commit=HASH compiler commit hash (default: "") + * stalls=on|off enable RAW stall model (default: off) + * area_opt=0|1|2 FPU latency variant (default: 0) + * verbose=on|off per-opcode breakdown appended to JSON (default: off) + * + * OUTPUT + * ------ + * One JSON object per benchmark run at BENCH_EXIT_PASS/FAIL. If no + * BENCH_RESTART is ever seen, one JSON record covering the full run is emitted + * at plugin exit. + * + * NOT MODELLED + * ------------ + * I-cache, D-cache, DDR/BRAM wait states, PLB/AXI bus, branch prediction, + * taken/not-taken penalties, interrupts, DMA, timer effects. */ #include @@ -44,122 +82,84 @@ QEMU_PLUGIN_EXPORT int qemu_plugin_version = QEMU_PLUGIN_VERSION; -/* - * Base throughput cost per opcode[31:26]. - * Source: UG984 v2026.1 §5 latency tables, C_AREA_OPTIMIZED=0. +/* ========================================================================= + * Instruction encoding + * ========================================================================= */ + +/* ori r0, r0, IMM: (0x28<<26)|(0<<21)|(0<<16)|IMM16 = 0xA0000000 | IMM */ +#define MB_ORI_R0_MASK 0xFFFF0000u +#define MB_ORI_R0_BASE 0xA0000000u +#define MB_MAGIC_MASK 0x0000FF00u +#define MB_MAGIC_BYTE 0x00004D00u /* 'M' in the IMM high byte */ + +#define MBEV_RESTART 0x4D01u +#define MBEV_STOP 0x4D02u +#define MBEV_PASS 0x4D03u +#define MBEV_FAIL 0x4D04u + +/* or r0, r0, r0 = 0x80000000: canonical delay-slot NOP */ +#define MB_NOP_WORD 0x80000000u + +/* ========================================================================= + * Weighted cost table (UG984 v2026.1 §5, C_AREA_OPTIMIZED=0) + * ========================================================================= */ +static const uint8_t mb_weight[64] = { + /* 0x00–0x0F: integer add/sub variants */ + 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, + /* 0x10 mul */ 1, /* 0x11 bsrl/a/l */ 1, /* 0x12 idiv */ 34, /* 0x13 getd */ 1, + /* 0x14 rsv */ 1, /* 0x15 rsv */ 1, /* 0x16 fpu */ 6, /* 0x17 rsv */ 1, + /* 0x18 muli */ 1, /* 0x19 bsrli */ 1, /* 0x1A rsv */ 1, /* 0x1B get */ 1, + /* 0x1C–0x1F reserved */ 1, 1, 1, 1, + /* 0x20 or */ 1, /* 0x21 and */ 1, /* 0x22 xor */ 1, /* 0x23 andn */ 1, + /* 0x24 sra */ 1, /* 0x25 mts/mfs */ 2, /* 0x26 br */ 1, /* 0x27 beq */ 1, + /* 0x28 ori */ 1, /* 0x29 andi */ 1, /* 0x2A xori */ 1, /* 0x2B andni*/ 1, + /* 0x2C imm */ 1, /* 0x2D rtsd */ 1, /* 0x2E bri */ 1, /* 0x2F beqi */ 1, + /* 0x30–0x37: loads/stores */ 1, 1, 1, 1, 1, 1, 1, 1, + /* 0x38–0x3F: loads/stores */ 1, 1, 1, 1, 1, 1, 1, 1, +}; + +/* Per-instruction FPU latency [area_opt 0/1/2][func bits[9:7]]: + * idx func name opt=0 opt=1 opt=2 + * 0 0x000 fadd/frsub 4 6 1 + * 2 0x100 fmul 4 6 1 + * 3 0x180 fdiv 30 32 29 + * 4 0x200 fcmp.* 4 6 1 + * 5 0x280 flt 5 7 2 + * 6 0x300 fint 4 6 1 + * 7 0x380 fsqrt 27 29 23 */ -static const uint8_t mb_cycle_table[64] = { - /* 0x00 */ 1, /* add */ - /* 0x01 */ 1, /* rsub */ - /* 0x02 */ 1, /* addc */ - /* 0x03 */ 1, /* rsubc */ - /* 0x04 */ 1, /* addk */ - /* 0x05 */ 1, /* rsubk / cmp / cmpu */ - /* 0x06 */ 1, /* addkc */ - /* 0x07 */ 1, /* rsubkc */ - /* 0x08 */ 1, /* addi */ - /* 0x09 */ 1, /* rsubi */ - /* 0x0A */ 1, /* addic */ - /* 0x0B */ 1, /* rsubic */ - /* 0x0C */ 1, /* addik */ - /* 0x0D */ 1, /* rsubik */ - /* 0x0E */ 1, /* addikc */ - /* 0x0F */ 1, /* rsubikc */ - /* 0x10 */ 1, /* mul / mulh / mulhu / mulhsu (pipelined) */ - /* 0x11 */ 1, /* bsrl / bsra / bsll */ - /* 0x12 */34, /* idiv / idivu (UG984: 34 cycles, blocking) */ - /* 0x13 */ 1, /* getd / putd (FSL) */ - /* 0x14 */ 1, /* (reserved) */ - /* 0x15 */ 1, /* (reserved) */ - /* 0x16 */ 6, /* fadd/frsub/fmul/fdiv/flt/fint/fcmp (6 cyc) */ - /* 0x17 */ 1, /* (reserved) */ - /* 0x18 */ 1, /* muli */ - /* 0x19 */ 1, /* bsrli / bsrai / bslli / bsefi / bsifi */ - /* 0x1A */ 1, /* (reserved) */ - /* 0x1B */ 1, /* get / put (FSL) */ - /* 0x1C */ 1, /* (reserved) */ - /* 0x1D */ 1, /* (reserved) */ - /* 0x1E */ 1, /* (reserved) */ - /* 0x1F */ 1, /* (reserved) */ - /* 0x20 */ 1, /* or / pcmpbf */ - /* 0x21 */ 1, /* and */ - /* 0x22 */ 1, /* xor / pcmpeq */ - /* 0x23 */ 1, /* andn / pcmpne */ - /* 0x24 */ 1, /* sra/src/srl/sext8/sext16/clz/swapb/swaph */ - /* 0x25 */ 2, /* mts / mfs / msrset / msrclr (UG984: 2 cyc) */ - /* 0x26 */ 1, /* br/bra/brd/brad/brld/brald/brk */ - /* 0x27 */ 1, /* beq-bge and delay-slot variants */ - /* 0x28 */ 1, /* ori */ - /* 0x29 */ 1, /* andi */ - /* 0x2A */ 1, /* xori */ - /* 0x2B */ 1, /* andni */ - /* 0x2C */ 1, /* imm */ - /* 0x2D */ 1, /* rtsd / rtid / rtbd / rted */ - /* 0x2E */ 1, /* bri/brai/brid/braid/brlid/bralid/brki/mbar */ - /* 0x2F */ 1, /* beqi-bgei and delay-slot variants */ - /* 0x30 */ 1, /* lbu / lbur / lbuea */ - /* 0x31 */ 1, /* lhu / lhur / lhuea */ - /* 0x32 */ 1, /* lw / lwr / lwea / lwx */ - /* 0x33 */ 1, /* (reserved) */ - /* 0x34 */ 1, /* sb / sbr / sbea */ - /* 0x35 */ 1, /* sh / shr / shea */ - /* 0x36 */ 1, /* sw / swr / swea / swx */ - /* 0x37 */ 1, /* (reserved) */ - /* 0x38 */ 1, /* lbui */ - /* 0x39 */ 1, /* lhui */ - /* 0x3A */ 1, /* lwi */ - /* 0x3B */ 1, /* (reserved) */ - /* 0x3C */ 1, /* sbi */ - /* 0x3D */ 1, /* shi */ - /* 0x3E */ 1, /* swi */ - /* 0x3F */ 1, /* (reserved) */ +static const uint8_t fpu_latency[3][8] = { + { 4, 4, 4, 30, 4, 5, 4, 27 }, + { 6, 6, 6, 32, 6, 7, 6, 29 }, + { 1, 1, 1, 29, 1, 2, 1, 23 }, }; +static unsigned fpu_area_opt; + +static inline uint8_t mb_fpu_lat(uint32_t word) +{ + return fpu_latency[fpu_area_opt][(word >> 7) & 7]; +} /* - * Forwarding latency table for the stall model. - * - * For each opcode: number of cycles after the instruction ISSUES before - * its GPR result is available for forwarding to the next instruction's - * execute stage. 0 means the instruction writes no GPR. - * - * For blocking instructions (idiv=34, fpu=6, mts/mfs=2): the latency equals - * the base throughput cost, so the next instruction (which can only start - * after the blocking instruction completes) finds the result already ready — - * no additional stall is incurred beyond what is already in mb_cycle_table. + * RAW forwarding latency for the stall model. + * 0 = no GPR written (store, branch, imm-prefix, reserved). + * Blocking ops: latency == base cost → consumer starts right after → 0 extra stall. + * FPU (0x16): handled separately via mb_fpu_lat(). */ static const uint8_t mb_result_latency[64] = { - /* 0x00 add */ 1, /* 0x01 rsub */ 1, /* 0x02 addc */ 1, /* 0x03 rsubc */ 1, - /* 0x04 addk */ 1, /* 0x05 rsubk */ 1, /* 0x06 addkc */ 1, /* 0x07 rsubkc */ 1, - /* 0x08 addi */ 1, /* 0x09 rsubi */ 1, /* 0x0A addic */ 1, /* 0x0B rsubic */ 1, - /* 0x0C addik */ 1, /* 0x0D rsubik */ 1, /* 0x0E addikc */ 1, /* 0x0F rsubikc*/ 1, - /* 0x10 mul */ 1, /* 0x11 bsrl */ 1, /* 0x12 idiv */34, /* 0x13 getd */ 1, - /* 0x14 rsv */ 0, /* 0x15 rsv */ 0, /* 0x16 fpu */ 6, /* 0x17 rsv */ 0, - /* 0x18 muli */ 1, /* 0x19 bsrli */ 1, /* 0x1A rsv */ 0, /* 0x1B get */ 1, - /* 0x1C rsv */ 0, /* 0x1D rsv */ 0, /* 0x1E rsv */ 0, /* 0x1F rsv */ 0, - /* 0x20 or */ 1, /* 0x21 and */ 1, /* 0x22 xor */ 1, /* 0x23 andn */ 1, - /* 0x24 sra */ 1, /* 0x25 mfs */ 2, /* 0x26 br */ 1, /* 0x27 beq */ 0, - /* 0x28 ori */ 1, /* 0x29 andi */ 1, /* 0x2A xori */ 1, /* 0x2B andni */ 1, - /* 0x2C imm */ 0, /* 0x2D rtsd */ 0, /* 0x2E bri */ 1, /* 0x2F beqi */ 0, - /* 0x30 lbu */ 2, /* 0x31 lhu */ 2, /* 0x32 lw */ 2, /* 0x33 rsv */ 0, - /* 0x34 sb */ 0, /* 0x35 sh */ 0, /* 0x36 sw */ 0, /* 0x37 rsv */ 0, - /* 0x38 lbui */ 2, /* 0x39 lhui */ 2, /* 0x3A lwi */ 2, /* 0x3B rsv */ 0, - /* 0x3C sbi */ 0, /* 0x3D shi */ 0, /* 0x3E swi */ 0, /* 0x3F rsv */ 0, + 1, 1, 1, 1, 1, 1, 1, 1, /* 0x00–0x07 */ + 1, 1, 1, 1, 1, 1, 1, 1, /* 0x08–0x0F */ + 1, 1,34, 1, 0, 0, 4, 0, /* 0x10–0x17 mul=1, idiv=34, fpu=4(base, overridden) */ + 1, 1, 0, 1, 0, 0, 0, 0, /* 0x18–0x1F muli=1 */ + 1, 1, 1, 1, 1, 2, 1, 0, /* 0x20–0x27 mts/mfs=2, beq=0 */ + 1, 1, 1, 1, 0, 0, 1, 0, /* 0x28–0x2F imm=0, rtsd=0, bri writes LR=1 */ + 2, 2, 2, 0, 0, 0, 0, 0, /* 0x30–0x37 loads=2, stores=0 */ + 2, 2, 2, 0, 0, 0, 0, 0, /* 0x38–0x3F loads=2, stores=0 */ }; /* - * For these opcodes, bits[25:21] encode a SOURCE register rather than (or in - * addition to) the usual Rd destination field: - * - * 0x27 beq-bge : bits[25:21] = Ra (condition register) - * 0x2F beqi-bgei : bits[25:21] = Ra (condition register) - * 0x34 sb : bits[25:21] = rD (data to store) - * 0x35 sh : bits[25:21] = rD (data to store) - * 0x36 sw : bits[25:21] = rD (data to store) - * 0x3C sbi : bits[25:21] = rD (data to store, imm addressing) - * 0x3D shi : bits[25:21] = rD (data to store, imm addressing) - * 0x3E swi : bits[25:21] = rD (data to store, imm addressing) - * - * The stall model checks reg_ready for these bits[25:21] as an extra source. + * Opcodes where bits[25:21] are a SOURCE register, not the destination. + * Store rD = data register; branch 0x27/0x2F rD field = condition register. */ static const bool mb_rd_is_src[64] = { [0x27] = true, [0x2F] = true, @@ -167,69 +167,113 @@ static const bool mb_rd_is_src[64] = { [0x3C] = true, [0x3D] = true, [0x3E] = true, }; -/* Human-readable class label for each opcode slot (verbose output). */ -static const char *const mb_class_name[64] = { - "add", "rsub", "addc", "rsubc", - "addk", "rsubk/cmp","addkc", "rsubkc", - "addi", "rsubi", "addic", "rsubic", - "addik", "rsubik", "addikc", "rsubikc", - "mul", "bsrl/a/l", "idiv", "getd/putd", - "rsv", "rsv", "fpu", "rsv", - "muli", "bsrli/ai", "rsv", "get/put", - "rsv", "rsv", "rsv", "rsv", - "or", "and", "xor", "andn", - "sra/ext", "mts/mfs", "br/brld", "beq-bge", - "ori", "andi", "xori", "andni", - "imm", "rtsd", "bri/braid","beqi-bgei", - "lbu", "lhu", "lw", "rsv", - "sb", "sh", "sw", "rsv", - "lbui", "lhui", "lwi", "rsv", - "sbi", "shi", "swi", "rsv", +/* Human-readable opcode names for verbose output. */ +static const char *const mb_op_name[64] = { + "add", "rsub", "addc", "rsubc", "addk", "rsubk/cmp", "addkc", "rsubkc", + "addi", "rsubi", "addic", "rsubic", "addik","rsubik","addikc","rsubikc", + "mul","bsrl/a","idiv","getd","rsv","rsv","fpu","rsv", + "muli","bsrli","rsv","get","rsv","rsv","rsv","rsv", + "or","and","xor","andn","sra/ext","mts/mfs","br/brld","beq-bge", + "ori","andi","xori","andni","imm","rtsd","bri/braid","beqi-bgei", + "lbu","lhu","lw","rsv","sb","sh","sw","rsv", + "lbui","lhui","lwi","rsv","sbi","shi","swi","rsv", +}; + +/* ========================================================================= + * Instruction class + * ========================================================================= */ +typedef enum { + CLS_ALU, CLS_MUL, CLS_DIV, CLS_FPU, + CLS_LOAD, CLS_STORE, CLS_BRANCH, CLS_MTS_MFS, CLS_NOP, + CLS_OTHER +} MBClass; + +static const uint8_t mb_class[64] = { + /* 0x00–0x0F: integer ALU */ + CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU, + CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU, + /* 0x10 mul */ CLS_MUL, /* 0x11 bsrl */ CLS_ALU, + /* 0x12 idiv */ CLS_DIV, /* 0x13 getd */ CLS_OTHER, + /* 0x14 rsv */ CLS_OTHER, /* 0x15 rsv */ CLS_OTHER, + /* 0x16 fpu */ CLS_FPU, /* 0x17 rsv */ CLS_OTHER, + /* 0x18 muli */ CLS_MUL, /* 0x19 bsrli */ CLS_ALU, + CLS_OTHER,CLS_OTHER,CLS_OTHER,CLS_OTHER,CLS_OTHER,CLS_OTHER, + /* 0x20–0x27 */ + CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_MTS_MFS,CLS_BRANCH,CLS_BRANCH, + /* 0x28–0x2F */ + CLS_ALU,CLS_ALU,CLS_ALU,CLS_ALU,CLS_OTHER,CLS_BRANCH,CLS_BRANCH,CLS_BRANCH, + /* 0x30–0x37 */ + CLS_LOAD,CLS_LOAD,CLS_LOAD,CLS_OTHER,CLS_STORE,CLS_STORE,CLS_STORE,CLS_OTHER, + /* 0x38–0x3F */ + CLS_LOAD,CLS_LOAD,CLS_LOAD,CLS_OTHER,CLS_STORE,CLS_STORE,CLS_STORE,CLS_OTHER, }; -/* ---- Counters (single-core MicroBlaze; not thread-safe for SMP) ---- */ -static uint64_t total_cycles; -static uint64_t total_insns; -static uint64_t total_stall_cycles; -static uint64_t opcode_cycles[64]; -static uint64_t opcode_count[64]; +/* ========================================================================= + * Counter block — used for totals, snapshot, and bench delta + * ========================================================================= */ +typedef struct { + uint64_t insns; + uint64_t wcycles; /* weighted cycles (no stalls) */ + uint64_t scycles; /* RAW stall cycles (stalls=on) */ + /* per-class */ + uint64_t alu, mul, div_i, fpu, load, store, branch, mts_mfs, nop; + /* delay slot quality */ + uint64_t delayed_branch; + uint64_t delay_slot_nop; + uint64_t delay_slot_filled; + /* per-opcode */ + uint64_t op_count[64]; + uint64_t op_cycles[64]; +} MBCounts; + +static MBCounts total; /* always-accumulating totals */ +static MBCounts snap; /* snapshot at BENCH_RESTART */ +static MBCounts bench; /* delta (total - snap) */ + static uint64_t trans_count; -static bool verbose; -static bool model_stalls; -static bool big_endian_target; +/* Pointer table: opcode → class counter in 'total'. Built at install time. */ +static uint64_t *op_class_ctr[64]; -/* ---- Stall model state ---- */ -static uint64_t reg_ready[32]; /* cycle when each GPR result is forwardable */ -static uint64_t pipeline_cycle; /* logical pipeline cycle counter */ +/* ========================================================================= + * Stall model state + * ========================================================================= */ +static uint64_t reg_ready[32]; +static uint64_t pipeline_cycle; -/* - * Extract the 6-bit opcode from the MSB byte of a MicroBlaze instruction. - * Works for both endian variants (see insn_bytes() comment). - */ -static inline uint32_t insn_opcode(const uint8_t *data) -{ - return (big_endian_target ? data[0] : data[3]) >> 2; -} +/* ========================================================================= + * Plugin configuration + * ========================================================================= */ +static bool model_stalls; +static bool verbose; +static bool big_endian_target; -/* - * Reconstruct the full 32-bit instruction word from guest memory bytes. - * MicroBlaze bit-31 = MSB. In little-endian ELF (microblazeel): byte[3] is MSB. - */ +static char bench_name[256] = "unknown"; +static char compiler_name[256] = "unknown"; +static char opt_flags[512] = ""; +static char target_config[256] = ""; +static char commit_hash[128] = ""; + +/* ========================================================================= + * Measurement state machine + * ========================================================================= */ +typedef enum { ST_IDLE, ST_MEASURING, ST_STOPPED } MBState; +static MBState state = ST_IDLE; +static bool bench_event_seen; +static uint64_t bench_seq; /* incremented on each EXIT_PASS or EXIT_FAIL */ + +/* ========================================================================= + * Helpers + * ========================================================================= */ static inline uint32_t insn_word32(const uint8_t *data) { if (big_endian_target) return ((uint32_t)data[0] << 24) | ((uint32_t)data[1] << 16) - | ((uint32_t)data[2] << 8) | (uint32_t)data[3]; - else - return ((uint32_t)data[3] << 24) | ((uint32_t)data[2] << 16) - | ((uint32_t)data[1] << 8) | (uint32_t)data[0]; + | ((uint32_t)data[2] << 8) | (uint32_t)data[3]; + return ((uint32_t)data[3] << 24) | ((uint32_t)data[2] << 16) + | ((uint32_t)data[1] << 8) | (uint32_t)data[0]; } -/* - * Return host-virtual pointer to the 4 bytes of a guest instruction. - * qemu_plugin_insn_haddr() is always valid for RAM-backed pages. - */ static inline const uint8_t *insn_bytes(const struct qemu_plugin_insn *insn) { if (qemu_plugin_insn_size(insn) >= 4) @@ -237,72 +281,208 @@ static inline const uint8_t *insn_bytes(const struct qemu_plugin_insn *insn) return (const uint8_t *)qemu_plugin_insn_haddr(insn); } -/* Return the number of stall cycles register R would impose right now. */ +static inline bool is_bench_magic(uint32_t w) +{ + return (w & MB_ORI_R0_MASK) == MB_ORI_R0_BASE + && (w & MB_MAGIC_MASK) == MB_MAGIC_BYTE; +} + +/* Return true if this is a delayed-branch (D-form) opcode. */ +static inline bool is_delayed_branch(uint32_t w) +{ + uint8_t op = (w >> 26) & 0x3F; + switch (op) { + case 0x26: return (w >> 8) & 1; /* brd/brad/brld/brald */ + case 0x27: return (w >> 20) & 1; /* beqd-bged */ + case 0x2D: return true; /* rtsd/rtid/rtbd/rted always have delay slot */ + case 0x2E: return (w >> 8) & 1; /* brid/braid/brlid/bralid */ + case 0x2F: return (w >> 20) & 1; /* beqid-bgeid */ + default: return false; + } +} + static inline uint64_t stall_for(uint8_t r, uint64_t now) { return (r && reg_ready[r] > now) ? reg_ready[r] - now : 0; } +/* ========================================================================= + * Bench state machine operations + * ========================================================================= */ +static void do_restart(void) +{ + memcpy(&snap, &total, sizeof(total)); + state = ST_MEASURING; + bench_event_seen = true; +} + +static void do_stop(void) +{ + if (state != ST_MEASURING) + return; + state = ST_STOPPED; + +#define DELTA(f) bench.f = total.f - snap.f + DELTA(insns); DELTA(wcycles); DELTA(scycles); + DELTA(alu); DELTA(mul); DELTA(div_i); DELTA(fpu); + DELTA(load); DELTA(store); DELTA(branch); DELTA(mts_mfs); DELTA(nop); + DELTA(delayed_branch); DELTA(delay_slot_nop); DELTA(delay_slot_filled); + for (int i = 0; i < 64; i++) { + bench.op_count[i] = total.op_count[i] - snap.op_count[i]; + bench.op_cycles[i] = total.op_cycles[i] - snap.op_cycles[i]; + } +#undef DELTA +} + +static void emit_json(const MBCounts *c, const char *status) +{ + uint64_t raw_cycles = c->wcycles + c->scycles; + double cpi_w = c->insns ? (double)c->wcycles / c->insns : 0.0; + double cpi_r = c->insns ? (double)raw_cycles / c->insns : 0.0; + double fill = c->delayed_branch + ? (double)c->delay_slot_filled / c->delayed_branch : 0.0; + + /* Emit in ~4 KB chunks via a local buffer; avoids heap allocation. */ + char buf[8192]; + int n = 0; + +#define A(fmt, ...) n += snprintf(buf + n, (int)sizeof(buf) - n, fmt, ##__VA_ARGS__) + + A("{\n"); + A(" \"benchmark\": \"%s\",\n", bench_name); + A(" \"compiler\": \"%s\",\n", compiler_name); + A(" \"opt\": \"%s\",\n", opt_flags); + A(" \"target_config\": \"%s\",\n", target_config); + A(" \"compiler_commit\": \"%s\",\n", commit_hash); + A(" \"status\": \"%s\",\n", status); + A(" \"stall_model\": \"%s\",\n", model_stalls ? "raw-stalls" : "weighted"); + A(" \"instructions\": %" PRIu64 ",\n", c->insns); + A(" \"weighted_cycles\": %" PRIu64 ",\n", c->wcycles); + A(" \"raw_stall_cycles\": %" PRIu64 ",\n", c->scycles); + A(" \"raw_cycles\": %" PRIu64 ",\n", raw_cycles); + A(" \"cpi_weighted\": %.3f,\n", cpi_w); + A(" \"cpi_raw\": %.3f,\n", cpi_r); + A(" \"alu_count\": %" PRIu64 ",\n", c->alu); + A(" \"load_count\": %" PRIu64 ",\n", c->load); + A(" \"store_count\": %" PRIu64 ",\n", c->store); + A(" \"branch_count\": %" PRIu64 ",\n", c->branch); + A(" \"nop_count\": %" PRIu64 ",\n", c->nop); + A(" \"div_count\": %" PRIu64 ",\n", c->div_i); + A(" \"mul_count\": %" PRIu64 ",\n", c->mul); + A(" \"fpu_count\": %" PRIu64 ",\n", c->fpu); + A(" \"mts_mfs_count\": %" PRIu64 ",\n", c->mts_mfs); + A(" \"delayed_branch_count\": %" PRIu64 ",\n", c->delayed_branch); + A(" \"delay_slot_nop_count\": %" PRIu64 ",\n", c->delay_slot_nop); + A(" \"delay_slot_filled_count\": %" PRIu64 ",\n", c->delay_slot_filled); + A(" \"delay_slot_fill_rate\": %.3f,\n", fill); + A(" \"bench_seq\": %" PRIu64, bench_seq); + + if (verbose) { + A(",\n \"opcode_breakdown\": [\n"); + bool first = true; + for (int op = 0; op < 64; op++) { + if (!c->op_count[op]) continue; + if (!first) A(",\n"); + first = false; + A(" {\"op\": \"0x%02x\", \"name\": \"%s\", \"count\": %" PRIu64 + ", \"cycles\": %" PRIu64 ", \"cpi\": %.2f}", + op, mb_op_name[op], c->op_count[op], c->op_cycles[op], + (double)c->op_cycles[op] / c->op_count[op]); + } + A("\n ]"); + } + + A("\n}\n"); +#undef A + qemu_plugin_outs(buf); +} + +/* ========================================================================= + * Execution callbacks + * ========================================================================= */ + +/* + * Bench event callback — registered on magic NOP instructions in both modes. + * Userdata: low byte = event nibble (MBEV_xxx & 0xFF). + */ +static void vcpu_insn_exec_bench(unsigned int vcpu_idx, void *userdata) +{ + uint8_t ev = (uint8_t)(uintptr_t)userdata; + switch (ev) { + case (MBEV_RESTART & 0xFF): do_restart(); break; + case (MBEV_STOP & 0xFF): do_stop(); break; + case (MBEV_PASS & 0xFF): + emit_json(&bench, "pass"); + bench_seq++; + state = ST_IDLE; + break; + case (MBEV_FAIL & 0xFF): + emit_json(&bench, "fail"); + bench_seq++; + state = ST_IDLE; + break; + } +} + /* - * Full callback for the RAW-stall model (stalls=on). + * RAW-stall callback for normal instructions (stalls=on). * - * Userdata is a packed 25-bit value: - * bits [ 5: 0] op (6-bit opcode) - * bits [10: 6] rd (bits[25:21] of instruction word) - * bits [15:11] ra (bits[20:16]) - * bits [20:16] rb (bits[15:11], meaningful only if !type_b) - * bit [21] type_b (1 = Type-B instruction, no Rb register field) - * bit [22] rd_is_src (1 = also check rd as a source, per mb_rd_is_src) - * bit [23] valid (1 = instruction bytes were available at trans time) + * Packed metadata in userdata (uintptr_t, 64-bit host): + * bits [ 5: 0] op 6-bit opcode + * bits [10: 6] rd bits[25:21] + * bits [15:11] ra bits[20:16] + * bits [20:16] rb bits[15:11] + * bit [21] type_b 1 = no Rb field + * bit [22] rd_is_src 1 = rd is also a source + * bit [23] valid 1 = instruction bytes were available + * bits [31:24] base_cyc pre-computed weighted cost + * bit [32] is_nop 1 = canonical NOP word */ static void vcpu_insn_exec_stall(unsigned int vcpu_idx, void *userdata) { - uintptr_t meta = (uintptr_t)userdata; - uint8_t op = (meta >> 0) & 0x3f; - uint8_t rd = (meta >> 6) & 0x1f; - uint8_t ra = (meta >> 11) & 0x1f; - uint8_t rb = (meta >> 16) & 0x1f; - bool type_b = (meta >> 21) & 1; - bool rd_is_src = (meta >> 22) & 1; - bool valid = (meta >> 23) & 1; - - uint64_t base = mb_cycle_table[op]; - uint64_t stall = 0; + uintptr_t meta = (uintptr_t)userdata; + uint8_t op = (meta >> 0) & 0x3F; + uint8_t rd = (meta >> 6) & 0x1F; + uint8_t ra = (meta >> 11) & 0x1F; + uint8_t rb = (meta >> 16) & 0x1F; + bool type_b = (meta >> 21) & 1; + bool rd_src = (meta >> 22) & 1; + bool valid = (meta >> 23) & 1; + uint64_t base = (meta >> 24) & 0xFF; + bool is_nop = (meta >> 32) & 1; + uint64_t stall = 0; if (valid) { uint64_t now = pipeline_cycle; uint64_t s; - - s = stall_for(ra, now); - if (s > stall) stall = s; - - if (!type_b) { - s = stall_for(rb, now); - if (s > stall) stall = s; - } - - if (rd_is_src) { - s = stall_for(rd, now); - if (s > stall) stall = s; - } + s = stall_for(ra, now); if (s > stall) stall = s; + if (!type_b) { s = stall_for(rb, now); if (s > stall) stall = s; } + if (rd_src) { s = stall_for(rd, now); if (s > stall) stall = s; } } - uint64_t issue = pipeline_cycle + stall; - pipeline_cycle = issue + base; + uint64_t issue = pipeline_cycle + stall; + pipeline_cycle = issue + base; - /* Record when this instruction's GPR result will be forwardable. */ - uint8_t lat = mb_result_latency[op]; + uint8_t lat = (op == 0x16) ? (uint8_t)base : mb_result_latency[op]; if (lat && rd) reg_ready[rd] = issue + lat; - uint64_t cost = base + stall; - total_cycles += cost; - total_stall_cycles += stall; - total_insns++; - opcode_count[op]++; - opcode_cycles[op] += cost; + total.insns++; + total.wcycles += base; + total.scycles += stall; + total.op_count[op]++; + total.op_cycles[op] += base + stall; + + if (is_nop) { + total.nop++; + } else if (op_class_ctr[op]) { + (*op_class_ctr[op])++; + } } +/* ========================================================================= + * TB translation callback + * ========================================================================= */ static void vcpu_tb_trans(qemu_plugin_id_t id, struct qemu_plugin_tb *tb) { trans_count++; @@ -312,153 +492,179 @@ static void vcpu_tb_trans(qemu_plugin_id_t id, struct qemu_plugin_tb *tb) struct qemu_plugin_insn *insn = qemu_plugin_tb_get_insn(tb, i); const uint8_t *data = insn_bytes(insn); - if (model_stalls) { - /* --- RAW-stall mode: full callback with packed instruction metadata --- */ - uint8_t op = 63, rd = 0, ra = 0, rb = 0; - bool type_b = false, rd_is_src = false, valid = false; - - if (data) { - uint32_t w = insn_word32(data); - op = (w >> 26) & 0x3f; - rd = (w >> 21) & 0x1f; - ra = (w >> 16) & 0x1f; - rb = (w >> 11) & 0x1f; - type_b = (op & 0x08) != 0; - rd_is_src = mb_rd_is_src[op]; - valid = true; + if (!data) { + /* No bytes available: 1-cycle unknown. */ + if (model_stalls) { + uintptr_t meta = (0x3Fu) | (1u << 23) | ((uintptr_t)1u << 24); + qemu_plugin_register_vcpu_insn_exec_cb( + insn, vcpu_insn_exec_stall, + QEMU_PLUGIN_CB_NO_REGS, (void *)meta); + } else { + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.insns, 1); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.wcycles, 1); } + continue; + } + + uint32_t w = insn_word32(data); + uint8_t op = (w >> 26) & 0x3F; + + /* ---- Bench control magic NOPs ---- */ + if (is_bench_magic(w)) { + /* Magic instructions are hardware NOPs: zero pipeline cost, + * not counted in insns/cycles. Only the event fires. */ + qemu_plugin_register_vcpu_insn_exec_cb( + insn, vcpu_insn_exec_bench, + QEMU_PLUGIN_CB_NO_REGS, (void *)(uintptr_t)(w & 0xFF)); + continue; + } + + /* ---- Delay slot quality (look-ahead at next insn in same TB) ---- */ + if (is_delayed_branch(w) && i + 1 < n) { + struct qemu_plugin_insn *next = qemu_plugin_tb_get_insn(tb, i + 1); + const uint8_t *nd = insn_bytes(next); + bool slot_nop = nd && (insn_word32(nd) == MB_NOP_WORD); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.delayed_branch, 1); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, + slot_nop ? &total.delay_slot_nop : &total.delay_slot_filled, 1); + } + + bool is_nop = (w == MB_NOP_WORD); + + /* ---- Per-instruction counters ---- */ + if (model_stalls) { + uint8_t rd = (w >> 21) & 0x1F; + uint8_t ra = (w >> 16) & 0x1F; + uint8_t rb = (w >> 11) & 0x1F; + bool type_b = (op & 0x08) != 0; + bool rd_src = mb_rd_is_src[op]; + + uint8_t base_cyc; + if (op == 0x16) + base_cyc = mb_fpu_lat(w); + else if (op == 0x24 && (w & 8u)) + base_cyc = 2; /* WIC: func bit[3]=1 → 2 cycles */ + else + base_cyc = mb_weight[op]; uintptr_t meta = (uintptr_t)op - | ((uintptr_t)rd << 6) - | ((uintptr_t)ra << 11) - | ((uintptr_t)rb << 16) - | ((uintptr_t)type_b << 21) - | ((uintptr_t)rd_is_src << 22) - | ((uintptr_t)valid << 23); + | ((uintptr_t)rd << 6) + | ((uintptr_t)ra << 11) + | ((uintptr_t)rb << 16) + | ((uintptr_t)type_b << 21) + | ((uintptr_t)rd_src << 22) + | ((uintptr_t)1u << 23) /* valid */ + | ((uintptr_t)base_cyc << 24) + | ((uintptr_t)is_nop << 32); qemu_plugin_register_vcpu_insn_exec_cb( insn, vcpu_insn_exec_stall, QEMU_PLUGIN_CB_NO_REGS, (void *)meta); } else { - /* --- Throughput mode: four inline atomic counters per instruction --- */ - uint8_t op; - uint64_t cycles; - if (data) { - op = insn_opcode(data); - cycles = mb_cycle_table[op]; - } else { - op = 63; - cycles = 1; - } + /* Weighted inline mode. */ + uint64_t weight; + if (op == 0x16) + weight = mb_fpu_lat(w); + else if (op == 0x24 && (w & 8u)) + weight = 2; + else + weight = mb_weight[op]; qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total_insns, 1); + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.insns, 1); qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total_cycles, cycles); + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.wcycles, weight); qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &opcode_count[op], 1); + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.op_count[op], 1); qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &opcode_cycles[op],cycles); + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.op_cycles[op], weight); + + /* Class counter. */ + uint64_t *cls = is_nop ? &total.nop : op_class_ctr[op]; + if (cls) + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, cls, 1); } } } +/* ========================================================================= + * Plugin exit + * ========================================================================= */ static void plugin_exit(qemu_plugin_id_t id, void *userdata) { - char buf[320]; - const char *mode = model_stalls - ? "(C_AREA_OPTIMIZED=0, branch-predicted, RAW stalls modeled)" - : "(C_AREA_OPTIMIZED=0, branch-predicted, throughput only)"; - - if (model_stalls) { - snprintf(buf, sizeof(buf), - "MicroBlaze cycle model %s\n" - "Total instructions : %" PRIu64 "\n" - "Total cycles : %" PRIu64 "\n" - "Stall cycles : %" PRIu64 "\n" - "CPI : %.3f\n" - "Stall CPI : %.3f\n" - "TB translations : %" PRIu64 "\n", - mode, - total_insns, total_cycles, total_stall_cycles, - total_insns ? (double)total_cycles / total_insns : 0.0, - total_insns ? (double)total_stall_cycles / total_insns : 0.0, - trans_count); - } else { - snprintf(buf, sizeof(buf), - "MicroBlaze cycle model %s\n" - "Total instructions : %" PRIu64 "\n" - "Total cycles : %" PRIu64 "\n" - "CPI : %.3f\n" - "TB translations : %" PRIu64 "\n", - mode, - total_insns, total_cycles, - total_insns ? (double)total_cycles / total_insns : 0.0, - trans_count); - } - qemu_plugin_outs(buf); - - if (!verbose) + if (!bench_event_seen) { + /* Legacy fallback: no bench events — emit totals for full run. */ + emit_json(&total, "no_bench_events"); return; - - qemu_plugin_outs("Opcode class breakdown:\n" - " Op Class Insns Cycles CPI\n" - " ---- --------------- ------------ -------- ----\n"); - - for (int op = 0; op < 64; op++) { - if (!opcode_count[op]) - continue; - snprintf(buf, sizeof(buf), - " 0x%02x %-15s %12" PRIu64 " %8" PRIu64 " %.2f\n", - op, mb_class_name[op], - opcode_count[op], opcode_cycles[op], - (double)opcode_cycles[op] / opcode_count[op]); - qemu_plugin_outs(buf); } + if (state == ST_STOPPED) + emit_json(&bench, "incomplete"); + /* If state == ST_IDLE, BENCH_EXIT_PASS/FAIL already emitted the record. */ } +/* ========================================================================= + * Plugin install + * ========================================================================= */ QEMU_PLUGIN_EXPORT int qemu_plugin_install(qemu_plugin_id_t id, const qemu_info_t *info, int argc, char **argv) { for (int i = 0; i < argc; i++) { char *opt = argv[i]; - g_auto(GStrv) tokens = g_strsplit(opt, "=", 2); - const char *key = tokens[0]; - const char *val = tokens[1] ? tokens[1] : ""; - - if (!g_strcmp0(key, "verbose")) { - if (!g_strcmp0(val,"on")||!g_strcmp0(val,"yes")|| - !g_strcmp0(val,"true")||!g_strcmp0(val,"1")) - verbose = true; - else if (!g_strcmp0(val,"off")||!g_strcmp0(val,"no")|| - !g_strcmp0(val,"false")||!g_strcmp0(val,"0")) - verbose = false; - else { - fprintf(stderr, "mb_cycles: bad value for verbose: %s\n", opt); - return -1; - } - } else if (!g_strcmp0(key, "stalls")) { - if (!g_strcmp0(val,"on")||!g_strcmp0(val,"yes")|| - !g_strcmp0(val,"true")||!g_strcmp0(val,"1")) - model_stalls = true; - else if (!g_strcmp0(val,"off")||!g_strcmp0(val,"no")|| - !g_strcmp0(val,"false")||!g_strcmp0(val,"0")) - model_stalls = false; - else { - fprintf(stderr, "mb_cycles: bad value for stalls: %s\n", opt); - return -1; - } - } else { - fprintf(stderr, "mb_cycles: unknown option '%s'\n", opt); - return -1; + g_auto(GStrv) tok = g_strsplit(opt, "=", 2); + const char *k = tok[0]; + const char *v = tok[1] ? tok[1] : ""; + +#define BOOL_ARG(flag, field) \ + if (!g_strcmp0(k, flag)) { \ + field = !g_strcmp0(v,"on") || !g_strcmp0(v,"yes") || \ + !g_strcmp0(v,"true") || !g_strcmp0(v,"1"); \ + continue; \ } + BOOL_ARG("stalls", model_stalls) + BOOL_ARG("verbose", verbose) +#undef BOOL_ARG + + if (!g_strcmp0(k, "bench")) { snprintf(bench_name, sizeof(bench_name), "%s", v); continue; } + if (!g_strcmp0(k, "compiler")) { snprintf(compiler_name, sizeof(compiler_name), "%s", v); continue; } + if (!g_strcmp0(k, "opt")) { snprintf(opt_flags, sizeof(opt_flags), "%s", v); continue; } + if (!g_strcmp0(k, "target")) { snprintf(target_config, sizeof(target_config), "%s", v); continue; } + if (!g_strcmp0(k, "commit")) { snprintf(commit_hash, sizeof(commit_hash), "%s", v); continue; } + if (!g_strcmp0(k, "area_opt")) { + fpu_area_opt = (unsigned)atoi(v); + if (fpu_area_opt > 2) { fprintf(stderr, "mb_cycles: area_opt must be 0-2\n"); return -1; } + continue; + } + fprintf(stderr, "mb_cycles: unknown option '%s'\n", opt); + return -1; } big_endian_target = info->target_name && strcmp(info->target_name, "microblaze") == 0; + /* Build opcode → class counter pointer table. */ + for (int op = 0; op < 64; op++) { + switch ((MBClass)mb_class[op]) { + case CLS_ALU: op_class_ctr[op] = &total.alu; break; + case CLS_MUL: op_class_ctr[op] = &total.mul; break; + case CLS_DIV: op_class_ctr[op] = &total.div_i; break; + case CLS_FPU: op_class_ctr[op] = &total.fpu; break; + case CLS_LOAD: op_class_ctr[op] = &total.load; break; + case CLS_STORE: op_class_ctr[op] = &total.store; break; + case CLS_BRANCH: op_class_ctr[op] = &total.branch; break; + case CLS_MTS_MFS: op_class_ctr[op] = &total.mts_mfs; break; + case CLS_NOP: + case CLS_OTHER: + default: op_class_ctr[op] = NULL; break; + } + } + qemu_plugin_register_vcpu_tb_trans_cb(id, vcpu_tb_trans); qemu_plugin_register_atexit_cb(id, plugin_exit, NULL); return 0; diff --git a/hw/microblaze/petalogix_s3adsp1800_mmu.c b/hw/microblaze/petalogix_s3adsp1800_mmu.c index a5efb425a07..b7279fde740 100644 --- a/hw/microblaze/petalogix_s3adsp1800_mmu.c +++ b/hw/microblaze/petalogix_s3adsp1800_mmu.c @@ -31,6 +31,8 @@ #include "net/net.h" #include "hw/block/flash.h" #include "sysemu/sysemu.h" +#include "sysemu/runstate.h" +#include "qemu/timer.h" #include "hw/boards.h" #include "hw/misc/unimp.h" #include "exec/address-spaces.h" @@ -38,6 +40,47 @@ #include "boot.h" +/* Exit-and-counter device at 0xFF000000 (4 KB region): + * + * 0xFF000000 W: write 0 → clean shutdown, nonzero → panic exit + * 0xFF000004 R: low 32 bits of QEMU virtual clock (ns; = icount with -icount 0) + * 0xFF000008 R: high 32 bits of QEMU virtual clock (ns) + * + * The bare-metal runtime reads the counter at program start and end and + * prints the delta so GCC vs Clang instruction counts can be compared. + * With -icount 0 each instruction advances the clock by 1 ns, so the + * delta equals the instruction count exactly. + */ +#define MB_EXIT_BASEADDR 0xFF000000 + +static uint64_t mb_exit_read(void *opaque, hwaddr addr, unsigned int size) +{ + int64_t ns = qemu_clock_get_ns(QEMU_CLOCK_VIRTUAL); + switch (addr) { + case 4: return (uint64_t)(uint32_t)(ns); /* lo */ + case 8: return (uint64_t)(uint32_t)(ns >> 32); /* hi */ + default: return 0; + } +} + +static void mb_exit_write(void *opaque, hwaddr addr, + uint64_t val, unsigned int size) +{ + if (val == 0) { + qemu_system_shutdown_request_with_code(SHUTDOWN_CAUSE_GUEST_SHUTDOWN, 0); + } else { + qemu_system_shutdown_request_with_code(SHUTDOWN_CAUSE_GUEST_PANIC, + (int)(val & 0xFF)); + } +} + +static const MemoryRegionOps mb_exit_ops = { + .read = mb_exit_read, + .write = mb_exit_write, + .endianness = DEVICE_LITTLE_ENDIAN, + .valid = { .min_access_size = 4, .max_access_size = 4 }, +}; + #define LMB_BRAM_SIZE (128 * KiB) #define FLASH_SIZE (16 * MiB) @@ -125,6 +168,12 @@ petalogix_s3adsp1800_init(MachineState *machine) create_unimplemented_device("gpio", GPIO_BASEADDR, 0x10000); + /* Exit register: firmware writes here to terminate simulation early. */ + MemoryRegion *exit_mr = g_new(MemoryRegion, 1); + memory_region_init_io(exit_mr, NULL, &mb_exit_ops, NULL, + "mb-exit", 0x1000); + memory_region_add_subregion(sysmem, MB_EXIT_BASEADDR, exit_mr); + microblaze_load_kernel(cpu, ddr_base, ram_size, machine->initrd_filename, BINARY_DEVICE_TREE_FILE, From 8cdbd1789ad18b4d6962f82f2eb98ce7f5152a80 Mon Sep 17 00:00:00 2001 From: Samuel Price Date: Sun, 28 Jun 2026 18:20:36 -0400 Subject: [PATCH 3/6] contrib/plugins/mb_cycles: add branch penalty model and fix D-bit detection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Branch taken/not-taken penalties (per UG984 §5): - Non-D conditional (bgti, bnei, …): 1 cycle not-taken, 3 cycles taken. Implemented via retroactive accounting: save fallthrough_pc at the end of each TB; if the next TB starts elsewhere, add 2 extra wcycles. - Non-D unconditional (bri, bra, …): always 3 cycles. - D-form (bgtid, bneid, rtsd, …): 1 cycle; delay slot counted separately at its own cost. No branch-taken penalty (delay slot absorbs it). Fix is_delayed_branch() D-bit detection: - 0x26 (br/brd): D was at bit 8 (wrong), now bit 20 (Ra field MSB). - 0x27 (beq/beqd): D was at bit 20 (wrong), now bit 25 (rD field MSB). - 0x2E (bri/brid): D was at bit 8 (wrong), now bit 20. - 0x2F (beqi/beqid): D was at bit 20 (wrong), now bit 25. Also add: - Cross-TB load-use stall detection for loads at TB boundaries. - delay_slot_filled_count / delay_slot_nop_count metrics per benchmark. - stalls=on mode tracks pipeline_cycle and reg_ready per-register. Co-Authored-By: Claude Sonnet 4.6 --- contrib/plugins/mb_cycles.c | 350 ++++++++++++++++++++++-------------- 1 file changed, 217 insertions(+), 133 deletions(-) diff --git a/contrib/plugins/mb_cycles.c b/contrib/plugins/mb_cycles.c index 23322c0b9f5..584464f735e 100644 --- a/contrib/plugins/mb_cycles.c +++ b/contrib/plugins/mb_cycles.c @@ -70,7 +70,10 @@ * NOT MODELLED * ------------ * I-cache, D-cache, DDR/BRAM wait states, PLB/AXI bus, branch prediction, - * taken/not-taken penalties, interrupts, DMA, timer effects. + * interrupts, DMA, timer effects. + * Branch taken/not-taken penalties ARE modelled: non-D conditional = 1 cycle + * (not taken) or 3 cycles (taken); non-D unconditional = 3 cycles always; + * D-form = 1 cycle (delay slot counted separately at its own cost). */ #include @@ -208,6 +211,7 @@ static const uint8_t mb_class[64] = { CLS_LOAD,CLS_LOAD,CLS_LOAD,CLS_OTHER,CLS_STORE,CLS_STORE,CLS_STORE,CLS_OTHER, }; + /* ========================================================================= * Counter block — used for totals, snapshot, and bench delta * ========================================================================= */ @@ -236,15 +240,35 @@ static uint64_t trans_count; static uint64_t *op_class_ctr[64]; /* ========================================================================= - * Stall model state + * Cross-TB stall state + * ========================================================================= + * If the last instruction of TB[N] is a load/mfs (result latency 2), and + * the first instruction of TB[N+1] reads the written register, there is a + * 1-cycle load-use stall. We cannot detect this at translation time since + * we do not know which TB follows which. We carry the loaded register Rd + * here; vcpu_tb_exec_branch checks it against first_insn_reads at the start + * of each TB execution. * ========================================================================= */ -static uint64_t reg_ready[32]; -static uint64_t pipeline_cycle; +static uint8_t g_cross_tb_stall_rd; /* 0 = no pending cross-TB stall */ + +/* ========================================================================= + * Retroactive branch-taken accounting + * ========================================================================= + * When a non-D conditional branch ends a TB we cannot know at translation + * time whether it will be taken. We save the fallthrough PC here; at the + * start of the NEXT TB's exec callback, if next_tb_vaddr != fallthrough_pc + * the branch was taken and we add 2 extra cycles to total.wcycles. + * ========================================================================= */ +typedef struct { + bool valid; + uint64_t fallthrough_pc; +} PendingBranch; + +static PendingBranch g_pending_branch; /* ========================================================================= * Plugin configuration * ========================================================================= */ -static bool model_stalls; static bool verbose; static bool big_endian_target; @@ -287,20 +311,104 @@ static inline bool is_bench_magic(uint32_t w) && (w & MB_MAGIC_MASK) == MB_MAGIC_BYTE; } -/* Return true if this is a delayed-branch (D-form) opcode. */ +/* + * Return true if this is a delayed-branch (D-form) opcode. + * + * Encoding verified against insns.decode: + * 0x26 (br/brd/brald): Ra-field bits[20:16] encode D/A/L; D = bit 20 + * 0x27 (beq/beqd etc.): Rd-field bits[25:21] encode cond+D; D = bit 25 (MSB of rD) + * 0x2D (rtsd etc.): always has a delay slot + * 0x2E (bri/brid etc.): same layout as 0x26; D = bit 20 + * 0x2F (beqi/beqid): same layout as 0x27; D = bit 25 (MSB of rD) + */ static inline bool is_delayed_branch(uint32_t w) { uint8_t op = (w >> 26) & 0x3F; switch (op) { - case 0x26: return (w >> 8) & 1; /* brd/brad/brld/brald */ - case 0x27: return (w >> 20) & 1; /* beqd-bged */ - case 0x2D: return true; /* rtsd/rtid/rtbd/rted always have delay slot */ - case 0x2E: return (w >> 8) & 1; /* brid/braid/brlid/bralid */ - case 0x2F: return (w >> 20) & 1; /* beqid-bgeid */ + case 0x26: return (w >> 20) & 1; /* brd/brad/brld/brald: D at bit 20 */ + case 0x27: return (w >> 25) & 1; /* beqd-bged: D at bit 25 (MSB of rD) */ + case 0x2D: return true; /* rtsd/rtid/rtbd/rted: always */ + case 0x2E: return (w >> 20) & 1; /* brid/braid/brlid/bralid: D at 20 */ + case 0x2F: return (w >> 25) & 1; /* beqid-bgeid: D at bit 25 (MSB of rD) */ default: return false; } } +/* Conditional branch opcodes: 0x27 (beq/bne/… register) and 0x2F (imm form). */ +static inline bool is_cond_branch_op(uint8_t op) +{ + return op == 0x27 || op == 0x2F; +} + +/* + * Static load-use stall helpers. + * + * Returns the GPR written by this instruction if it has a 2-cycle result + * latency (loads, mfs). Returns 0 if no hazard is possible (r0 is always + * 0 so writing it never stalls a consumer). + */ +static inline uint8_t stall_source_rd(uint32_t w) +{ + uint8_t op = (w >> 26) & 0x3F; + if (mb_result_latency[op] < 2) + return 0; + return (w >> 21) & 0x1F; /* Rd; 0 means r0, never a hazard */ +} + +/* + * Returns true if instruction w reads GPR r (r != 0). + * Checks Ra, Rb (Type-A only), and Rd-as-source (stores/conditional branches). + */ +static inline bool insn_reads_reg(uint32_t w, uint8_t r) +{ + if (!r) return false; + uint8_t op = (w >> 26) & 0x3F; + if (((w >> 16) & 0x1F) == r) return true; /* Ra */ + if (!(op & 8) && ((w >> 11) & 0x1F) == r) return true; /* Rb (Type-A) */ + if (mb_rd_is_src[op] && ((w >> 21) & 0x1F) == r) return true; /* Rd-src */ + return false; +} + +/* Bitmask of all GPRs read by instruction w (bits 1–31; bit 0 always clear). */ +static inline uint32_t insn_reads_mask(uint32_t w) +{ + uint32_t m = 0; + uint8_t op = (w >> 26) & 0x3F; + uint8_t ra = (w >> 16) & 0x1F; if (ra) m |= 1u << ra; + if (!(op & 8)) { uint8_t rb = (w >> 11) & 0x1F; if (rb) m |= 1u << rb; } + if (mb_rd_is_src[op]) { uint8_t rd = (w >> 21) & 0x1F; if (rd) m |= 1u << rd; } + return m; +} + +/* + * Static cycle cost for a branch instruction. + * + * Per UG984 §5 branch latency table: + * not taken: 1 cycle + * taken + D-bit set: 2 cycles (branch + delay slot, no flush) + * taken + D-bit not set: 3 cycles (branch + 2-cycle pipeline flush) + * + * D-form (D-bit set): cost = 1. Delay slot always executes regardless of + * taken/not-taken and is counted separately. Total = 1 + delay_slot_cost. + * + * Non-D unconditional (bri, bra, bral, brl and register forms): always + * taken → cost = 3 (1 branch + 2-cycle flush). + * + * Non-D conditional (bnei, beqi, blti, …): static cost = 1. An extra 2 + * cycles is credited retroactively when the next TB's start address does not + * equal this branch's fallthrough PC (i.e. it was taken). Not-taken → 1 + * cycle total; taken → 3 cycles total. Accounting is done in + * vcpu_tb_exec_branch (stalls=off) / vcpu_tb_exec_stall (stalls=on). + */ +static inline uint8_t mb_branch_cost(uint32_t w) +{ + if (is_delayed_branch(w)) + return 1; + if (is_cond_branch_op((w >> 26) & 0x3F)) + return 1; /* +2 added retroactively on taken path */ + return 3; /* unconditional non-D: always taken, 3-cycle flush */ +} + static inline uint64_t stall_for(uint8_t r, uint64_t now) { return (r && reg_ready[r] > now) ? reg_ready[r] - now : 0; @@ -424,107 +532,102 @@ static void vcpu_insn_exec_bench(unsigned int vcpu_idx, void *userdata) } } -/* - * RAW-stall callback for normal instructions (stalls=on). - * - * Packed metadata in userdata (uintptr_t, 64-bit host): - * bits [ 5: 0] op 6-bit opcode - * bits [10: 6] rd bits[25:21] - * bits [15:11] ra bits[20:16] - * bits [20:16] rb bits[15:11] - * bit [21] type_b 1 = no Rb field - * bit [22] rd_is_src 1 = rd is also a source - * bit [23] valid 1 = instruction bytes were available - * bits [31:24] base_cyc pre-computed weighted cost - * bit [32] is_nop 1 = canonical NOP word - */ -static void vcpu_insn_exec_stall(unsigned int vcpu_idx, void *userdata) +/* ========================================================================= + * TB-exec callback — branch accounting and cross-TB stall resolution + * ========================================================================= */ +typedef struct { + uint64_t tb_vaddr; + bool has_cond_branch; + uint64_t fallthrough_pc; + uint8_t last_stall_rd; /* Rd of last stall-source insn in TB (0 = none) */ + uint32_t first_insn_reads; /* bitmask of GPRs read by TB's first real insn */ +} TBBranchMeta; + +static void vcpu_tb_exec_branch(unsigned int vcpu_idx, void *userdata) { - uintptr_t meta = (uintptr_t)userdata; - uint8_t op = (meta >> 0) & 0x3F; - uint8_t rd = (meta >> 6) & 0x1F; - uint8_t ra = (meta >> 11) & 0x1F; - uint8_t rb = (meta >> 16) & 0x1F; - bool type_b = (meta >> 21) & 1; - bool rd_src = (meta >> 22) & 1; - bool valid = (meta >> 23) & 1; - uint64_t base = (meta >> 24) & 0xFF; - bool is_nop = (meta >> 32) & 1; - - uint64_t stall = 0; - if (valid) { - uint64_t now = pipeline_cycle; - uint64_t s; - s = stall_for(ra, now); if (s > stall) stall = s; - if (!type_b) { s = stall_for(rb, now); if (s > stall) stall = s; } - if (rd_src) { s = stall_for(rd, now); if (s > stall) stall = s; } + const TBBranchMeta *info = (const TBBranchMeta *)userdata; + + /* Cross-TB load-use stall: previous TB ended with a stall-source insn. */ + if (g_cross_tb_stall_rd && + (info->first_insn_reads >> g_cross_tb_stall_rd) & 1) + total.scycles += 1; + g_cross_tb_stall_rd = info->last_stall_rd; + + /* Retroactive branch-taken credit. */ + if (g_pending_branch.valid) { + if (info->tb_vaddr != g_pending_branch.fallthrough_pc) + total.wcycles += 2; /* taken: 1 static + 2 extra = 3 total */ + g_pending_branch.valid = false; } - uint64_t issue = pipeline_cycle + stall; - pipeline_cycle = issue + base; - - uint8_t lat = (op == 0x16) ? (uint8_t)base : mb_result_latency[op]; - if (lat && rd) - reg_ready[rd] = issue + lat; - - total.insns++; - total.wcycles += base; - total.scycles += stall; - total.op_count[op]++; - total.op_cycles[op] += base + stall; - - if (is_nop) { - total.nop++; - } else if (op_class_ctr[op]) { - (*op_class_ctr[op])++; + if (info->has_cond_branch) { + g_pending_branch.valid = true; + g_pending_branch.fallthrough_pc = info->fallthrough_pc; } } /* ========================================================================= * TB translation callback + * ========================================================================= + * One path only: inline counter updates per instruction. Load-use stalls + * are detected statically by looking at adjacent instruction pairs; the + * stall penalty is baked in as a constant inline add at translation time. + * Branch taken/not-taken and cross-TB stall resolution are handled by one + * lightweight C callback per TB (vcpu_tb_exec_branch). * ========================================================================= */ static void vcpu_tb_trans(qemu_plugin_id_t id, struct qemu_plugin_tb *tb) { trans_count++; size_t n = qemu_plugin_tb_n_insns(tb); + uint64_t last_cond_nond_vaddr = 0; + uint8_t prev_stall_rd = 0; /* Rd of previous stall-source insn */ + uint8_t last_stall_rd = 0; /* final value → cross-TB state */ + uint32_t first_insn_reads = 0; /* reads mask of first real insn */ + bool first_real_seen = false; + for (size_t i = 0; i < n; i++) { struct qemu_plugin_insn *insn = qemu_plugin_tb_get_insn(tb, i); const uint8_t *data = insn_bytes(insn); if (!data) { - /* No bytes available: 1-cycle unknown. */ - if (model_stalls) { - uintptr_t meta = (0x3Fu) | (1u << 23) | ((uintptr_t)1u << 24); - qemu_plugin_register_vcpu_insn_exec_cb( - insn, vcpu_insn_exec_stall, - QEMU_PLUGIN_CB_NO_REGS, (void *)meta); - } else { - qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.insns, 1); - qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.wcycles, 1); - } + prev_stall_rd = 0; + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.insns, 1); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.wcycles, 1); continue; } uint32_t w = insn_word32(data); uint8_t op = (w >> 26) & 0x3F; - /* ---- Bench control magic NOPs ---- */ if (is_bench_magic(w)) { - /* Magic instructions are hardware NOPs: zero pipeline cost, - * not counted in insns/cycles. Only the event fires. */ + prev_stall_rd = 0; /* magic NOP breaks the stall chain */ qemu_plugin_register_vcpu_insn_exec_cb( insn, vcpu_insn_exec_bench, QEMU_PLUGIN_CB_NO_REGS, (void *)(uintptr_t)(w & 0xFF)); continue; } - /* ---- Delay slot quality (look-ahead at next insn in same TB) ---- */ + /* Capture reads mask of the first real instruction for cross-TB stall. */ + if (!first_real_seen) { + first_insn_reads = insn_reads_mask(w); + first_real_seen = true; + } + + /* Static load-use stall: previous insn wrote a register we read. */ + if (prev_stall_rd && insn_reads_reg(w, prev_stall_rd)) + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.scycles, 1); + + /* Update stall-source state for next iteration. */ + prev_stall_rd = stall_source_rd(w); + last_stall_rd = prev_stall_rd; + if (is_delayed_branch(w) && i + 1 < n) { - struct qemu_plugin_insn *next = qemu_plugin_tb_get_insn(tb, i + 1); - const uint8_t *nd = insn_bytes(next); + struct qemu_plugin_insn *nxt = qemu_plugin_tb_get_insn(tb, i + 1); + const uint8_t *nd = insn_bytes(nxt); bool slot_nop = nd && (insn_word32(nd) == MB_NOP_WORD); qemu_plugin_register_vcpu_insn_exec_inline( insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.delayed_branch, 1); @@ -535,62 +638,43 @@ static void vcpu_tb_trans(qemu_plugin_id_t id, struct qemu_plugin_tb *tb) bool is_nop = (w == MB_NOP_WORD); - /* ---- Per-instruction counters ---- */ - if (model_stalls) { - uint8_t rd = (w >> 21) & 0x1F; - uint8_t ra = (w >> 16) & 0x1F; - uint8_t rb = (w >> 11) & 0x1F; - bool type_b = (op & 0x08) != 0; - bool rd_src = mb_rd_is_src[op]; - - uint8_t base_cyc; - if (op == 0x16) - base_cyc = mb_fpu_lat(w); - else if (op == 0x24 && (w & 8u)) - base_cyc = 2; /* WIC: func bit[3]=1 → 2 cycles */ - else - base_cyc = mb_weight[op]; - - uintptr_t meta = (uintptr_t)op - | ((uintptr_t)rd << 6) - | ((uintptr_t)ra << 11) - | ((uintptr_t)rb << 16) - | ((uintptr_t)type_b << 21) - | ((uintptr_t)rd_src << 22) - | ((uintptr_t)1u << 23) /* valid */ - | ((uintptr_t)base_cyc << 24) - | ((uintptr_t)is_nop << 32); - - qemu_plugin_register_vcpu_insn_exec_cb( - insn, vcpu_insn_exec_stall, - QEMU_PLUGIN_CB_NO_REGS, (void *)meta); - - } else { - /* Weighted inline mode. */ - uint64_t weight; - if (op == 0x16) - weight = mb_fpu_lat(w); - else if (op == 0x24 && (w & 8u)) - weight = 2; - else - weight = mb_weight[op]; - - qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.insns, 1); - qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.wcycles, weight); + uint64_t weight; + if (op == 0x16) + weight = mb_fpu_lat(w); + else if (op == 0x24 && (w & 8u)) + weight = 2; + else if (op == 0x26 || op == 0x27 || op == 0x2E || op == 0x2F) + weight = mb_branch_cost(w); + else + weight = mb_weight[op]; + + /* Track last conditional non-D branch for retroactive accounting. */ + if (is_cond_branch_op(op) && !is_delayed_branch(w)) + last_cond_nond_vaddr = qemu_plugin_insn_vaddr(insn); + + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.insns, 1); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.wcycles, weight); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.op_count[op], 1); + qemu_plugin_register_vcpu_insn_exec_inline( + insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.op_cycles[op], weight); + + uint64_t *cls = is_nop ? &total.nop : op_class_ctr[op]; + if (cls) qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.op_count[op], 1); - qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, &total.op_cycles[op], weight); - - /* Class counter. */ - uint64_t *cls = is_nop ? &total.nop : op_class_ctr[op]; - if (cls) - qemu_plugin_register_vcpu_insn_exec_inline( - insn, QEMU_PLUGIN_INLINE_ADD_U64, cls, 1); - } + insn, QEMU_PLUGIN_INLINE_ADD_U64, cls, 1); } + + TBBranchMeta *binfo = g_malloc(sizeof(TBBranchMeta)); + binfo->tb_vaddr = qemu_plugin_tb_vaddr(tb); + binfo->has_cond_branch = last_cond_nond_vaddr != 0; + binfo->fallthrough_pc = last_cond_nond_vaddr + 4; + binfo->last_stall_rd = last_stall_rd; + binfo->first_insn_reads = first_insn_reads; + qemu_plugin_register_vcpu_tb_exec_cb( + tb, vcpu_tb_exec_branch, QEMU_PLUGIN_CB_NO_REGS, (void *)binfo); } /* ========================================================================= From e33e9f43b8202aa4be3d392f7c88b3e50afa949f Mon Sep 17 00:00:00 2001 From: Samuel Price Date: Thu, 2 Jul 2026 16:58:21 -0400 Subject: [PATCH 4/6] target/microblaze: fix gen_bsefi to use end-bit semantics per UG984 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit BSEFI (opcode 0x19, bit[14]=1) extracts bits [IMMW:IMMS] from rA, where IMMW is the end-bit index (inclusive) and IMMS is the start bit. The correct width is (IMMW - IMMS + 1). gen_bsefi was incorrectly passing imm_w directly as the length argument to tcg_gen_extract_i32, treating it as a field width rather than an end-bit index. For bsefi r22, r19, 27, 24 (extract nibble 6), QEMU was extracting 27 bits from bit 24 instead of 4 bits — reading well past the h[] table and returning garbage. gen_bsifi already uses the correct (imm_w - imm_s + 1) convention; align gen_bsefi to match. Co-Authored-By: Claude Sonnet 4.6 --- target/microblaze/translate.c | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/target/microblaze/translate.c b/target/microblaze/translate.c index e8cf543d60f..920f9af508c 100644 --- a/target/microblaze/translate.c +++ b/target/microblaze/translate.c @@ -372,13 +372,16 @@ static void gen_bsefi(TCGv_i32 out, TCGv_i32 ina, int32_t imm) /* Note that decodetree has extracted and reassembled imm_w/imm_s. */ int imm_w = extract32(imm, 5, 5); int imm_s = extract32(imm, 0, 5); + /* UG984 §5: imm_w is the end-bit index (inclusive), imm_s is the start + * bit. Width = imm_w - imm_s + 1. tcg_gen_extract_i32 takes (start, len). */ + int width = imm_w - imm_s + 1; - if (imm_w + imm_s > 32 || imm_w == 0) { + if (imm_w < imm_s || imm_w >= 32) { /* These inputs have an undefined behavior. */ qemu_log_mask(LOG_GUEST_ERROR, "bsefi: Bad input w=%d s=%d\n", imm_w, imm_s); } else { - tcg_gen_extract_i32(out, ina, imm_s, imm_w); + tcg_gen_extract_i32(out, ina, imm_s, width); } } From 6bdafcd9a93f0186f13c7150ab09bcc44e5e436b Mon Sep 17 00:00:00 2001 From: Samuel Price Date: Fri, 3 Jul 2026 10:23:23 -0400 Subject: [PATCH 5/6] target/microblaze: fix gen_bsefi WIDTH semantics (revert end-bit patch) The previous commit changed gen_bsefi to interpret bits[10:6] as an end-bit index (imm_w - imm_s + 1) to match the (wrong) LLVM backend encoding. That made QEMU agree with our wrong toolchain but diverge from real hardware and GAS. Upstream GAS (AMD/Xilinx 2023 tc-microblaze.c) and upstream QEMU master both store WIDTH directly in bits[10:6] for BSEFI. BSIFI is the asymmetric one: it stores end-bit = start+width-1 in bits[10:6]. The correct LLVM fix is in tryBSEFI (see companion llvm-project commit). Co-Authored-By: Claude Sonnet 4.6 --- target/microblaze/translate.c | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/target/microblaze/translate.c b/target/microblaze/translate.c index 920f9af508c..f216b421d96 100644 --- a/target/microblaze/translate.c +++ b/target/microblaze/translate.c @@ -372,16 +372,14 @@ static void gen_bsefi(TCGv_i32 out, TCGv_i32 ina, int32_t imm) /* Note that decodetree has extracted and reassembled imm_w/imm_s. */ int imm_w = extract32(imm, 5, 5); int imm_s = extract32(imm, 0, 5); - /* UG984 §5: imm_w is the end-bit index (inclusive), imm_s is the start - * bit. Width = imm_w - imm_s + 1. tcg_gen_extract_i32 takes (start, len). */ - int width = imm_w - imm_s + 1; - - if (imm_w < imm_s || imm_w >= 32) { + /* GAS / hardware encoding: bits[10:6] = WIDTH (number of bits to extract), + * bits[4:0] = START. tcg_gen_extract_i32(out, ina, start, len). */ + if (imm_w + imm_s > 32 || imm_w == 0) { /* These inputs have an undefined behavior. */ qemu_log_mask(LOG_GUEST_ERROR, "bsefi: Bad input w=%d s=%d\n", imm_w, imm_s); } else { - tcg_gen_extract_i32(out, ina, imm_s, width); + tcg_gen_extract_i32(out, ina, imm_s, imm_w); } } From 003fe7c244896cf2fc131fc57fc7ff8e216f04c0 Mon Sep 17 00:00:00 2001 From: Samuel Price Date: Mon, 6 Jul 2026 22:33:28 -0400 Subject: [PATCH 6/6] mb_cycles plugin: replace runtime RAW stall tracking with static load-use detection Detect the 1-cycle load-use stall at TB translation time by inspecting adjacent instruction pairs, baking the penalty into a constant inline add. Cross-TB pairs are resolved with one lightweight callback per TB. Removes the stalls=on/off mode split; the weighted model now includes load-use stalls by default. Co-Authored-By: Claude Fable 5 --- contrib/plugins/mb_cycles.c | 151 +++++++++++++++++++----------------- 1 file changed, 78 insertions(+), 73 deletions(-) diff --git a/contrib/plugins/mb_cycles.c b/contrib/plugins/mb_cycles.c index 584464f735e..09b820b80d2 100644 --- a/contrib/plugins/mb_cycles.c +++ b/contrib/plugins/mb_cycles.c @@ -8,19 +8,15 @@ * The goal is a reproducible measurement so that compiler changes produce clear * before/after diffs. * - * TWO MEASUREMENT MODES - * --------------------- - * weighted (stalls=off, default): - * Counts weighted instruction cycles from a fixed UG984 §5 table. No stalls, - * no memory effects, no branch penalties. Use as the primary regression baseline. - * - * raw-stalls (stalls=on): - * Adds in-order RAW forwarding stalls on top of weighted cycles. Tracks - * register readiness for all 32 GPRs. Loads add 1 stall on a direct consumer - * (load result latency = 2 cycles, forwarded from MEM stage). Blocking ops - * (idiv=34, FPU=per-func, mts/mfs=2) stall the pipeline for their base cost; - * no extra RAW stall after them. Ignores WAW/WAR hazards, memory hierarchy, - * and branch penalties. + * MEASUREMENT MODEL + * ----------------- + * Counts weighted instruction cycles from a fixed UG984 §5 table. Load-use + * stalls (1-cycle RAW hazard when a load result is consumed by the immediately + * following instruction) are detected statically at TB translation time by + * inspecting adjacent instruction pairs — no runtime tracking required. The + * stall penalty is baked in as a constant inline add, so overhead is zero. + * Cross-TB load-use stalls (load at end of TB[N], consumer at start of TB[N+1]) + * are resolved at TB-exec time via a single lightweight C callback per TB. * * BENCHMARK CONTROL (magic NOP instructions) * ------------------------------------------ @@ -57,7 +53,6 @@ * opt=FLAGS optimisation flags (default: "") * target=CONFIG target configuration (default: "") * commit=HASH compiler commit hash (default: "") - * stalls=on|off enable RAW stall model (default: off) * area_opt=0|1|2 FPU latency variant (default: 0) * verbose=on|off per-opcode breakdown appended to JSON (default: off) * @@ -257,14 +252,11 @@ static uint8_t g_cross_tb_stall_rd; /* 0 = no pending cross-TB stall */ * When a non-D conditional branch ends a TB we cannot know at translation * time whether it will be taken. We save the fallthrough PC here; at the * start of the NEXT TB's exec callback, if next_tb_vaddr != fallthrough_pc - * the branch was taken and we add 2 extra cycles to total.wcycles. + * the branch was taken → add 2 extra cycles. Not-taken → 1 cycle total; + * taken → 3 cycles total. * ========================================================================= */ -typedef struct { - bool valid; - uint64_t fallthrough_pc; -} PendingBranch; - -static PendingBranch g_pending_branch; +static bool g_pending_branch_valid; +static uint64_t g_pending_branch_fallthrough; /* ========================================================================= * Plugin configuration @@ -314,22 +306,31 @@ static inline bool is_bench_magic(uint32_t w) /* * Return true if this is a delayed-branch (D-form) opcode. * - * Encoding verified against insns.decode: - * 0x26 (br/brd/brald): Ra-field bits[20:16] encode D/A/L; D = bit 20 - * 0x27 (beq/beqd etc.): Rd-field bits[25:21] encode cond+D; D = bit 25 (MSB of rD) - * 0x2D (rtsd etc.): always has a delay slot - * 0x2E (bri/brid etc.): same layout as 0x26; D = bit 20 - * 0x2F (beqi/beqid): same layout as 0x27; D = bit 25 (MSB of rD) + * 0x26 (br* reg): rA field encodes {D,A,L,0,0}; D = bit 20 (rA[4]). + * This applies to ALL 0x26 variants including link (brld/brald), + * because even link branches store the link register in rD and + * put the modifier flags in rA. + * + * 0x27 (beq* reg): rD field encodes {D, cond[3:0]}; D = bit 25 (rD[4]). + * + * 0x2D (rtsd etc.): always D-form; no flag needed. + * + * 0x2E (bri* imm): ALL forms (bri/brai/brid/braid/brlid/bralid) encode {D,A,L} + * in rA[20:16]; D = bit 20 (rA[4]). rD is either the link + * register (brlid/bralid) or 0 (non-link forms). + * Verified against QEMU insns.decode and GNU AS output. + * + * 0x2F (beqi* imm): same layout as 0x27; D = bit 25 (rD[4]). */ static inline bool is_delayed_branch(uint32_t w) { uint8_t op = (w >> 26) & 0x3F; switch (op) { - case 0x26: return (w >> 20) & 1; /* brd/brad/brld/brald: D at bit 20 */ - case 0x27: return (w >> 25) & 1; /* beqd-bged: D at bit 25 (MSB of rD) */ - case 0x2D: return true; /* rtsd/rtid/rtbd/rted: always */ - case 0x2E: return (w >> 20) & 1; /* brid/braid/brlid/bralid: D at 20 */ - case 0x2F: return (w >> 25) & 1; /* beqid-bgeid: D at bit 25 (MSB of rD) */ + case 0x26: return (w >> 20) & 1; /* D in rA[4] */ + case 0x27: return (w >> 25) & 1; /* D in rD[4] */ + case 0x2D: return true; /* always D-form */ + case 0x2E: return (w >> 20) & 1; /* D in rA[4] for ALL 0x2E forms */ + case 0x2F: return (w >> 25) & 1; /* D in rD[4] */ default: return false; } } @@ -395,10 +396,9 @@ static inline uint32_t insn_reads_mask(uint32_t w) * taken → cost = 3 (1 branch + 2-cycle flush). * * Non-D conditional (bnei, beqi, blti, …): static cost = 1. An extra 2 - * cycles is credited retroactively when the next TB's start address does not - * equal this branch's fallthrough PC (i.e. it was taken). Not-taken → 1 - * cycle total; taken → 3 cycles total. Accounting is done in - * vcpu_tb_exec_branch (stalls=off) / vcpu_tb_exec_stall (stalls=on). + * cycles is added retroactively at the start of the next TB when the next + * TB's vaddr != this branch's fallthrough PC (i.e. the branch was taken). + * Not-taken → 1 cycle total; taken → 3 cycles total. */ static inline uint8_t mb_branch_cost(uint32_t w) { @@ -409,10 +409,6 @@ static inline uint8_t mb_branch_cost(uint32_t w) return 3; /* unconditional non-D: always taken, 3-cycle flush */ } -static inline uint64_t stall_for(uint8_t r, uint64_t now) -{ - return (r && reg_ready[r] > now) ? reg_ready[r] - now : 0; -} /* ========================================================================= * Bench state machine operations @@ -463,7 +459,7 @@ static void emit_json(const MBCounts *c, const char *status) A(" \"target_config\": \"%s\",\n", target_config); A(" \"compiler_commit\": \"%s\",\n", commit_hash); A(" \"status\": \"%s\",\n", status); - A(" \"stall_model\": \"%s\",\n", model_stalls ? "raw-stalls" : "weighted"); + A(" \"stall_model\": \"static\",\n"); A(" \"instructions\": %" PRIu64 ",\n", c->insns); A(" \"weighted_cycles\": %" PRIu64 ",\n", c->wcycles); A(" \"raw_stall_cycles\": %" PRIu64 ",\n", c->scycles); @@ -534,35 +530,44 @@ static void vcpu_insn_exec_bench(unsigned int vcpu_idx, void *userdata) /* ========================================================================= * TB-exec callback — branch accounting and cross-TB stall resolution + * + * One callback per TB execution. Computed at translation time; all fields + * are constant for the lifetime of the translated block. g_malloc is called + * once per unique TB (translation), not per execution. * ========================================================================= */ typedef struct { - uint64_t tb_vaddr; - bool has_cond_branch; - uint64_t fallthrough_pc; - uint8_t last_stall_rd; /* Rd of last stall-source insn in TB (0 = none) */ - uint32_t first_insn_reads; /* bitmask of GPRs read by TB's first real insn */ -} TBBranchMeta; - -static void vcpu_tb_exec_branch(unsigned int vcpu_idx, void *userdata) + uint64_t tb_vaddr; /* for retroactive branch comparison */ + uint64_t fallthrough_pc; /* PC after the conditional non-D branch */ + uint32_t first_insn_reads; /* GPR read bitmask of TB's first real insn */ + uint8_t last_stall_rd; /* Rd of last stall-source insn (0 = none) */ + bool has_cond_branch; /* TB ends with a non-D conditional branch */ +} TBMeta; + +static void vcpu_tb_exec_cb(unsigned int vcpu_idx, void *userdata) { - const TBBranchMeta *info = (const TBBranchMeta *)userdata; + const TBMeta *m = (const TBMeta *)userdata; + + /* Fast path: nothing pending in either direction. */ + if (!g_pending_branch_valid && !g_cross_tb_stall_rd && + !m->has_cond_branch && !m->last_stall_rd) + return; - /* Cross-TB load-use stall: previous TB ended with a stall-source insn. */ + /* Cross-TB load-use stall: previous TB ended with a load/mfs. */ if (g_cross_tb_stall_rd && - (info->first_insn_reads >> g_cross_tb_stall_rd) & 1) + (m->first_insn_reads >> g_cross_tb_stall_rd) & 1) total.scycles += 1; - g_cross_tb_stall_rd = info->last_stall_rd; + g_cross_tb_stall_rd = m->last_stall_rd; /* Retroactive branch-taken credit. */ - if (g_pending_branch.valid) { - if (info->tb_vaddr != g_pending_branch.fallthrough_pc) + if (g_pending_branch_valid) { + if (m->tb_vaddr != g_pending_branch_fallthrough) total.wcycles += 2; /* taken: 1 static + 2 extra = 3 total */ - g_pending_branch.valid = false; + g_pending_branch_valid = false; } - if (info->has_cond_branch) { - g_pending_branch.valid = true; - g_pending_branch.fallthrough_pc = info->fallthrough_pc; + if (m->has_cond_branch) { + g_pending_branch_valid = true; + g_pending_branch_fallthrough = m->fallthrough_pc; } } @@ -572,18 +577,19 @@ static void vcpu_tb_exec_branch(unsigned int vcpu_idx, void *userdata) * One path only: inline counter updates per instruction. Load-use stalls * are detected statically by looking at adjacent instruction pairs; the * stall penalty is baked in as a constant inline add at translation time. - * Branch taken/not-taken and cross-TB stall resolution are handled by one - * lightweight C callback per TB (vcpu_tb_exec_branch). + * Cross-TB stall resolution and retroactive branch-taken accounting are + * handled by one lightweight C callback per TB (vcpu_tb_exec_cb); its + * fast-path exit makes it near-zero cost when no state is pending. * ========================================================================= */ static void vcpu_tb_trans(qemu_plugin_id_t id, struct qemu_plugin_tb *tb) { trans_count++; size_t n = qemu_plugin_tb_n_insns(tb); - uint64_t last_cond_nond_vaddr = 0; - uint8_t prev_stall_rd = 0; /* Rd of previous stall-source insn */ - uint8_t last_stall_rd = 0; /* final value → cross-TB state */ - uint32_t first_insn_reads = 0; /* reads mask of first real insn */ + uint64_t last_cond_nond_vaddr = 0; /* vaddr of last non-D conditional branch */ + uint8_t prev_stall_rd = 0; /* Rd of previous stall-source insn */ + uint8_t last_stall_rd = 0; /* final value → cross-TB state */ + uint32_t first_insn_reads = 0; /* reads mask of first real insn */ bool first_real_seen = false; for (size_t i = 0; i < n; i++) { @@ -667,14 +673,14 @@ static void vcpu_tb_trans(qemu_plugin_id_t id, struct qemu_plugin_tb *tb) insn, QEMU_PLUGIN_INLINE_ADD_U64, cls, 1); } - TBBranchMeta *binfo = g_malloc(sizeof(TBBranchMeta)); - binfo->tb_vaddr = qemu_plugin_tb_vaddr(tb); - binfo->has_cond_branch = last_cond_nond_vaddr != 0; - binfo->fallthrough_pc = last_cond_nond_vaddr + 4; - binfo->last_stall_rd = last_stall_rd; - binfo->first_insn_reads = first_insn_reads; + TBMeta *m = g_malloc(sizeof(TBMeta)); + m->tb_vaddr = qemu_plugin_tb_vaddr(tb); + m->has_cond_branch = last_cond_nond_vaddr != 0; + m->fallthrough_pc = last_cond_nond_vaddr + 4; + m->last_stall_rd = last_stall_rd; + m->first_insn_reads = first_insn_reads; qemu_plugin_register_vcpu_tb_exec_cb( - tb, vcpu_tb_exec_branch, QEMU_PLUGIN_CB_NO_REGS, (void *)binfo); + tb, vcpu_tb_exec_cb, QEMU_PLUGIN_CB_NO_REGS, (void *)m); } /* ========================================================================= @@ -711,7 +717,6 @@ int qemu_plugin_install(qemu_plugin_id_t id, const qemu_info_t *info, !g_strcmp0(v,"true") || !g_strcmp0(v,"1"); \ continue; \ } - BOOL_ARG("stalls", model_stalls) BOOL_ARG("verbose", verbose) #undef BOOL_ARG