#!/usr/bin/env python3 """How much of the bus does the 68000 decoder actually leave for a DMAC? python3 tools/analysis/15_bus_occupancy.py [container.dlx] [--nframes N] FINDINGS 29.6's DMAC idea only pays if the DMAC can find bus slots the CPU is not using. That is not a cycle count, it is a BUS count, and nothing in the tree had one. Two sources, and the point is that they check each other: DATA accesses MEASURED by tools/bench/c68k/c68k_bench, which counts every Read/Write callback the C68K core makes. Exact. INSTRUCTION DERIVED here by walking src/player/decode.s's straight-line prefetch paths in tools/bench/decode.lst and multiplying by the mode histogram. Not measurable from either emulator: MAME's core does not expose a fetch count and C68K reads opcodes straight through a host pointer with no callback. If the derived DATA figure matches the measured one, the derived PREFETCH figure from the same walk is trustworthy too. That check is the first thing printed, and this script exits non-zero if it fails. A 68000 bus cycle is 4 clocks, so a frame of C clocks holds C/4 bus slots. """ import sys, os, argparse, csv sys.path.insert(0, "tools/encoder") sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import numpy as np from dlx import DLX import buscost as B from buscost import V7_FRAME_PREF, V7_FRAME_DATA BUS_CLK = 4 # --- straight-line path costs, read off tools/bench/decode.lst ------------- # (instruction words, data bus cycles). A long access is two bus cycles on the # 68000's 16-bit bus; movem.l of N registers is 2N. # # dispatch move.b (a1),d0 / lsr.b / and.w #3 / beq .sk 6w, 1 read # + subq / beq .v1 -> 8w # + subq / bne .rw -> 10w # V4 body $10090..$100E2 = 82 B = 41w; 4 x (1 byte read # + movem.l 2 regs = 4 reads + 2 move.l = 4 writes) = 36 # V1 body $100E2..$10106 = 36 B = 18w; 1 byte read # + movem.l 8 regs = 16 reads + 4 x movem.l 2 = 16 w = 33 # RAW body $10106..$10164 = 94 B = 47w; 8 x (2 byte reads # + 1 move.l = 2 writes) = 32 # .sk tail addq.l #8,a4 1w # BLOCK 0 has no lsr.b, so one of the four dispatches in a group is 1w cheaper. DISPATCH_SK, DISPATCH_V1, DISPATCH_V4 = 6, 8, 10 BODY = {0: (0, 0), 1: (18, 33), 2: (41, 36), 3: (47, 32)} DISPATCH = {0: DISPATCH_SK, 1: DISPATCH_V1, 2: DISPATCH_V4, 3: DISPATCH_V4} SK_TAIL = 1 GROUP_HEAD = 3 # tst.b (a1) 1w + beq allskip 2w GROUP_TAIL = 4 # addq.l #1,a1 / cmpa.l a5,a4 / bne byteloop ALLSKIP = 9 # the whole four-block fast path, tst.b included ROW_HEAD, ROW_TAIL = 3, 7 ap = argparse.ArgumentParser() ap.add_argument("container", nargs="?", default="tmp/rc_fr_singe_scsi_cpufit.dlx") ap.add_argument("--csv", default="tmp/c68k_frames.csv", help="per-frame output of tools/bench/c68k/run.sh") ap.add_argument("--nframes", type=int, default=None) a = ap.parse_args() if not os.path.exists(a.container): sys.exit(f"missing {a.container}") d = DLX(a.container) meas = {} if os.path.exists(a.csv): for r in csv.DictReader(open(a.csv)): meas[int(r["frame"])] = (int(r["cycles"]), int(r["bus_reads"]) + int(r["bus_writes"])) NF = a.nframes or (max(meas) + 1 if meas else d.nframes) pref_t, data_t, cyc_t = [], [], [] for f in range(NF): m = d.modes(f).reshape(d.nby, d.nbx) pref = d.nby * (ROW_HEAD + ROW_TAIL) data = 0 for by in range(d.nby): row = m[by] for gi in range(0, d.nbx, 4): g = row[gi:gi+4] if (g == 0).all(): pref += ALLSKIP; data += 1 continue pref += GROUP_HEAD + GROUP_TAIL - 1 # BLOCK 0 has no lsr.b data += 1 for b in g: b = int(b) pw, pd = BODY[b] pref += DISPATCH[b] + pw + SK_TAIL data += 1 + pd # The span section is bus traffic too, and it is most of the frame's data # accesses in a span-heavy container: 48 per 24-pixel chain unit. Leaving it # out would not merely understate the total -- it would break the CHECK # below, which is the whole licence for the prefetch figure. sp, _ = d.spans(f) if sp: pref += V7_FRAME_PREF; data += V7_FRAME_DATA for _, _, px in sp: sp_p, sp_d = B.v7_span_split(len(px)) pref += sp_p; data += sp_d pref_t.append(pref); data_t.append(data) cyc_t.append(meas.get(f, (0, 0))[0]) pref_t, data_t, cyc_t = map(np.array, (pref_t, data_t, cyc_t)) print(f"{a.container}: {NF} frames, {d.nb} blocks/frame\n") if meas: md = np.array([meas[f][1] for f in range(NF)]) err = 100 * (data_t - md) / md print("CHECK -- derived DATA bus cycles against the C68K harness's measurement") print(f" measured mean {md.mean():>10,.0f} /frame") print(f" derived mean {data_t.mean():>10,.0f} /frame " f"error {err.mean():+.2f}% mean, {np.abs(err).max():.2f}% worst") if np.abs(err).max() > 2.0: sys.exit("\nFAIL: the path walk does not reproduce the measured data " "accesses, so its prefetch figure cannot be trusted either.") print(" the walk reproduces the measurement, so its prefetch count stands\n") slots = cyc_t / BUS_CLK tot = pref_t + data_t print(f"{'':<22}{'mean':>12}{'median':>12}{'worst frame':>14}") for label, v in (("bus slots in a frame", slots), (" data accesses", data_t), (" instruction prefetch", pref_t), (" total bus cycles", tot)): print(f"{label:<22}{v.mean():>12,.0f}{np.median(v):>12,.0f}{v.max():>14,.0f}") occ = 100 * tot / slots print(f"{'bus OCCUPANCY':<22}{occ.mean():>11.1f}%{np.median(occ):>11.1f}%" f"{occ.max():>13.1f}%") free = slots - tot print(f"{'slots left for a DMAC':<22}{free.mean():>12,.0f}{np.median(free):>12,.0f}" f"{free.min():>14,.0f} (worst = fewest)") print(f"\nprefetch is {100*pref_t.sum()/tot.sum():.0f}% of the decoder's bus traffic: " f"the data-only\nfigure the harness prints understates occupancy by about 2x.") print(f"A DMAC painting spans at 8 clocks (2 bus cycles) per pixel could use at\n" f"most {free.mean()/2:,.0f} pixels' worth of the mean frame's spare slots " f"-- against {d.nb*16:,} pixels\nin a whole screen.")