#!/usr/bin/env python3 """How much of the bus does the 68000 decoder actually leave for a DMAC? python3 tools/analysis/15_bus_occupancy.py [container.dlx] [--nframes N] FINDINGS 29.6's DMAC idea only pays if the DMAC can find bus slots the CPU is not using. That is not a cycle count, it is a BUS count, and nothing in the tree had one. Two sources, and the point is that they check each other: DATA accesses MEASURED by tools/bench/c68k/c68k_bench, which counts every Read/Write callback the C68K core makes. Exact. INSTRUCTION DERIVED here by walking src/player/decode.s's straight-line prefetch paths in tools/bench/decode.lst and multiplying by the mode histogram. Not measurable from either emulator: MAME's core does not expose a fetch count and C68K reads opcodes straight through a host pointer with no callback. If the derived DATA figure matches the measured one, the derived PREFETCH figure from the same walk is trustworthy too. That check is the first thing printed, and this script exits non-zero if it fails. A 68000 bus cycle is 4 clocks, so a frame of C clocks holds C/4 bus slots. """ import sys, os, argparse, csv sys.path.insert(0, "tools/encoder") sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import numpy as np from dlx import DLX import buscost as B from buscost import V7_FRAME_PREF, V7_FRAME_DATA BUS_CLK = 4 # --- straight-line path costs, read off tools/bench/decode.lst ------------- # (instruction words, data bus cycles). A long access is two bus cycles on the # 68000's 16-bit bus; movem.l of N registers is 2N. # # dispatch move.b (a1),d0 / lsr.b / and.w #3 / beq .sk 6w, 1 read # + subq / beq .v1 -> 8w # + subq / bne .rw -> 10w # V4 body $10090..$100E2 = 82 B = 41w; 4 x (1 byte read # + movem.l 2 regs = 4 reads + 2 move.l = 4 writes) = 36 # V1 body $100E2..$10106 = 36 B = 18w; 1 byte read # + movem.l 8 regs = 16 reads + 4 x movem.l 2 = 16 w = 33 # RAW body $10106..$10164 = 94 B = 47w; 8 x (2 byte reads # + 1 move.l = 2 writes) = 32 # .sk tail addq.l #8,a4 1w # BLOCK 0 has no lsr.b, so one of the four dispatches in a group is 1w cheaper. DISPATCH_SK, DISPATCH_V1, DISPATCH_V4 = 6, 8, 10 BODY = {0: (0, 0), 1: (18, 33), 2: (41, 36), 3: (47, 32)} DISPATCH = {0: DISPATCH_SK, 1: DISPATCH_V1, 2: DISPATCH_V4, 3: DISPATCH_V4} SK_TAIL = 1 GROUP_HEAD = 3 # tst.b (a1) 1w + beq allskip 2w GROUP_TAIL = 4 # addq.l #1,a1 / cmpa.l a5,a4 / bne byteloop ALLSKIP = 9 # the whole four-block fast path, tst.b included ROW_HEAD, ROW_TAIL = 3, 7 ap = argparse.ArgumentParser() ap.add_argument("container", nargs="?", default="tmp/rc_fr_singe_scsi_cpufit.dlx") ap.add_argument("--csv", default="tmp/c68k_frames.csv", help="per-frame output of tools/bench/c68k/run.sh") ap.add_argument("--nframes", type=int, default=None) a = ap.parse_args() if not os.path.exists(a.container): sys.exit(f"missing {a.container}") d = DLX(a.container) meas = {} if os.path.exists(a.csv): for r in csv.DictReader(open(a.csv)): meas[int(r["frame"])] = (int(r["cycles"]), int(r["bus_reads"]) + int(r["bus_writes"])) NF = a.nframes or (max(meas) + 1 if meas else d.nframes) pref_t, data_t, cyc_t = [], [], [] for f in range(NF): m = d.modes(f).reshape(d.nby, d.nbx) pref = d.nby * (ROW_HEAD + ROW_TAIL) data = 0 for by in range(d.nby): row = m[by] for gi in range(0, d.nbx, 4): g = row[gi:gi+4] if (g == 0).all(): pref += ALLSKIP; data += 1 continue pref += GROUP_HEAD + GROUP_TAIL - 1 # BLOCK 0 has no lsr.b data += 1 for b in g: b = int(b) pw, pd = BODY[b] pref += DISPATCH[b] + pw + SK_TAIL data += 1 + pd # The span section is bus traffic too, and it is most of the frame's data # accesses in a span-heavy container: 48 per 24-pixel chain unit. Leaving it # out would not merely understate the total -- it would break the CHECK # below, which is the whole licence for the prefetch figure. sp, _ = d.spans(f) if sp: pref += V7_FRAME_PREF; data += V7_FRAME_DATA for _, _, px in sp: sp_p, sp_d = B.v7_span_split(len(px)) pref += sp_p; data += sp_d pref_t.append(pref); data_t.append(data) cyc_t.append(meas.get(f, (0, 0))[0]) pref_t, data_t, cyc_t = map(np.array, (pref_t, data_t, cyc_t)) print(f"{a.container}: {NF} frames, {d.nb} blocks/frame\n") if meas: md = np.array([meas[f][1] for f in range(NF)]) err = 100 * (data_t - md) / md print("CHECK -- derived DATA bus cycles against the C68K harness's measurement") print(f" measured mean {md.mean():>10,.0f} /frame") print(f" derived mean {data_t.mean():>10,.0f} /frame " f"error {err.mean():+.2f}% mean, {np.abs(err).max():.2f}% worst") if np.abs(err).max() > 2.0: sys.exit("\nFAIL: the path walk does not reproduce the measured data " "accesses, so its prefetch figure cannot be trusted either.") print(" the walk reproduces the measurement, so its prefetch count stands\n") slots = cyc_t / BUS_CLK tot = pref_t + data_t print(f"{'':<22}{'mean':>12}{'median':>12}{'worst frame':>14}") for label, v in (("bus slots in a frame", slots), (" data accesses", data_t), (" instruction prefetch", pref_t), (" total bus cycles", tot)): print(f"{label:<22}{v.mean():>12,.0f}{np.median(v):>12,.0f}{v.max():>14,.0f}") occ = 100 * tot / slots print(f"{'bus OCCUPANCY':<22}{occ.mean():>11.1f}%{np.median(occ):>11.1f}%" f"{occ.max():>13.1f}%") free = slots - tot print(f"{'slots left for a DMAC':<22}{free.mean():>12,.0f}{np.median(free):>12,.0f}" f"{free.min():>14,.0f} (worst = fewest)") print(f"\nprefetch is {100*pref_t.sum()/tot.sum():.0f}% of the decoder's bus traffic: " f"the data-only\nfigure the harness prints understates occupancy by about 2x.") print(f"A DMAC painting spans at 8 clocks (2 bus cycles) per pixel could use at\n" f"most {free.mean()/2:,.0f} pixels' worth of the mean frame's spare slots " f"-- against {d.nb*16:,} pixels\nin a whole screen.") # --------------------------------------------------------------------------- # THE OTHER TWO MASTERS. Everything above is the 68000's own traffic, and it # was the whole of this tool until session 20. The frame also has to carry the # bitstream in off the disk and a byte of ADPCM out to $E92003 every 128 us, # and neither has ever appeared in a bus figure -- FINDINGS 35's lesson, which # was about the CLOCK budget, had never been applied to the BUS one. # # The DMAC does not overlap with the CPU (buscost.DMA_OVERLAPS = False): the # 68000 has no cache and a two-word prefetch queue that empties at once, so a # stolen bus cycle is a stopped CPU. The three demands therefore ADD. # # Audio's per-byte figure is SETTLED, not bracketed by taste: # tools/analysis/21_iplrom_dmac.py reads the IPL ROM's own HD63450 setup and # finds channel 3 dual-address, 8-bit port, cycle steal without hold, external # request -- one arbitration per byte, no burst. Video's is NOT settled: it is # ROADMAP B3 / FINDINGS 42.4-42.6's W, so it is swept rather than picked. print("\n" + "=" * 72) print("THE OTHER TWO MASTERS -- what the DMAC takes out of the same frame\n") FPS = d.fps CPUHZ = 10e6 # stock X68000, MAME 0.277 x68k.cpp:1133 FRAME_CLK = CPUHZ / FPS vid_bpf = sum(n + 4 for (_, n) in d.frames[:NF]) / NF # DLX2 record padding aud_bpf = B.ADPCM_BYTES_PER_S / FPS a_lo = aud_bpf * B.ADPCM_CLK_BYTE_BEST a_hi = aud_bpf * B.ADPCM_CLK_BYTE_WORST cpu_clk = cyc_t.mean() if cyc_t.any() else float("nan") print(f"frame period at {FPS:g} fps on a 10 MHz 68000: {FRAME_CLK:,.0f} clocks") if cyc_t.any(): print(f" decoder, MEASURED (C68K) {cpu_clk:>10,.0f} clk " f"{100*cpu_clk/FRAME_CLK:5.1f}% worst frame " f"{100*cyc_t.max()/FRAME_CLK:.1f}%") print(f" audio DMA, {aud_bpf:,.1f} B/frame {a_lo:>10,.0f} clk " f"{100*a_lo/FRAME_CLK:5.2f}% .. {a_hi:,.0f} clk " f"({100*a_hi/FRAME_CLK:.2f}%)") print(f" {B.ADPCM_CLK_BYTE_BEST}..{B.ADPCM_CLK_BYTE_WORST} clk/byte, " f"from the ROM's own DCR/OCR (21_iplrom_dmac.py). NOT a guess, and\n" f" not the disk's rate: audio arbitrates for the bus once per byte " f"and cannot burst.") print(f"\n video DMA, {vid_bpf:,.0f} B/frame, swept over W -- ROADMAP B3 is " f"still open:") print(f" {'W (clk/byte)':<16}{'clk/frame':>12}{'% of frame':>12} " f"{'CPU+audio+video':>18}") for W, note in ((5.0, "single address, bus held (11_cpu_budget.py default)"), (8.0, "FINDINGS 5's long-standing per-word ESTIMATE"), (12.0, "single address, arbitrated per byte"), (16.0, "what the ROM programs for SASI (best case)"), (19.0, "what the ROM programs for SASI (worst case)")): v = vid_bpf * W tot_clk = (cpu_clk if cyc_t.any() else 0) + a_lo + v print(f" {W:<16.0f}{v:>12,.0f}{100*v/FRAME_CLK:>11.1f}% " f"{100*tot_clk/FRAME_CLK:>17.1f}% {note}") print(f"\n (the last column adds the MEASURED mean decode and the BEST-CASE " f"audio, so it is\n the optimistic end of every row. 100% is the frame " f"deadline at {FPS:g} fps.)") print(f""" Audio is {100*a_lo/FRAME_CLK:.2f}%..{100*a_hi/FRAME_CLK:.2f}% of the frame and video is {vid_bpf*5/FRAME_CLK*100:.0f}%..{vid_bpf*19/FRAME_CLK*100:.0f}%. The unpriced audio stream was never the risk P6 called it -- ON THE BUS. What the same reading of the ROM found is that the DISK's per-byte cost has a worked example on this machine, it is 16..19 clocks, and at that price this design does not fit at any container size. W is the number to attack, and it is a PLAYER decision.""")