#!/usr/bin/env python3 """How much of the bus does the 68000 decoder actually leave for a DMAC? python3 tools/analysis/15_bus_occupancy.py [container.dlx] [--nframes N] FINDINGS 29.6's DMAC idea only pays if the DMAC can find bus slots the CPU is not using. That is not a cycle count, it is a BUS count, and nothing in the tree had one. Two sources, and the point is that they check each other: DATA accesses MEASURED by tools/bench/c68k/c68k_bench, which counts every Read/Write callback the C68K core makes. Exact. INSTRUCTION DERIVED here by walking src/player/decode.s's straight-line prefetch paths in tools/bench/decode.lst and multiplying by the mode histogram. Not measurable from either emulator: MAME's core does not expose a fetch count and C68K reads opcodes straight through a host pointer with no callback. If the derived DATA figure matches the measured one, the derived PREFETCH figure from the same walk is trustworthy too. That check is the first thing printed, and this script exits non-zero if it fails. A 68000 bus cycle is 4 clocks, so a frame of C clocks holds C/4 bus slots. """ import sys, os, argparse, csv sys.path.insert(0, "tools/encoder") sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import numpy as np from dlx import DLX import buscost as B from buscost import V7_FRAME_PREF, V7_FRAME_DATA BUS_CLK = 4 # --- straight-line path costs, read off tools/bench/decode.lst ------------- # (instruction words, data bus cycles). A long access is two bus cycles on the # 68000's 16-bit bus; movem.l of N registers is 2N. # # dispatch move.b (a1),d0 / lsr.b / and.w #3 / beq .sk 6w, 1 read # + subq / beq .v1 -> 8w # + subq / bne .rw -> 10w # V4 body $10090..$100E2 = 82 B = 41w; 4 x (1 byte read # + movem.l 2 regs = 4 reads + 2 move.l = 4 writes) = 36 # V1 body $100E2..$10106 = 36 B = 18w; 1 byte read # + movem.l 8 regs = 16 reads + 4 x movem.l 2 = 16 w = 33 # RAW body $10106..$10164 = 94 B = 47w; 8 x (2 byte reads # + 1 move.l = 2 writes) = 32 # .sk tail addq.l #8,a4 1w # BLOCK 0 has no lsr.b, so one of the four dispatches in a group is 1w cheaper. DISPATCH_SK, DISPATCH_V1, DISPATCH_V4 = 6, 8, 10 BODY = {0: (0, 0), 1: (18, 33), 2: (41, 36), 3: (47, 32)} DISPATCH = {0: DISPATCH_SK, 1: DISPATCH_V1, 2: DISPATCH_V4, 3: DISPATCH_V4} SK_TAIL = 1 GROUP_HEAD = 3 # tst.b (a1) 1w + beq allskip 2w GROUP_TAIL = 4 # addq.l #1,a1 / cmpa.l a5,a4 / bne byteloop ALLSKIP = 9 # the whole four-block fast path, tst.b included ROW_HEAD, ROW_TAIL = 3, 7 ap = argparse.ArgumentParser() ap.add_argument("container", nargs="?", default="tmp/rc_fr_singe_scsi_cpufit.dlx") ap.add_argument("--csv", default="tmp/c68k_frames.csv", help="per-frame output of tools/bench/c68k/run.sh") ap.add_argument("--nframes", type=int, default=None) ap.add_argument("--kbps", type=float, default=None, help="delivery rate in KB/s. OPTIONAL and there is no default " "(FINDINGS 50): supply it and the AUTO-REQUEST rows are " "added, which are the only rows whose cost depends on how " "long the record takes to arrive (59.3).") a = ap.parse_args() if not os.path.exists(a.container): sys.exit(f"missing {a.container}") d = DLX(a.container) meas = {} if os.path.exists(a.csv): for r in csv.DictReader(open(a.csv)): meas[int(r["frame"])] = (int(r["cycles"]), int(r["bus_reads"]) + int(r["bus_writes"])) NF = a.nframes or (max(meas) + 1 if meas else d.nframes) pref_t, data_t, cyc_t = [], [], [] for f in range(NF): m = d.modes(f).reshape(d.nby, d.nbx) pref = d.nby * (ROW_HEAD + ROW_TAIL) data = 0 for by in range(d.nby): row = m[by] for gi in range(0, d.nbx, 4): g = row[gi:gi+4] if (g == 0).all(): pref += ALLSKIP; data += 1 continue pref += GROUP_HEAD + GROUP_TAIL - 1 # BLOCK 0 has no lsr.b data += 1 for b in g: b = int(b) pw, pd = BODY[b] pref += DISPATCH[b] + pw + SK_TAIL data += 1 + pd # The span section is bus traffic too, and it is most of the frame's data # accesses in a span-heavy container: 48 per 24-pixel chain unit. Leaving it # out would not merely understate the total -- it would break the CHECK # below, which is the whole licence for the prefetch figure. sp, _ = d.spans(f) if sp: pref += V7_FRAME_PREF; data += V7_FRAME_DATA for _, _, px in sp: sp_p, sp_d = B.v7_span_split(len(px)) pref += sp_p; data += sp_d pref_t.append(pref); data_t.append(data) cyc_t.append(meas.get(f, (0, 0))[0]) pref_t, data_t, cyc_t = map(np.array, (pref_t, data_t, cyc_t)) print(f"{a.container}: {NF} frames, {d.nb} blocks/frame\n") if meas: md = np.array([meas[f][1] for f in range(NF)]) err = 100 * (data_t - md) / md print("CHECK -- derived DATA bus cycles against the C68K harness's measurement") print(f" measured mean {md.mean():>10,.0f} /frame") print(f" derived mean {data_t.mean():>10,.0f} /frame " f"error {err.mean():+.2f}% mean, {np.abs(err).max():.2f}% worst") if np.abs(err).max() > 2.0: sys.exit("\nFAIL: the path walk does not reproduce the measured data " "accesses, so its prefetch figure cannot be trusted either.") print(" the walk reproduces the measurement, so its prefetch count stands\n") slots = cyc_t / BUS_CLK tot = pref_t + data_t print(f"{'':<22}{'mean':>12}{'median':>12}{'worst frame':>14}") for label, v in (("bus slots in a frame", slots), (" data accesses", data_t), (" instruction prefetch", pref_t), (" total bus cycles", tot)): print(f"{label:<22}{v.mean():>12,.0f}{np.median(v):>12,.0f}{v.max():>14,.0f}") occ = 100 * tot / slots print(f"{'bus OCCUPANCY':<22}{occ.mean():>11.1f}%{np.median(occ):>11.1f}%" f"{occ.max():>13.1f}%") free = slots - tot print(f"{'slots left for a DMAC':<22}{free.mean():>12,.0f}{np.median(free):>12,.0f}" f"{free.min():>14,.0f} (worst = fewest)") print(f"\nprefetch is {100*pref_t.sum()/tot.sum():.0f}% of the decoder's bus traffic: " f"the data-only\nfigure the harness prints understates occupancy by about 2x.") print(f"A DMAC painting spans at 8 clocks (2 bus cycles) per pixel could use at\n" f"most {free.mean()/2:,.0f} pixels' worth of the mean frame's spare slots " f"-- against {d.nb*16:,} pixels\nin a whole screen.") # --------------------------------------------------------------------------- # THE OTHER TWO MASTERS. Everything above is the 68000's own traffic, and it # was the whole of this tool until session 20. The frame also has to carry the # bitstream in off the disk and a byte of ADPCM out to $E92003 every 128 us, # and neither has ever appeared in a bus figure -- FINDINGS 35's lesson, which # was about the CLOCK budget, had never been applied to the BUS one. # # The DMAC does not overlap with the CPU (buscost.DMA_OVERLAPS = False): the # 68000 has no cache and a two-word prefetch queue that empties at once, so a # stolen bus cycle is a stopped CPU. The three demands therefore ADD. # # Audio's per-byte figure is SETTLED, not bracketed by taste: # tools/analysis/21_iplrom_dmac.py reads the IPL ROM's own HD63450 setup and # finds channel 3 dual-address, 8-bit port, cycle steal without hold, external # request -- one arbitration per byte, no burst. Video's is NOT settled: it is # ROADMAP B3 / FINDINGS 42.4-42.6's W, so it is swept rather than picked. print("\n" + "=" * 72) print("THE OTHER TWO MASTERS -- what the DMAC takes out of the same frame\n") FPS = d.fps CPUHZ = 10e6 # stock X68000, MAME 0.277 x68k.cpp:1133 FRAME_CLK = CPUHZ / FPS vid_bpf = sum(n + 4 for (_, n) in d.frames[:NF]) / NF # DLX2 record padding aud_bpf = B.ADPCM_BYTES_PER_S / FPS a_lo = aud_bpf * B.ADPCM_CLK_BYTE_BEST a_hi = aud_bpf * B.ADPCM_CLK_BYTE_WORST cpu_clk = cyc_t.mean() if cyc_t.any() else float("nan") print(f"frame period at {FPS:g} fps on a 10 MHz 68000: {FRAME_CLK:,.0f} clocks") if cyc_t.any(): print(f" decoder, MEASURED (C68K) {cpu_clk:>10,.0f} clk " f"{100*cpu_clk/FRAME_CLK:5.1f}% worst frame " f"{100*cyc_t.max()/FRAME_CLK:.1f}%") print(f" audio DMA, {aud_bpf:,.1f} B/frame {a_lo:>10,.0f} clk " f"{100*a_lo/FRAME_CLK:5.2f}% .. {a_hi:,.0f} clk " f"({100*a_hi/FRAME_CLK:.2f}%)") print(f" {B.ADPCM_CLK_BYTE_BEST}..{B.ADPCM_CLK_BYTE_WORST} clk/byte, " f"from the ROM's own DCR/OCR (21_iplrom_dmac.py). NOT a guess, and\n" f" not the disk's rate: audio arbitrates for the bus once per byte " f"and cannot burst.") print(f"\n video DMA, {vid_bpf:,.0f} B/frame, swept over W -- ROADMAP B3 is " f"still open:") print(f" {'W (clk/byte)':<16}{'clk/frame':>12}{'% of frame':>12} " f"{'CPU+audio+video':>18}") for W, note in ((5.0, "single address, bus held (11_cpu_budget.py default)"), (8.0, "FINDINGS 5's long-standing per-word ESTIMATE"), (9.0, "DUAL address, bus held -- and the FLOOR of every " "dual-address\n " " configuration, auto-request included (59.3)"), (12.0, "single address, arbitrated per byte"), (16.0, "what the ROM programs for SASI (best case)"), (19.0, "what the ROM programs for SASI (worst case)"), (87.28, "PIO -- MEASURED, FINDINGS 58.2, the CPU doing it itself")): v = vid_bpf * W tot_clk = (cpu_clk if cyc_t.any() else 0) + a_lo + v print(f" {W:<16.6g}{v:>12,.0f}{100*v/FRAME_CLK:>11.1f}% " f"{100*tot_clk/FRAME_CLK:>17.1f}% {note}") print(f"\n (the last column adds the MEASURED mean decode and the BEST-CASE " f"audio, so it is\n the optimistic end of every row. 100% is the frame " f"deadline at {FPS:g} fps.)") print(f""" Audio is {100*a_lo/FRAME_CLK:.2f}%..{100*a_hi/FRAME_CLK:.2f}% of the frame and video is {vid_bpf*5/FRAME_CLK*100:.0f}%..{vid_bpf*19/FRAME_CLK*100:.0f}% over the ladder, against {vid_bpf*87.28/FRAME_CLK*100:.0f}% for the PIO transport FINDINGS 58.2 measured. The unpriced audio stream was never the risk P6 called it -- ON THE BUS.""") # --- HEADROOM, AND THE FLOOR UNDER THE LADDER ----------------------------- # Added session 27. The sweep above answers "what does each W cost"; it never # answered "what can this frame afford", and the two are not the same question. # FINDINGS 59.2 is why it matters now: with no external request line the only # configurations that can be run are dual-address, and a dual-address byte has # a FLOOR -- one 4-clock read of the device plus one 5-clock write to memory, # buscost.DMA_DUAL_BYTE_CLK. No GCR share and no delivery rate goes under it. print("\n" + "=" * 72) print("WHAT THE FRAME CAN AFFORD, AND THE FLOOR UNDER THE LADDER\n") head_clk = FRAME_CLK - (cpu_clk if cyc_t.any() else 0) - a_lo head_wb = head_clk / vid_bpf print(f" headroom after the MEASURED decode and best-case audio: " f"{head_clk:,.0f} clk = {100*head_clk/FRAME_CLK:.1f}%") print(f" at {vid_bpf:,.0f} B a frame that is {head_wb:.2f} CLOCKS PER BYTE, and " f"that is the number\n a transport has to come in under.\n") floor = B.DMA_DUAL_BYTE_CLK print(f" dual-address floor {floor} clk/B ({B.DMA_READ_CLK} read of the " f"device + {B.DMA_WRITE_CLK} write to memory, Fig 4-25)") print(f" single-address held {B.DMA_DISK_CLK_WORD_HELD} clk/B (one memory " f"write; needs the device to ACK, i.e. a REQUEST LINE)") if head_wb < floor: print(f""" SO DUAL ADDRESS DOES NOT FIT THIS CONTAINER AT {FPS:g} fps -- not at any delivery rate and not at any GCR share, because {head_wb:.2f} < {floor}. A share decides whether the channel sits AT the floor or above it; it cannot go under it. That is FINDINGS 59.2's three bounds arriving in the budget: the configurations this machine can run are exactly the ones the frame cannot afford, and the one it can afford -- single address, {B.DMA_DISK_CLK_WORD_HELD} clk/B, {100*vid_bpf*B.DMA_DISK_CLK_WORD_HELD/FRAME_CLK:.1f}% -- needs the request line ROADMAP B3 asks about.""") for w, what in ((floor, "dual address"), (B.DMA_DISK_CLK_WORD_HELD, "single address")): tgt = head_clk / w print(f"\n TO FIT AT {w} clk/B ({what}) THIS CONTAINER MUST COME DOWN TO") print(f" {tgt:,.0f} B a frame = {tgt*FPS/1024:,.0f} KB/s of payload " f"(it is {vid_bpf:,.0f} B, {vid_bpf*FPS/1024:,.0f} KB/s)" + (" -- already met" if vid_bpf <= tgt else f" -- {100*(vid_bpf/tgt-1):.0f}% too big")) print(f""" AND THAT IS THE PESSIMISTIC READING OF THE ENCODER LEVER: a lighter container also DECODES cheaper, so the decode term above falls with the byte term. The figure to re-derive it against is this tool run on the lighter container -- with its OWN C68K measurement, because the cross-check at the top is what licenses every number below it.""") else: print(f"\n The frame affords {head_wb:.2f} clk/B, which is at or above the " f"{floor} clk/B dual-address floor.") # --- AUTO-REQUEST, and only when a rate is supplied ------------------------ # These are the rows 59.3 added and they are the only ones here whose cost is # not a property of the transfer: an auto-requested channel spends its share of # the bus whether or not a byte is there, so what a record costs depends on how # long it takes to ARRIVE. No default rate, deliberately (FINDINGS 50). if a.kbps: RATE = a.kbps * 1024.0 wire_clk = vid_bpf / RATE * CPUHZ cap = 0.5 * CPUHZ / B.DMA_DUAL_BYTE_CLK # the 50% share's ceiling print("\n" + "=" * 72) print(f"AUTO-REQUEST AT {a.kbps:g} KB/s -- charged by TIME, not by byte " f"(59.3)\n") print(f" the record takes {wire_clk:,.0f} clk to arrive = " f"{100*wire_clk/FRAME_CLK:.1f}% of a frame\n") print(f" {'configuration':<34}{'clk/B':>8}{'% of frame':>12}" f"{'CPU+audio+video':>18}") rows = [("REQG 01, max rate (100% of the bus)", 1.0, None)] for br, share in ((0, .5), (1, .25), (2, .125), (3, .0625)): rows.append((f"REQG 00, LRAR BR={br:02b}, {share*100:g}% share", share, share * CPUHZ / B.DMA_DUAL_BYTE_CLK)) for name, share, sustains in rows: v = share * wire_clk tot = (cpu_clk if cyc_t.any() else 0) + a_lo + v flag = "" if sustains is not None and sustains < RATE: flag = f" cannot carry the rate ({sustains/1024:.0f} KB/s max)" print(f" {name:<34}{v/vid_bpf:>8.2f}{100*v/FRAME_CLK:>11.1f}%" f"{100*tot/FRAME_CLK:>17.1f}%{flag}") print(f""" A FASTER DISC MAKES AUTO-REQUEST CHEAPER, which no W does -- the share is spent over a shorter wire time. But it cannot reach the floor: a 50% share tops out at {cap/1024:,.0f} KB/s, above which the CHANNEL is the bottleneck and the delivered rate falls back to it. At that ceiling the cost is exactly the {B.DMA_DUAL_BYTE_CLK} clk/B floor, which is where the section above already put it.""")