#!/usr/bin/env python3 """What does spending the idle bus bandwidth buy back in CPU cycles? python3 tools/analysis/12_span_tradeoff.py [container.dlx] --bus FINDINGS 28 leaves the decoder CPU-bound at 110 KB/s on a much wider pipe. Every codec decision was made when bytes were scarce, so each one trades cycles to save them -- and the cheapest thing a 68000 can be handed is the most expensive thing to store: word-expanded pixels in row-linear runs. This prices ONE new mode against the real mode maps: a per-row SPAN of word-expanded literals, `movem.l`-ed straight from the stream buffer into GVRAM. A run of L horizontally adjacent dirty blocks becomes 4 spans of 4L pixels. MEASURED as of session 8 (FINDINGS 30), on the 68000, with the span decoder in tools/bench/blit.s v6 and the streams in tools/bench/prep_spans.py: 43.7 cycles per span + 9.152 per pixel, fitting eleven span lengths to within 0.3%. That is the ENCODER-ASSISTED format: the record is an absolute GVRAM address and a jump displacement into an unrolled copy chain, so the decoder does no arithmetic per span. The obvious decoder -- handed (x, npix) and left to work the copy out -- measures 97.9 + 10.46 and is 2.2x dearer on a 24-pixel span (v5). Span length is therefore a multiple of 24 pixels, and a run pads up to it; the padding is free of cycles beyond its pixels and correct on screen, because a literal span carries true pixels of the current frame. The mode maps are NOT re-optimised: this only re-codes regions the encoder already chose to redraw, so it is a lower bound on what a cost-aware encoder would find. """ import sys, os, argparse sys.path.insert(0, "tools/encoder") import numpy as np from dlx import DLX FRAME_CYC = 833333.0 # 12fps at 10 MHz AUDIO_KBPS = 7.8 C_V1, C_V4, C_RAW = 299.9, 448.2, 400.4 # FINDINGS 28.2 (measured) C_SKIP_CLUSTERED, C_SKIP_MIXED = 13.25, 45.0 SPAN_OVERHEAD = 43.7 # per span, MEASURED, FINDINGS 30 CYC_PX_ROWLIN = 9.152 # per pixel, MEASURED, FINDINGS 30 SPAN_UNIT_PX = 24 # 12 registers of movem.l, one chain unit SPAN_BYTES_PX = 2 # word-expanded: 1 pixel = 1 word SPAN_HDR = 6 # u32 GVRAM address + u16 jump displacement def span_px(npix): # a span is a whole number of units return -(-npix // SPAN_UNIT_PX) * SPAN_UNIT_PX ap = argparse.ArgumentParser() ap.add_argument("container", nargs="?", default="tmp/rc_fr_singe_sasi_rcprofile.dlx") ap.add_argument("--bus", type=float, required=True, help="REQUIRED. There is no default: the delivery rate is a property of the medium and this project has never measured it. FINDINGS 42.1 -- the figure this tool used to default to was a user-supplied '4 Mbps' with no provenance, was a tenth of SCSI-1's asynchronous rating, and was never a bus measurement at all. A default let every table in FINDINGS 30-49 be scored against it without anyone restating it. Pass one explicitly.") ap.add_argument("--fps", type=float, default=12.0) a = ap.parse_args() if not os.path.exists(a.container): sys.exit(f"missing {a.container}") BYTE_BUD = (a.bus - AUDIO_KBPS) * 1024 / a.fps d = DLX(a.container) BLK_C = {1: C_V1, 2: C_V4, 3: C_RAW} BLK_B = {1: 1, 2: 4, 3: 16} rows = [] for f in range(d.nframes): mode = d.modes(f) g = mode.reshape(-1, 4) allskip = (g == 0).all(1) base = allskip.sum() * 4 * C_SKIP_CLUSTERED mm = g[~allskip] base += (mm == 0).sum() * C_SKIP_MIXED for k, c in BLK_C.items(): base += (mm == k).sum() * c base_b = d.mode_bytes + sum(BLK_B.get(int(x), 0) for x in mode) m = mode.reshape(d.nby, d.nbx) cand = [] for by in range(d.nby): dirty = m[by] != 0 i = 0 while i < d.nbx: if not dirty[i]: i += 1 continue j = i while j < d.nbx and dirty[j]: j += 1 L = j - i cur_c = sum(BLK_C[int(b)] for b in m[by][i:j]) cur_b = sum(BLK_B[int(b)] for b in m[by][i:j]) sp = span_px(4 * L) # padded to the chain's 24-pixel unit span_c = 4 * (SPAN_OVERHEAD + sp * CYC_PX_ROWLIN) span_b = 4 * (SPAN_HDR + sp * SPAN_BYTES_PX) if span_c < cur_c: cand.append((cur_c - span_c, span_b - cur_b, L)) i = j cand.sort(key=lambda s: -(s[0] / max(s[1], 1))) # best cycles per byte cyc, byt, taken = base, base_b, 0 for dc, db, L in cand: if byt + db <= BYTE_BUD: cyc -= dc; byt += db; taken += 1 rows.append((base, cyc, base_b, byt, len(cand), taken)) base, new, bb, nb, ncand, ntaken = map(np.array, list(zip(*rows))) pc = lambda v: 100 * v / FRAME_CYC print(f"{a.container}: {d.nframes} frames") print(f"bus {a.bus:.0f} KB/s - {AUDIO_KBPS} audio -> {BYTE_BUD:,.0f} B/frame " f"at {a.fps:g}fps\n") print(f"{'':<26}{'today':>12}{'+ literal spans':>18}") for label, fn in (("median frame", np.median), ("p90 frame", lambda v: np.percentile(v, 90)), ("worst frame", np.max)): print(f" {label:<24}{pc(fn(base)):>11.1f}%{pc(fn(new)):>17.1f}%") print(f" {'frames missing budget':<24}{int((base>FRAME_CYC).sum()):>8}/{d.nframes}" f"{int((new>FRAME_CYC).sum()):>14}/{d.nframes}") print(f" {'bitrate':<24}{bb.mean()*a.fps/1024:>10.1f} KB/s" f"{nb.mean()*a.fps/1024:>13.1f} KB/s") print(f"\nspans taken: {ntaken.sum()} of {ncand.sum()} candidate runs " f"({100*ntaken.sum()/max(ncand.sum(),1):.0f}%) -- the rest priced out by the bus") brk = next(L for L in range(1, 65) if 4*(SPAN_OVERHEAD + span_px(4*L)*CYC_PX_ROWLIN) < L*C_V1) print(f"\nspan cost MEASURED (FINDINGS 30): {SPAN_OVERHEAD:.1f}/span + " f"{CYC_PX_ROWLIN:.3f}/pixel, {SPAN_UNIT_PX}-pixel units.") print(f"a run of L blocks beats all-V1 from L={brk} blocks up " f"({4*(SPAN_OVERHEAD + span_px(4*brk)*CYC_PX_ROWLIN)/brk:.0f} vs {C_V1:.0f} " f"cycles/block); the floor at a full row is " f"{4*(SPAN_OVERHEAD + span_px(256)*CYC_PX_ROWLIN)/64:.0f}.") print("The mode maps are NOT re-optimised, so this is a lower bound on a " "cost-aware encoder.") # FINDINGS 28.5 said a scene cut cannot fit at 12fps: the cheapest full redraw # the codec's mode set allows is all-V1 at 110.5% of budget. 29.4 reopened that # on derived span costs; this is the same arithmetic on measured ones. Mix a # fraction x of a 100%-changed frame as full-row spans, V1 for the rest. NB = d.nb row_c = 4 * (SPAN_OVERHEAD + span_px(4 * d.nbx) * CYC_PX_ROWLIN) / d.nbx row_b = 4 * (SPAN_HDR + span_px(4 * d.nbx) * SPAN_BYTES_PX) / d.nbx x_cpu = (NB * C_V1 - FRAME_CYC) / (NB * (C_V1 - row_c)) x_bus = (BYTE_BUD - d.mode_bytes - NB * BLK_B[1]) / (NB * (row_b - BLK_B[1])) print(f"\nscene cut (100% of blocks change), spans at full row width " f"({row_c:.0f} cyc, {row_b:.1f} B per block):") print(f" all-V1 costs {100*NB*C_V1/FRAME_CYC:.1f}% of the frame -- FINDINGS 28.5") print(f" CPU needs x >= {x_cpu:.3f} of the frame as spans; " f"the bus allows x <= {x_bus:.3f}") print(" " + ("the interval is NOT empty: a cut fits at 12fps (FINDINGS 29.4 holds)" if x_cpu <= x_bus else "the interval IS empty: a cut does not fit (FINDINGS 28.5 stands)"))