#!/usr/bin/env python3 """What do the spans the ENCODER actually emitted cost, and what do they buy? python3 tools/analysis/17_span_delivered.py a.dlx [b.dlx ...] [--bus 488] Every span figure before this one -- FINDINGS 29 through 40, and tools/analysis/12 and 14 -- was scored by SIMULATING span selection over mode maps that were chosen without spans available. FINDINGS 39.3 flagged that as a lower bound on what a span-aware encoder would find, and docs/STATUS.md's item 2 asks for the figures to be re-run "against a container the encoder actually emits with spans in it". This is that script: it reads the span section out of a DLX3 container and prices exactly those spans, with no selection model at all. THE MODEL IS 14_dmac_chain.py's, deliberately unchanged, so the columns are comparable: frame clocks = block decode + span painting + disk DMA additive, because a 68000 has no cache and a two-word prefetch queue and stalls the moment another master takes the bus (FINDINGS 38.3). Block cost is vq_hybrid.cycles(), which reads a spanned block as SKIP -- correct, because the span section is what paints it, and its cost is the second term. The span term is the MEASURED v7 fit (FINDINGS 40), and as of session 12 that fit is confirmed inside src/player/decode.s itself rather than only in tools/bench/blit.s: the synthetic all-SPAN anchors of tools/bench/prep_dlx.py reproduce it to 0.23% on both emulators (FINDINGS 41.3). """ import argparse, os, sys sys.path.insert(0, "tools/encoder") sys.path.insert(0, "tools/analysis") import numpy as np import vq_hybrid as H import spans as SP import buscost as B from dlx import DLX FRAME_CYC = 833333.0 AUDIO_KBPS = 7.8 ap = argparse.ArgumentParser() ap.add_argument("containers", nargs="+") ap.add_argument("--bus", type=float, default=488.0, help="SCSI pipe, KB/s") ap.add_argument("--fps", type=float, default=12.0) ap.add_argument("--disk-clk-word", type=float, default=8.0, help="clocks the SCSI DMA steals per word (FINDINGS 39.7 " "brackets it at 5..12; 8 is the midpoint)") a = ap.parse_args() def score(path): d = DLX(path) rows = [] for f in range(d.nframes): mode = d.modes(f) sp, _ = d.spans(f) _, n = d.frames[f] blk = H.cycles(mode) spc = sum(SP.clocks(len(p)) for _, _, p in sp) disk = n / 2.0 * a.disk_clk_word rows.append((blk, spc, disk, n, len(sp), sum(len(p) for _, _, p in sp))) return d, np.array(rows).T print(f"{'container':<34}{'KB/s':>8}{'spans':>9}{'span px':>9}" f"{'median':>9}{'worst':>9}{'over':>9}") print(f"{'':<34}{'':>8}{'/frame':>9}{'%':>9}" f"{'% frame':>9}{'% frame':>9}{'budget':>9}") for path in a.containers: if not os.path.exists(path): print(f"{path:<34} missing"); continue d, r = score(path) blk, spc, disk, byt, nsp, spx = r tot = blk + spc + disk kbps = byt.mean() * a.fps / 1024 + AUDIO_KBPS print(f"{os.path.basename(path):<34}{kbps:>8.1f}{nsp.mean():>9.0f}" f"{100*spx.mean()/(d.W*d.H):>9.1f}" f"{100*np.median(tot)/FRAME_CYC:>9.1f}" f"{100*tot.max()/FRAME_CYC:>9.1f}" f"{int((tot > FRAME_CYC).sum()):>6}/{d.nframes:<3}") print(f"\n ADDITIVE: frame = block decode + span painting + disk DMA, the model" f"\n of 14_dmac_chain.py. Disk debited at {a.disk_clk_word:g} clocks/word " f"over the\n container's own byte count; CPU budget {FRAME_CYC:,.0f} " f"clocks at {a.fps:g} fps.") # The decomposition is the point: a span moves work out of the block loop and # into the span section, and it pays for it in bytes -- which the disk term # then charges back. A design that only counted the CPU would show a win that # the I/O it created takes away again (docs/FINDINGS.md 33). print(f"\nWHERE EACH FRAME'S CLOCKS GO, mean over the container") print(f" {'container':<34}{'blocks':>12}{'spans':>12}{'disk':>12}{'total':>12}") for path in a.containers: if not os.path.exists(path): continue d, r = score(path) blk, spc, disk = r[0], r[1], r[2] print(f" {os.path.basename(path):<34}{blk.mean():>12,.0f}{spc.mean():>12,.0f}" f"{disk.mean():>12,.0f}{(blk+spc+disk).mean():>12,.0f}")