11_cpu_budget.py takes --machine. Clocks confirmed from MAME 0.277
x68k.cpp:1133/1194/1200, not recalled: x68000 AND x68ksupr are both
40_MHz_XTAL/4 = 10 MHz; only the XVI is faster at 33.33_MHz_XTAL/2.
sasi scsi
stock 10MHz 31% miss 42% miss
XVI 16.7MHz 0% miss 0% miss
sasi is the cheaper profile but it does not fit either at 10 MHz. The XVI
column is headroom, not a target: the profiles are an I/O-bandwidth axis and
say nothing about CPU, and the locked target CPU is a stock 10 MHz 68000 for
both of them. So both profiles have to fit the same 833,333-cycle budget, and
the cycle ceiling has to be enforced in the encoder regardless of which one
ships.
Model comparisons are now gated to the clock and framerate they were stated
at: quoting 24.5's 76.6% or the stock-machine 68000 timings against an XVI
budget compares a model to a measurement of a different machine.
Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6
128 lines
6.1 KiB
Python
128 lines
6.1 KiB
Python
#!/usr/bin/env python3
|
|
"""Per-frame CPU cost of the real decoder, from MEASURED per-mode block costs.
|
|
|
|
python3 tools/analysis/11_cpu_budget.py [container.dlx]
|
|
|
|
FINDINGS 24.5 priced the display path as "76.6% of a 12fps frame x the non-SKIP
|
|
block fraction", i.e. every non-SKIP block costs the same. It does not: the four
|
|
block modes were measured separately on the 68000 (synthetic single-mode frames,
|
|
tools/bench/prep_dlx.py) and V4 costs 1.5x V1. Since V4 is roughly half of all
|
|
non-SKIP blocks on hard content, the old model runs ~1.8x optimistic exactly
|
|
where it matters.
|
|
|
|
This applies the measured costs to a real container's mode histograms and
|
|
reports what fraction of frames actually fit 833,333 cycles.
|
|
|
|
Costs are MEASURED (tools/bench/decode.lua), cross-checked against hand-derived
|
|
MC68000 timings in FINDINGS 28.4. They are instruction cycles against
|
|
zero-wait-state memory, so like every figure in this project since FINDINGS 24
|
|
they are a LOWER BOUND -- real GVRAM stalls the CPU.
|
|
"""
|
|
import sys, os, argparse
|
|
sys.path.insert(0, "tools/encoder")
|
|
import numpy as np
|
|
from dlx import DLX
|
|
|
|
# Machine clocks, confirmed from MAME 0.277 src/mame/sharp/x68k.cpp:1133/1194/
|
|
# 1200 -- not recalled. x68000 and x68ksupr are BOTH 40_MHz_XTAL/4 = 10 MHz;
|
|
# only the XVI is faster, at 33.33_MHz_XTAL/2. So "has SCSI" and "has a faster
|
|
# CPU" are different sets of machines: the Super has SCSI at 10 MHz.
|
|
CLOCKS = {"stock": 10.0, "super": 10.0, "xvi": 33.33 / 2, "x68030": 25.0}
|
|
FPS = 12
|
|
|
|
# cycles per block, measured on the emulated 68000 (synthetic single-mode frames)
|
|
C_V1, C_V4, C_RAW = 299.9, 448.2, 400.4
|
|
C_SKIP_FAST = 53.0 / 4 # all-SKIP header byte: one tst.b for 4
|
|
C_SKIP_MIXED = 45.0 # a SKIP block inside a mixed byte
|
|
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("container", nargs="?",
|
|
default="tmp/rc_fr_singe_sasi_rcprofile.dlx")
|
|
ap.add_argument("--machine", default="stock", choices=list(CLOCKS),
|
|
help="which X68000's clock to budget against (default stock)")
|
|
ap.add_argument("--fps", type=float, default=FPS)
|
|
a = ap.parse_args()
|
|
CPUHZ = CLOCKS[a.machine] * 1e6
|
|
FPS = a.fps
|
|
FRAME = CPUHZ / FPS
|
|
if not os.path.exists(a.container):
|
|
sys.exit(f"missing {a.container}")
|
|
|
|
d = DLX(a.container)
|
|
|
|
def cycles(mode):
|
|
g = mode.reshape(-1, 4) # one header byte = four blocks
|
|
allskip = (g == 0).all(1)
|
|
c = allskip.sum() * 4 * C_SKIP_FAST
|
|
m = g[~allskip]
|
|
c += (m == 0).sum() * C_SKIP_MIXED
|
|
c += (m == 1).sum() * C_V1
|
|
c += (m == 2).sum() * C_V4
|
|
c += (m == 3).sum() * C_RAW
|
|
return c
|
|
|
|
modes = [d.modes(f) for f in range(d.nframes)]
|
|
cyc = np.array([cycles(m) for m in modes])
|
|
pct = 100 * cyc / FRAME
|
|
ns = np.array([100 * (m != 0).mean() for m in modes])
|
|
|
|
print(f"{a.container}: {d.nframes} frames, {d.nb} blocks/frame")
|
|
print(f"budget: {a.machine} @ {CLOCKS[a.machine]:.2f} MHz, {FPS:g} fps "
|
|
f"-> {FRAME:,.0f} cycles/frame")
|
|
if a.machine != "stock":
|
|
print(" (derived: scaled by clock from cycles measured on the 10 MHz core.\n"
|
|
" MAME 0.277 marks x68ksupr/x68kxvi/x68030 MACHINE_NOT_WORKING, so\n"
|
|
" this is not measured on those machines and ignores any difference\n"
|
|
" in memory timing.)")
|
|
print(f"measured block costs: SKIP {C_SKIP_FAST*4:.0f}/4 (clustered) "
|
|
f"{C_SKIP_MIXED:.0f} (mixed) V1 {C_V1:.0f} V4 {C_V4:.0f} RAW {C_RAW:.0f} cycles\n")
|
|
|
|
# --- validation against the four real frames timed on the 68000. These
|
|
# timings belong to ONE container; quoting them against any other would be
|
|
# comparing a model of this stream to a measurement of a different one.
|
|
TIMED = "tmp/rc_fr_singe_sasi_rcprofile.dlx"
|
|
TIMED_FRAMES = (("min non-SKIP", 15.4, 31.5), ("median", 48.1, 73.8),
|
|
("p90", 82.5, 116.4), ("max non-SKIP", 100.0, 135.8))
|
|
if (os.path.abspath(a.container) == os.path.abspath(TIMED)
|
|
and a.machine == "stock" and a.fps == 12):
|
|
print("model vs the frames actually timed on the 68000:")
|
|
for label, frac, meas in TIMED_FRAMES:
|
|
i = int(np.argmin(abs(ns - frac)))
|
|
print(f" {label:<14} non-SKIP {ns[i]:5.1f}% model {pct[i]:6.1f}% "
|
|
f"measured {meas:5.1f}% error {pct[i]-meas:+.1f} pt")
|
|
else:
|
|
print(f"(no 68000 timings for this container/machine -- the model was "
|
|
f"validated to\n within 1 pt on {TIMED} at stock/12fps;\n"
|
|
f" run tools/bench/decode.lua to time another container)")
|
|
|
|
print(f"\nper-frame cost, % of a {FPS:g}fps frame budget:")
|
|
print(f" measured-cost model: median {np.median(pct):5.1f} "
|
|
f"p90 {np.percentile(pct,90):5.1f} max {pct.max():5.1f}")
|
|
if a.machine == "stock" and a.fps == 12:
|
|
# 24.5's 76.6% is a 10 MHz / 12 fps figure; quoting it at another clock or
|
|
# framerate would be comparing against a model that was never stated there.
|
|
old = 76.6 * ns / 100
|
|
print(f" FINDINGS 24.5 model: median {np.median(old):5.1f} "
|
|
f"p90 {np.percentile(old,90):5.1f} max {old.max():5.1f} "
|
|
f"(optimistic by {np.median(pct)/np.median(old):.2f}x at the median)")
|
|
|
|
miss = pct > 100
|
|
print(f"\nframes that do NOT fit {FRAME:,.0f} cycles: {miss.sum()}/{d.nframes} "
|
|
f"({100*miss.mean():.0f}%)")
|
|
print(f" sustainable framerate if EVERY frame must fit: "
|
|
f"{CPUHZ/cyc.max():.1f} fps; at the mean frame {CPUHZ/cyc.mean():.1f} fps")
|
|
if miss.any():
|
|
print(f" worst {pct.max():.1f}% -- {(pct.max()-100)/100*1000/FPS:.0f} ms late "
|
|
f"on an {1000/FPS:.0f} ms frame")
|
|
print(f" the budget is first missed at {ns[miss].min():.1f}% non-SKIP blocks")
|
|
|
|
# Where do the cycles go? This is what a cost-aware mode decision would act on.
|
|
tot = np.array([[(m == k).sum() for k in range(4)] for m in modes]).sum(0)
|
|
spend = tot * np.array([C_SKIP_MIXED, C_V1, C_V4, C_RAW])
|
|
print(f"\nwhere the cycles go, over the whole window:")
|
|
for k, n in enumerate(("SKIP", "V1", "V4", "RAW")):
|
|
print(f" {n:<5} {100*tot[k]/tot.sum():5.1f}% of blocks "
|
|
f"{100*spend[k]/spend.sum():5.1f}% of the cycles")
|
|
print(f"\nV4 is {C_V4/C_V1:.2f}x a V1 block for {4}x the payload bytes -- the mode "
|
|
f"decision\nin vq_hybrid.py charges it the bytes but not the cycles.")
|