Files
Dragon-s-Lair-X68k/tools/analysis/11_cpu_budget.py
T
prosolis ed353d24a9 Budget both profiles against both clocks: the CPU limit is the clock, not the profile
11_cpu_budget.py takes --machine. Clocks confirmed from MAME 0.277
x68k.cpp:1133/1194/1200, not recalled: x68000 AND x68ksupr are both
40_MHz_XTAL/4 = 10 MHz; only the XVI is faster at 33.33_MHz_XTAL/2.

                sasi            scsi
  stock 10MHz   31% miss        42% miss
  XVI 16.7MHz   0% miss         0% miss

sasi is the cheaper profile but it does not fit either at 10 MHz. The XVI
column is headroom, not a target: the profiles are an I/O-bandwidth axis and
say nothing about CPU, and the locked target CPU is a stock 10 MHz 68000 for
both of them. So both profiles have to fit the same 833,333-cycle budget, and
the cycle ceiling has to be enforced in the encoder regardless of which one
ships.

Model comparisons are now gated to the clock and framerate they were stated
at: quoting 24.5's 76.6% or the stock-machine 68000 timings against an XVI
budget compares a model to a measurement of a different machine.

Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6
2026-08-23 15:13:34 -07:00

128 lines
6.1 KiB
Python

#!/usr/bin/env python3
"""Per-frame CPU cost of the real decoder, from MEASURED per-mode block costs.
python3 tools/analysis/11_cpu_budget.py [container.dlx]
FINDINGS 24.5 priced the display path as "76.6% of a 12fps frame x the non-SKIP
block fraction", i.e. every non-SKIP block costs the same. It does not: the four
block modes were measured separately on the 68000 (synthetic single-mode frames,
tools/bench/prep_dlx.py) and V4 costs 1.5x V1. Since V4 is roughly half of all
non-SKIP blocks on hard content, the old model runs ~1.8x optimistic exactly
where it matters.
This applies the measured costs to a real container's mode histograms and
reports what fraction of frames actually fit 833,333 cycles.
Costs are MEASURED (tools/bench/decode.lua), cross-checked against hand-derived
MC68000 timings in FINDINGS 28.4. They are instruction cycles against
zero-wait-state memory, so like every figure in this project since FINDINGS 24
they are a LOWER BOUND -- real GVRAM stalls the CPU.
"""
import sys, os, argparse
sys.path.insert(0, "tools/encoder")
import numpy as np
from dlx import DLX
# Machine clocks, confirmed from MAME 0.277 src/mame/sharp/x68k.cpp:1133/1194/
# 1200 -- not recalled. x68000 and x68ksupr are BOTH 40_MHz_XTAL/4 = 10 MHz;
# only the XVI is faster, at 33.33_MHz_XTAL/2. So "has SCSI" and "has a faster
# CPU" are different sets of machines: the Super has SCSI at 10 MHz.
CLOCKS = {"stock": 10.0, "super": 10.0, "xvi": 33.33 / 2, "x68030": 25.0}
FPS = 12
# cycles per block, measured on the emulated 68000 (synthetic single-mode frames)
C_V1, C_V4, C_RAW = 299.9, 448.2, 400.4
C_SKIP_FAST = 53.0 / 4 # all-SKIP header byte: one tst.b for 4
C_SKIP_MIXED = 45.0 # a SKIP block inside a mixed byte
ap = argparse.ArgumentParser()
ap.add_argument("container", nargs="?",
default="tmp/rc_fr_singe_sasi_rcprofile.dlx")
ap.add_argument("--machine", default="stock", choices=list(CLOCKS),
help="which X68000's clock to budget against (default stock)")
ap.add_argument("--fps", type=float, default=FPS)
a = ap.parse_args()
CPUHZ = CLOCKS[a.machine] * 1e6
FPS = a.fps
FRAME = CPUHZ / FPS
if not os.path.exists(a.container):
sys.exit(f"missing {a.container}")
d = DLX(a.container)
def cycles(mode):
g = mode.reshape(-1, 4) # one header byte = four blocks
allskip = (g == 0).all(1)
c = allskip.sum() * 4 * C_SKIP_FAST
m = g[~allskip]
c += (m == 0).sum() * C_SKIP_MIXED
c += (m == 1).sum() * C_V1
c += (m == 2).sum() * C_V4
c += (m == 3).sum() * C_RAW
return c
modes = [d.modes(f) for f in range(d.nframes)]
cyc = np.array([cycles(m) for m in modes])
pct = 100 * cyc / FRAME
ns = np.array([100 * (m != 0).mean() for m in modes])
print(f"{a.container}: {d.nframes} frames, {d.nb} blocks/frame")
print(f"budget: {a.machine} @ {CLOCKS[a.machine]:.2f} MHz, {FPS:g} fps "
f"-> {FRAME:,.0f} cycles/frame")
if a.machine != "stock":
print(" (derived: scaled by clock from cycles measured on the 10 MHz core.\n"
" MAME 0.277 marks x68ksupr/x68kxvi/x68030 MACHINE_NOT_WORKING, so\n"
" this is not measured on those machines and ignores any difference\n"
" in memory timing.)")
print(f"measured block costs: SKIP {C_SKIP_FAST*4:.0f}/4 (clustered) "
f"{C_SKIP_MIXED:.0f} (mixed) V1 {C_V1:.0f} V4 {C_V4:.0f} RAW {C_RAW:.0f} cycles\n")
# --- validation against the four real frames timed on the 68000. These
# timings belong to ONE container; quoting them against any other would be
# comparing a model of this stream to a measurement of a different one.
TIMED = "tmp/rc_fr_singe_sasi_rcprofile.dlx"
TIMED_FRAMES = (("min non-SKIP", 15.4, 31.5), ("median", 48.1, 73.8),
("p90", 82.5, 116.4), ("max non-SKIP", 100.0, 135.8))
if (os.path.abspath(a.container) == os.path.abspath(TIMED)
and a.machine == "stock" and a.fps == 12):
print("model vs the frames actually timed on the 68000:")
for label, frac, meas in TIMED_FRAMES:
i = int(np.argmin(abs(ns - frac)))
print(f" {label:<14} non-SKIP {ns[i]:5.1f}% model {pct[i]:6.1f}% "
f"measured {meas:5.1f}% error {pct[i]-meas:+.1f} pt")
else:
print(f"(no 68000 timings for this container/machine -- the model was "
f"validated to\n within 1 pt on {TIMED} at stock/12fps;\n"
f" run tools/bench/decode.lua to time another container)")
print(f"\nper-frame cost, % of a {FPS:g}fps frame budget:")
print(f" measured-cost model: median {np.median(pct):5.1f} "
f"p90 {np.percentile(pct,90):5.1f} max {pct.max():5.1f}")
if a.machine == "stock" and a.fps == 12:
# 24.5's 76.6% is a 10 MHz / 12 fps figure; quoting it at another clock or
# framerate would be comparing against a model that was never stated there.
old = 76.6 * ns / 100
print(f" FINDINGS 24.5 model: median {np.median(old):5.1f} "
f"p90 {np.percentile(old,90):5.1f} max {old.max():5.1f} "
f"(optimistic by {np.median(pct)/np.median(old):.2f}x at the median)")
miss = pct > 100
print(f"\nframes that do NOT fit {FRAME:,.0f} cycles: {miss.sum()}/{d.nframes} "
f"({100*miss.mean():.0f}%)")
print(f" sustainable framerate if EVERY frame must fit: "
f"{CPUHZ/cyc.max():.1f} fps; at the mean frame {CPUHZ/cyc.mean():.1f} fps")
if miss.any():
print(f" worst {pct.max():.1f}% -- {(pct.max()-100)/100*1000/FPS:.0f} ms late "
f"on an {1000/FPS:.0f} ms frame")
print(f" the budget is first missed at {ns[miss].min():.1f}% non-SKIP blocks")
# Where do the cycles go? This is what a cost-aware mode decision would act on.
tot = np.array([[(m == k).sum() for k in range(4)] for m in modes]).sum(0)
spend = tot * np.array([C_SKIP_MIXED, C_V1, C_V4, C_RAW])
print(f"\nwhere the cycles go, over the whole window:")
for k, n in enumerate(("SKIP", "V1", "V4", "RAW")):
print(f" {n:<5} {100*tot[k]/tot.sum():5.1f}% of blocks "
f"{100*spend[k]/spend.sum():5.1f}% of the cycles")
print(f"\nV4 is {C_V4/C_V1:.2f}x a V1 block for {4}x the payload bytes -- the mode "
f"decision\nin vq_hybrid.py charges it the bytes but not the cycles.")