From ed353d24a9a0f656cbdcc88670582e8f5f5a0a27 Mon Sep 17 00:00:00 2001 From: prosolis <5590409+prosolis@users.noreply.github.com> Date: Sun, 23 Aug 2026 15:12:50 -0700 Subject: [PATCH] Budget both profiles against both clocks: the CPU limit is the clock, not the profile 11_cpu_budget.py takes --machine. Clocks confirmed from MAME 0.277 x68k.cpp:1133/1194/1200, not recalled: x68000 AND x68ksupr are both 40_MHz_XTAL/4 = 10 MHz; only the XVI is faster at 33.33_MHz_XTAL/2. sasi scsi stock 10MHz 31% miss 42% miss XVI 16.7MHz 0% miss 0% miss sasi is the cheaper profile but it does not fit either at 10 MHz. The XVI column is headroom, not a target: the profiles are an I/O-bandwidth axis and say nothing about CPU, and the locked target CPU is a stock 10 MHz 68000 for both of them. So both profiles have to fit the same 833,333-cycle budget, and the cycle ceiling has to be enforced in the encoder regardless of which one ships. Model comparisons are now gated to the clock and framerate they were stated at: quoting 24.5's 76.6% or the stock-machine 68000 timings against an XVI budget compares a model to a measurement of a different machine. Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6 --- tools/analysis/11_cpu_budget.py | 45 +++++++++++++++++++++++++-------- 1 file changed, 34 insertions(+), 11 deletions(-) diff --git a/tools/analysis/11_cpu_budget.py b/tools/analysis/11_cpu_budget.py index 008c2ee..7c6910d 100644 --- a/tools/analysis/11_cpu_budget.py +++ b/tools/analysis/11_cpu_budget.py @@ -23,9 +23,12 @@ sys.path.insert(0, "tools/encoder") import numpy as np from dlx import DLX -CPUHZ = 10_000_000 +# Machine clocks, confirmed from MAME 0.277 src/mame/sharp/x68k.cpp:1133/1194/ +# 1200 -- not recalled. x68000 and x68ksupr are BOTH 40_MHz_XTAL/4 = 10 MHz; +# only the XVI is faster, at 33.33_MHz_XTAL/2. So "has SCSI" and "has a faster +# CPU" are different sets of machines: the Super has SCSI at 10 MHz. +CLOCKS = {"stock": 10.0, "super": 10.0, "xvi": 33.33 / 2, "x68030": 25.0} FPS = 12 -FRAME = CPUHZ / FPS # 833,333 cycles # cycles per block, measured on the emulated 68000 (synthetic single-mode frames) C_V1, C_V4, C_RAW = 299.9, 448.2, 400.4 @@ -35,7 +38,13 @@ C_SKIP_MIXED = 45.0 # a SKIP block inside a mixed byte ap = argparse.ArgumentParser() ap.add_argument("container", nargs="?", default="tmp/rc_fr_singe_sasi_rcprofile.dlx") +ap.add_argument("--machine", default="stock", choices=list(CLOCKS), + help="which X68000's clock to budget against (default stock)") +ap.add_argument("--fps", type=float, default=FPS) a = ap.parse_args() +CPUHZ = CLOCKS[a.machine] * 1e6 +FPS = a.fps +FRAME = CPUHZ / FPS if not os.path.exists(a.container): sys.exit(f"missing {a.container}") @@ -58,6 +67,13 @@ pct = 100 * cyc / FRAME ns = np.array([100 * (m != 0).mean() for m in modes]) print(f"{a.container}: {d.nframes} frames, {d.nb} blocks/frame") +print(f"budget: {a.machine} @ {CLOCKS[a.machine]:.2f} MHz, {FPS:g} fps " + f"-> {FRAME:,.0f} cycles/frame") +if a.machine != "stock": + print(" (derived: scaled by clock from cycles measured on the 10 MHz core.\n" + " MAME 0.277 marks x68ksupr/x68kxvi/x68030 MACHINE_NOT_WORKING, so\n" + " this is not measured on those machines and ignores any difference\n" + " in memory timing.)") print(f"measured block costs: SKIP {C_SKIP_FAST*4:.0f}/4 (clustered) " f"{C_SKIP_MIXED:.0f} (mixed) V1 {C_V1:.0f} V4 {C_V4:.0f} RAW {C_RAW:.0f} cycles\n") @@ -67,27 +83,34 @@ print(f"measured block costs: SKIP {C_SKIP_FAST*4:.0f}/4 (clustered) " TIMED = "tmp/rc_fr_singe_sasi_rcprofile.dlx" TIMED_FRAMES = (("min non-SKIP", 15.4, 31.5), ("median", 48.1, 73.8), ("p90", 82.5, 116.4), ("max non-SKIP", 100.0, 135.8)) -if os.path.abspath(a.container) == os.path.abspath(TIMED): +if (os.path.abspath(a.container) == os.path.abspath(TIMED) + and a.machine == "stock" and a.fps == 12): print("model vs the frames actually timed on the 68000:") for label, frac, meas in TIMED_FRAMES: i = int(np.argmin(abs(ns - frac))) print(f" {label:<14} non-SKIP {ns[i]:5.1f}% model {pct[i]:6.1f}% " f"measured {meas:5.1f}% error {pct[i]-meas:+.1f} pt") else: - print(f"(no 68000 timings for this container -- the model was validated to " - f"within\n 1 pt on {TIMED}; run tools/bench/decode.lua to time this one)") + print(f"(no 68000 timings for this container/machine -- the model was " + f"validated to\n within 1 pt on {TIMED} at stock/12fps;\n" + f" run tools/bench/decode.lua to time another container)") -old = 76.6 * ns / 100 -print(f"\nper-frame cost, % of a {FPS}fps frame budget:") +print(f"\nper-frame cost, % of a {FPS:g}fps frame budget:") print(f" measured-cost model: median {np.median(pct):5.1f} " f"p90 {np.percentile(pct,90):5.1f} max {pct.max():5.1f}") -print(f" FINDINGS 24.5 model: median {np.median(old):5.1f} " - f"p90 {np.percentile(old,90):5.1f} max {old.max():5.1f} " - f"(optimistic by {np.median(pct)/np.median(old):.2f}x at the median)") +if a.machine == "stock" and a.fps == 12: + # 24.5's 76.6% is a 10 MHz / 12 fps figure; quoting it at another clock or + # framerate would be comparing against a model that was never stated there. + old = 76.6 * ns / 100 + print(f" FINDINGS 24.5 model: median {np.median(old):5.1f} " + f"p90 {np.percentile(old,90):5.1f} max {old.max():5.1f} " + f"(optimistic by {np.median(pct)/np.median(old):.2f}x at the median)") miss = pct > 100 -print(f"\nframes that do NOT fit 833,333 cycles: {miss.sum()}/{d.nframes} " +print(f"\nframes that do NOT fit {FRAME:,.0f} cycles: {miss.sum()}/{d.nframes} " f"({100*miss.mean():.0f}%)") +print(f" sustainable framerate if EVERY frame must fit: " + f"{CPUHZ/cyc.max():.1f} fps; at the mean frame {CPUHZ/cyc.mean():.1f} fps") if miss.any(): print(f" worst {pct.max():.1f}% -- {(pct.max()-100)/100*1000/FPS:.0f} ms late " f"on an {1000/FPS:.0f} ms frame")