#!/usr/bin/env python3 """Per-frame CPU cost of the real decoder, from MEASURED per-mode block costs. python3 tools/analysis/11_cpu_budget.py [container.dlx] FINDINGS 24.5 priced the display path as "76.6% of a 12fps frame x the non-SKIP block fraction", i.e. every non-SKIP block costs the same. It does not: the four block modes were measured separately on the 68000 (synthetic single-mode frames, tools/bench/prep_dlx.py) and V4 costs 1.5x V1. Since V4 is roughly half of all non-SKIP blocks on hard content, the old model runs ~1.8x optimistic exactly where it matters. This applies the measured costs to a real container's mode histograms and reports what fraction of frames actually fit 833,333 cycles. Costs are MEASURED (tools/bench/decode.lua), cross-checked against hand-derived MC68000 timings in FINDINGS 28.4. They are instruction cycles against zero-wait-state memory, so like every figure in this project since FINDINGS 24 they are a LOWER BOUND -- real GVRAM stalls the CPU. """ import sys, os, argparse sys.path.insert(0, "tools/encoder") import numpy as np from dlx import DLX CPUHZ = 10_000_000 FPS = 12 FRAME = CPUHZ / FPS # 833,333 cycles # cycles per block, measured on the emulated 68000 (synthetic single-mode frames) C_V1, C_V4, C_RAW = 299.9, 448.2, 400.4 C_SKIP_FAST = 53.0 / 4 # all-SKIP header byte: one tst.b for 4 C_SKIP_MIXED = 45.0 # a SKIP block inside a mixed byte ap = argparse.ArgumentParser() ap.add_argument("container", nargs="?", default="tmp/rc_fr_singe_sasi_rcprofile.dlx") a = ap.parse_args() if not os.path.exists(a.container): sys.exit(f"missing {a.container}") d = DLX(a.container) def cycles(mode): g = mode.reshape(-1, 4) # one header byte = four blocks allskip = (g == 0).all(1) c = allskip.sum() * 4 * C_SKIP_FAST m = g[~allskip] c += (m == 0).sum() * C_SKIP_MIXED c += (m == 1).sum() * C_V1 c += (m == 2).sum() * C_V4 c += (m == 3).sum() * C_RAW return c modes = [d.modes(f) for f in range(d.nframes)] cyc = np.array([cycles(m) for m in modes]) pct = 100 * cyc / FRAME ns = np.array([100 * (m != 0).mean() for m in modes]) print(f"{a.container}: {d.nframes} frames, {d.nb} blocks/frame") print(f"measured block costs: SKIP {C_SKIP_FAST*4:.0f}/4 (clustered) " f"{C_SKIP_MIXED:.0f} (mixed) V1 {C_V1:.0f} V4 {C_V4:.0f} RAW {C_RAW:.0f} cycles\n") # --- validation against the four real frames timed on the 68000. These # timings belong to ONE container; quoting them against any other would be # comparing a model of this stream to a measurement of a different one. TIMED = "tmp/rc_fr_singe_sasi_rcprofile.dlx" TIMED_FRAMES = (("min non-SKIP", 15.4, 31.5), ("median", 48.1, 73.8), ("p90", 82.5, 116.4), ("max non-SKIP", 100.0, 135.8)) if os.path.abspath(a.container) == os.path.abspath(TIMED): print("model vs the frames actually timed on the 68000:") for label, frac, meas in TIMED_FRAMES: i = int(np.argmin(abs(ns - frac))) print(f" {label:<14} non-SKIP {ns[i]:5.1f}% model {pct[i]:6.1f}% " f"measured {meas:5.1f}% error {pct[i]-meas:+.1f} pt") else: print(f"(no 68000 timings for this container -- the model was validated to " f"within\n 1 pt on {TIMED}; run tools/bench/decode.lua to time this one)") old = 76.6 * ns / 100 print(f"\nper-frame cost, % of a {FPS}fps frame budget:") print(f" measured-cost model: median {np.median(pct):5.1f} " f"p90 {np.percentile(pct,90):5.1f} max {pct.max():5.1f}") print(f" FINDINGS 24.5 model: median {np.median(old):5.1f} " f"p90 {np.percentile(old,90):5.1f} max {old.max():5.1f} " f"(optimistic by {np.median(pct)/np.median(old):.2f}x at the median)") miss = pct > 100 print(f"\nframes that do NOT fit 833,333 cycles: {miss.sum()}/{d.nframes} " f"({100*miss.mean():.0f}%)") if miss.any(): print(f" worst {pct.max():.1f}% -- {(pct.max()-100)/100*1000/FPS:.0f} ms late " f"on an {1000/FPS:.0f} ms frame") print(f" the budget is first missed at {ns[miss].min():.1f}% non-SKIP blocks") # Where do the cycles go? This is what a cost-aware mode decision would act on. tot = np.array([[(m == k).sum() for k in range(4)] for m in modes]).sum(0) spend = tot * np.array([C_SKIP_MIXED, C_V1, C_V4, C_RAW]) print(f"\nwhere the cycles go, over the whole window:") for k, n in enumerate(("SKIP", "V1", "V4", "RAW")): print(f" {n:<5} {100*tot[k]/tot.sum():5.1f}% of blocks " f"{100*spend[k]/spend.sum():5.1f}% of the cycles") print(f"\nV4 is {C_V4/C_V1:.2f}x a V1 block for {4}x the payload bytes -- the mode " f"decision\nin vq_hybrid.py charges it the bytes but not the cycles.")