#!/usr/bin/env python3 """Per-frame CPU cost of the real decoder, from MEASURED per-mode block costs. python3 tools/analysis/11_cpu_budget.py [container.dlx] FINDINGS 24.5 priced the display path as "76.6% of a 12fps frame x the non-SKIP block fraction", i.e. every non-SKIP block costs the same. It does not: the four block modes were measured separately on the 68000 (synthetic single-mode frames, tools/bench/prep_dlx.py) and V4 costs 1.5x V1. Since V4 is roughly half of all non-SKIP blocks on hard content, the old model runs ~1.8x optimistic exactly where it matters. This applies the measured costs to a real container's mode histograms and reports what fraction of frames actually fit 833,333 cycles. Costs are MEASURED (tools/bench/decode.lua), cross-checked against hand-derived MC68000 timings in FINDINGS 28.4. They are instruction cycles against zero-wait-state memory, so like every figure in this project since FINDINGS 24 they are a LOWER BOUND -- real GVRAM stalls the CPU. """ import sys, os, argparse sys.path.insert(0, "tools/encoder") import numpy as np from dlx import DLX import vq_hybrid as H import ratectl as RC sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import buscost as B # The audio byte rate is now DERIVED, not restated: 15.6 kHz mono MSM6258V is # 15,625 4-bit samples/s, two to a byte. RC.AUDIO_KBPS's 7.8 is that figure in # DECIMAL kB, and was being multiplied by 1024 here -- a 2.4% overstatement, # harmless, but it hid which unit the constant was in. RC_AUDIO_BPS = B.ADPCM_BYTES_PER_S # Machine clocks, confirmed from MAME 0.277 src/mame/sharp/x68k.cpp:1133/1194/ # 1200 -- not recalled. x68000 and x68ksupr are BOTH 40_MHz_XTAL/4 = 10 MHz; # only the XVI is faster, at 33.33_MHz_XTAL/2. So "has SCSI" and "has a faster # CPU" are different sets of machines: the Super has SCSI at 10 MHz. CLOCKS = {"stock": 10.0, "super": 10.0, "xvi": 33.33 / 2, "x68030": 25.0} FPS = 12 # Cycles per block, measured on the emulated 68000 (synthetic single-mode # frames). Defined in tools/encoder/vq_hybrid.py, which is where the mode # decision needs them too -- one copy, not two, so a re-measurement cannot # leave the encoder and the scorer disagreeing. C_V1, C_V4, C_RAW = H.C_V1, H.C_V4, H.C_RAW C_SKIP_FAST, C_SKIP_MIXED = H.C_SKIP_CLUSTERED, H.C_SKIP_MIXED cycles = H.cycles ap = argparse.ArgumentParser() ap.add_argument("container", nargs="?", default="tmp/rc_fr_singe_sasi_rcprofile.dlx") ap.add_argument("--machine", default="stock", choices=list(CLOCKS), help="which X68000's clock to budget against (default stock)") ap.add_argument("--fps", type=float, default=FPS) # FINDINGS 35: the frame budget has never had the disk in it. The bitstream has # to be moved off SCSI into the ring buffer, and on this machine that costs CPU # whether it is DMA (the HD63450 cycle-steals) or PIO (the 68000 moves every # byte). Default ON, because scoring a decoder against a budget that assumes the # data arrives for free is exactly the mistake 35 was raised to stop. ap.add_argument("--io", default="dma", choices=["dma", "pio", "none"], help="how the bitstream reaches RAM (default dma)") ap.add_argument("--dma-clocks-per-word", type=float, default=8.0, help="HD63450 cycle-steal. ESTIMATE from FINDINGS 5, NEVER " "MEASURED, and the most load-bearing unmeasured number " "in the project (FINDINGS 35.3)") ap.add_argument("--dma-clocks-per-byte", type=float, default=5.0, help="what the SCSI DMA costs per DELIVERED BYTE. The MB89352 " "is an 8-bit port, so the DMAC pays per byte and the " "per-word denominator of FINDINGS 5/39.7 was half the " "real debit (FINDINGS 43). 5 = single-address, bus held, " "no drive wait; 9 = dual-address") ap.add_argument("--pio-clocks-per-byte", type=float, default=12.0, help="hand-derived floor for a 68000 register-to-RAM copy") # Audio is NOT the disk, and charging it the disk's rate was charging it the # favourable side of an open question. tools/analysis/21_iplrom_dmac.py reads # the IPL ROM's own HD63450 setup: channel 3 is dual address, 8-bit port, cycle # steal WITHOUT hold, external request -- one full arbitration per byte, no # burst to amortise it over. 16 is the datasheet best case, 19 the worst. ap.add_argument("--adpcm-clocks-per-byte", type=float, default=B.ADPCM_CLK_BYTE_BEST, help="what an ADPCM byte costs. READ OUT OF THE IPL ROM's DMAC " "configuration (21_iplrom_dmac.py), not assumed: dual " "address + per-byte arbitration = 16 best, 19 worst. The " "audio stream always DMAs, whatever --io says about the " "disk") a = ap.parse_args() CPUHZ = CLOCKS[a.machine] * 1e6 FPS = a.fps FRAME = CPUHZ / FPS if not os.path.exists(a.container): sys.exit(f"missing {a.container}") d = DLX(a.container) # --- what the transfer costs, from the container's own byte rate vid_bps = sum(n + 4 for (_, n) in d.frames) / d.nframes * d.fps io_bps = vid_bps + RC_AUDIO_BPS aud_cycles_per_s = RC_AUDIO_BPS * a.adpcm_clocks_per_byte if a.io == "dma": io_cycles_per_s = vid_bps * a.dma_clocks_per_byte + aud_cycles_per_s elif a.io == "pio": io_cycles_per_s = vid_bps * a.pio_clocks_per_byte + aud_cycles_per_s else: io_cycles_per_s = 0.0 io_pct = 100 * io_cycles_per_s / CPUHZ aud_pct = 100 * aud_cycles_per_s / CPUHZ FRAME_NET = FRAME * (1 - io_pct / 100) modes = [d.modes(f) for f in range(d.nframes)] cyc = np.array([cycles(m) for m in modes]) pct = 100 * cyc / FRAME_NET ns = np.array([100 * (m != 0).mean() for m in modes]) print(f"{a.container}: {d.nframes} frames, {d.nb} blocks/frame") print(f"budget: {a.machine} @ {CLOCKS[a.machine]:.2f} MHz, {FPS:g} fps " f"-> {FRAME:,.0f} cycles/frame") print(f" I/O ({a.io}): {io_bps/1024:.1f} KB/s costs {io_pct:.1f}% of the CPU " f"-> {FRAME_NET:,.0f} cycles/frame left for decoding") if a.io != "none": print(f" video {vid_bps/1024:6.1f} KB/s x " f"{(a.dma_clocks_per_byte if a.io=='dma' else a.pio_clocks_per_byte):g}" f" clk/B = {io_pct-aud_pct:5.2f}% " f"(W: still open, ROADMAP B3 / FINDINGS 42.4)\n" f" audio {RC_AUDIO_BPS/1024:6.2f} KB/s x {a.adpcm_clocks_per_byte:g}" f" clk/B = {aud_pct:5.2f}% " f"(SETTLED: read out of the IPL ROM, FINDINGS 52)") if a.io == "dma": print(f" {a.dma_clocks_per_byte:g} clocks/BYTE, the MC68450 datasheet " f"floor for an 8-bit port (FINDINGS 43).\n It is not measured on " f"hardware; what IS settled is that the per-word denominator this\n" f" used before session 14 was physically impossible -- 2.5 " f"clocks/byte is below\n the 68000's 4-clock minimum bus cycle.") elif a.io == "none": print(" WARNING: --io none scores the decoder as if the disk were free. " "That is the\n premise FINDINGS 35 overturned; every 'N frames miss' " "figure before session 9\n was computed this way.") if a.machine != "stock": print(" (derived: scaled by clock from cycles measured on the 10 MHz core.\n" " MAME 0.277 marks x68ksupr/x68kxvi/x68030 MACHINE_NOT_WORKING, so\n" " this is not measured on those machines and ignores any difference\n" " in memory timing.)") print(f"measured block costs: SKIP {C_SKIP_FAST*4:.0f}/4 (clustered) " f"{C_SKIP_MIXED:.0f} (mixed) V1 {C_V1:.0f} V4 {C_V4:.0f} RAW {C_RAW:.0f} cycles\n") # --- validation against the four real frames timed on the 68000. These # timings belong to ONE container; quoting them against any other would be # comparing a model of this stream to a measurement of a different one. TIMED = "tmp/rc_fr_singe_sasi_rcprofile.dlx" TIMED_FRAMES = (("min non-SKIP", 15.4, 31.5), ("median", 48.1, 73.8), ("p90", 82.5, 116.4), ("max non-SKIP", 100.0, 135.8)) if (os.path.abspath(a.container) == os.path.abspath(TIMED) and a.machine == "stock" and a.fps == 12): print("model vs the frames actually timed on the 68000 " "(the model reads HIGH, and by more\n as the frame gets harder -- " "so a 'does not fit' from it is the safe direction):") for label, frac, meas in TIMED_FRAMES: i = int(np.argmin(abs(ns - frac))) print(f" {label:<14} non-SKIP {ns[i]:5.1f}% model {pct[i]:6.1f}% " f"measured {meas:5.1f}% error {pct[i]-meas:+.1f} pt") else: print(f"(no 68000 timings for this container/machine. The model is " f"validated against four\n frames timed on the 68000, and only on " f"{TIMED}\n at stock/12fps -- run it on that container to see the " f"errors, which are a few points\n CONSERVATIVE and grow with the " f"non-SKIP fraction. Run tools/bench/decode.lua to\n time another " f"container.)") print(f"\nper-frame cost, % of a {FPS:g}fps frame budget:") print(f" measured-cost model: median {np.median(pct):5.1f} " f"p90 {np.percentile(pct,90):5.1f} max {pct.max():5.1f}") if a.machine == "stock" and a.fps == 12: # 24.5's 76.6% is a 10 MHz / 12 fps figure; quoting it at another clock or # framerate would be comparing against a model that was never stated there. old = 76.6 * ns / 100 print(f" FINDINGS 24.5 model: median {np.median(old):5.1f} " f"p90 {np.percentile(old,90):5.1f} max {old.max():5.1f} " f"(optimistic by {np.median(pct)/np.median(old):.2f}x at the median)") miss = pct > 100 print(f"\nframes that do NOT fit {FRAME_NET:,.0f} cycles: {miss.sum()}/{d.nframes} " f"({100*miss.mean():.0f}%)") print(f" sustainable framerate if EVERY frame must fit: " f"{CPUHZ*(1-io_pct/100)/cyc.max():.1f} fps; at the mean frame " f"{CPUHZ*(1-io_pct/100)/cyc.mean():.1f} fps") if miss.any(): print(f" worst {pct.max():.1f}% -- {(pct.max()-100)/100*1000/FPS:.0f} ms late " f"on an {1000/FPS:.0f} ms frame") print(f" the budget is first missed at {ns[miss].min():.1f}% non-SKIP blocks") # Where do the cycles go? This is what a cost-aware mode decision would act on. tot = np.array([[(m == k).sum() for k in range(4)] for m in modes]).sum(0) spend = tot * np.array([C_SKIP_MIXED, C_V1, C_V4, C_RAW]) print(f"\nwhere the cycles go, over the whole window:") for k, n in enumerate(("SKIP", "V1", "V4", "RAW")): print(f" {n:<5} {100*tot[k]/tot.sum():5.1f}% of blocks " f"{100*spend[k]/spend.sum():5.1f}% of the cycles") print(f"\nV4 is {C_V4/C_V1:.2f}x a V1 block for {4}x the payload bytes. Since " f"session 8 the mode\ndecision charges it BOTH (decide(ctx, lam, mu), " f"FINDINGS 31), which is why V4 is now\nthe rarest non-SKIP mode here -- " f"a byte-rich profile buys its way out to RAW instead.")