Files
Dragon-s-Lair-X68k/tools/bench/decode.lua
T
prosolis e1aa26bb57 The 68000 decoder draws pixel-exact frames, and does not fit
src/player/decode.s parses DLX1 and decodes straight into GVRAM. Verified
pixel-exact over a 120-frame sequential run of the worst sustained window on
the disc -- all four block modes, full temporal recursion, so the last frame
is only right if all 120 were. In check.sh.

It costs a mean of 81.7% of a 12fps frame budget, and 31% of frames exceed
100% (42% at scsi). CPU is now the binding constraint. FINDINGS 28.

Three things that were believed and are not true:

- The dual-display-path plan of FINDINGS 24.5/25.6 is incoherent. The compose
  path needs a RAM copy of the previous reconstruction; the direct path's
  selling point is that it keeps none. Mixing them shows stale pixels on 70 of
  120 frames, worst frame 18.8% of the screen. Every coherent repair is dearer
  than not mixing, and 24.5's two figures were both copies with no decode in
  either, so there was never a crossover to find. One path ships, and the 96KB
  reference frame is gone. tools/analysis/10_pathmix_drift.py keeps the
  counterexample runnable; check.sh asserts it still reproduces.

- The four block modes do not cost the same. V1 300, V4 448, RAW 400 cycles
  against the old model's flat 207.8. V4 is 25% of blocks and 50% of the
  cycles, and the mode decision charges it bytes it does not charge cycles for.
  tools/analysis/11_cpu_budget.py reproduces all four frames timed on the
  68000 to within 1 point. Hand-derived timings agree to 0.5% on V1.

- The container is big-endian but not aligned. Variable-length records laid end
  to end put frame 1's length field at an odd address, and move.l (a0)+ there
  is an address error: frame 0 decoded perfectly and then vectored into the
  IPL for 59 emulated seconds looking like a hang. Found by dumping PC, not by
  reading the source.

Also: an all-V1 frame, the cheapest possible full redraw, is 110.5% of budget.
No mode assignment fits a scene cut at 12fps. That one needs a decision, not a
measurement.

Next: charge cycles in the mode decision and bisect against 833,333 per frame,
the way session 6 bisects lam against bytes -- but with no bucket, because a
late frame cannot be banked.

Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6
2026-08-23 15:04:38 -07:00

173 lines
6.9 KiB
Lua

-- Time and verify src/player/decode.s on the emulated 68000.
--
-- Two questions, one run:
-- 1. CORRECTNESS. Decode the whole window frame by frame and snapshot the
-- last frame. tools/bench/verify_decode.py checks it against the Python
-- reference decoder (tools/encoder/dlx.py) pixel-for-pixel. Every SKIP
-- block in every frame is a claim about the previous frame still being on
-- screen, so a sequential run is the only honest test -- decoding one
-- frame in isolation would prove nothing about the temporal recursion.
-- 2. COST. Time individual frames chosen across the non-SKIP distribution,
-- not its mean (FINDINGS 25.6), plus one full 120-frame pass.
--
-- MEASUREMENT SCOPE, unchanged from blit.lua: MAME's gvram_w/gvram_r carry no
-- timing at all, so these are pure 68000 instruction cycles against
-- zero-wait-state memory -- a LOWER BOUND on real hardware, not a prediction.
-- Interrupts are masked (SR=$2700) so the IPL cannot steal cycles.
--
-- Codebook expansion and palette packing are done host-side by prep_dlx.py:
-- they are load-time costs, not per-frame ones, and including them would
-- flatter or damn the inner loop for no reason.
M = manager.machine
SP = M.devices[":maincpu"].spaces["program"]
local function findfile(n)
for _,p in ipairs{"../tools/bench/"..n, "tools/bench/"..n, n} do
local f = io.open(p,"rb"); if f then f:close(); return p end
end
error(n.." not found")
end
local MODE = loadfile(findfile("crtc_mode.lua"))()
local META = loadfile("decode_meta.lua")()
local FLAG, ITER, NFR, FPTR = 0x18000, 0x18008, 0x1800C, 0x18010
local CB1, CB4, STREAM = 0x20000, 0x22000, 0x30000
local GVRAM, GPAL = 0xC00000, 0xE82000
local CPUHZ = 10000000 -- x68k.cpp:1133, 40_MHz_XTAL/4
local FRAME12 = CPUHZ / META.fps
local code do local f=io.open("decode.bin","rb"); code=f:read("a"); f:close() end
local data do local f=io.open("decode_data.bin","rb"); data=f:read("a"); f:close() end
local YOFF = (MODE.height - META.H) // 2
local function T() local t=M.time; return t.seconds + t.attoseconds/1e18 end
local function P(s) print("[DEC] "..s) end
-- Bulk-load a slice of the blob as big-endian longwords. 1 MB one byte at a
-- time is 1M Lua->C calls; longwords cut that by four.
local function push(addr, s, from, len)
local i, n = from, len
while n >= 4 do
SP:write_u32(addr, (string.unpack(">I4", s, i)))
addr, i, n = addr+4, i+4, n-4
end
while n > 0 do
SP:write_u8(addr, string.byte(s,i)); addr, i, n = addr+1, i+1, n-1
end
end
local function setup()
MODE.apply(SP)
local o = 1
push(CB1, data, o, META.cb1_len); o = o + META.cb1_len
push(CB4, data, o, META.cb4_len); o = o + META.cb4_len
local palo = o; o = o + META.pal_len
push(STREAM, data, o, META.stream_len)
for c = 0, 255 do
SP:write_u16(GPAL + c*2, (string.unpack(">I2", data, palo + c*2)))
end
-- Active area starts at index 0, exactly as the reference decoder's canvas
-- does; the letterbox gets the palette's darkest entry because the encoder
-- does not yet reserve a black one (docs/STATUS.md, encoder gaps).
for y = 0, MODE.height-1 do
local base, v = GVRAM + y*1024, 0
if y < YOFF or y >= YOFF+META.H then v = META.dark end
for x = 0, MODE.width-1 do SP:write_u16(base + x*2, v) end
end
for i = 1, #code do SP:write_u8(0x10000+i-1, string.byte(code,i)) end
P(string.format("loaded decode.bin=%d B, codebooks %d+%d B, stream %d B, %d frames",
#code, META.cb1_len, META.cb4_len, META.stream_len, META.nframes))
end
local function launch(off, nfr, iter)
SP:write_u32(FLAG, 0)
SP:write_u32(ITER, iter)
SP:write_u32(NFR, nfr)
SP:write_u32(FPTR, STREAM + off)
local cpu = M.devices[":maincpu"]
cpu.state["SR"].value = 0x2700 -- supervisor, ALL interrupts masked
cpu.state["SP"].value = 0x8000
cpu.state["PC"].value = 0x10000
end
-- The plan: one sequential correctness pass, then the cost anchors, then a
-- full pass timed. Iteration counts target ~4 emulated seconds each so the
-- 1/55.46 s timing granularity costs under 0.5%.
-- DLX_VERIFY_ONLY=1 drops the cost anchors and runs only the correctness pass,
-- so tools/bench/check.sh can gate the decoder without paying for ~2 minutes of
-- timing runs that would make the green light sensitive to host load anyway.
local VERIFY_ONLY = os.getenv("DLX_VERIFY_ONLY") == "1"
local PLAN = { {name="sequential decode of all "..META.nframes.." frames (correctness)",
off=0, nfr=META.nframes, iter=1, snap=true} }
for _,an in ipairs(VERIFY_ONLY and {} or META.anchors) do
local est = math.max(0.06, an.frac/100) * 1.30 * FRAME12
PLAN[#PLAN+1] = {name="frame @ "..an.name, off=an.off, nfr=1,
iter=math.max(20, math.floor(4*CPUHZ/est)), frac=an.frac}
end
if not VERIFY_ONLY then
PLAN[#PLAN+1] = {name="full "..META.nframes.."-frame pass (mean over the window)",
off=0, nfr=META.nframes, iter=1, seq=true}
end
local step, st, t0 = 0, "boot", nil
local results = {}
local function report(p, dt)
local per = p.nfr * p.iter
local cyc = dt * CPUHZ / per
local pct = 100 * cyc / FRAME12
if p.snap then return end -- correctness pass, iter=1, too coarse
results[#results+1] = {p=p, cyc=cyc, pct=pct}
P(string.format("%s", p.name))
P(string.format(" %d frames in %.4f s -> %.0f cycles/frame = %.1f%% of a %dfps frame",
per, dt, cyc, pct, META.fps))
end
SUB = emu.add_machine_frame_notifier(function()
local ok, err = pcall(function()
local t = T()
if st == "boot" then
if t < 3.0 then return end
setup(); step = 1; launch(PLAN[1].off, PLAN[1].nfr, PLAN[1].iter)
st, t0 = "running", nil; return
end
if st == "running" then
local fl = SP:read_u32(FLAG)
if fl == 1 and not t0 then t0 = t; return end
if fl == 0xEE then
P("BITSTREAM DESYNC -- decoder consumed the wrong number of payload bytes")
M:exit(); return
end
if fl == 0xFF then
report(PLAN[step], t - (t0 or t))
if PLAN[step].snap then st = "snap"; return end
step = step + 1
if PLAN[step] then
launch(PLAN[step].off, PLAN[step].nfr, PLAN[step].iter)
st, t0 = "running", nil
else st = "finish" end
return
end
if t > 400 then P("TIMEOUT flag="..string.format("%08X",fl)); M:exit() end
return
end
if st == "snap" then
M.video:snapshot()
P("snapshot taken after the sequential pass -- last frame, 68000-decoded")
step = step + 1
launch(PLAN[step].off, PLAN[step].nfr, PLAN[step].iter)
st, t0 = "running", nil; return
end
if st == "finish" then
P("---- summary (instruction cycles only; real GVRAM adds wait states) ----")
for _,r in ipairs(results) do
P(string.format(" %-46s %8.0f cyc %5.1f%% of a frame", r.p.name, r.cyc, r.pct))
end
M:exit()
end
end)
if not ok then print("[DEC] LUA ERROR: "..tostring(err)); M:exit() end
end)