The 68000 decoder draws pixel-exact frames, and does not fit
src/player/decode.s parses DLX1 and decodes straight into GVRAM. Verified pixel-exact over a 120-frame sequential run of the worst sustained window on the disc -- all four block modes, full temporal recursion, so the last frame is only right if all 120 were. In check.sh. It costs a mean of 81.7% of a 12fps frame budget, and 31% of frames exceed 100% (42% at scsi). CPU is now the binding constraint. FINDINGS 28. Three things that were believed and are not true: - The dual-display-path plan of FINDINGS 24.5/25.6 is incoherent. The compose path needs a RAM copy of the previous reconstruction; the direct path's selling point is that it keeps none. Mixing them shows stale pixels on 70 of 120 frames, worst frame 18.8% of the screen. Every coherent repair is dearer than not mixing, and 24.5's two figures were both copies with no decode in either, so there was never a crossover to find. One path ships, and the 96KB reference frame is gone. tools/analysis/10_pathmix_drift.py keeps the counterexample runnable; check.sh asserts it still reproduces. - The four block modes do not cost the same. V1 300, V4 448, RAW 400 cycles against the old model's flat 207.8. V4 is 25% of blocks and 50% of the cycles, and the mode decision charges it bytes it does not charge cycles for. tools/analysis/11_cpu_budget.py reproduces all four frames timed on the 68000 to within 1 point. Hand-derived timings agree to 0.5% on V1. - The container is big-endian but not aligned. Variable-length records laid end to end put frame 1's length field at an odd address, and move.l (a0)+ there is an address error: frame 0 decoded perfectly and then vectored into the IPL for 59 emulated seconds looking like a hang. Found by dumping PC, not by reading the source. Also: an all-V1 frame, the cheapest possible full redraw, is 110.5% of budget. No mode assignment fits a scene cut at 12fps. That one needs a decision, not a measurement. Next: charge cycles in the mode decision and bisect against 833,333 per frame, the way session 6 bisects lam against bytes -- but with no bucket, because a late frame cannot be banked. Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6
This commit is contained in:
@@ -0,0 +1,172 @@
|
||||
-- Time and verify src/player/decode.s on the emulated 68000.
|
||||
--
|
||||
-- Two questions, one run:
|
||||
-- 1. CORRECTNESS. Decode the whole window frame by frame and snapshot the
|
||||
-- last frame. tools/bench/verify_decode.py checks it against the Python
|
||||
-- reference decoder (tools/encoder/dlx.py) pixel-for-pixel. Every SKIP
|
||||
-- block in every frame is a claim about the previous frame still being on
|
||||
-- screen, so a sequential run is the only honest test -- decoding one
|
||||
-- frame in isolation would prove nothing about the temporal recursion.
|
||||
-- 2. COST. Time individual frames chosen across the non-SKIP distribution,
|
||||
-- not its mean (FINDINGS 25.6), plus one full 120-frame pass.
|
||||
--
|
||||
-- MEASUREMENT SCOPE, unchanged from blit.lua: MAME's gvram_w/gvram_r carry no
|
||||
-- timing at all, so these are pure 68000 instruction cycles against
|
||||
-- zero-wait-state memory -- a LOWER BOUND on real hardware, not a prediction.
|
||||
-- Interrupts are masked (SR=$2700) so the IPL cannot steal cycles.
|
||||
--
|
||||
-- Codebook expansion and palette packing are done host-side by prep_dlx.py:
|
||||
-- they are load-time costs, not per-frame ones, and including them would
|
||||
-- flatter or damn the inner loop for no reason.
|
||||
|
||||
M = manager.machine
|
||||
SP = M.devices[":maincpu"].spaces["program"]
|
||||
|
||||
local function findfile(n)
|
||||
for _,p in ipairs{"../tools/bench/"..n, "tools/bench/"..n, n} do
|
||||
local f = io.open(p,"rb"); if f then f:close(); return p end
|
||||
end
|
||||
error(n.." not found")
|
||||
end
|
||||
local MODE = loadfile(findfile("crtc_mode.lua"))()
|
||||
local META = loadfile("decode_meta.lua")()
|
||||
|
||||
local FLAG, ITER, NFR, FPTR = 0x18000, 0x18008, 0x1800C, 0x18010
|
||||
local CB1, CB4, STREAM = 0x20000, 0x22000, 0x30000
|
||||
local GVRAM, GPAL = 0xC00000, 0xE82000
|
||||
local CPUHZ = 10000000 -- x68k.cpp:1133, 40_MHz_XTAL/4
|
||||
local FRAME12 = CPUHZ / META.fps
|
||||
|
||||
local code do local f=io.open("decode.bin","rb"); code=f:read("a"); f:close() end
|
||||
local data do local f=io.open("decode_data.bin","rb"); data=f:read("a"); f:close() end
|
||||
|
||||
local YOFF = (MODE.height - META.H) // 2
|
||||
local function T() local t=M.time; return t.seconds + t.attoseconds/1e18 end
|
||||
local function P(s) print("[DEC] "..s) end
|
||||
|
||||
-- Bulk-load a slice of the blob as big-endian longwords. 1 MB one byte at a
|
||||
-- time is 1M Lua->C calls; longwords cut that by four.
|
||||
local function push(addr, s, from, len)
|
||||
local i, n = from, len
|
||||
while n >= 4 do
|
||||
SP:write_u32(addr, (string.unpack(">I4", s, i)))
|
||||
addr, i, n = addr+4, i+4, n-4
|
||||
end
|
||||
while n > 0 do
|
||||
SP:write_u8(addr, string.byte(s,i)); addr, i, n = addr+1, i+1, n-1
|
||||
end
|
||||
end
|
||||
|
||||
local function setup()
|
||||
MODE.apply(SP)
|
||||
local o = 1
|
||||
push(CB1, data, o, META.cb1_len); o = o + META.cb1_len
|
||||
push(CB4, data, o, META.cb4_len); o = o + META.cb4_len
|
||||
local palo = o; o = o + META.pal_len
|
||||
push(STREAM, data, o, META.stream_len)
|
||||
for c = 0, 255 do
|
||||
SP:write_u16(GPAL + c*2, (string.unpack(">I2", data, palo + c*2)))
|
||||
end
|
||||
-- Active area starts at index 0, exactly as the reference decoder's canvas
|
||||
-- does; the letterbox gets the palette's darkest entry because the encoder
|
||||
-- does not yet reserve a black one (docs/STATUS.md, encoder gaps).
|
||||
for y = 0, MODE.height-1 do
|
||||
local base, v = GVRAM + y*1024, 0
|
||||
if y < YOFF or y >= YOFF+META.H then v = META.dark end
|
||||
for x = 0, MODE.width-1 do SP:write_u16(base + x*2, v) end
|
||||
end
|
||||
for i = 1, #code do SP:write_u8(0x10000+i-1, string.byte(code,i)) end
|
||||
P(string.format("loaded decode.bin=%d B, codebooks %d+%d B, stream %d B, %d frames",
|
||||
#code, META.cb1_len, META.cb4_len, META.stream_len, META.nframes))
|
||||
end
|
||||
|
||||
local function launch(off, nfr, iter)
|
||||
SP:write_u32(FLAG, 0)
|
||||
SP:write_u32(ITER, iter)
|
||||
SP:write_u32(NFR, nfr)
|
||||
SP:write_u32(FPTR, STREAM + off)
|
||||
local cpu = M.devices[":maincpu"]
|
||||
cpu.state["SR"].value = 0x2700 -- supervisor, ALL interrupts masked
|
||||
cpu.state["SP"].value = 0x8000
|
||||
cpu.state["PC"].value = 0x10000
|
||||
end
|
||||
|
||||
-- The plan: one sequential correctness pass, then the cost anchors, then a
|
||||
-- full pass timed. Iteration counts target ~4 emulated seconds each so the
|
||||
-- 1/55.46 s timing granularity costs under 0.5%.
|
||||
-- DLX_VERIFY_ONLY=1 drops the cost anchors and runs only the correctness pass,
|
||||
-- so tools/bench/check.sh can gate the decoder without paying for ~2 minutes of
|
||||
-- timing runs that would make the green light sensitive to host load anyway.
|
||||
local VERIFY_ONLY = os.getenv("DLX_VERIFY_ONLY") == "1"
|
||||
|
||||
local PLAN = { {name="sequential decode of all "..META.nframes.." frames (correctness)",
|
||||
off=0, nfr=META.nframes, iter=1, snap=true} }
|
||||
for _,an in ipairs(VERIFY_ONLY and {} or META.anchors) do
|
||||
local est = math.max(0.06, an.frac/100) * 1.30 * FRAME12
|
||||
PLAN[#PLAN+1] = {name="frame @ "..an.name, off=an.off, nfr=1,
|
||||
iter=math.max(20, math.floor(4*CPUHZ/est)), frac=an.frac}
|
||||
end
|
||||
if not VERIFY_ONLY then
|
||||
PLAN[#PLAN+1] = {name="full "..META.nframes.."-frame pass (mean over the window)",
|
||||
off=0, nfr=META.nframes, iter=1, seq=true}
|
||||
end
|
||||
|
||||
local step, st, t0 = 0, "boot", nil
|
||||
local results = {}
|
||||
|
||||
local function report(p, dt)
|
||||
local per = p.nfr * p.iter
|
||||
local cyc = dt * CPUHZ / per
|
||||
local pct = 100 * cyc / FRAME12
|
||||
if p.snap then return end -- correctness pass, iter=1, too coarse
|
||||
results[#results+1] = {p=p, cyc=cyc, pct=pct}
|
||||
P(string.format("%s", p.name))
|
||||
P(string.format(" %d frames in %.4f s -> %.0f cycles/frame = %.1f%% of a %dfps frame",
|
||||
per, dt, cyc, pct, META.fps))
|
||||
end
|
||||
|
||||
SUB = emu.add_machine_frame_notifier(function()
|
||||
local ok, err = pcall(function()
|
||||
local t = T()
|
||||
if st == "boot" then
|
||||
if t < 3.0 then return end
|
||||
setup(); step = 1; launch(PLAN[1].off, PLAN[1].nfr, PLAN[1].iter)
|
||||
st, t0 = "running", nil; return
|
||||
end
|
||||
if st == "running" then
|
||||
local fl = SP:read_u32(FLAG)
|
||||
if fl == 1 and not t0 then t0 = t; return end
|
||||
if fl == 0xEE then
|
||||
P("BITSTREAM DESYNC -- decoder consumed the wrong number of payload bytes")
|
||||
M:exit(); return
|
||||
end
|
||||
if fl == 0xFF then
|
||||
report(PLAN[step], t - (t0 or t))
|
||||
if PLAN[step].snap then st = "snap"; return end
|
||||
step = step + 1
|
||||
if PLAN[step] then
|
||||
launch(PLAN[step].off, PLAN[step].nfr, PLAN[step].iter)
|
||||
st, t0 = "running", nil
|
||||
else st = "finish" end
|
||||
return
|
||||
end
|
||||
if t > 400 then P("TIMEOUT flag="..string.format("%08X",fl)); M:exit() end
|
||||
return
|
||||
end
|
||||
if st == "snap" then
|
||||
M.video:snapshot()
|
||||
P("snapshot taken after the sequential pass -- last frame, 68000-decoded")
|
||||
step = step + 1
|
||||
launch(PLAN[step].off, PLAN[step].nfr, PLAN[step].iter)
|
||||
st, t0 = "running", nil; return
|
||||
end
|
||||
if st == "finish" then
|
||||
P("---- summary (instruction cycles only; real GVRAM adds wait states) ----")
|
||||
for _,r in ipairs(results) do
|
||||
P(string.format(" %-46s %8.0f cyc %5.1f%% of a frame", r.p.name, r.cyc, r.pct))
|
||||
end
|
||||
M:exit()
|
||||
end
|
||||
end)
|
||||
if not ok then print("[DEC] LUA ERROR: "..tostring(err)); M:exit() end
|
||||
end)
|
||||
Reference in New Issue
Block a user