The decoder has been CPU-bound since FINDINGS 28 while the mode decision
minimised D + lam*R -- distortion against BYTES. decide() now minimises
D + lam*bytes + mu*cycles, and ratectl bisects mu per frame against the
833,333-cycle budget with the lam bisection nested inside it. On the worst
sustained window:
sasi 27.22 -> 26.95 dB, 109.5 -> 109.4 KB/s, 37/120 misses -> 1
scsi 29.90 -> 29.27 dB, 280.0 -> 278.6 KB/s, 51/120 misses -> 1
Bitrate does not move: the byte controller still binds, and mu changes WHICH
modes are bought. V4 is what it stops buying -- 25.2 -> 20.3% of blocks at sasi
and 15.0 -> 5.3% at scsi, where RAW takes it. That is 28.8's inversion in
practice: RAW is dearer in bytes and cheaper in cycles, so only the byte-rich
profile can buy its way out of V4.
Three things worth knowing beyond the headline:
- The one frame that still misses, at both profiles, is FRAME 0 -- no previous
reconstruction, so 100% changed by definition, which is also what a scene
cut is. It comes out at the all-V1 floor of 110.6% and is emitted late on
purpose. Freezing a cut to make a deadline is the worse failure.
- 28.7's "11 frames are impossible" was too pessimistic. That floor held the
SKIP set fixed and asked how cheaply the drawn blocks could be drawn; the
real decision can also MOVE a block to SKIP, which above ~90% non-SKIP is
the only lever left.
- SKIP's price depends on its neighbours (13.25 cycles clustered, 45 mixed),
which a per-block lagrangian cannot see. The way out is that the two uses
need not share a cost function: a ranking constant inside decide(), the
exact clustered rule for the frame-level bisection. vq_hybrid.cycles() is
now the one definition of that rule and 11_cpu_budget.py imports it.
Gated: 09_ratectl_drift.py runs both controllers, both 0/120 drifting frames.
The cost-aware container decodes pixel-exact on the 68000 (120 frames). ON by
default in encode.py; --no-cpu-fit restores session 7. check.sh ALL GREEN.
Still a model, not a measurement, for THIS container: FINDINGS 31's cycle
figures come from vq_hybrid.cycles (within 1 point of the 68000 on four frames
of the session-7 container). Timing this one on the machine is step 1 of the
next session -- it was started and killed for time, and it is slow.
FINDINGS 31. tools/analysis/13_cpu_ratectl.py.
Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6
178 lines
7.3 KiB
Lua
178 lines
7.3 KiB
Lua
-- Time and verify src/player/decode.s on the emulated 68000.
|
|
--
|
|
-- Two questions, one run:
|
|
-- 1. CORRECTNESS. Decode the whole window frame by frame and snapshot the
|
|
-- last frame. tools/bench/verify_decode.py checks it against the Python
|
|
-- reference decoder (tools/encoder/dlx.py) pixel-for-pixel. Every SKIP
|
|
-- block in every frame is a claim about the previous frame still being on
|
|
-- screen, so a sequential run is the only honest test -- decoding one
|
|
-- frame in isolation would prove nothing about the temporal recursion.
|
|
-- 2. COST. Time individual frames chosen across the non-SKIP distribution,
|
|
-- not its mean (FINDINGS 25.6), plus one full 120-frame pass.
|
|
--
|
|
-- MEASUREMENT SCOPE, unchanged from blit.lua: MAME's gvram_w/gvram_r carry no
|
|
-- timing at all, so these are pure 68000 instruction cycles against
|
|
-- zero-wait-state memory -- a LOWER BOUND on real hardware, not a prediction.
|
|
-- Interrupts are masked (SR=$2700) so the IPL cannot steal cycles.
|
|
--
|
|
-- Codebook expansion and palette packing are done host-side by prep_dlx.py:
|
|
-- they are load-time costs, not per-frame ones, and including them would
|
|
-- flatter or damn the inner loop for no reason.
|
|
|
|
M = manager.machine
|
|
SP = M.devices[":maincpu"].spaces["program"]
|
|
|
|
local function findfile(n)
|
|
for _,p in ipairs{"../tools/bench/"..n, "tools/bench/"..n, n} do
|
|
local f = io.open(p,"rb"); if f then f:close(); return p end
|
|
end
|
|
error(n.." not found")
|
|
end
|
|
local MODE = loadfile(findfile("crtc_mode.lua"))()
|
|
local META = loadfile("decode_meta.lua")()
|
|
|
|
local FLAG, ITER, NFR, FPTR = 0x18000, 0x18008, 0x1800C, 0x18010
|
|
local CB1, CB4, STREAM = 0x20000, 0x22000, 0x30000
|
|
local GVRAM, GPAL = 0xC00000, 0xE82000
|
|
local CPUHZ = 10000000 -- x68k.cpp:1133, 40_MHz_XTAL/4
|
|
local FRAME12 = CPUHZ / META.fps
|
|
|
|
local code do local f=io.open("decode.bin","rb"); code=f:read("a"); f:close() end
|
|
local data do local f=io.open("decode_data.bin","rb"); data=f:read("a"); f:close() end
|
|
|
|
local YOFF = (MODE.height - META.H) // 2
|
|
local function T() local t=M.time; return t.seconds + t.attoseconds/1e18 end
|
|
local function P(s) print("[DEC] "..s) end
|
|
|
|
-- Bulk-load a slice of the blob as big-endian longwords. 1 MB one byte at a
|
|
-- time is 1M Lua->C calls; longwords cut that by four.
|
|
local function push(addr, s, from, len)
|
|
local i, n = from, len
|
|
while n >= 4 do
|
|
SP:write_u32(addr, (string.unpack(">I4", s, i)))
|
|
addr, i, n = addr+4, i+4, n-4
|
|
end
|
|
while n > 0 do
|
|
SP:write_u8(addr, string.byte(s,i)); addr, i, n = addr+1, i+1, n-1
|
|
end
|
|
end
|
|
|
|
local function setup()
|
|
MODE.apply(SP)
|
|
local o = 1
|
|
push(CB1, data, o, META.cb1_len); o = o + META.cb1_len
|
|
push(CB4, data, o, META.cb4_len); o = o + META.cb4_len
|
|
local palo = o; o = o + META.pal_len
|
|
push(STREAM, data, o, META.stream_len)
|
|
for c = 0, 255 do
|
|
SP:write_u16(GPAL + c*2, (string.unpack(">I2", data, palo + c*2)))
|
|
end
|
|
-- Active area starts at index 0, exactly as the reference decoder's canvas
|
|
-- does; the letterbox gets the palette's darkest entry because the encoder
|
|
-- does not yet reserve a black one (docs/STATUS.md, encoder gaps).
|
|
for y = 0, MODE.height-1 do
|
|
local base, v = GVRAM + y*1024, 0
|
|
if y < YOFF or y >= YOFF+META.H then v = META.dark end
|
|
for x = 0, MODE.width-1 do SP:write_u16(base + x*2, v) end
|
|
end
|
|
for i = 1, #code do SP:write_u8(0x10000+i-1, string.byte(code,i)) end
|
|
P(string.format("loaded decode.bin=%d B, codebooks %d+%d B, stream %d B, %d frames",
|
|
#code, META.cb1_len, META.cb4_len, META.stream_len, META.nframes))
|
|
end
|
|
|
|
local function launch(off, nfr, iter)
|
|
SP:write_u32(FLAG, 0)
|
|
SP:write_u32(ITER, iter)
|
|
SP:write_u32(NFR, nfr)
|
|
SP:write_u32(FPTR, STREAM + off)
|
|
local cpu = M.devices[":maincpu"]
|
|
cpu.state["SR"].value = 0x2700 -- supervisor, ALL interrupts masked
|
|
cpu.state["SP"].value = 0x8000
|
|
cpu.state["PC"].value = 0x10000
|
|
end
|
|
|
|
-- The plan: one sequential correctness pass, then the cost anchors, then a
|
|
-- full pass timed. Iteration counts target ~4 emulated seconds each so the
|
|
-- 1/55.46 s timing granularity costs under 0.5%.
|
|
-- DLX_VERIFY_ONLY=1 drops the cost anchors and runs only the correctness pass,
|
|
-- so tools/bench/check.sh can gate the decoder without paying for ~2 minutes of
|
|
-- timing runs that would make the green light sensitive to host load anyway.
|
|
local VERIFY_ONLY = os.getenv("DLX_VERIFY_ONLY") == "1"
|
|
|
|
local PLAN = { {name="sequential decode of all "..META.nframes.." frames (correctness)",
|
|
off=0, nfr=META.nframes, iter=1, snap=true} }
|
|
for _,an in ipairs(VERIFY_ONLY and {} or META.anchors) do
|
|
local est = math.max(0.06, an.frac/100) * 1.30 * FRAME12
|
|
PLAN[#PLAN+1] = {name="frame @ "..an.name, off=an.off, nfr=1,
|
|
iter=math.max(20, math.floor(4*CPUHZ/est)), frac=an.frac}
|
|
end
|
|
if not VERIFY_ONLY then
|
|
PLAN[#PLAN+1] = {name="full "..META.nframes.."-frame pass (mean over the window)",
|
|
off=0, nfr=META.nframes, iter=1, seq=true}
|
|
end
|
|
|
|
local step, st, t0 = 0, "boot", nil
|
|
local results = {}
|
|
|
|
local function report(p, dt)
|
|
local per = p.nfr * p.iter
|
|
local cyc = dt * CPUHZ / per
|
|
local pct = 100 * cyc / FRAME12
|
|
if p.snap then return end -- correctness pass, iter=1, too coarse
|
|
results[#results+1] = {p=p, cyc=cyc, pct=pct}
|
|
P(string.format("%s", p.name))
|
|
P(string.format(" %d frames in %.4f s -> %.0f cycles/frame = %.1f%% of a %dfps frame",
|
|
per, dt, cyc, pct, META.fps))
|
|
end
|
|
|
|
SUB = emu.add_machine_frame_notifier(function()
|
|
local ok, err = pcall(function()
|
|
local t = T()
|
|
if st == "boot" then
|
|
if t < 3.0 then return end
|
|
setup(); step = 1; launch(PLAN[1].off, PLAN[1].nfr, PLAN[1].iter)
|
|
st, t0 = "running", nil; return
|
|
end
|
|
if st == "running" then
|
|
local fl = SP:read_u32(FLAG)
|
|
if fl == 1 and not t0 then t0 = t; return end
|
|
if fl == 0xEE then
|
|
P("BITSTREAM DESYNC -- decoder consumed the wrong number of payload bytes")
|
|
M:exit(); return
|
|
end
|
|
if fl == 0xFF then
|
|
report(PLAN[step], t - (t0 or t))
|
|
if PLAN[step].snap then st = "snap"; return end
|
|
step = step + 1
|
|
if PLAN[step] then
|
|
launch(PLAN[step].off, PLAN[step].nfr, PLAN[step].iter)
|
|
st, t0 = "running", nil
|
|
else st = "finish" end
|
|
return
|
|
end
|
|
if t > 400 then P("TIMEOUT flag="..string.format("%08X",fl)); M:exit() end
|
|
return
|
|
end
|
|
if st == "snap" then
|
|
M.video:snapshot()
|
|
P("snapshot taken after the sequential pass -- last frame, 68000-decoded")
|
|
step = step + 1
|
|
-- DLX_VERIFY_ONLY leaves nothing after the correctness pass, and this
|
|
-- used to walk off the end of PLAN and raise a Lua error AFTER the
|
|
-- snapshot was already on disk -- harmless to check.sh, and exactly the
|
|
-- kind of thing that gets mistaken for a decoder failure later.
|
|
if not PLAN[step] then st = "finish"; return end
|
|
launch(PLAN[step].off, PLAN[step].nfr, PLAN[step].iter)
|
|
st, t0 = "running", nil; return
|
|
end
|
|
if st == "finish" then
|
|
P("---- summary (instruction cycles only; real GVRAM adds wait states) ----")
|
|
for _,r in ipairs(results) do
|
|
P(string.format(" %-46s %8.0f cyc %5.1f%% of a frame", r.p.name, r.cyc, r.pct))
|
|
end
|
|
M:exit()
|
|
end
|
|
end)
|
|
if not ok then print("[DEC] LUA ERROR: "..tostring(err)); M:exit() end
|
|
end)
|