ROADMAP P3 said "needs MFP timer or VBL" and neither can do it. The MFP's timer clock is 16 MHz/4, its prescalers stop at 200 and its data register is 8 bits, so the slowest tick any single timer can make is 78.125 Hz -- 6.5x faster than a frame -- and 4e6/12 is not an integer, so no setting reaches 12 Hz at all. The raster has no whole divide near 12 either: 4 refreshes is 13.86 fps and 5 is 11.09. tools/analysis/23_frame_clock.py walks all 7x256 timer settings rather than asserting it. src/player/clock.i takes the V-DISP falling edge on MFP GPIP4 -- the start of vertical blanking, which is when a player would present -- and adds fps*VTOTAL per edge to a 16-bit accumulator, emitting a tick at 31,500 and keeping the remainder. The long-run rate is fps*VTOTAL/VTOTAL = 12.000000 fps exactly, and both constants are read out of the CRTC at init, so the clock is derived from the registers that generate the raster it counts. Measured over 3,000 refreshes: 3,000 interrupts, 649 ticks where 649.1429 were due. It costs 181.35 clocks per V-DISP, 838 per frame, 0.1006% of the budget -- timed by the 68000 itself, because the host's granularity is 17.64 ms and the interrupt is microseconds. The loop's own cost was calibrated rather than looked up and landed on 38.000002 clocks, which both licenses the subtraction and confirms buscost.py's model; the 181.35 then decomposes exactly, leaving 43.99 clocks for the interrupt exception -- the textbook 44, measured. THE ONE THAT MOVES SOMETHING: 12 fps on a 55.4577 Hz raster is 4.6215 refreshes, so a frame is shown for 4 refreshes (72.13 ms) or 5 (90.16 ms), 37.9% of them short. The 833,333-clock budget every figure in this project is priced against is the MEAN slot, and the short one is 13.4% under it. The cadence was already in the tree unnamed: stream.lua's tick is sampled at frame boundaries, so its gaps were always 4 or 5, and every host-paced result in FINDINGS 49/51 carried it. P3 moved who produces it onto the machine and made it visible. It is not a dropped frame -- the pace gate lets an overrun eat the next frame's idle -- and on the gate container it costs 4 frames of 120 their idle against 1 for the nominal model, most of that the frame-0 transient at 111% of budget. stream.s counts it now, and the rig matches an offline model of the divider exactly. Also struck: MAME's raster runs 2.22% fast. refresh_mode() builds the frame period from scr.max_x*scr.max_y with scr.max_x = m_htotal - 8, one character cell short and an inclusive bound used as a count, so it runs at 56.6901 Hz where the registers say 55.4577 -- agreeing to six digits with the arithmetic. Every "1/55.46 s granularity" note in this tree was wrong and is 1/56.69 s, corrected in six files with the derivation put once in crtc_mode.lua. No conclusion changes and no 68000 cycle figure moves; the CPU clock is unrelated to the screen. But anything paced by the raster runs fast under MAME, so the rig reports both rates and prices the interrupt against the hardware's. decode.s and frame.i are unchanged; decode.bin is still 1,296 B at the same MD5. The pace gate's wait loop is byte-for-byte the one FINDINGS 51 measured and the free-running path executes none of the new code. check.sh gains two stages: the clock's own measurement, and 120 frames decoded pixel-exact with nothing outside the machine deciding when a frame may start. Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6
266 lines
10 KiB
Lua
266 lines
10 KiB
Lua
-- Measure the cost of a row-linear literal SPAN on the 68000 (FINDINGS 29.5.1).
|
|
--
|
|
-- FINDINGS 29 proposes one new decoder mode and prices it at
|
|
-- 4 * (50 + 4L*9.08) cycles for a run of L blocks
|
|
-- then labels the whole section DERIVED, NOT MEASURED, because both terms are
|
|
-- extrapolations: the 50-cycle per-span overhead is hand-derived, and the 9.08
|
|
-- cycles/pixel is a FINDINGS 24 measurement taken at FULL ROW WIDTH with
|
|
-- 12-register bursts. A 4-pixel span cannot burst at all. Everything session
|
|
-- 8 wants to do downstream optimises over the mode set this number decides, so
|
|
-- it goes first.
|
|
--
|
|
-- Method: v5 in tools/bench/blit.s walks a stream of per-row span records and
|
|
-- copies each span into GVRAM. tools/bench/prep_spans.py emits one stream per
|
|
-- span length, every one covering the same whole frame, so the work differs
|
|
-- only in how finely it is cut. Regressing
|
|
-- cycles = A*spans + B*pixels
|
|
-- over the set reads A (the per-span overhead) and B (the per-pixel cost)
|
|
-- straight off, and every config also draws a verifiable picture: the frame is
|
|
-- cleared before each run and snapshotted after, so a config that timed fast
|
|
-- by not writing pixels fails tools/bench/verify_frame256.py.
|
|
--
|
|
-- MEASUREMENT SCOPE, unchanged from blit.lua: MAME's gvram_w carries no timing,
|
|
-- so these are 68000 instruction cycles against zero-wait-state memory -- a
|
|
-- LOWER BOUND on real hardware. Interrupts are masked (SR=$2700).
|
|
|
|
M = manager.machine
|
|
SP = M.devices[":maincpu"].spaces["program"]
|
|
|
|
local function findfile(n)
|
|
for _,p in ipairs{"../tools/bench/"..n, "tools/bench/"..n, n} do
|
|
local f = io.open(p,"rb"); if f then f:close(); return p end
|
|
end
|
|
error(n.." not found")
|
|
end
|
|
local MODE = loadfile(findfile("crtc_mode.lua"))()
|
|
local SPEC = loadfile("spans_meta.lua")()
|
|
|
|
local FLAG, VAR, ITER, SPTR = 0x18000, 0x18004, 0x18008, 0x1800C
|
|
local STREAM = 0x90000
|
|
local GVRAM, GPAL = 0xC00000, 0xE82000
|
|
local CPUHZ = 10000000 -- x68k.cpp:1133, 40_MHz_XTAL/4
|
|
local FRAME12 = CPUHZ / 12
|
|
|
|
local code do local f=io.open("blit.bin","rb"); code=f:read("a"); f:close() end
|
|
local blob do local f=io.open("spans.bin","rb"); blob=f:read("a"); f:close() end
|
|
local frame do local f=io.open("frame256.bin","rb"); frame=f:read("a"); f:close() end
|
|
|
|
local function B(i) return string.byte(frame,i) end
|
|
local W, H = B(5)*256+B(6), B(7)*256+B(8)
|
|
local PAL0 = 9
|
|
local YOFF = (MODE.height - H) // 2
|
|
|
|
-- Identical packing to blit.lua / show_frame256.lua: shared LSB I per entry.
|
|
local function pal6(v) return ((v<<2)|(v>>4)) & 0xff end
|
|
local function pack(r,g,b)
|
|
local f = {r>>3, g>>3, b>>3}
|
|
local best, bestI = nil, 1
|
|
for I = 0,1 do
|
|
local e = 0
|
|
for c = 1,3 do
|
|
local want = ({r,g,b})[c]
|
|
local d = pal6((f[c]<<1)|I) - want
|
|
e = e + d*d
|
|
end
|
|
if best == nil or e < best then best, bestI = e, I end
|
|
end
|
|
return (f[2]<<11)|(f[1]<<6)|(f[3]<<1)|bestI
|
|
end
|
|
|
|
local function T() local t=M.time; return t.seconds + t.attoseconds/1e18 end
|
|
local function P(s) print("[SPAN] "..s) end
|
|
|
|
-- 148 KB one byte at a time is 148k Lua->C calls; longwords cut that by four.
|
|
local function push(addr, s, from, len)
|
|
local i, n = from, len
|
|
while n >= 4 do
|
|
SP:write_u32(addr, (string.unpack(">I4", s, i)))
|
|
addr, i, n = addr+4, i+4, n-4
|
|
end
|
|
while n > 0 do
|
|
SP:write_u8(addr, string.byte(s,i)); addr, i, n = addr+1, i+1, n-1
|
|
end
|
|
end
|
|
|
|
local function clear_picture() -- so a config that writes nothing is caught
|
|
for y = YOFF, YOFF+H-1 do
|
|
local base = GVRAM + y*1024
|
|
for x = 0, MODE.width-1, 2 do SP:write_u32(base + x*2, 0) end
|
|
end
|
|
end
|
|
|
|
local function setup()
|
|
MODE.apply(SP)
|
|
for y = 0, MODE.height-1 do
|
|
local base = GVRAM + y*1024
|
|
for x = 0, MODE.width-1, 2 do SP:write_u32(base + x*2, 0) end
|
|
end
|
|
for c = 0, 255 do
|
|
local o = PAL0 + c*3
|
|
SP:write_u16(GPAL + c*2, pack(B(o), B(o+1), B(o+2)))
|
|
end
|
|
for i = 1, #code do SP:write_u8(0x10000+i-1, string.byte(code,i)) end
|
|
P(string.format("loaded blit.bin=%d B, %d span configs, picture %dx%d at yoff=%d",
|
|
#code, #SPEC.configs, W, H, YOFF))
|
|
end
|
|
|
|
local function launch(cfg)
|
|
push(STREAM, blob, cfg.off+1, cfg.len)
|
|
clear_picture()
|
|
-- ~4 emulated seconds per config: 1/56.69 s granularity (crtc_mode.lua)
|
|
-- costs under 0.5%.
|
|
local est = cfg.nspans*(cfg.var == 5 and 60 or (cfg.var == 6 and 50 or 70))
|
|
+ cfg.npix*10
|
|
cfg.iter = math.max(4, math.floor(4*CPUHZ/est))
|
|
SP:write_u32(FLAG, 0)
|
|
SP:write_u32(VAR, cfg.var)
|
|
SP:write_u32(ITER, cfg.iter)
|
|
SP:write_u32(SPTR, STREAM)
|
|
local cpu = M.devices[":maincpu"]
|
|
cpu.state["SR"].value = 0x2700
|
|
cpu.state["SP"].value = 0x8000
|
|
cpu.state["PC"].value = 0x10000
|
|
end
|
|
|
|
local results = {}
|
|
local function report(cfg, dt)
|
|
local cyc = dt * CPUHZ / cfg.iter
|
|
results[#results+1] = {cfg=cfg, cyc=cyc}
|
|
P(string.format("v%d span %4s px: %5d spans %6d px %d iter in %.4f s -> %8.0f cyc/frame"
|
|
.." %5.2f cyc/px %5.1f%% of a 12fps frame",
|
|
cfg.var, cfg.name, cfg.nspans, cfg.npix, cfg.iter, dt, cyc,
|
|
cyc/cfg.npix, 100*cyc/FRAME12))
|
|
end
|
|
|
|
-- Ordinary least squares on cycles = A*spans + B*pixels, no intercept: the
|
|
-- 192 row headers and the outer loop are the only work not attributable to a
|
|
-- span or a pixel, and at ~10 cycles a row they are 0.2% of the smallest run.
|
|
local function fit(rs)
|
|
local ss,sp,pp,sy,py = 0,0,0,0,0
|
|
for _,r in ipairs(rs) do
|
|
local s,p,y = r.cfg.nspans, r.cfg.npix, r.cyc
|
|
ss=ss+s*s; sp=sp+s*p; pp=pp+p*p; sy=sy+s*y; py=py+p*y
|
|
end
|
|
local det = ss*pp - sp*sp
|
|
return (sy*pp - py*sp)/det, (ss*py - sp*sy)/det
|
|
end
|
|
|
|
-- v7 has two per-pixel costs -- the 24-pixel coarse chain and the 2-pixel fine
|
|
-- chain -- so its fit is cycles = A*spans + Bc*coarse_px + Bf*fine_px, solved
|
|
-- by plain Gaussian elimination on the 3x3 normal equations. prep_spans.py
|
|
-- picks span lengths so every fine remainder a real span can have (0,4,..,20)
|
|
-- appears, which is what makes the three terms separable.
|
|
local function fit3(rs)
|
|
local M3 = {{0,0,0,0},{0,0,0,0},{0,0,0,0}}
|
|
for _,r in ipairs(rs) do
|
|
local x = {r.cfg.nspans, r.cfg.cpx, r.cfg.fpx}
|
|
for i=1,3 do
|
|
for j=1,3 do M3[i][j] = M3[i][j] + x[i]*x[j] end
|
|
M3[i][4] = M3[i][4] + x[i]*r.cyc
|
|
end
|
|
end
|
|
for c=1,3 do
|
|
local piv = c
|
|
for r=c+1,3 do if math.abs(M3[r][c]) > math.abs(M3[piv][c]) then piv=r end end
|
|
M3[c], M3[piv] = M3[piv], M3[c]
|
|
for r=1,3 do
|
|
if r ~= c then
|
|
local f = M3[r][c]/M3[c][c]
|
|
for k=c,4 do M3[r][k] = M3[r][k] - f*M3[c][k] end
|
|
end
|
|
end
|
|
end
|
|
return M3[1][4]/M3[1][1], M3[2][4]/M3[2][2], M3[3][4]/M3[3][3]
|
|
end
|
|
|
|
local step, st, t0 = 0, "boot", nil
|
|
|
|
SUB = emu.add_machine_frame_notifier(function()
|
|
local ok, err = pcall(function()
|
|
local t = T()
|
|
if st == "boot" then
|
|
if t < 3.0 then return end
|
|
setup(); step = 1; launch(SPEC.configs[1]); st, t0 = "running", nil; return
|
|
end
|
|
if st == "running" then
|
|
local fl = SP:read_u32(FLAG)
|
|
if fl == 1 and not t0 then t0 = t; return end
|
|
if fl == 0xFF then
|
|
report(SPEC.configs[step], t - (t0 or t))
|
|
st = "snap"; return
|
|
end
|
|
if t > 900 then P("TIMEOUT flag="..string.format("%08X",fl)); M:exit() end
|
|
return
|
|
end
|
|
if st == "snap" then
|
|
M.video:snapshot() -- verified by tools/bench/span.sh
|
|
step = step + 1
|
|
if SPEC.configs[step] then
|
|
launch(SPEC.configs[step]); st, t0 = "running", nil
|
|
else
|
|
st = "finish"
|
|
end
|
|
return
|
|
end
|
|
if st == "finish" then
|
|
P("---- measured (instruction cycles only; real GVRAM adds wait states) ----")
|
|
for _,v in ipairs{5,6,7} do
|
|
local sub = {}
|
|
for _,r in ipairs(results) do if r.cfg.var == v then sub[#sub+1] = r end end
|
|
-- v5's fit is over its BURSTING configs only (span length a multiple of
|
|
-- the 16-pixel burst). Mixing the remainder-path configs in would hide
|
|
-- the two costs behind one bad line; they are reported against the fit
|
|
-- instead, which is where the remainder shows up as error.
|
|
local fitset = {}
|
|
for _,r in ipairs(sub) do
|
|
if v ~= 5 or r.cfg.p % 16 == 0 then fitset[#fitset+1] = r end
|
|
end
|
|
local A, Bp, Bf
|
|
if v == 7 then
|
|
A, Bp, Bf = fit3(fitset)
|
|
P(string.format("-- v7: cycles = %.1f per span + %.3f per COARSE pixel"
|
|
.." + %.3f per FINE pixel (fitted on %d of %d configs)",
|
|
A, Bp, Bf, #fitset, #sub))
|
|
else
|
|
A, Bp = fit(fitset)
|
|
Bf = Bp
|
|
P(string.format("-- v%d: cycles = %.1f per span + %.3f per pixel"
|
|
.." (fitted on %d of %d configs)", v, A, Bp, #fitset, #sub))
|
|
end
|
|
for _,r in ipairs(sub) do
|
|
local model = A*r.cfg.nspans + Bp*r.cfg.cpx + Bf*r.cfg.fpx
|
|
P(string.format(" span %4s px %8.0f cyc %5.2f cyc/px %6.1f cyc/span"
|
|
.." vs fit %+6.1f%%", r.cfg.name, r.cyc,
|
|
r.cyc/r.cfg.npix, r.cyc/r.cfg.nspans, 100*(model/r.cyc-1)))
|
|
end
|
|
-- What the mode decision actually needs: a run of L horizontally
|
|
-- adjacent 4x4 blocks is 4 spans of 4L pixels, one per pixel row, and
|
|
-- v6 pads each to a whole 24-pixel chain unit.
|
|
local line = " -> cycles per 4x4 block in a run of L blocks: "
|
|
for _,L in ipairs{1,2,4,8,16,64} do
|
|
local px, cyc = 4*L, nil
|
|
if v == 6 then
|
|
px = math.ceil(px/24)*24
|
|
cyc = A + px*Bp
|
|
elseif v == 7 then
|
|
local c = math.floor(px/24)*24
|
|
cyc = A + c*Bp + (px-c)*Bf -- a multiple of 4 pads to nothing
|
|
else
|
|
cyc = A + px*Bp
|
|
end
|
|
line = line..string.format("L=%d %.0f ", L, 4*cyc/L)
|
|
end
|
|
P(line.."(V1 is 299.9)")
|
|
if v == 5 then
|
|
P(" v5's fit only holds where 4L is a whole number of 16-pixel bursts.")
|
|
P(" L=1 and L=2 are extrapolations its own measured spans"
|
|
.." contradict: 721 and 482.")
|
|
end
|
|
end
|
|
P(" FINDINGS 29 assumed 50.0 per span + 9.080 per pixel, 4 spans per run")
|
|
M:exit()
|
|
end
|
|
end)
|
|
if not ok then print("[SPAN] LUA ERROR: "..tostring(err)); M:exit() end
|
|
end)
|