From 48e912de8bf27568165152e56f181f5c363cc86b Mon Sep 17 00:00:00 2001 From: prosolis <5590409+prosolis@users.noreply.github.com> Date: Sun, 23 Aug 2026 12:12:34 -0700 Subject: [PATCH] Size against 4 Mbps: peaks break the scsi profile; DMA steal is not free User clarified the bandwidth figure is 4 Mbps (488 KB/s), not 4 MB/s -- ~8x tighter than the previous commit reasoned against. Two consequences, plus a correction to session 1. 1. The scsi profile committed in f0f2f80 DOES NOT FIT. Its mean is a comfortable 52% of the pipe but it PEAKS at 96.4% (470.8 KB/s on scene 00020), and a frame that arrives late is a dropped frame, not a slow one. Peak/mean is 1.4-1.9x even on 1.2-1.7s clips. Sizing a real-time stream on the mean was the error. Flagged in STATUS rather than silently retuned, because the fix is rate control, not a lower lam. This promotes ratectl.py -- written in session 2, never wired into encode.py -- from a loose end to the highest-value work in the repo. It is worth a full step on the quality ladder (lam=25 -> lam=10, +0.7/+1.2 dB) because it allows sizing for the mean instead of the peak. 2. Pixel-exact is off the table at this bandwidth: lam=0 needs 92-97% of the pipe. The previous commit's "if SCSI sustains >=800 KB/s, ship transparent" conclusion only applies at roughly double the user's figure. 3. FINDINGS 5 said that because transfers are DMA, streaming "costs essentially no CPU" and the 68000 is "nearly idle". That is wrong. The HD63450 steals ~8 clocks per 16-bit word: 10-20% of the machine at the rates the profiles now use, on top of a 38% full-frame blit. Bandwidth and CPU are one budget. Adds tools/encoder/profile_gen.py, which derives lam FROM a bandwidth figure (accounting for audio, peak/mean and DMA steal) instead of reading it off the knee of the RD curve, and docs/BENCHMARK.md covering how to actually measure the subsystem -- including why MAME cannot answer the bandwidth question and would be the same class of error as the FINDINGS 4 traps. The 4 Mbps figure is user-supplied and its provenance is not recorded; every profile now hangs off it. Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6 --- docs/BENCHMARK.md | 133 +++++++++++++++++++++++++++++++++++ docs/FINDINGS.md | 61 ++++++++++++++++ docs/STATUS.md | 18 +++-- tools/encoder/profile_gen.py | 117 ++++++++++++++++++++++++++++++ 4 files changed, 324 insertions(+), 5 deletions(-) create mode 100644 docs/BENCHMARK.md create mode 100644 tools/encoder/profile_gen.py diff --git a/docs/BENCHMARK.md b/docs/BENCHMARK.md new file mode 100644 index 0000000..e825dc2 --- /dev/null +++ b/docs/BENCHMARK.md @@ -0,0 +1,133 @@ +# Benchmarking the storage subsystem, and deriving profiles from it + +Written session 2, in answer to "how do we benchmark the SCSI subsystem itself +and base our performance profiles around that?" + +## The short answer + +**You cannot set a bitrate profile from MAME.** MAME's `x68k_hdc` (SASI) and +`mb89352`/`cz6bs1` (SCSI) are *functional* models — they move the right bytes +and raise the right interrupts, but they are not transfer-timing accurate. A +throughput number out of MAME measures how fast the emulator's device model +hands over a buffer, which is an artefact of MAME's scheduling, not of a +Fujitsu MB89352 on a 10MHz bus. + +So split the question in two, because they need different instruments: + +| question | instrument | what it settles | +|---|---|---| +| does our read path work at all? | MAME | correctness, IOCS vs direct SPC, DMA setup | +| what rate does the hardware sustain? | derivation + real hardware | the profile bitrates | + +Using MAME for the second is the same class of error as FINDINGS 4: a number +that looks like a measurement but is an artefact of the apparatus. + +## Tier 1 — MAME: validate the path, not the speed + +This is what `tools/bench/` already does, and what is currently blocked +(`IOCS _B_READ` returns -1 uniformly). Its value is that it proves the +request/DMA/completion loop is correct before any of it is burned into 68000 +player code. + +Next moves, in order — the SCSI path was never tried and is more relevant to +the target anyway: + +1. **SCSI instead of SASI.** `-exp1 cz6bs1 -hard disk.chd`, with + `exp1:cz6bs1:scsi:0 harddisk`. Use IOCS `_S_READ` ($F5) rather than + `_B_READ` ($46). +2. **Move the stack.** `SP=$8000` may sit on top of the IOCS work area in low + RAM; put it at $200000+ (hypothesis 3 from session 1). +3. **Format the image.** Hypothesis 1 — a raw image has no X68000 partition + structure, so the IPL's boot scan never registers a drive and IOCS refuses. + Needs a Human68k image, which this machine does not have. +4. **Bypass IOCS entirely** and drive the MB89352 SPC registers directly. This + is what the shipping player will do anyway, since we want DMA straight into + a ring buffer with no OS in the path. If direct SPC works while IOCS does + not, that is a complete answer to the blocker and we simply skip IOCS. + +Record from MAME: bytes transferred, completion status, and whether DMA or PIO +was used. **Do not record KB/s and treat it as a hardware figure.** + +## Tier 2 — derivation: the defensible ceiling + +Already partly in FINDINGS 5. Bounds worth tightening from datasheets: + +- 68000 bus cycle: 4 clocks @ 10MHz, 16-bit => **5 MB/s** absolute ceiling +- HD63450 single-address DMA, ~8 clocks/word => **~2.5 MB/s** practical ceiling +- SCSI-1 asynchronous REQ/ACK handshake per byte, plus MB89352 FIFO depth + => the real limiter, and the number we do not have from a primary source + +The user's working figure is **4 Mbps = 488 KB/s**, which sits sensibly between +the derived DMA ceiling and observed period-drive rates. **Provenance not yet +recorded — worth pinning down, because every profile now hangs off it.** + +### The coupling nobody had counted +Cycle-stealing DMA is not free DMA. At ~8 clocks per 16-bit word: + +| stream | CPU stolen | + full-frame blit (38.3%) | +|---|---|---| +| 110 KB/s | 4.5% | 42.8% | +| 250 KB/s | 10.2% | 48.5% | +| 450 KB/s | 18.4% | 56.7% | +| 488 KB/s | 20.0% | 58.3% | + +FINDINGS 5 concluded that because transfers are DMA, "streaming costs +essentially no CPU". **That is wrong.** It costs up to a fifth of the machine at +the rates we now care about. Bandwidth and CPU are one budget, not two. + +## Tier 3 — real hardware: the only thing that settles it + +An X68000 (ACE/EXPERT for SASI, Super/XVI or a CZ-6BS1-equipped 10MHz machine +for SCSI) with a **BlueSCSI or SCSI2SD**, which is the realistic deployment +anyway and removes mechanical seek from the measurement. + +The benchmark must measure **what the player actually does**, not a synthetic +bulk read: + +1. Sequential read into a ring buffer, in the chunk size the player will use. +2. **With the decoder running** — so DMA/CPU contention is included. An idle-CPU + bulk read will overstate the sustained rate by roughly the blit percentage. +3. Timed with the machine's own timer (MFP timer-C or the 1/100s system clock), + not a stopwatch. +4. Reported as sustained KB/s over >=30s, plus the worst 1-second window. The + worst window is what the profile must survive, since a frame that arrives + late is a dropped frame. + +Deliverable: a `.x` executable and its source in `tools/bench/`, runnable on +real hardware and reporting a single number. + +## Feeding the result back into the profiles + +`tools/encoder/profile_gen.py` inverts the dependency — give it a bandwidth and +it returns the lam that fits, from the MEASURED rate-distortion points in +FINDINGS 17.4: + +``` +python3 tools/encoder/profile_gen.py --bw-mbps 4 --name scsi +``` + +It accounts for the three things that eat the pipe before video sees any of it: +audio (7.8 KB/s), peak-to-mean burstiness (measured 1.4-1.9x), and it reports +the DMA cycle-steal so the CPU coupling stays visible. + +At 4 Mbps it currently returns: + +| | lam | mean | peak | 00020 | 00146 | CPU | +|---|---|---|---|---|---|---| +| today (no rate control) | 25 | 194 KB/s | 368 KB/s | -1.22 dB | -4.21 dB | 53% | +| with rate control wired | 10 | 305 KB/s | 305 KB/s | -0.52 dB | -2.98 dB | 51% | + +**Rate control is worth a full step on the quality ladder** — it is not a +tidiness feature, it is the difference between sizing for the peak and sizing +for the mean. That is the strongest argument yet for wiring up `ratectl.py`. + +## What would change the design + +- **If sustained is much below 4 Mbps** (say 2 Mbps / 244 KB/s): `scsi` + collapses toward today's `sasi`, and the two profiles stop being meaningfully + different. At that point reconsider 10 fps, or a narrower active area. +- **If sustained is much above** (>=8 Mbps / 976 KB/s): `lam=0` fits with + margin and the port ships **pixel-exact** video on SCSI. +- **If DMA cannot be used** and transfers fall back to PIO, the CPU cost rises + from ~15% to something far larger and CPU becomes the binding constraint. + This is the single worst outcome and is worth checking early in Tier 1. diff --git a/docs/FINDINGS.md b/docs/FINDINGS.md index 6337d73..eba283b 100644 --- a/docs/FINDINGS.md +++ b/docs/FINDINGS.md @@ -450,3 +450,64 @@ now sit at 110 and 280 KB/s, close enough to the folklore ceilings that the error bars matter, and **if SCSI sustains >=800 KB/s the correct `scsi` profile is lam=0 — pixel-exact video.** Whether this port ships transparent or lossy on SCSI is now waiting on one measurement. + +## 18. Peak-to-mean burstiness — the mean was hiding the problem + +Prompted by the user clarifying that the bandwidth figure is **4 Mbps = 488 KB/s**, +not 4 MB/s. That is ~8x tighter than what 17 was reasoning against, and it +changes the answer. + +Per-frame instantaneous rate (video + 7.8 KB/s audio), 12 fps: + +| scene | lam | mean | p90 | **max** | peak/mean | max as % of 488 KB/s | +|---|---|---|---|---|---|---| +| 00010 | 60 | 95.0 | 127.3 | 138.8 | 1.46 | 28.4% | +| 00010 | 10 | 198.9 | 266.1 | 284.0 | 1.43 | 58.2% | +| 00020 | 60 | 115.8 | 155.4 | 222.3 | 1.92 | 45.5% | +| 00020 | 10 | 255.9 | 391.2 | **470.8** | 1.84 | **96.4%** | + +**The `scsi` profile as committed in f0f2f80 does not fit 4 Mbps.** Its mean is a +comfortable 52% of the pipe, but it peaks at 96.4% — and a frame that arrives +late is a *dropped frame*, not a slow one. Sizing a real-time stream on the mean +is the mistake; peak/mean is 1.4-1.9x on 1.2-1.7s clips and will be worse across +a full scene. + +Two ways out, and only one is good: +- Size for the peak: `lam=25`, mean 194 KB/s. Costs a full step of quality. +- **Rate-control to the mean and carry a leaky bucket:** `lam=10` fits, and buys + back +0.7 dB (00020) / +1.2 dB (00146). + +`ratectl.py` was written in session 2 but **never wired into `encode.py`**. This +demotes that from a loose end to the highest-value unfinished work in the repo. + +## 19. Cycle-stealing DMA is not free DMA — 5 was wrong + +FINDINGS 5 concluded "because it's DMA, streaming costs essentially no CPU — +this stacks with the 8% blit utilisation. The 68000 really is nearly idle." + +The HD63450 steals bus cycles from the 68000 at roughly 8 clocks per 16-bit word: + +| stream | words/s | clocks/s | CPU stolen | + full-frame blit | +|---|---|---|---|---| +| 110 KB/s | 56,320 | 450,560 | 4.5% | 42.8% | +| 250 KB/s | 128,000 | 1,024,000 | 10.2% | 48.5% | +| 450 KB/s | 230,400 | 1,843,200 | 18.4% | 56.7% | +| 488 KB/s | 249,856 | 1,998,848 | 20.0% | 58.3% | + +At the rates the profiles now use, streaming costs **10-20% of the machine**. +Still affordable — nothing here breaks — but **bandwidth and CPU are one budget, +not two**, and any future headroom argument has to spend from both. The +"nearly idle" framing should not be reused. + +(The 8 clocks/word figure is session 1's ESTIMATE from HD63450 timing, not a +measurement. It is the weakest link in this table.) + +## 20. Where the profiles should come from + +`tools/encoder/profile_gen.py` now derives lam from a bandwidth figure rather +than from the shape of the RD curve, accounting for audio, peak/mean, and +reporting DMA steal. Full benchmarking methodology — and why MAME cannot answer +the bandwidth question — is in `docs/BENCHMARK.md`. + +The 4 Mbps figure itself is **user-supplied and its provenance is not recorded**. +Every profile now hangs off it, so it is worth pinning down. diff --git a/docs/STATUS.md b/docs/STATUS.md index b073a9e..7d8780f 100644 --- a/docs/STATUS.md +++ b/docs/STATUS.md @@ -18,10 +18,16 @@ Session 1 left "which machine do we target" open. The user's answer: **ship both as two quality profiles. This is now implemented rather than hypothetical — the bitrate ceiling is a build parameter in `tools/encoder/ratectl.py`: -| profile | target | lam | quality (00020 / 00146) | bus utilisation | machine | +| profile | target | lam | quality (00020 / 00146) | machine | |---|---|---|---|---|---| -| `sasi` | 110 KB/s | 60 | 36.9 / 29.6 dB | 35% of 300 KB/s | stock 10MHz ACE/EXPERT | -| `scsi` | 280 KB/s | 10 | 39.4 / 32.3 dB | 28% of 1 MB/s | Super/XVI, or CZ-6BS1 board | +| `sasi` | 110 KB/s | 60 | 36.9 / 29.6 dB | stock 10MHz ACE/EXPERT | +| `scsi` | 280 KB/s | 10 | 39.4 / 32.3 dB | Super/XVI, or CZ-6BS1 board | + +**WARNING — `scsi` does not currently fit 4 Mbps.** The user's working bandwidth +figure is **4 Mbps = 488 KB/s**. `scsi` means 52% of that but **peaks at 96.4%** +(FINDINGS 18), and a late frame is a dropped frame. Until `ratectl.py` is wired +into `encode.py`, `scsi` must either drop to `lam=25` (194 KB/s mean) or not +ship. Derive profiles with `tools/encoder/profile_gen.py --bw-mbps 4`, not by eye. `scsi` is now within **0.5 dB of the palette ceiling** on 00020. These were initially set at 45 / 75 KB/s, which was 12% / 7% bus utilisation — read off the @@ -126,8 +132,10 @@ Next move is the untried SCSI path: `-exp1 cz6bs1 -hard disk.chd`. ## Next steps, in priority order -1. **Wire rate control into `encode.py`** and validate that the hard ceiling - actually holds on an action scene (the whole point of choosing VQ). +1. **Wire rate control into `encode.py`** — now the highest-value work in the + repo, not a loose end. It is worth a full step on the quality ladder + (lam=25 -> lam=10, +0.7/+1.2 dB) because it lets us size for the mean + instead of the peak. Validate the bucket holds on an action scene. 2. ~~Entropy-code the payload~~ — **ABANDONED, see FINDINGS 17.2.** Deflate decode is ~216% of the frame budget on a 68000 and LZ4 is ~54%; there is no room beside a 38% blit. All bitrates are raw payload. This also demotes the diff --git a/tools/encoder/profile_gen.py b/tools/encoder/profile_gen.py new file mode 100644 index 0000000..48d318a --- /dev/null +++ b/tools/encoder/profile_gen.py @@ -0,0 +1,117 @@ +#!/usr/bin/env python3 +"""Derive quality profiles FROM a measured bandwidth, instead of guessing lam. + + python3 tools/encoder/profile_gen.py --bw-kbps 488 --name scsi + python3 tools/encoder/profile_gen.py --bw-mbps 4 # same thing + +Session 2 set the profile bitrates by eye off the rate-distortion knee, which +was wrong twice over (FINDINGS 17.1). This inverts the dependency: give it a +bandwidth and it returns the lam that fits, with the headroom accounted for. + +Three things eat the pipe before video gets any: + + 1. AUDIO -- 7.8 KB/s of MSM6258 ADPCM, constant. + 2. PEAK/MEAN -- measured 1.4-1.9x (FINDINGS 18). The disk delivers a + SUSTAINED rate; a frame that overruns is a DROPPED frame. + Either size for the peak, or rate-control to the mean and + carry a bucket. We do the latter, so we need bucket depth + rather than peak headroom -- but until rate control is + actually wired in (it is not), size for the peak. + 3. DMA CYCLE-STEAL -- the HD63450 steals ~8 clocks per 16-bit word from the + 68000. At 488 KB/s that is 20% of the CPU, on top of the + blit. Bandwidth and CPU are NOT independent budgets. + FINDINGS 5 said streaming "costs essentially no CPU"; + that is wrong -- cycle-stealing DMA is not free DMA. + +The rate-distortion points are MEASURED (FINDINGS 17.4), not modelled, so this +interpolates real data rather than fitting a curve to a guess. +""" +import argparse + +AUDIO_KBPS = 7.8 +CLK = 10_000_000 +FPS = 12 +BLIT_FULL_FRAME_PCT = 38.3 # FINDINGS 17.2 +DMA_CLOCKS_PER_WORD = 8 # FINDINGS 5 (ESTIMATE, from HD63450 timing) + +# (lam, KB/s, PSNR) measured on the two probe scenes -- FINDINGS 17.4. +# 00146 is the harder scene; we size against it so profiles are not tuned to +# the easy case. Rates are RAW payload: entropy coding is ruled out (17.2). +CURVE = [ + # lam 00020 KB/s 00020 dB 00146 KB/s 00146 dB + ( 0, 442.1, 39.90, 467.6, 35.25), + ( 10, 248.1, 39.38, 305.2, 32.27), + ( 25, 182.2, 38.68, 193.5, 31.04), + ( 60, 108.0, 36.94, 103.1, 29.61), + ( 150, 55.6, 35.31, 56.1, 28.63), + ( 300, 44.1, 34.80, 44.4, 28.28), + ( 800, 32.5, 33.87, 36.1, 27.77), +] +CEILING = {"00020": 39.90, "00146": 35.25} +PEAK_OVER_MEAN = 1.9 # measured worst case, FINDINGS 18 + + +def dma_steal_pct(kbps): + return (kbps * 1024 / 2) * DMA_CLOCKS_PER_WORD / CLK * 100 + + +def pick(bw_kbps, peak_factor=PEAK_OVER_MEAN, margin=0.85, rate_controlled=False): + """Largest-quality lam whose worst-case demand fits inside bw_kbps.""" + usable = bw_kbps * margin - AUDIO_KBPS + factor = 1.0 if rate_controlled else peak_factor + allow_mean = usable / factor + for lam, k20, d20, k146, d146 in CURVE: + worst = max(k20, k146) + if worst <= allow_mean: + return dict(lam=lam, mean_kbps=worst, peak_kbps=worst * factor, + psnr20=d20, psnr146=d146, + loss20=CEILING["00020"] - d20, + loss146=CEILING["00146"] - d146, + allow_mean=allow_mean, usable=usable) + return None + + +def main(): + ap = argparse.ArgumentParser() + g = ap.add_mutually_exclusive_group(required=True) + g.add_argument("--bw-kbps", type=float) + g.add_argument("--bw-mbps", type=float, help="megaBITS/sec") + ap.add_argument("--name", default="profile") + ap.add_argument("--margin", type=float, default=0.85, + help="fraction of the pipe we allow ourselves (seeks, " + "container overhead, and the fact that the bandwidth " + "figure itself is folklore)") + ap.add_argument("--rate-controlled", action="store_true", + help="assume the leaky bucket absorbs peaks (NOT YET TRUE " + "-- ratectl.py is written but not wired into encode.py)") + a = ap.parse_args() + + bw = a.bw_kbps if a.bw_kbps else a.bw_mbps * 1_000_000 / 8 / 1024 + src = f"{a.bw_mbps} Mbps" if a.bw_mbps else f"{a.bw_kbps} KB/s" + print(f"bandwidth {src} = {bw:.0f} KB/s sustained") + print(f" usable at {a.margin:.0%} margin : {bw*a.margin:.0f} KB/s") + print(f" less audio ({AUDIO_KBPS}) : {bw*a.margin-AUDIO_KBPS:.0f} KB/s for video") + if not a.rate_controlled: + print(f" less peak/mean {PEAK_OVER_MEAN}x : " + f"{(bw*a.margin-AUDIO_KBPS)/PEAK_OVER_MEAN:.0f} KB/s mean allowance") + else: + print(" peaks absorbed by rate control (bucket depth must be validated)") + + r = pick(bw, margin=a.margin, rate_controlled=a.rate_controlled) + if r is None: + print("\n NO PROFILE FITS -- even lam=800 overruns. Lower the framerate,") + print(" the resolution, or get more bandwidth.") + return + steal = dma_steal_pct(r["peak_kbps"]) + print(f"\n -> {a.name}: lam={r['lam']}, {r['mean_kbps']:.0f} KB/s mean, " + f"{r['peak_kbps']:.0f} KB/s peak") + print(f" quality 00020 {r['psnr20']:.2f} dB (-{r['loss20']:.2f} from ceiling)") + print(f" 00146 {r['psnr146']:.2f} dB (-{r['loss146']:.2f} from ceiling)") + print(f" CPU blit {BLIT_FULL_FRAME_PCT:.0f}% + DMA steal {steal:.1f}% " + f"= {BLIT_FULL_FRAME_PCT+steal:.0f}% of the frame budget") + if BLIT_FULL_FRAME_PCT + steal > 85: + print(" WARNING: CPU is now the binding constraint, not the bus.") + + +if __name__ == "__main__": + main()