From e00264a058559488d3ab7df42db52beccede82b8 Mon Sep 17 00:00:00 2001 From: prosolis <5590409+prosolis@users.noreply.github.com> Date: Sun, 23 Aug 2026 14:00:12 -0700 Subject: [PATCH] Find the sustained action sequence: it breaks both profiles The open risk since session 2 was "a sustained action sequence could still break the bitrate", with every clip measured so far being 1.2-1.7 s. Closed by measurement rather than by sampling clips by hand. 07_motion_survey.py scans a whole stream at 96x72 for the hottest sliding window of inter-frame difference. On 00223 the spread between the quietest and hottest sustained 10 s windows is 10.6x, which is the argument for not eyeballing it. Hottest is t=539.4s, the Singe endgame. There, with the fixed lam the CLI uses, sasi overshoots 110 -> 129.6 KB/s (+18%) and scsi 280 -> 373.8 KB/s (+34%). Rate control moves from "insurance, not a fix" to required, and is promoted above the full-disc survey. The bus is not broken -- 381.6 KB/s still fits the 488 KB/s figure -- so FINDINGS 21 survives, at 78% of the pipe instead of a comfortable margin. Three further corrections fall out: - The two largest streams on the disc are bonus material. 00216 is the feature with a burned-in commentary PiP; 00215 is the commentary. 00223 is the clean 9.4 min. A size-ranked survey would have encoded live action. - On hard content the 256-colour scene palette (31.33 dB) binds well before the X68000 display (40.81 dB); scsi is already within 0.51 dB of it. - FINDINGS 24.5's architecture question resolves to "both paths, chosen per frame": 30-53% of frames sit above the 70% crossover. Picking per frame costs a median 37.0% of the frame budget and caps at 53.6%. Reporting for this is wired into encode.py, which previously only printed a mean over all frames -- the one statistic that cannot answer a per-frame question. extract.py takes optional start/dur; 08_mode_map.py renders source | decoded | block-mode map to .webm. Claude-Session: https://claude.ai/code/session_01194oWYW8DQXK1SZ2DnChW6 --- README.md | 10 +++ docs/FINDINGS.md | 113 +++++++++++++++++++++++++++++ docs/STATUS.md | 78 +++++++++++++++++--- tools/analysis/07_motion_survey.py | 57 +++++++++++++++ tools/analysis/08_mode_map.py | 107 +++++++++++++++++++++++++++ tools/encoder/encode.py | 24 ++++++ tools/encoder/extract.py | 8 +- 7 files changed, 385 insertions(+), 12 deletions(-) create mode 100644 tools/analysis/07_motion_survey.py create mode 100644 tools/analysis/08_mode_map.py diff --git a/README.md b/README.md index d38d031..24b744e 100644 --- a/README.md +++ b/README.md @@ -26,8 +26,13 @@ docs/ findings, status, hardware reference tools/analysis/ measurement scripts, numbered in the order they were written (01/02 marked BROKEN deliberately, kept as regression refs). Run from the repo root — they import from tools/encoder/. + 07 finds the hottest sustained window in a stream; 08 renders + source | decoded | block-mode map as .webm. tools/bench/ MAME Lua injection harness + 68000 benchmark sources. `check.sh` re-runs both display regression tests (~40 s). + `blit.s`/`blit.lua` time the full-frame GVRAM blit on the + 68000 itself (FINDINGS 24) — not part of check.sh, because + wall timings would make the green-light check host-sensitive. `crtc_mode.lua` is the single source of truth for CRTC R00-R08 and R20 — do not write CRTC values anywhere else. tools/vasm/ vasm m68k assembler (built from source) @@ -62,3 +67,8 @@ python3 tools/encoder/profile_gen.py --bw-mbps 4 --name scsi > pointing to the correction — heed those, especially 18 (reversed by 21). Source media (`DRAGONS_LAIR.iso`) and ROMs are gitignored — supply your own. + +**Not every large stream is game footage.** `00216` is the feature with a +burned-in commentary picture-in-picture and `00215` is the commentary itself — +the two largest files on the disc. The clean 9.4-minute animation is **`00223`**. +See FINDINGS 25.1 before running any size-ranked survey. diff --git a/docs/FINDINGS.md b/docs/FINDINGS.md index 7fbc20d..5bdd4d1 100644 --- a/docs/FINDINGS.md +++ b/docs/FINDINGS.md @@ -890,3 +890,116 @@ V1's output was snapshotted and passes `verify_frame256.py` unchanged: `256x512 native, double-scan exact, active 256x192 pixel-exact, letterbox true black`, 40.81 dB. So 68000 code drives the mode of FINDINGS 23 correctly, and 23.5 is now closed. + +--- + +## 25. The sustained action sequence, found and measured (session 5) + +STATUS has carried "a *sustained* action sequence is the one thing that could +still break the bitrate" as the open risk since session 2. Every clip measured +before this was 1.2-1.7 s. This section closes it: **it does break the profiles, +though not the bus.** + +### 25.1 The two largest streams on the disc are not game footage +A survey that sorts 224 streams by size and encodes the biggest would have +measured **live action**: + +| stream | size | what it actually is | +|---|---:|---| +| 00216 | 3777 MB | the feature with a **burned-in picture-in-picture commentary** | +| 00215 | 3475 MB | the commentary itself, full-screen live action | +| **00223** | **1802 MB** | **clean animation, 9.4 min — the one to use** | + +The PiP in 00216 is burned into video stream 0, not a selectable secondary +stream, so there is no ffmpeg flag that recovers a clean frame from it. This +extends FINDINGS 13's menu-vs-content warning: the classification needed is +**content / menu / bonus**, and bonus material is the one that looks most like +content by every cheap metric (size, duration, bitrate). + +### 25.2 Picking the worst window by measurement, not by eye +`tools/analysis/07_motion_survey.py` scans a whole stream at 96x72 and reports +the highest-mean sliding window of inter-frame absolute difference. On 00223: + +``` +6793 frames @12fps = 566.1s +motion energy mean 9.40 median 5.60 p90 21.70 max 112.39 +hottest sustained 10s window: t = 539.4s (2.01x stream mean) +quietest 10s window: t = 144.2s (0.19x stream mean) +``` + +The 10.6x spread between the quietest and hottest sustained windows is the whole +argument for not sampling clips by hand. `t = 539.4s` is the Singe endgame. + +### 25.3 Both profiles overshoot on that window — rate control is now required +Encoding those 120 frames at the shipping profiles, with the fixed `lam` the CLI +currently uses: + +| profile | target | measured | overshoot | PSNR | palette ceiling | +|---|---:|---:|---:|---:|---:| +| `sasi` | 110 KB/s | **129.6 KB/s** | **+18%** | 27.82 dB | 31.33 dB | +| `scsi` | 280 KB/s | **373.8 KB/s** | **+34%** | 30.81 dB | 31.33 dB | +| *(00020 baseline, `sasi`)* | 110 KB/s | 108.0 KB/s | -2% | 36.94 dB | 39.90 dB | + +**This reclassifies rate control from insurance to a requirement.** STATUS has +had "wire rate control into `encode.py`" at priority 3-4 since session 2 with the +note "no longer a blocker (FINDINGS 21)". That was true of the clips measured +then. It is not true of this one. `ratectl.encode_rate_controlled()` already +exists and builds a per-frame lam ladder; it has simply never been hooked up. + +Note what did **not** break: 373.8 + 7.8 = 381.6 KB/s is still under the 488 KB/s +working figure, so FINDINGS 21's ring-buffer conclusion survives — but at 78% of +the pipe sustained over ten seconds rather than the comfortable margin implied by +1.7 s clips. + +### 25.4 The palette ceiling is content-dependent, and on hard content it binds +The 256-colour scene palette costs **31.33 dB** on this window against **39.90 dB** +on 00020 — 8.6 dB worse. Fire, lava and smoke gradients are exactly what a +256-entry mediancut palette handles worst. + +This inverts an assumption the project has been carrying. FINDINGS 23.3 put the +X68000 display ceiling at 40.81 dB and treated it as comfortably clear of the +codec's own error. On this content the **scene palette (31.33 dB), not the +display hardware (40.81 dB), is the binding constraint** — and `scsi` is already +within 0.51 dB of it. Spending bits to close that last half-dB is spending them +against a ceiling that is not the display's. + +### 25.5 `scsi` collapses to RAW under stress +Mode distribution on this window is qualitatively different from anything +measured before: + +| profile | SKIP | V1 | V4 | RAW | +|---|---:|---:|---:|---:| +| `sasi` (lam=60) | 45.6% | 16.3% | 24.2% | 13.9% | +| `scsi` (lam=10) | 26.2% | 5.5% | 7.1% | **61.2%** | +| *00020, `sasi`* | 46.9% | 24.1% | 17.8% | 11.2% | + +At `lam=10` the rate-distortion decision finds literal pixels cheaper than any +codeword for 61% of blocks — the codebooks are simply not describing this +content. That is the mechanism behind the +34% overshoot in 25.3, and it is a +rate-control problem, not a codec-structure problem: the RD decision is behaving +correctly for the lam it was given. + +### 25.6 The decoder needs BOTH display paths, chosen per frame +Applying FINDINGS 24.5's crossover to the real per-frame distribution: + +| | median non-SKIP | p90 | frames over the 70% crossover | +|---|---:|---:|---:| +| `sasi`, Singe window | 48.4% | 82.8% | 36 / 120 (30%) | +| `scsi`, Singe window | 70.8% | 92.4% | 64 / 120 (53%) | +| `sasi`, 00020 | 54.0% | 88.8% | 3 / 14 (21%) | + +Neither path wins outright: **30-53% of frames want the flat blit and the rest +want direct-to-GVRAM.** A player that implements both and picks per frame — the +mode headers are parsed before any pixel is written, so the count is free — pays +a median of **37.0%** of the frame budget and is capped at **53.6%**. A player +that implements only direct-to-GVRAM pays up to 76.6% and would miss frames on +the scene cuts. + +So the answer to 24.5 is "both", and the selection is a one-line comparison +against a block count the decoder already has in hand. + +### 25.7 What this does not measure +One 10 s window of one stream, at fixed lam, with `_paint` still a Python loop. +The full-disc survey is still not done, and the numbers above are the *worst* +window rather than a distribution over content. What has changed is that the +worst case is now a measurement rather than a worry. diff --git a/docs/STATUS.md b/docs/STATUS.md index 99d5437..0b40fa4 100644 --- a/docs/STATUS.md +++ b/docs/STATUS.md @@ -96,7 +96,29 @@ rate-distortion curve, not two codecs. 3. **Reading the source frame is exactly half the blit cost** (V1 53.6% vs a write-only floor V3 of 27.1%). That is what makes the architecture question below live. -4. **The decoder architecture now hinges on one unmeasured number.** Writing +4. **That number is now measured, and the answer is "implement both paths".** + On the worst sustained window found on the disc, 30% of frames (`sasi`) to + 53% (`scsi`) sit above the 70% crossover and want the flat blit; the rest + want direct-to-GVRAM. A player that picks per frame — the mode headers are + parsed before any pixel is written, so the count is free — pays a **median + 37.0%** and is **capped at 53.6%**. FINDINGS 25.6. +5. **The sustained action sequence exists, was found by measurement, and breaks + both profiles.** `tools/analysis/07_motion_survey.py` scans a whole stream + for the hottest sliding window; on 00223 it is t=539.4s, the Singe endgame, + at 2.01x the stream mean. There, fixed-lam `sasi` overshoots 110 -> 129.6 + KB/s (+18%) and `scsi` 280 -> 373.8 KB/s (+34%). **Rate control is no longer + insurance — it is required.** FINDINGS 25.3. +6. **The two largest streams on the disc are bonus material, not game footage.** + 00216 is the feature with a burned-in commentary PiP; 00215 is the commentary + itself. **00223 (9.4 min) is the clean one.** A size-ranked survey would have + encoded live action. FINDINGS 25.1. +7. **On hard content the scene palette, not the display, is the binding + ceiling** — 31.33 dB on the Singe window against 39.90 dB on 00020 and 40.81 + dB for the X68000 display. `scsi` is already within 0.51 dB of it. + FINDINGS 25.4. + +### Superseded within session 5 +4a. **The decoder architecture hinged on one unmeasured number.** Writing codewords straight into GVRAM costs 76.6% of the frame budget for a *full* frame (V4 — the 1024-byte stride kills the `movem.l` burst), but scales with the non-SKIP block fraction and needs **no RAM reference frame at all**, @@ -284,7 +306,11 @@ SDL_VIDEODRIVER=dummy mame x68000 -bios ipl10 -video soft -window \ ## Next steps, in priority order -1. **Measure the non-SKIP block fraction.** *(new top priority, session 5)* +1. ~~**Measure the non-SKIP block fraction.**~~ **DONE, session 5** — FINDINGS + 25.6. Answer: implement **both** display paths and pick per frame; median + 37.0% of the frame budget, capped at 53.6%. Reporting is wired into + `encode.py`. Original framing kept below because the reasoning still governs + the decoder's inner loop: FINDINGS 24.5: compose-in-RAM-then-blit costs a flat 53.6% of the frame budget; decode-direct-to-GVRAM costs 76.6% x (fraction of blocks that are not SKIP) and needs no RAM reference frame. **They cross at 70%.** Which side of @@ -297,6 +323,15 @@ SDL_VIDEODRIVER=dummy mame x68000 -bios ipl10 -video soft -window \ scene-cut frame is ~100% non-SKIP and a held frame near 0%, and the mean of those two is a number describing no actual frame. +1b. **Wire rate control into `encode.py`. NOW REQUIRED (was priority 4).** + FINDINGS 25.3: on the worst sustained window both profiles overshoot their + targets with the fixed `lam` the CLI uses — `sasi` by 18%, `scsi` by 34%. + The note that used to sit here, "no longer a blocker (FINDINGS 21)", was true + of the 1.2-1.7 s clips measured at the time and is not true of this one. + `ratectl.encode_rate_controlled()` already builds a per-frame lam ladder; it + has never been hooked up. Do this before the full-disc survey or the survey + measures an encoder nobody will ship. + 2. **68000 decoder skeleton**, with the inner loop chosen by (1). Parse `DLX1`, expand codebooks, blit per block mode. The display path is verified *by 68000 code* now (FINDINGS 24) and the harness pattern is `tools/bench/blit.s` + @@ -312,16 +347,18 @@ SDL_VIDEODRIVER=dummy mame x68000 -bios ipl10 -video soft -window \ rejection, which was argued as "54% LZ4 with no room beside a 38% blit". The conclusion gets *stronger*, not weaker, but the arithmetic should be restated. -3. **Full-disc survey.** Only 4 clips of 1.2-1.7 s out of 224 streams have been - measured, and 00146 already runs 23% hotter than 00020. A *sustained* action - sequence is the one thing that could still break the bitrate. Classify menu - vs content first (FINDINGS 13) or the averages are diluted by static menus. - **Vectorise `_paint` before this run** — it is a Python per-block loop. - Pairs naturally with (1): the same run produces both numbers. +3. **Full-disc survey.** Now scoped by session 5 rather than open-ended: the + worst *sustained* window is measured (FINDINGS 25), so what remains is the + distribution over content, not the worst case. + - Classify **content / menu / bonus** — not just menu vs content. FINDINGS + 25.1: the two largest streams are bonus material and look like content by + size, duration and bitrate alike. + - Run `tools/analysis/07_motion_survey.py` per stream first; it is cheap + (96x72 greyscale) and gives a hot-window shortlist so the expensive encode + only runs where it matters. + - **Vectorise `_paint` before this run** — it is a Python per-block loop. + - Do it **after** rate control (1b), or it measures an encoder nobody ships. -4. **Wire rate control into `encode.py`.** No longer a blocker (FINDINGS 21), but - it is what gives a deterministic ceiling over content not yet measured, which - was the original reason for choosing VQ. Insurance, not a fix. Pairs with (1). 5. **Confirm DMA vs PIO in MAME** (see the benchmark section above) — cheap, and the only thing that could still move CPU into the binding position. 6. **Resolve the framing question** (FINDINGS 12: crop vs squash vs wide). @@ -425,3 +462,22 @@ V1's output. To check that snapshot is still pixel-exact: Not added to `check.sh`: `check.sh` asserts pixel-exactness, and asserting wall timings there would make the green-light check sensitive to host load. + +## Reproducing the sustained-action result (session 5) + +``` +python3 tools/analysis/07_motion_survey.py 00223 10 # -> hottest window t=539.4s +python3 tools/encoder/extract.py 00223 tmp/fr_singe 12 crop 539.4 10.0 +python3 tools/encoder/encode.py tmp/fr_singe tmp/singe_sasi.dlx --profile sasi +python3 tools/encoder/encode.py tmp/fr_singe tmp/singe_scsi.dlx --profile scsi +python3 tools/analysis/08_mode_map.py tmp/fr_singe tmp/singe_modes.webm \ + --profile sasi --scale 2 +``` +`extract.py` now takes optional `[start_s] [dur_s]` — needed because 00223 is +9.4 min and the windows that stress the codec are seconds long. + +`08_mode_map.py` renders palettised source | decoded | block-mode map at 12fps. +Output format follows the extension; **prefer `.webm`** — GIF re-quantises to +256 colours, which is a poor fit for output whose subject is colour fidelity, +and runs larger. It uses `yuv444p` because the mode map is flat saturated colour +on a 4-pixel grid and chroma subsampling smears exactly those edges. diff --git a/tools/analysis/07_motion_survey.py b/tools/analysis/07_motion_survey.py new file mode 100644 index 0000000..2bd438b --- /dev/null +++ b/tools/analysis/07_motion_survey.py @@ -0,0 +1,57 @@ +#!/usr/bin/env python3 +"""Find the worst sustained-motion window in a stream, cheaply. + +STATUS lists "a sustained action sequence is the one thing that could still +break the bitrate" as the open risk, and every clip measured so far has been +1.2-1.7 s. Picking a hot clip by eye is how you get a comfortable answer, so +this scans the whole stream instead. + +Proxy: mean absolute inter-frame difference at 96x72, decimated to the target +12 fps. It is a proxy, not a bitrate -- but the codec's cost is dominated by +how many blocks fail SKIP, and that is what frame difference measures. The +window it picks then gets encoded for real. + +Usage: python3 tools/analysis/07_motion_survey.py 00223 [window_seconds] +""" +import subprocess, sys +import numpy as np + +STREAM_DIR = "/media/reala-misaki/BDROM/BDMV/STREAM" +W, H, FPS = 96, 72, 12 + +def frames(stream): + src = f"{STREAM_DIR}/{stream}.m2ts" + vf = f"fps={FPS},crop=1440:1080:240:0,scale={W}:{H}:flags=bilinear" + p = subprocess.Popen(["ffmpeg", "-v", "error", "-i", src, "-vf", vf, + "-f", "rawvideo", "-pix_fmt", "gray", "-"], + stdout=subprocess.PIPE) + buf = p.stdout.read() + p.wait() + n = len(buf) // (W * H) + return np.frombuffer(buf[:n*W*H], np.uint8).reshape(n, H, W).astype(np.int16) + +def main(): + stream = sys.argv[1] + win_s = float(sys.argv[2]) if len(sys.argv) > 2 else 10.0 + f = frames(stream) + d = np.abs(np.diff(f, axis=0)).mean(axis=(1, 2)) # per-frame motion energy + print(f"{stream}: {len(f)} frames @ {FPS}fps = {len(f)/FPS:.1f}s") + print(f" motion energy mean {d.mean():.2f} median {np.median(d):.2f} " + f"p90 {np.percentile(d,90):.2f} max {d.max():.2f}") + + w = int(win_s * FPS) + if len(d) < w: + print("stream shorter than the window"); return + # sustained = highest mean over a sliding window, not the single hottest frame + k = np.convolve(d, np.ones(w) / w, mode="valid") + best = int(np.argmax(k)) + print(f" hottest sustained {win_s:.0f}s window: t = {best/FPS:.1f}s " + f"(mean {k[best]:.2f}, {k[best]/d.mean():.2f}x stream mean)") + quiet = int(np.argmin(k)) + print(f" quietest {win_s:.0f}s window: t = {quiet/FPS:.1f}s " + f"(mean {k[quiet]:.2f}, {k[quiet]/d.mean():.2f}x stream mean)") + np.save(f"tmp/motion_{stream}.npy", d) + print(f" per-frame energy -> tmp/motion_{stream}.npy") + +if __name__ == "__main__": + main() diff --git a/tools/analysis/08_mode_map.py b/tools/analysis/08_mode_map.py new file mode 100644 index 0000000..8f4cfb1 --- /dev/null +++ b/tools/analysis/08_mode_map.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""Render what the codec is actually DOING, per block, per frame. + +Three panels at 12 fps: the palettised source (the real quality ceiling, not +1080p -- FINDINGS 11), the decoded output, and a block-mode map. + +The mode map is not decoration. FINDINGS 24.5 makes the per-frame non-SKIP +fraction the number that selects the decoder's inner loop, and a percentile +cannot show you that the non-SKIP blocks are CLUSTERED (a moving character on a +held background) rather than scattered. Clustering is what a run-length over +the mode headers would exploit. + + SKIP left as the previous frame, costs the 68000 nothing + V1 one 4x4 codeword, 1 byte + V4 four 2x2 codewords, 4 bytes + RAW 16 literal palette indices -- the escape that makes lam=0 pixel-exact + +Usage: python3 tools/analysis/08_mode_map.py [--profile p] + [--scale N] +Output format follows the extension. Prefer .webm: GIF re-quantises to 256 +colours, which is a poor fit for output whose subject is colour fidelity. +""" +import sys, os +sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "encoder")) +import numpy as np +from PIL import Image +import vq as VQ, vq_hybrid as H, ratectl as RC + +MODE_RGB = np.array([[ 20, 22, 30], # SKIP - near black, costs nothing + [ 60, 150, 230], # V1 - blue + [ 80, 200, 120], # V4 - green + [235, 90, 70]], # RAW - red, the expensive escape + dtype=np.uint8) +LABEL = ["SKIP", "V1", "V4", "RAW"] + +SCALE = 1 +LOSSLESS = False + +def main(): + global SCALE, LOSSLESS + LOSSLESS = "--lossless" in sys.argv + src, out = sys.argv[1], sys.argv[2] + if "--scale" in sys.argv: + SCALE = int(sys.argv[sys.argv.index("--scale")+1]) + prof = RC.PROFILES[sys.argv[sys.argv.index("--profile")+1] + if "--profile" in sys.argv else "sasi"] + m = H.build(src, k1=prof["k1"], k4=prof["k4"]) + enc = H.encode(m, lam=prof["lam"]) + pal, H_, W_ = m["pal"], m["H"], m["W"] + nbx, nby = W_ // 4, H_ // 4 + + frames, stats = [], [] + for f, (rec, mode) in enumerate(zip(enc["recon"], enc["modes"])): + srcp = pal[m["idx"][f]] + decp = pal[rec] + mmap = MODE_RGB[mode.reshape(nby, nbx)].repeat(4, 0).repeat(4, 1) + # tint the mode map with the decoded luma so the action stays legible + luma = decp.mean(2, keepdims=True) / 255.0 + mmap = (mmap * (0.45 + 0.55 * luma)).astype(np.uint8) + gap = np.full((H_, 3, 3), 60, np.uint8) + panel = np.hstack([srcp, gap, decp, gap, mmap]) + im = Image.fromarray(panel) + if SCALE != 1: + im = im.resize((panel.shape[1]*SCALE, H_*SCALE), Image.NEAREST) + frames.append(im) + stats.append([(mode == i).mean() for i in range(4)]) + + if out.endswith(".webm") or out.endswith(".mp4"): + # Preferred. GIF would impose its own 256-colour palette on top of + # output whose entire subject is colour fidelity, and costs ~4x the + # bytes doing it. -lossless keeps the panels pixel-exact. + import subprocess + w, h = frames[0].size + cmd = ["ffmpeg", "-v", "error", "-y", "-f", "rawvideo", "-pix_fmt", "rgb24", + "-s", f"{w}x{h}", "-r", "12", "-i", "-"] + # yuv444p, not 420: the mode map is flat saturated colour on a 4-pixel + # grid, and chroma subsampling smears exactly those edges. Lossless is + # available but runs larger than the GIF on this content; crf 18 in 444 + # is visually clean at a quarter the size. + if out.endswith(".webm"): + cmd += ["-c:v", "libvpx-vp9", "-pix_fmt", "yuv444p", "-row-mt", "1"] + cmd += ["-lossless", "1"] if LOSSLESS else ["-crf", "18", "-b:v", "0"] + else: + cmd += ["-c:v", "libx264", "-crf", "12", "-pix_fmt", "yuv444p"] + p = subprocess.Popen(cmd + [out], stdin=subprocess.PIPE) + for f in frames: + p.stdin.write(np.asarray(f.convert("RGB")).tobytes()) + p.stdin.close(); p.wait() + else: + # One shared adaptive palette: per-frame palettes are what make a naive + # GIF of this enormous, and a stable palette also stops the mode-map + # colours shimmering between frames. + shared = frames[0].quantize(colors=192, method=Image.MEDIANCUT) + q = [f.quantize(palette=shared, dither=Image.NONE) for f in frames] + q[0].save(out, save_all=True, append_images=q[1:], + duration=1000//12, loop=0, optimize=True) + st = np.array(stats) + print(f"{len(frames)} frames -> {out} ({os.path.getsize(out)/1024:.0f} KB)") + print(" panels: palettised source | decoded | block mode map") + for i, n in enumerate(LABEL): + print(f" {n:4s} mean {100*st[:,i].mean():5.1f}% " + f"per-frame range {100*st[:,i].min():5.1f}% .. {100*st[:,i].max():5.1f}%") + ns = 100 * (1 - st[:, 0]) + print(f" non-SKIP: median {np.median(ns):.1f}% p90 {np.percentile(ns,90):.1f}%") + +if __name__ == "__main__": + main() diff --git a/tools/encoder/encode.py b/tools/encoder/encode.py index bdb0d1b..1557d86 100644 --- a/tools/encoder/encode.py +++ b/tools/encoder/encode.py @@ -31,6 +31,12 @@ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) import numpy as np import vq as VQ, vq_hybrid as H, ratectl as RC +# Measured on the emulated 68000, FINDINGS 24. Instruction cycles against +# zero-wait-state memory, so these are floors, not hardware predictions. +BLIT_PCT = 53.6 # V1: compose in RAM, then a row-linear movem.l blit +DIRECT_PCT = 76.6 # V4: write every block straight into GVRAM +CROSSOVER_PCT = 100 * BLIT_PCT / DIRECT_PCT + def pack_modes(mode): """2 bits per block, MSB-first -- cheap for the 68000 to shift out.""" @@ -133,6 +139,24 @@ def main(): print(f" modes: SKIP {r['skip']:.1f}% V1 {r['v1']:.1f}% " f"V4 {r['v4']:.1f}% RAW {r['raw']:.1f}%") + # PER-FRAME non-SKIP distribution. The mean above cannot answer the + # decoder-architecture question (FINDINGS 24.5): decode-direct-to-GVRAM + # costs 76.6% of a 12fps frame budget x (non-SKIP fraction), while + # compose-in-RAM-then-blit is a flat 53.6% regardless. They cross at 70%, + # and that is a decision taken FRAME BY FRAME -- a scene cut is ~100% + # non-SKIP and a held frame near 0%, so their mean describes no real frame. + ns = np.array([100 * (mm != 0).mean() for mm in enc["modes"]]) + over = int((ns > CROSSOVER_PCT).sum()) + print(f" non-SKIP blocks/frame: median {np.median(ns):.1f}% " + f"p90 {np.percentile(ns, 90):.1f}% max {ns.max():.1f}%") + print(f" frames above the {CROSSOVER_PCT:.0f}% blit crossover: " + f"{over}/{len(ns)} ({100*over/len(ns):.1f}%) -> " + f"{'compose+blit wins on those' if over else 'direct-to-GVRAM wins throughout'}") + cost = np.minimum(BLIT_PCT, DIRECT_PCT * ns / 100) + print(f" display cost if the player picks the cheaper path per frame: " + f"median {np.median(cost):.1f}% p90 {np.percentile(cost, 90):.1f}% " + f"max {cost.max():.1f}% of a 12fps frame") + if a.preview: from PIL import Image f = len(idx) // 2 diff --git a/tools/encoder/extract.py b/tools/encoder/extract.py index 5edb42a..4ee78ab 100644 --- a/tools/encoder/extract.py +++ b/tools/encoder/extract.py @@ -35,7 +35,13 @@ def extract(stream, outdir, fps=12, mode="crop", start=None, dur=None): return n if __name__ == "__main__": + # extract.py [fps] [mode] [start_s] [dur_s] + # start/dur matter for the long streams: 00223 is 9.4 min and the parts + # that stress the codec are a few seconds each (tools/analysis/07 finds + # them). Without them a survey silently averages action into idle scenes. stream, outdir = sys.argv[1], sys.argv[2] fps = int(sys.argv[3]) if len(sys.argv) > 3 else 12 mode = sys.argv[4] if len(sys.argv) > 4 else "crop" - extract(stream, outdir, fps, mode) + start = float(sys.argv[5]) if len(sys.argv) > 5 else None + dur = float(sys.argv[6]) if len(sys.argv) > 6 else None + extract(stream, outdir, fps, mode, start, dur)