From e652810de77bfb4e893eb711b328a28b9d0945d0 Mon Sep 17 00:00:00 2001 From: Viktor Vaczi Date: Fri, 14 Aug 2026 12:07:18 +0200 Subject: [PATCH] =?UTF-8?q?bench:=20jetson=20FPS=20battery=20=E2=80=94=20b?= =?UTF-8?q?oth=20backends=20saturate=20at=2080.9=20MB=20scale?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Parameterizes the throttle-sweep FPS battery (fpsBattery helper) and adds a jetson-agx-thor run: asyncify raf 2.1-5.3 / distinct 0.2-0.8 vs JSPI raf 1.9-5.0 / distinct 0.1-0.7 — indistinguishable. Suspension overhead is a per-event-loop-turn cost; at hundreds of ms of GAL work per frame it stops discriminating. vme-wren (~27 MB) remains the discriminating size class. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_016X9eh1s5sTx1o9Em9KBuwR --- .../perf-bench-asyncify.ndjson | 7 ++++ .../bench-data-2026-08/perf-bench-jspi.ndjson | 7 ++++ .../jspi-vs-asyncify-bench-2026-08.md | 10 ++++++ tests/kicad/pcbnew-large-perf.spec.ts | 35 +++++++++++++++---- 4 files changed, 53 insertions(+), 6 deletions(-) diff --git a/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-asyncify.ndjson b/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-asyncify.ndjson index 7fe0e5a..1123c72 100644 --- a/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-asyncify.ndjson +++ b/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-asyncify.ndjson @@ -18,3 +18,10 @@ {"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T09:48:29.439Z","section":"fps-vme","throttle":6,"rep":2,"rafFps":34.3,"distinctFps":2.8} {"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T09:48:29.442Z","section":"fps-vme-postmem","postFpsMem":{"wasmHeapBytes":962330624,"jsHeapBytes":60300000}} {"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T09:48:46.855Z","section":"open-jetson","bytes":84806775,"outcome":"loaded","openMs":14584,"openPeakHeap":1904214016,"postOpenMem":{"wasmHeapBytes":1904214016,"jsHeapBytes":116000000}} +{"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T10:05:39.841Z","section":"fps-jetson","throttle":1,"rep":1,"rafFps":4.4,"distinctFps":0.5} +{"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T10:05:46.092Z","section":"fps-jetson","throttle":1,"rep":2,"rafFps":5.3,"distinctFps":0.8} +{"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T10:05:52.980Z","section":"fps-jetson","throttle":4,"rep":1,"rafFps":2.1,"distinctFps":0.4} +{"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T10:06:00.712Z","section":"fps-jetson","throttle":4,"rep":2,"rafFps":3.4,"distinctFps":0.6} +{"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T10:06:06.843Z","section":"fps-jetson","throttle":6,"rep":1,"rafFps":2.5,"distinctFps":0.2} +{"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T10:06:14.282Z","section":"fps-jetson","throttle":6,"rep":2,"rafFps":3.9,"distinctFps":0.7} +{"arm":"asyncify","sha256":"6dab493196d060d7","when":"2026-08-14T10:06:14.285Z","section":"fps-jetson-postmem","postFpsMem":{"wasmHeapBytes":1904214016,"jsHeapBytes":177000000}} diff --git a/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-jspi.ndjson b/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-jspi.ndjson index c1988d9..22e1e71 100644 --- a/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-jspi.ndjson +++ b/docs/features/async/migration-evidence/bench-data-2026-08/perf-bench-jspi.ndjson @@ -25,3 +25,10 @@ {"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T09:52:24.000Z","section":"fps-vme","throttle":6,"rep":1,"rafFps":34.3,"distinctFps":3.3,"note":"glcanvas sampler re-run; when approximate"} {"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T09:52:30.000Z","section":"fps-vme","throttle":6,"rep":2,"rafFps":41.8,"distinctFps":5.3,"note":"glcanvas sampler re-run; when approximate"} {"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T09:52:30.100Z","section":"fps-vme-postmem","postFpsMem":{"wasmHeapBytes":801898496,"jsHeapBytes":64000000}} +{"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T10:04:14.509Z","section":"fps-jetson","throttle":1,"rep":1,"rafFps":5,"distinctFps":0.7} +{"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T10:04:21.557Z","section":"fps-jetson","throttle":1,"rep":2,"rafFps":4.1,"distinctFps":0.7} +{"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T10:04:32.709Z","section":"fps-jetson","throttle":4,"rep":1,"rafFps":2.3,"distinctFps":0.4} +{"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T10:04:40.639Z","section":"fps-jetson","throttle":4,"rep":2,"rafFps":1.9,"distinctFps":0.1} +{"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T10:04:48.685Z","section":"fps-jetson","throttle":6,"rep":1,"rafFps":2.1,"distinctFps":0.1} +{"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T10:04:55.534Z","section":"fps-jetson","throttle":6,"rep":2,"rafFps":4.3,"distinctFps":0.3} +{"arm":"jspi","sha256":"6f03e62d6d5619c3","when":"2026-08-14T10:04:55.537Z","section":"fps-jetson-postmem","postFpsMem":{"wasmHeapBytes":1730805760,"jsHeapBytes":157000000}} diff --git a/docs/features/async/migration-evidence/jspi-vs-asyncify-bench-2026-08.md b/docs/features/async/migration-evidence/jspi-vs-asyncify-bench-2026-08.md index 3d89b0f..0f374e2 100644 --- a/docs/features/async/migration-evidence/jspi-vs-asyncify-bench-2026-08.md +++ b/docs/features/async/migration-evidence/jspi-vs-asyncify-bench-2026-08.md @@ -156,6 +156,16 @@ The distinct-frame gap grows from +21 % at 1× to +68 % at 4× — the same "advantage widens under throttle" signature doc-12 used to prove lower CPU-per-operation (as opposed to just a smaller module). +**FPS on jetson-agx-thor (80.9 MB)** — added 2026-08-14 pm: at this scale BOTH +arms saturate outright and the backend difference disappears into noise +(asyncify raf 2.1–5.3 / distinct 0.2–0.8; JSPI raf 1.9–5.0 / distinct +0.1–0.7; post-FPS heap 1.90 vs 1.73 GB). With <1 real redraw/s the 6 s +windows count 1–5 frames, i.e. quantization noise. Reading: suspension +overhead is a per-event-loop-turn cost — once per-frame GAL/geometry work is +hundreds of ms, it no longer discriminates. vme-wren (~27 MB) is the size +class where the backend visibly matters for interaction; jetson is the +"both need render-side optimization" regime. + **Memory checkpoints** (wasm linear memory; Chromium `usedJSHeapSize` tracked alongside, differences <10 %): boot 557 vs 387 MB; after vme-wren open 962 vs 802 MB; unchanged after the FPS sweep on both arms. diff --git a/tests/kicad/pcbnew-large-perf.spec.ts b/tests/kicad/pcbnew-large-perf.spec.ts index 8a71058..fe2526c 100644 --- a/tests/kicad/pcbnew-large-perf.spec.ts +++ b/tests/kicad/pcbnew-large-perf.spec.ts @@ -118,11 +118,18 @@ test.describe('pcbnew large-board bench', () => { }); } - test('FPS on vme-wren across throttles', async ({ page, testLogger }) => { - test.setTimeout(900000); + // Shared throttle-sweep FPS battery: load editor, fetch+open the board, + // then rAF + distinct-glcanvas-frame FPS at each throttle rate. + async function fpsBattery( + page: Parameters[2]>[0]['page'], + testLogger: { consoleLogs: string[]; errors: string[] }, + boardUrl: string, + memfsPath: string, + section: string, + ): Promise { await measureLoad(page, '/kicad/pcbnew.html'); - await fetchIntoMemfs(page, VME_URL, '/home/kicad/documents/vme-wren.kicad_pcb'); - await openAndWait(page, '/home/kicad/documents/vme-wren.kicad_pcb', 'board', testLogger, 480000); + await fetchIntoMemfs(page, boardUrl, memfsPath); + await openAndWait(page, memfsPath, 'board', testLogger, 780000); await page.keyboard.press('Escape').catch(() => {}); // eslint-disable-line -- best-effort Escape const cdp = await page.context().newCDPSession(page); @@ -130,13 +137,29 @@ test.describe('pcbnew large-board bench', () => { await setThrottle(cdp, rate); for (let rep = 0; rep < FPS_REPS; rep++) { const f = await measureFpsDetailed(page, FPS_SECS); - record('fps-vme', { throttle: rate, rep: rep + 1, ...f }); + record(section, { throttle: rate, rep: rep + 1, ...f }); expect(f.rafFps).toBeGreaterThan(0); } } await setThrottle(cdp, 1); const mem = await sampleMemory(page); - record('fps-vme-postmem', { postFpsMem: mem }); + record(`${section}-postmem`, { postFpsMem: mem }); + } + + test('FPS on vme-wren across throttles', async ({ page, testLogger }) => { + test.setTimeout(900000); + await fpsBattery(page, testLogger, VME_URL, '/home/kicad/documents/vme-wren.kicad_pcb', 'fps-vme'); + }); + + test('FPS on jetson-agx-thor across throttles', async ({ page, testLogger }) => { + test.setTimeout(900000); + await fpsBattery( + page, + testLogger, + JETSON_URL, + '/home/kicad/documents/jetson-agx-thor-baseboard.kicad_pcb', + 'fps-jetson', + ); }); test('open jetson-agx-thor (80.9 MB) — outcome, OOM allowed', async ({ page, testLogger }) => {