diff --git a/bench-baseline.log b/bench-baseline.log
new file mode 100644
index 0000000..802217b
--- /dev/null
+++ b/bench-baseline.log
@@ -0,0 +1,35 @@
+fixture: http://127.0.0.1:52273/product/prd-bench
+device: mobile reps: 2 variants: baseline
+
+rep 1/2 baseline wall 7485ms cpu 2660ms task 1758ms eval 15 cdp 435 reviews 461
+rep 2/2 baseline wall 7456ms cpu 2680ms task 1741ms eval 15 cdp 435 reviews 461
+
+## cost — median of 2 reps (delta vs baseline)
+
+variant wall cpu(tree) task task-other proc-cpu script v8compile layout style settle postProc evals cdp-sent passes height elements
+-------- ---- --------- ---- ---------- -------- ------ --------- ------ ----- ------ -------- ----- -------- ------ ------ --------
+baseline 7471 2670 1750 0 0 23 0 49 35 7126 225 15 435 3 18358 21864
+
+## fidelity — must match baseline, or the cost win is not a win
+
+variant reviews revealed lazyTiles imgGated offers shadow bytes
+-------- ------- -------- --------- -------- ------ ------ ------
+baseline 461 1 288 1 6 1788 945084
+
+
+## baseline CDP methods sent (one representative run, 435 total)
+
+ 385 Fetch.fulfillRequest
+ 15 Runtime.callFunctionOn
+ 4 Fetch.continueRequest
+ 2 Target.setAutoAttach
+ 2 Runtime.runIfWaitingForDebugger
+ 2 Page.addScriptToEvaluateOnNewDocument
+ 2 Emulation.setDeviceMetricsOverride
+ 2 Emulation.setTouchEmulationEnabled
+ 1 Target.createBrowserContext
+ 1 Browser.setDownloadBehavior
+ 1 Target.createTarget
+ 1 Network.enable
+ 1 Page.enable
+ 1 Page.getFrameTree
diff --git a/bench-load.log b/bench-load.log
new file mode 100644
index 0000000..aa8cde8
--- /dev/null
+++ b/bench-load.log
@@ -0,0 +1,45 @@
+fixture: http://127.0.0.1:53872/product/prd-bench
+device: mobile reps: 2 concurrency: 1,8
+cores: 14
+
+c= 1 baseline batch 7386ms per-render 7386ms cpu/render 2585ms 0.14/s reviews 461
+c= 1 monitor batch 7380ms per-render 7379ms cpu/render 2605ms 0.14/s reviews 461
+c= 1 tall-viewport batch 3331ms per-render 3330ms cpu/render 1595ms 0.3/s reviews 461
+c= 1 step-raf batch 3079ms per-render 3078ms cpu/render 1575ms 0.32/s reviews 461
+c= 1 images-off batch 7362ms per-render 7362ms cpu/render 2325ms 0.14/s reviews 461
+c= 1 no-prune-css batch 7241ms per-render 7240ms cpu/render 2390ms 0.14/s reviews 461
+c= 1 best-of batch 1697ms per-render 1697ms cpu/render 1090ms 0.59/s reviews 461
+c= 1 cpu-stack-safe batch 1665ms per-render 1664ms cpu/render 1090ms 0.6/s reviews 461
+
+c= 8 baseline batch 7588ms per-render 7563ms cpu/render 2206ms 1.05/s reviews 461
+c= 8 monitor batch 7561ms per-render 7534ms cpu/render 2171ms 1.06/s reviews 461
+c= 8 tall-viewport batch 4533ms per-render 3482ms cpu/render 1883ms 1.76/s reviews 461
+c= 8 step-raf batch 3343ms per-render 3316ms cpu/render 1725ms 2.39/s reviews 461
+c= 8 images-off batch 7454ms per-render 7445ms cpu/render 1973ms 1.07/s reviews 461
+c= 8 no-prune-css batch 7395ms per-render 7373ms cpu/render 2076ms 1.08/s reviews 461
+c= 8 best-of batch 2015ms per-render 1976ms cpu/render 1348ms 3.97/s reviews 461
+c= 8 cpu-stack-safe batch 1967ms per-render 1924ms cpu/render 1358ms 4.07/s reviews 461
+
+
+## throughput under load (median of reps)
+
+conc variant batch per-render max-wall cpu/render renders/s vs base /s reviews
+---- -------------- ------ ---------- -------- ---------- --------- ---------- -------
+1 baseline 7386ms 7386ms 7386ms 2585ms 0.14 1.00x 461
+1 monitor 7380ms 7379ms 7379ms 2605ms 0.14 1.00x 461
+1 tall-viewport 3331ms 3330ms 3330ms 1595ms 0.3 2.14x 461
+1 step-raf 3079ms 3078ms 3078ms 1575ms 0.32 2.29x 461
+1 images-off 7362ms 7362ms 7362ms 2325ms 0.14 1.00x 461
+1 no-prune-css 7241ms 7240ms 7240ms 2390ms 0.14 1.00x 461
+1 best-of 1697ms 1697ms 1697ms 1090ms 0.59 4.21x 461
+1 cpu-stack-safe 1665ms 1664ms 1664ms 1090ms 0.6 4.29x 461
+8 baseline 7588ms 7563ms 7588ms 2206ms 1.05 1.00x 461
+8 monitor 7561ms 7534ms 7560ms 2171ms 1.06 1.01x 461
+8 tall-viewport 4533ms 3482ms 4532ms 1883ms 1.76 1.68x 461
+8 step-raf 3343ms 3316ms 3342ms 1725ms 2.39 2.28x 461
+8 images-off 7454ms 7445ms 7453ms 1973ms 1.07 1.02x 461
+8 no-prune-css 7395ms 7373ms 7394ms 2076ms 1.08 1.03x 461
+8 best-of 2015ms 1976ms 2015ms 1348ms 3.97 3.78x 461
+8 cpu-stack-safe 1967ms 1924ms 1967ms 1358ms 4.07 3.88x 461
+
+wrote /private/tmp/prerender-plugin/perf-cdp/bench/render-cpu/results/load.json
diff --git a/bench-mobile-final.log b/bench-mobile-final.log
new file mode 100644
index 0000000..ddb8ab8
--- /dev/null
+++ b/bench-mobile-final.log
@@ -0,0 +1,232 @@
+fixture: http://127.0.0.1:53114/product/prd-bench
+device: mobile reps: 5 variants: baseline, combine-scroll-count, combine-waitfor, combine-tail, install-helpers, combines-all, exists-shortcircuit, native-count, monitor, tall-viewport, viewport-2000, viewport-10000, viewport-taller-than-page, step-20ms, step-raf, idle-tight, poll-60ms, shared-context, images-off, stub-off, no-prune-css, no-flatten-shadow, no-minify-css, no-remove-attrs, noindex-bail, tall-plus-raf, best-of, reduced-motion
+
+rep 1/5 baseline wall 7466ms cpu 2390ms task 1529ms eval 15 cdp 435 reviews 461
+rep 1/5 combine-scroll-count wall 9761ms cpu 2910ms task 1871ms eval 13 cdp 433 reviews 461
+rep 1/5 combine-waitfor wall 7453ms cpu 2520ms task 1650ms eval 14 cdp 434 reviews 461
+rep 1/5 combine-tail wall 7475ms cpu 2620ms task 1723ms eval 14 cdp 435 reviews 461
+rep 1/5 install-helpers wall 7462ms cpu 2650ms task 1754ms eval 15 cdp 436 reviews 461
+rep 1/5 combines-all wall 7453ms cpu 2510ms task 1646ms eval 13 cdp 434 reviews 461
+rep 1/5 exists-shortcircuit wall 7467ms cpu 2450ms task 1569ms eval 15 cdp 435 reviews 461
+rep 1/5 native-count wall 7461ms cpu 2510ms task 1663ms eval 15 cdp 436 reviews 461
+rep 1/5 monitor wall 7478ms cpu 2640ms task 1733ms eval 15 cdp 436 reviews 461
+rep 1/5 tall-viewport wall 3399ms cpu 1560ms task 877ms eval 15 cdp 435 reviews 461
+rep 1/5 viewport-2000 wall 4674ms cpu 1820ms task 992ms eval 15 cdp 435 reviews 461
+rep 1/5 viewport-10000 wall 2969ms cpu 1690ms task 877ms eval 15 cdp 435 reviews 461
+rep 1/5 viewport-taller-than-page wall 2762ms cpu 1810ms task 823ms eval 15 cdp 435 reviews 461
+rep 1/5 step-20ms wall 4161ms cpu 1940ms task 1127ms eval 15 cdp 435 reviews 461
+rep 1/5 step-raf wall 3138ms cpu 1630ms task 989ms eval 15 cdp 435 reviews 461
+rep 1/5 idle-tight wall 6548ms cpu 2510ms task 1604ms eval 15 cdp 435 reviews 461
+rep 1/5 poll-60ms wall 7451ms cpu 2550ms task 1669ms eval 15 cdp 435 reviews 461
+rep 1/5 shared-context wall 7436ms cpu 2420ms task 1555ms eval 15 cdp 432 reviews 461
+rep 1/5 images-off wall 7417ms cpu 2290ms task 1610ms eval 15 cdp 50 reviews 461
+rep 1/5 stub-off wall 7434ms cpu 2510ms task 1669ms eval 15 cdp 435 reviews 461
+rep 1/5 no-prune-css wall 7294ms cpu 2390ms task 1518ms eval 15 cdp 435 reviews 461
+rep 1/5 no-flatten-shadow wall 7363ms cpu 2400ms task 1562ms eval 15 cdp 435 reviews 461
+rep 1/5 no-minify-css wall 7456ms cpu 2390ms task 1544ms eval 15 cdp 435 reviews 461
+rep 1/5 no-remove-attrs wall 7446ms cpu 2520ms task 1636ms eval 15 cdp 435 reviews 461
+rep 1/5 noindex-bail wall 7445ms cpu 2460ms task 1606ms eval 15 cdp 435 reviews 461
+rep 1/5 tall-plus-raf wall 2673ms cpu 1550ms task 770ms eval 15 cdp 435 reviews 461
+rep 1/5 best-of wall 1733ms cpu 1170ms task 673ms eval 14 cdp 434 reviews 461
+rep 1/5 reduced-motion wall 7440ms cpu 2570ms task 1677ms eval 15 cdp 436 reviews 461
+rep 2/5 baseline wall 7444ms cpu 2610ms task 1726ms eval 15 cdp 435 reviews 461
+rep 2/5 combine-scroll-count wall 9755ms cpu 3260ms task 2152ms eval 13 cdp 433 reviews 461
+rep 2/5 combine-waitfor wall 7434ms cpu 2450ms task 1580ms eval 14 cdp 434 reviews 461
+rep 2/5 combine-tail wall 7470ms cpu 2560ms task 1664ms eval 14 cdp 435 reviews 461
+rep 2/5 install-helpers wall 7464ms cpu 2640ms task 1753ms eval 15 cdp 436 reviews 461
+rep 2/5 combines-all wall 7474ms cpu 2670ms task 1754ms eval 13 cdp 434 reviews 461
+rep 2/5 exists-shortcircuit wall 7456ms cpu 2610ms task 1693ms eval 15 cdp 435 reviews 461
+rep 2/5 native-count wall 7461ms cpu 2660ms task 1771ms eval 15 cdp 436 reviews 461
+rep 2/5 monitor wall 7447ms cpu 2490ms task 1632ms eval 15 cdp 436 reviews 461
+rep 2/5 tall-viewport wall 3364ms cpu 1540ms task 866ms eval 15 cdp 435 reviews 461
+rep 2/5 viewport-2000 wall 4672ms cpu 1830ms task 1022ms eval 15 cdp 435 reviews 461
+rep 2/5 viewport-10000 wall 2953ms cpu 1740ms task 881ms eval 15 cdp 435 reviews 461
+rep 2/5 viewport-taller-than-page wall 2759ms cpu 1760ms task 812ms eval 15 cdp 435 reviews 461
+rep 2/5 step-20ms wall 4164ms cpu 1940ms task 1154ms eval 15 cdp 435 reviews 461
+rep 2/5 step-raf wall 3143ms cpu 1580ms task 966ms eval 15 cdp 435 reviews 461
+rep 2/5 idle-tight wall 6551ms cpu 2500ms task 1612ms eval 15 cdp 435 reviews 461
+rep 2/5 poll-60ms wall 7441ms cpu 2610ms task 1691ms eval 15 cdp 435 reviews 461
+rep 2/5 shared-context wall 7476ms cpu 2560ms task 1670ms eval 15 cdp 432 reviews 461
+rep 2/5 images-off wall 7371ms cpu 2170ms task 1495ms eval 15 cdp 50 reviews 461
+rep 2/5 stub-off wall 7420ms cpu 2440ms task 1593ms eval 15 cdp 435 reviews 461
+rep 2/5 no-prune-css wall 7301ms cpu 2400ms task 1531ms eval 15 cdp 435 reviews 461
+rep 2/5 no-flatten-shadow wall 7356ms cpu 2450ms task 1549ms eval 15 cdp 435 reviews 461
+rep 2/5 no-minify-css wall 7440ms cpu 2570ms task 1682ms eval 15 cdp 435 reviews 461
+rep 2/5 no-remove-attrs wall 7453ms cpu 2550ms task 1686ms eval 15 cdp 435 reviews 461
+rep 2/5 noindex-bail wall 7442ms cpu 2450ms task 1584ms eval 15 cdp 435 reviews 461
+rep 2/5 tall-plus-raf wall 2699ms cpu 1560ms task 799ms eval 15 cdp 435 reviews 461
+rep 2/5 best-of wall 1762ms cpu 1150ms task 654ms eval 14 cdp 434 reviews 461
+rep 2/5 reduced-motion wall 7451ms cpu 2610ms task 1690ms eval 15 cdp 436 reviews 461
+rep 3/5 baseline wall 7441ms cpu 2490ms task 1610ms eval 15 cdp 435 reviews 461
+rep 3/5 combine-scroll-count wall 9747ms cpu 2930ms task 1923ms eval 13 cdp 433 reviews 461
+rep 3/5 combine-waitfor wall 7449ms cpu 2550ms task 1656ms eval 14 cdp 434 reviews 461
+rep 3/5 combine-tail wall 7441ms cpu 2570ms task 1700ms eval 14 cdp 435 reviews 461
+rep 3/5 install-helpers wall 7465ms cpu 2570ms task 1674ms eval 15 cdp 436 reviews 461
+rep 3/5 combines-all wall 7463ms cpu 2500ms task 1609ms eval 13 cdp 434 reviews 461
+rep 3/5 exists-shortcircuit wall 7437ms cpu 2510ms task 1630ms eval 15 cdp 435 reviews 461
+rep 3/5 native-count wall 7460ms cpu 2510ms task 1662ms eval 15 cdp 436 reviews 461
+rep 3/5 monitor wall 7455ms cpu 2460ms task 1618ms eval 15 cdp 436 reviews 461
+rep 3/5 tall-viewport wall 3383ms cpu 1590ms task 889ms eval 15 cdp 435 reviews 461
+rep 3/5 viewport-2000 wall 4690ms cpu 1910ms task 1081ms eval 15 cdp 435 reviews 461
+rep 3/5 viewport-10000 wall 2939ms cpu 1710ms task 845ms eval 15 cdp 435 reviews 461
+rep 3/5 viewport-taller-than-page wall 2772ms cpu 1770ms task 802ms eval 15 cdp 435 reviews 461
+rep 3/5 step-20ms wall 4160ms cpu 1880ms task 1126ms eval 15 cdp 435 reviews 461
+rep 3/5 step-raf wall 3104ms cpu 1540ms task 923ms eval 15 cdp 435 reviews 461
+rep 3/5 idle-tight wall 6551ms cpu 2430ms task 1571ms eval 15 cdp 435 reviews 461
+rep 3/5 poll-60ms wall 7434ms cpu 2450ms task 1589ms eval 15 cdp 435 reviews 461
+rep 3/5 shared-context wall 7456ms cpu 2580ms task 1668ms eval 15 cdp 432 reviews 461
+rep 3/5 images-off wall 7393ms cpu 2300ms task 1574ms eval 15 cdp 50 reviews 461
+rep 3/5 stub-off wall 7424ms cpu 2310ms task 1510ms eval 15 cdp 435 reviews 461
+rep 3/5 no-prune-css wall 7289ms cpu 2170ms task 1334ms eval 15 cdp 435 reviews 461
+rep 3/5 no-flatten-shadow wall 7336ms cpu 2250ms task 1408ms eval 15 cdp 435 reviews 461
+rep 3/5 no-minify-css wall 7451ms cpu 2530ms task 1636ms eval 15 cdp 435 reviews 461
+rep 3/5 no-remove-attrs wall 7450ms cpu 2580ms task 1691ms eval 15 cdp 435 reviews 461
+rep 3/5 noindex-bail wall 7435ms cpu 2570ms task 1702ms eval 15 cdp 435 reviews 461
+rep 3/5 tall-plus-raf wall 2669ms cpu 1560ms task 792ms eval 15 cdp 435 reviews 461
+rep 3/5 best-of wall 1720ms cpu 1170ms task 654ms eval 14 cdp 434 reviews 461
+rep 3/5 reduced-motion wall 7453ms cpu 2580ms task 1686ms eval 15 cdp 436 reviews 461
+rep 4/5 baseline wall 7450ms cpu 2660ms task 1723ms eval 15 cdp 435 reviews 461
+rep 4/5 combine-scroll-count wall 9754ms cpu 3270ms task 2191ms eval 13 cdp 433 reviews 461
+rep 4/5 combine-waitfor wall 7438ms cpu 2630ms task 1719ms eval 14 cdp 434 reviews 461
+rep 4/5 combine-tail wall 7461ms cpu 2690ms task 1783ms eval 14 cdp 435 reviews 461
+rep 4/5 install-helpers wall 7459ms cpu 2630ms task 1716ms eval 15 cdp 436 reviews 461
+rep 4/5 combines-all wall 7451ms cpu 2680ms task 1765ms eval 13 cdp 434 reviews 461
+rep 4/5 exists-shortcircuit wall 7451ms cpu 2650ms task 1708ms eval 15 cdp 435 reviews 461
+rep 4/5 native-count wall 7439ms cpu 2550ms task 1653ms eval 15 cdp 436 reviews 461
+rep 4/5 monitor wall 7428ms cpu 2520ms task 1654ms eval 15 cdp 436 reviews 461
+rep 4/5 tall-viewport wall 3381ms cpu 1650ms task 948ms eval 15 cdp 435 reviews 461
+rep 4/5 viewport-2000 wall 4675ms cpu 1910ms task 1069ms eval 15 cdp 435 reviews 461
+rep 4/5 viewport-10000 wall 2963ms cpu 1750ms task 883ms eval 15 cdp 435 reviews 461
+rep 4/5 viewport-taller-than-page wall 2760ms cpu 1750ms task 785ms eval 15 cdp 435 reviews 461
+rep 4/5 step-20ms wall 4169ms cpu 1900ms task 1105ms eval 15 cdp 435 reviews 461
+rep 4/5 step-raf wall 3131ms cpu 1570ms task 942ms eval 15 cdp 435 reviews 461
+rep 4/5 idle-tight wall 6544ms cpu 2500ms task 1610ms eval 15 cdp 435 reviews 461
+rep 4/5 poll-60ms wall 7443ms cpu 2560ms task 1671ms eval 15 cdp 435 reviews 461
+rep 4/5 shared-context wall 7436ms cpu 2600ms task 1711ms eval 15 cdp 432 reviews 461
+rep 4/5 images-off wall 7377ms cpu 2380ms task 1653ms eval 15 cdp 50 reviews 461
+rep 4/5 stub-off wall 7419ms cpu 2600ms task 1708ms eval 15 cdp 435 reviews 461
+rep 4/5 no-prune-css wall 7303ms cpu 2540ms task 1602ms eval 15 cdp 435 reviews 461
+rep 4/5 no-flatten-shadow wall 7354ms cpu 2510ms task 1620ms eval 15 cdp 435 reviews 461
+rep 4/5 no-minify-css wall 7449ms cpu 2550ms task 1658ms eval 15 cdp 435 reviews 461
+rep 4/5 no-remove-attrs wall 7475ms cpu 2690ms task 1762ms eval 15 cdp 435 reviews 461
+rep 4/5 noindex-bail wall 7450ms cpu 2410ms task 1549ms eval 15 cdp 435 reviews 461
+rep 4/5 tall-plus-raf wall 2683ms cpu 1560ms task 789ms eval 15 cdp 435 reviews 461
+rep 4/5 best-of wall 1711ms cpu 1170ms task 655ms eval 14 cdp 434 reviews 461
+rep 4/5 reduced-motion wall 7450ms cpu 2640ms task 1710ms eval 15 cdp 436 reviews 461
+rep 5/5 baseline wall 7454ms cpu 2520ms task 1648ms eval 15 cdp 435 reviews 461
+rep 5/5 combine-scroll-count wall 9750ms cpu 3050ms task 2021ms eval 13 cdp 433 reviews 461
+rep 5/5 combine-waitfor wall 7452ms cpu 2550ms task 1664ms eval 14 cdp 434 reviews 461
+rep 5/5 combine-tail wall 7466ms cpu 2650ms task 1726ms eval 14 cdp 435 reviews 461
+rep 5/5 install-helpers wall 7473ms cpu 2680ms task 1751ms eval 15 cdp 436 reviews 461
+rep 5/5 combines-all wall 7462ms cpu 2500ms task 1620ms eval 13 cdp 434 reviews 461
+rep 5/5 exists-shortcircuit wall 7456ms cpu 2400ms task 1517ms eval 15 cdp 435 reviews 461
+rep 5/5 native-count wall 7428ms cpu 2320ms task 1469ms eval 15 cdp 436 reviews 461
+rep 5/5 monitor wall 7432ms cpu 2320ms task 1483ms eval 15 cdp 436 reviews 461
+rep 5/5 tall-viewport wall 3366ms cpu 1520ms task 840ms eval 15 cdp 435 reviews 461
+rep 5/5 viewport-2000 wall 4673ms cpu 2010ms task 1135ms eval 15 cdp 435 reviews 461
+rep 5/5 viewport-10000 wall 2956ms cpu 1760ms task 893ms eval 15 cdp 435 reviews 461
+rep 5/5 viewport-taller-than-page wall 2762ms cpu 1800ms task 826ms eval 15 cdp 435 reviews 461
+rep 5/5 step-20ms wall 4164ms cpu 1930ms task 1148ms eval 15 cdp 435 reviews 461
+rep 5/5 step-raf wall 3137ms cpu 1630ms task 984ms eval 15 cdp 435 reviews 461
+rep 5/5 idle-tight wall 6543ms cpu 2450ms task 1585ms eval 15 cdp 435 reviews 461
+rep 5/5 poll-60ms wall 7445ms cpu 2600ms task 1711ms eval 15 cdp 435 reviews 461
+rep 5/5 shared-context wall 7460ms cpu 2620ms task 1728ms eval 15 cdp 432 reviews 461
+rep 5/5 images-off wall 7394ms cpu 2360ms task 1643ms eval 15 cdp 50 reviews 461
+rep 5/5 stub-off wall 7430ms cpu 2620ms task 1734ms eval 15 cdp 435 reviews 461
+rep 5/5 no-prune-css wall 7302ms cpu 2500ms task 1585ms eval 15 cdp 435 reviews 461
+rep 5/5 no-flatten-shadow wall 7358ms cpu 2510ms task 1619ms eval 15 cdp 435 reviews 461
+rep 5/5 no-minify-css wall 7441ms cpu 2700ms task 1758ms eval 15 cdp 435 reviews 461
+rep 5/5 no-remove-attrs wall 7453ms cpu 2640ms task 1722ms eval 15 cdp 435 reviews 461
+rep 5/5 noindex-bail wall 7422ms cpu 2610ms task 1700ms eval 15 cdp 435 reviews 461
+rep 5/5 tall-plus-raf wall 2681ms cpu 1560ms task 797ms eval 15 cdp 435 reviews 461
+rep 5/5 best-of wall 1759ms cpu 1200ms task 685ms eval 14 cdp 434 reviews 461
+rep 5/5 reduced-motion wall 7443ms cpu 2650ms task 1747ms eval 15 cdp 436 reviews 461
+
+## cost — median of 5 reps (delta vs baseline)
+
+variant wall cpu(tree) task task-other proc-cpu script v8compile devtools layout style settle scroll idle count gate postProc evals cdp-sent passes height elements
+------------------------- ----------- ----------- ----------- ----------- ----------- --------- --------- ---------- --------- --------- ----------- ----------- ----------- --------- ------------ ---------- --------- --------- -------- ------------ ------------
+baseline 7450 2520 1648 1334 2241 21 0 213 46 32 7110 4968 1503 18 23 218 15 435 3 18375 21866
+combine-scroll-count 9754 (+31%) 3050 (+21%) 2021 (+23%) 1697 (+27%) 2727 (+22%) 25 (+19%) 0 215 (+1%) 52 (+13%) 33 (+3%) 9418 (+32%) 0 (-100%) 2005 (+33%) 0 (-100%) 20 (-13%) 222 (+2%) 13 (-13%) 433 (0%) 4 (+33%) 18375 (0%) 21866 (0%)
+combine-waitfor 7449 (0%) 2550 (+1%) 1656 (0%) 1340 (0%) 2253 (+1%) 21 (0%) 0 209 (-2%) 45 (-2%) 33 (+3%) 7106 (0%) 4968 (0%) 1503 (0%) 16 (-11%) 11 (-52%) 216 (-1%) 14 (-7%) 434 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+combine-tail 7466 (0%) 2620 (+4%) 1723 (+5%) 1386 (+4%) 2321 (+4%) 33 (+57%) 0 217 (+2%) 46 (0%) 33 (+3%) 7118 (0%) 4968 (0%) 1503 (0%) 17 (-6%) 23 (0%) 225 (+3%) 14 (-7%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+install-helpers 7464 (0%) 2640 (+5%) 1751 (+6%) 1418 (+6%) 2353 (+5%) 33 (+57%) 0 217 (+2%) 46 (0%) 34 (+6%) 7117 (0%) 4968 (0%) 1503 (0%) 15 (-17%) 24 (+4%) 223 (+2%) 15 (0%) 436 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+combines-all 7462 (0%) 2510 (0%) 1646 (0%) 1321 (-1%) 2241 (0%) 33 (+57%) 0 214 (0%) 45 (-2%) 34 (+6%) 7105 (0%) 4968 (0%) 1504 (0%) 16 (-11%) 6 (-74%) 225 (+3%) 13 (-13%) 434 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+exists-shortcircuit 7456 (0%) 2510 (0%) 1630 (-1%) 1321 (-1%) 2223 (-1%) 21 (0%) 0 216 (+1%) 46 (0%) 32 (0%) 7120 (0%) 4967 (0%) 1504 (0%) 16 (-11%) 25 (+9%) 218 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+native-count 7460 (0%) 2510 (0%) 1662 (+1%) 1348 (+1%) 2239 (0%) 33 (+57%) 0 203 (-5%) 45 (-2%) 33 (+3%) 7102 (0%) 4968 (0%) 1504 (0%) 4 (-78%) 21 (-9%) 222 (+2%) 15 (0%) 436 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+monitor 7447 (0%) 2490 (-1%) 1632 (-1%) 1309 (-2%) 2218 (-1%) 34 (+62%) 0 208 (-2%) 46 (0%) 34 (+6%) 7088 (0%) 4968 (0%) 1503 (0%) 4 (-78%) 12 (-48%) 235 (+8%) 15 (0%) 436 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+tall-viewport 3381 (-55%) 1560 (-38%) 877 (-47%) 595 (-55%) 1237 (-45%) 10 (-52%) 0 205 (-4%) 35 (-24%) 31 (-3%) 3042 (-57%) 903 (-82%) 1503 (0%) 12 (-33%) 11 (-52%) 218 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+viewport-2000 4674 (-37%) 1910 (-24%) 1069 (-35%) 773 (-42%) 1539 (-31%) 14 (-33%) 0 213 (0%) 37 (-20%) 31 (-3%) 4342 (-39%) 2206 (-56%) 1503 (0%) 14 (-22%) 8 (-65%) 223 (+2%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+viewport-10000 2956 (-60%) 1740 (-31%) 881 (-47%) 597 (-55%) 1273 (-43%) 9 (-57%) 0 207 (-3%) 35 (-24%) 29 (-9%) 2620 (-63%) 472 (-90%) 1503 (0%) 13 (-28%) 17 (-26%) 217 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+viewport-taller-than-page 2762 (-63%) 1770 (-30%) 812 (-51%) 545 (-59%) 1157 (-48%) 9 (-57%) 0 196 (-8%) 34 (-26%) 30 (-6%) 2430 (-66%) 295 (-94%) 1503 (0%) 23 (+28%) 6 (-74%) 214 (-2%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+step-20ms 4164 (-44%) 1930 (-23%) 1127 (-32%) 838 (-37%) 1624 (-28%) 15 (-29%) 0 213 (0%) 36 (-22%) 30 (-6%) 3835 (-46%) 1686 (-66%) 1503 (0%) 26 (+44%) 13 (-43%) 219 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+step-raf 3137 (-58%) 1580 (-37%) 966 (-41%) 679 (-49%) 1341 (-40%) 12 (-43%) 0 212 (0%) 37 (-20%) 30 (-6%) 2795 (-61%) 641 (-87%) 1503 (0%) 25 (+39%) 17 (-26%) 226 (+4%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+idle-tight 6548 (-12%) 2500 (-1%) 1604 (-3%) 1295 (-3%) 2210 (-1%) 20 (-5%) 0 213 (0%) 44 (-4%) 32 (0%) 6213 (-13%) 4968 (0%) 604 (-60%) 17 (-6%) 19 (-17%) 218 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+poll-60ms 7443 (0%) 2560 (+2%) 1671 (+1%) 1362 (+2%) 2275 (+2%) 21 (0%) 0 211 (-1%) 45 (-2%) 32 (0%) 7106 (0%) 4966 (0%) 1503 (0%) 14 (-22%) 18 (-22%) 218 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+shared-context 7456 (0%) 2580 (+2%) 1670 (+1%) 1360 (+2%) 2259 (+1%) 21 (0%) 0 214 (0%) 46 (0%) 33 (+3%) 7072 (-1%) 4933 (-1%) 1503 (0%) 14 (-22%) 23 (0%) 222 (+2%) 15 (0%) 432 (-1%) 3 (0%) 18375 (0%) 21866 (0%)
+images-off 7393 (-1%) 2300 (-9%) 1610 (-2%) 1287 (-4%) 2135 (-5%) 21 (0%) 0 204 (-4%) 49 (+7%) 33 (+3%) 7124 (0%) 4972 (0%) 1503 (0%) 19 (+6%) 22 (-4%) 191 (-12%) 15 (0%) 50 (-89%) 3 (0%) 18358 (0%) 21865 (0%)
+stub-off 7424 (0%) 2510 (0%) 1669 (+1%) 1357 (+2%) 2216 (-1%) 21 (0%) 0 209 (-2%) 47 (+2%) 33 (+3%) 7109 (0%) 4967 (0%) 1503 (0%) 14 (-22%) 22 (-4%) 216 (-1%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21865 (0%)
+no-prune-css 7301 (-2%) 2400 (-5%) 1531 (-7%) 1373 (+3%) 2120 (-5%) 21 (0%) 0 67 (-69%) 47 (+2%) 23 (-28%) 7112 (0%) 4967 (0%) 1503 (0%) 16 (-11%) 18 (-22%) 76 (-65%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+no-flatten-shadow 7356 (-1%) 2450 (-3%) 1562 (-5%) 1347 (+1%) 2124 (-5%) 21 (0%) 0 126 (-41%) 45 (-2%) 23 (-28%) 7116 (0%) 4967 (0%) 1504 (0%) 18 (0%) 14 (-39%) 127 (-42%) 15 (0%) 435 (0%) 3 (0%) 25105 (+37%) 14174 (-35%)
+no-minify-css 7449 (0%) 2550 (+1%) 1658 (+1%) 1347 (+1%) 2264 (+1%) 21 (0%) 0 207 (-3%) 45 (-2%) 32 (0%) 7114 (0%) 4967 (0%) 1503 (0%) 13 (-28%) 24 (+4%) 219 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+no-remove-attrs 7453 (0%) 2580 (+2%) 1691 (+3%) 1378 (+3%) 2292 (+2%) 22 (+5%) 0 216 (+1%) 46 (0%) 33 (+3%) 7114 (0%) 4967 (0%) 1503 (0%) 15 (-17%) 24 (+4%) 220 (+1%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+noindex-bail 7442 (0%) 2460 (-2%) 1606 (-3%) 1295 (-3%) 2186 (-2%) 21 (0%) 0 210 (-1%) 45 (-2%) 32 (0%) 7110 (0%) 4967 (0%) 1503 (0%) 15 (-17%) 19 (-17%) 218 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+tall-plus-raf 2681 (-64%) 1560 (-38%) 792 (-52%) 507 (-62%) 1130 (-50%) 10 (-52%) 0 205 (-4%) 37 (-20%) 32 (0%) 2345 (-67%) 202 (-96%) 1503 (0%) 12 (-33%) 19 (-17%) 218 (0%) 15 (0%) 435 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+best-of 1733 (-77%) 1170 (-54%) 655 (-60%) 387 (-71%) 943 (-58%) 9 (-57%) 0 203 (-5%) 33 (-28%) 29 (-9%) 1395 (-80%) 119 (-98%) 402 (-73%) 6 (-67%) 263 (+1043%) 220 (+1%) 14 (-7%) 434 (0%) 2 (-33%) 18375 (0%) 21866 (0%)
+reduced-motion 7450 (0%) 2610 (+4%) 1690 (+3%) 1375 (+3%) 2309 (+3%) 21 (0%) 0 212 (0%) 46 (0%) 33 (+3%) 7113 (0%) 4967 (0%) 1503 (0%) 17 (-6%) 22 (-4%) 218 (0%) 15 (0%) 436 (0%) 3 (0%) 18375 (0%) 21866 (0%)
+
+## fidelity — must match baseline, or the cost win is not a win
+
+variant reviews revealed lazyTiles imgGated scrollGated offers shadow bytes
+------------------------- ------- -------- --------- -------- ----------- ------ ------ -------
+baseline 461 1 288 1 1 6 1788 945164
+combine-scroll-count 461 1 288 1 1 6 1788 945164
+combine-waitfor 461 1 288 1 1 6 1788 945164
+combine-tail 461 1 288 1 1 6 1788 945164
+install-helpers 461 1 288 1 1 6 1788 945164
+combines-all 461 1 288 1 1 6 1788 945164
+exists-shortcircuit 461 1 288 1 1 6 1788 945164
+native-count 461 1 288 1 1 6 1788 945164
+monitor 461 1 288 1 1 6 1788 945164
+tall-viewport 461 1 288 1 1 6 1788 945164
+viewport-2000 461 1 288 1 1 6 1788 945164
+viewport-10000 461 1 288 1 1 6 1788 945164
+viewport-taller-than-page 461 1 288 1 1 6 1788 945164
+step-20ms 461 1 288 1 1 6 1788 945164
+step-raf 461 1 288 1 1 6 1788 945164
+idle-tight 461 1 288 1 1 6 1788 945164
+poll-60ms 461 1 288 1 1 6 1788 945164
+shared-context 461 1 288 1 1 6 1788 945164
+images-off 461 1 288 0 1 6 1788 945118
+stub-off 461 1 288 0 1 6 1788 945137
+no-prune-css 461 1 288 1 1 6 1788 1046597
+no-flatten-shadow 461 1 288 1 1 6 0 618274
+no-minify-css 461 1 288 1 1 6 1788 945164
+no-remove-attrs 461 1 288 1 1 6 1788 1162180
+noindex-bail 461 1 288 1 1 6 1788 945159
+tall-plus-raf 461 1 288 1 1 6 1788 945164
+best-of 461 1 288 1 1 6 1788 945164
+reduced-motion 461 1 288 1 1 6 1788 945164
+
+FIDELITY DIVERGENCE images-off: imgGated 1 -> 0
+FIDELITY DIVERGENCE stub-off: imgGated 1 -> 0
+FIDELITY DIVERGENCE no-flatten-shadow: shadow 1788 -> 0
+
+## baseline CDP methods sent (one representative run, 435 total)
+
+ 385 Fetch.fulfillRequest
+ 15 Runtime.callFunctionOn
+ 4 Fetch.continueRequest
+ 2 Target.setAutoAttach
+ 2 Runtime.runIfWaitingForDebugger
+ 2 Page.addScriptToEvaluateOnNewDocument
+ 2 Emulation.setDeviceMetricsOverride
+ 2 Emulation.setTouchEmulationEnabled
+ 1 Target.createBrowserContext
+ 1 Browser.setDownloadBehavior
+ 1 Target.createTarget
+ 1 Network.enable
+ 1 Page.enable
+ 1 Page.getFrameTree
+
+wrote /private/tmp/prerender-plugin/perf-cdp/bench/render-cpu/results/mobile.json
diff --git a/bench-mobile.log b/bench-mobile.log
new file mode 100644
index 0000000..1b2c989
--- /dev/null
+++ b/bench-mobile.log
@@ -0,0 +1,118 @@
+fixture: http://127.0.0.1:52289/product/prd-bench
+device: mobile reps: 5 variants: baseline, combine-scroll-count, combine-waitfor, combine-tail, install-helpers, combines-all, exists-shortcircuit, native-count, monitor, tall-viewport, images-off, reduced-motion
+
+rep 1/5 baseline wall 7438ms cpu 2500ms task 1627ms eval 15 cdp 435 reviews 461
+rep 1/5 combine-scroll-count wall 9752ms cpu 3030ms task 2007ms eval 13 cdp 433 reviews 461
+rep 1/5 combine-waitfor wall 7450ms cpu 2510ms task 1616ms eval 14 cdp 434 reviews 461
+rep 1/5 combine-tail wall 7461ms cpu 2610ms task 1700ms eval 14 cdp 435 reviews 461
+rep 1/5 install-helpers wall 7464ms cpu 2620ms task 1723ms eval 15 cdp 436 reviews 461
+rep 1/5 combines-all wall 9764ms cpu 3210ms task 2135ms eval 11 cdp 432 reviews 461
+rep 1/5 exists-shortcircuit wall 7458ms cpu 2550ms task 1665ms eval 15 cdp 435 reviews 461
+rep 1/5 native-count wall 7442ms cpu 2420ms task 1569ms eval 15 cdp 436 reviews 461
+rep 1/5 monitor wall 7423ms cpu 2350ms task 1501ms eval 15 cdp 436 reviews 461
+rep 1/5 tall-viewport wall 3405ms cpu 1630ms task 934ms eval 15 cdp 435 reviews 461
+rep 1/5 images-off wall 7386ms cpu 2420ms task 1672ms eval 15 cdp 50 reviews 461
+rep 1/5 reduced-motion wall 7453ms cpu 2650ms task 1740ms eval 15 cdp 436 reviews 461
+rep 2/5 baseline wall 7452ms cpu 2580ms task 1677ms eval 15 cdp 435 reviews 461
+rep 2/5 combine-scroll-count wall 9752ms cpu 3260ms task 2178ms eval 13 cdp 433 reviews 461
+rep 2/5 combine-waitfor wall 7443ms cpu 2700ms task 1752ms eval 14 cdp 434 reviews 461
+rep 2/5 combine-tail wall 7460ms cpu 2650ms task 1729ms eval 14 cdp 435 reviews 461
+rep 2/5 install-helpers wall 7462ms cpu 2680ms task 1759ms eval 15 cdp 436 reviews 461
+rep 2/5 combines-all wall 9763ms cpu 3190ms task 2112ms eval 11 cdp 432 reviews 461
+rep 2/5 exists-shortcircuit wall 7448ms cpu 2530ms task 1629ms eval 15 cdp 435 reviews 461
+rep 2/5 native-count wall 7462ms cpu 2630ms task 1713ms eval 15 cdp 436 reviews 461
+rep 2/5 monitor wall 7455ms cpu 2430ms task 1565ms eval 15 cdp 436 reviews 461
+rep 2/5 tall-viewport wall 3379ms cpu 1630ms task 924ms eval 15 cdp 435 reviews 461
+rep 2/5 images-off wall 7386ms cpu 2310ms task 1583ms eval 15 cdp 50 reviews 461
+rep 2/5 reduced-motion wall 7441ms cpu 2500ms task 1618ms eval 15 cdp 436 reviews 461
+rep 3/5 baseline wall 7417ms cpu 2340ms task 1478ms eval 15 cdp 435 reviews 461
+rep 3/5 combine-scroll-count wall 9737ms cpu 3070ms task 2030ms eval 13 cdp 433 reviews 461
+rep 3/5 combine-waitfor wall 7438ms cpu 2620ms task 1710ms eval 14 cdp 434 reviews 461
+rep 3/5 combine-tail wall 7464ms cpu 2680ms task 1752ms eval 14 cdp 435 reviews 461
+rep 3/5 install-helpers wall 7464ms cpu 2690ms task 1773ms eval 15 cdp 436 reviews 461
+rep 3/5 combines-all wall 9773ms cpu 3340ms task 2219ms eval 11 cdp 432 reviews 461
+rep 3/5 exists-shortcircuit wall 7453ms cpu 2610ms task 1720ms eval 15 cdp 435 reviews 461
+rep 3/5 native-count wall 7430ms cpu 2670ms task 1747ms eval 15 cdp 436 reviews 461
+rep 3/5 monitor wall 7452ms cpu 2630ms task 1723ms eval 15 cdp 436 reviews 461
+rep 3/5 tall-viewport wall 3379ms cpu 1600ms task 927ms eval 15 cdp 435 reviews 461
+rep 3/5 images-off wall 7381ms cpu 2450ms task 1696ms eval 15 cdp 50 reviews 461
+rep 3/5 reduced-motion wall 7448ms cpu 2650ms task 1733ms eval 15 cdp 436 reviews 461
+rep 4/5 baseline wall 7450ms cpu 2650ms task 1744ms eval 15 cdp 435 reviews 461
+rep 4/5 combine-scroll-count wall 9758ms cpu 3250ms task 2162ms eval 13 cdp 433 reviews 461
+rep 4/5 combine-waitfor wall 7450ms cpu 2660ms task 1753ms eval 14 cdp 434 reviews 461
+rep 4/5 combine-tail wall 7456ms cpu 2680ms task 1744ms eval 14 cdp 435 reviews 461
+rep 4/5 install-helpers wall 7462ms cpu 2690ms task 1758ms eval 15 cdp 436 reviews 461
+rep 4/5 combines-all wall 9759ms cpu 3350ms task 2241ms eval 11 cdp 432 reviews 461
+rep 4/5 exists-shortcircuit wall 7442ms cpu 2630ms task 1712ms eval 15 cdp 435 reviews 461
+rep 4/5 native-count wall 7460ms cpu 2720ms task 1787ms eval 15 cdp 436 reviews 461
+rep 4/5 monitor wall 7446ms cpu 2690ms task 1764ms eval 15 cdp 436 reviews 461
+rep 4/5 tall-viewport wall 3382ms cpu 1620ms task 926ms eval 15 cdp 435 reviews 461
+rep 4/5 images-off wall 7378ms cpu 2440ms task 1711ms eval 15 cdp 50 reviews 461
+rep 4/5 reduced-motion wall 7451ms cpu 2700ms task 1761ms eval 15 cdp 436 reviews 461
+rep 5/5 baseline wall 7426ms cpu 2590ms task 1703ms eval 15 cdp 435 reviews 461
+rep 5/5 combine-scroll-count wall 9760ms cpu 3330ms task 2228ms eval 13 cdp 433 reviews 461
+rep 5/5 combine-waitfor wall 7452ms cpu 2630ms task 1723ms eval 14 cdp 434 reviews 461
+rep 5/5 combine-tail wall 7460ms cpu 2720ms task 1783ms eval 14 cdp 435 reviews 461
+rep 5/5 install-helpers wall 7460ms cpu 2710ms task 1776ms eval 15 cdp 436 reviews 461
+rep 5/5 combines-all wall 9760ms cpu 3320ms task 2207ms eval 11 cdp 432 reviews 461
+rep 5/5 exists-shortcircuit wall 7445ms cpu 2650ms task 1736ms eval 15 cdp 435 reviews 461
+rep 5/5 native-count wall 7452ms cpu 2710ms task 1784ms eval 15 cdp 436 reviews 461
+rep 5/5 monitor wall 7460ms cpu 2640ms task 1729ms eval 15 cdp 436 reviews 461
+rep 5/5 tall-viewport wall 3393ms cpu 1630ms task 945ms eval 15 cdp 435 reviews 461
+rep 5/5 images-off wall 7369ms cpu 2400ms task 1687ms eval 15 cdp 50 reviews 461
+rep 5/5 reduced-motion wall 7443ms cpu 2660ms task 1713ms eval 15 cdp 436 reviews 461
+
+## cost — median of 5 reps (delta vs baseline)
+
+variant wall cpu(tree) task task-other proc-cpu script v8compile devtools layout style settle postProc evals cdp-sent passes height elements
+-------------------- ----------- ----------- ----------- ----------- ----------- --------- --------- --------- --------- -------- ----------- ---------- --------- --------- -------- ---------- ----------
+baseline 7438 2580 1677 1363 2286 21 0 210 46 33 7104 218 15 435 3 18358 21864
+combine-scroll-count 9752 (+31%) 3250 (+26%) 2162 (+29%) 1831 (+34%) 2912 (+27%) 25 (+19%) 0 215 (+2%) 54 (+17%) 33 (0%) 9418 (+33%) 223 (+2%) 13 (-13%) 433 (0%) 4 (+33%) 18358 (0%) 21864 (0%)
+combine-waitfor 7450 (0%) 2630 (+2%) 1723 (+3%) 1412 (+4%) 2330 (+2%) 21 (0%) 0 212 (+1%) 47 (+2%) 32 (-3%) 7114 (0%) 216 (-1%) 14 (-7%) 434 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+combine-tail 7460 (0%) 2680 (+4%) 1744 (+4%) 1419 (+4%) 2372 (+4%) 33 (+57%) 0 214 (+2%) 47 (+2%) 34 (+3%) 7112 (0%) 223 (+2%) 14 (-7%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+install-helpers 7462 (0%) 2690 (+4%) 1759 (+5%) 1431 (+5%) 2394 (+5%) 34 (+62%) 0 214 (+2%) 47 (+2%) 33 (0%) 7116 (0%) 221 (+1%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+combines-all 9763 (+31%) 3320 (+29%) 2207 (+32%) 1863 (+37%) 2974 (+30%) 38 (+81%) 0 220 (+5%) 51 (+11%) 35 (+6%) 9418 (+33%) 224 (+3%) 11 (-27%) 432 (-1%) 4 (+33%) 18358 (0%) 21864 (0%)
+exists-shortcircuit 7448 (0%) 2610 (+1%) 1712 (+2%) 1398 (+3%) 2325 (+2%) 21 (0%) 0 215 (+2%) 46 (0%) 33 (0%) 7120 (0%) 216 (-1%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+native-count 7452 (0%) 2670 (+3%) 1747 (+4%) 1437 (+5%) 2372 (+4%) 33 (+57%) 0 202 (-4%) 47 (+2%) 33 (0%) 7097 (0%) 221 (+1%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+monitor 7452 (0%) 2630 (+2%) 1723 (+3%) 1409 (+3%) 2343 (+2%) 33 (+57%) 0 198 (-6%) 47 (+2%) 33 (0%) 7098 (0%) 223 (+2%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+tall-viewport 3382 (-55%) 1630 (-37%) 927 (-45%) 645 (-53%) 1295 (-43%) 11 (-48%) 0 209 (0%) 35 (-24%) 31 (-6%) 3053 (-57%) 219 (0%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+images-off 7381 (-1%) 2420 (-6%) 1687 (+1%) 1380 (+1%) 2252 (-1%) 21 (0%) 0 204 (-3%) 48 (+4%) 33 (0%) 7124 (0%) 187 (-14%) 15 (0%) 50 (-89%) 3 (0%) 18341 (0%) 21863 (0%)
+reduced-motion 7448 (0%) 2650 (+3%) 1733 (+3%) 1412 (+4%) 2351 (+3%) 21 (0%) 0 215 (+2%) 47 (+2%) 32 (-3%) 7114 (0%) 217 (0%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+
+## fidelity — must match baseline, or the cost win is not a win
+
+variant reviews revealed lazyTiles imgGated offers shadow bytes
+-------------------- ------- -------- --------- -------- ------ ------ ------
+baseline 461 1 288 1 6 1788 945084
+combine-scroll-count 461 1 288 1 6 1788 945084
+combine-waitfor 461 1 288 1 6 1788 945084
+combine-tail 461 1 288 1 6 1788 945084
+install-helpers 461 1 288 1 6 1788 945084
+combines-all 461 1 288 1 6 1788 945084
+exists-shortcircuit 461 1 288 1 6 1788 945084
+native-count 461 1 288 1 6 1788 945084
+monitor 461 1 288 1 6 1788 945084
+tall-viewport 461 1 288 1 6 1788 945084
+images-off 461 1 288 0 6 1788 945038
+reduced-motion 461 1 288 1 6 1788 945084
+
+FIDELITY DIVERGENCE images-off: imgGated 1 -> 0
+
+## baseline CDP methods sent (one representative run, 435 total)
+
+ 385 Fetch.fulfillRequest
+ 15 Runtime.callFunctionOn
+ 4 Fetch.continueRequest
+ 2 Target.setAutoAttach
+ 2 Runtime.runIfWaitingForDebugger
+ 2 Page.addScriptToEvaluateOnNewDocument
+ 2 Emulation.setDeviceMetricsOverride
+ 2 Emulation.setTouchEmulationEnabled
+ 1 Target.createBrowserContext
+ 1 Browser.setDownloadBehavior
+ 1 Target.createTarget
+ 1 Network.enable
+ 1 Page.enable
+ 1 Page.getFrameTree
+
+wrote /private/tmp/prerender-plugin/perf-cdp/bench/render-cpu/results/mobile.json
diff --git a/bench-mobile2.log b/bench-mobile2.log
new file mode 100644
index 0000000..5ceafdf
--- /dev/null
+++ b/bench-mobile2.log
@@ -0,0 +1,189 @@
+fixture: http://127.0.0.1:52549/product/prd-bench
+device: mobile reps: 5 variants: baseline, combine-scroll-count, combine-waitfor, combine-tail, install-helpers, combines-all, exists-shortcircuit, native-count, monitor, tall-viewport, viewport-2000, viewport-10000, viewport-taller-than-page, step-20ms, step-raf, idle-tight, poll-60ms, shared-context, images-off, stub-off, best-of, reduced-motion
+
+rep 1/5 baseline wall 7465ms cpu 2680ms task 1777ms eval 15 cdp 435 reviews 461
+rep 1/5 combine-scroll-count wall 9756ms cpu 3340ms task 2238ms eval 13 cdp 433 reviews 461
+rep 1/5 combine-waitfor wall 7454ms cpu 2700ms task 1773ms eval 14 cdp 434 reviews 461
+rep 1/5 combine-tail wall 7479ms cpu 2750ms task 1800ms eval 14 cdp 435 reviews 461
+rep 1/5 install-helpers wall 7471ms cpu 2690ms task 1771ms eval 15 cdp 436 reviews 461
+rep 1/5 combines-all wall 7460ms cpu 2650ms task 1742ms eval 13 cdp 434 reviews 461
+rep 1/5 exists-shortcircuit wall 7444ms cpu 2620ms task 1722ms eval 15 cdp 435 reviews 461
+rep 1/5 native-count wall 7448ms cpu 2700ms task 1782ms eval 15 cdp 436 reviews 461
+rep 1/5 monitor wall 7460ms cpu 2650ms task 1740ms eval 15 cdp 436 reviews 461
+rep 1/5 tall-viewport wall 3390ms cpu 1640ms task 925ms eval 15 cdp 435 reviews 461
+rep 1/5 viewport-2000 wall 4643ms cpu 2010ms task 1113ms eval 15 cdp 435 reviews 461
+rep 1/5 viewport-10000 wall 2954ms cpu 1760ms task 890ms eval 15 cdp 435 reviews 461
+rep 1/5 viewport-taller-than-page wall 2778ms cpu 1770ms task 816ms eval 15 cdp 435 reviews 461
+rep 1/5 step-20ms wall 4160ms cpu 1880ms task 1086ms eval 15 cdp 435 reviews 461
+rep 1/5 step-raf wall 3130ms cpu 1640ms task 974ms eval 15 cdp 435 reviews 461
+rep 1/5 idle-tight wall 6544ms cpu 2550ms task 1647ms eval 15 cdp 435 reviews 461
+rep 1/5 poll-60ms wall 7447ms cpu 2620ms task 1695ms eval 15 cdp 435 reviews 461
+rep 1/5 shared-context wall 7485ms cpu 2650ms task 1737ms eval 15 cdp 432 reviews 461
+rep 1/5 images-off wall 7383ms cpu 2390ms task 1657ms eval 15 cdp 50 reviews 461
+rep 1/5 stub-off wall 7432ms cpu 2560ms task 1690ms eval 15 cdp 435 reviews 461
+rep 1/5 best-of wall 1751ms cpu 1170ms task 671ms eval 14 cdp 434 reviews 461
+rep 1/5 reduced-motion wall 7444ms cpu 2600ms task 1701ms eval 15 cdp 436 reviews 461
+rep 2/5 baseline wall 7444ms cpu 2660ms task 1732ms eval 15 cdp 435 reviews 461
+rep 2/5 combine-scroll-count wall 9749ms cpu 3200ms task 2120ms eval 13 cdp 433 reviews 461
+rep 2/5 combine-waitfor wall 7444ms cpu 2620ms task 1691ms eval 14 cdp 434 reviews 461
+rep 2/5 combine-tail wall 7460ms cpu 2670ms task 1759ms eval 14 cdp 435 reviews 461
+rep 2/5 install-helpers wall 7454ms cpu 2670ms task 1739ms eval 15 cdp 436 reviews 461
+rep 2/5 combines-all wall 7460ms cpu 2630ms task 1730ms eval 13 cdp 434 reviews 461
+rep 2/5 exists-shortcircuit wall 7458ms cpu 2670ms task 1777ms eval 15 cdp 435 reviews 461
+rep 2/5 native-count wall 7456ms cpu 2630ms task 1724ms eval 15 cdp 436 reviews 461
+rep 2/5 monitor wall 7442ms cpu 2630ms task 1725ms eval 15 cdp 436 reviews 461
+rep 2/5 tall-viewport wall 3392ms cpu 1630ms task 933ms eval 15 cdp 435 reviews 461
+rep 2/5 viewport-2000 wall 4628ms cpu 1930ms task 1062ms eval 15 cdp 435 reviews 461
+rep 2/5 viewport-10000 wall 2951ms cpu 1760ms task 894ms eval 15 cdp 435 reviews 461
+rep 2/5 viewport-taller-than-page wall 2766ms cpu 1710ms task 780ms eval 15 cdp 435 reviews 461
+rep 2/5 step-20ms wall 4162ms cpu 1890ms task 1140ms eval 15 cdp 435 reviews 461
+rep 2/5 step-raf wall 3129ms cpu 1600ms task 955ms eval 15 cdp 435 reviews 461
+rep 2/5 idle-tight wall 6543ms cpu 2400ms task 1528ms eval 15 cdp 435 reviews 461
+rep 2/5 poll-60ms wall 7439ms cpu 2540ms task 1659ms eval 15 cdp 435 reviews 461
+rep 2/5 shared-context wall 7474ms cpu 2580ms task 1667ms eval 15 cdp 432 reviews 461
+rep 2/5 images-off wall 7391ms cpu 2320ms task 1608ms eval 15 cdp 50 reviews 461
+rep 2/5 stub-off wall 7400ms cpu 2470ms task 1602ms eval 15 cdp 435 reviews 461
+rep 2/5 best-of wall 1721ms cpu 1120ms task 629ms eval 14 cdp 434 reviews 461
+rep 2/5 reduced-motion wall 7434ms cpu 2310ms task 1468ms eval 15 cdp 436 reviews 461
+rep 3/5 baseline wall 7453ms cpu 2580ms task 1690ms eval 15 cdp 435 reviews 461
+rep 3/5 combine-scroll-count wall 9750ms cpu 3210ms task 2137ms eval 13 cdp 433 reviews 461
+rep 3/5 combine-waitfor wall 7441ms cpu 2610ms task 1697ms eval 14 cdp 434 reviews 461
+rep 3/5 combine-tail wall 7461ms cpu 2610ms task 1711ms eval 14 cdp 435 reviews 461
+rep 3/5 install-helpers wall 7465ms cpu 2630ms task 1709ms eval 15 cdp 436 reviews 461
+rep 3/5 combines-all wall 7462ms cpu 2600ms task 1696ms eval 13 cdp 434 reviews 461
+rep 3/5 exists-shortcircuit wall 7439ms cpu 2480ms task 1608ms eval 15 cdp 435 reviews 461
+rep 3/5 native-count wall 7438ms cpu 2570ms task 1679ms eval 15 cdp 436 reviews 461
+rep 3/5 monitor wall 7451ms cpu 2570ms task 1666ms eval 15 cdp 436 reviews 461
+rep 3/5 tall-viewport wall 3402ms cpu 1650ms task 943ms eval 15 cdp 435 reviews 461
+rep 3/5 viewport-2000 wall 4622ms cpu 1880ms task 1055ms eval 15 cdp 435 reviews 461
+rep 3/5 viewport-10000 wall 2955ms cpu 1720ms task 861ms eval 15 cdp 435 reviews 461
+rep 3/5 viewport-taller-than-page wall 2762ms cpu 1710ms task 784ms eval 15 cdp 435 reviews 461
+rep 3/5 step-20ms wall 4166ms cpu 1860ms task 1111ms eval 15 cdp 435 reviews 461
+rep 3/5 step-raf wall 3134ms cpu 1590ms task 959ms eval 15 cdp 435 reviews 461
+rep 3/5 idle-tight wall 6545ms cpu 2400ms task 1518ms eval 15 cdp 435 reviews 461
+rep 3/5 poll-60ms wall 7437ms cpu 2450ms task 1574ms eval 15 cdp 435 reviews 461
+rep 3/5 shared-context wall 7423ms cpu 2290ms task 1454ms eval 15 cdp 432 reviews 461
+rep 3/5 images-off wall 7400ms cpu 2270ms task 1577ms eval 15 cdp 50 reviews 461
+rep 3/5 stub-off wall 7417ms cpu 2610ms task 1724ms eval 15 cdp 435 reviews 461
+rep 3/5 best-of wall 1752ms cpu 1140ms task 654ms eval 14 cdp 434 reviews 461
+rep 3/5 reduced-motion wall 7439ms cpu 2670ms task 1751ms eval 15 cdp 436 reviews 461
+rep 4/5 baseline wall 7450ms cpu 2650ms task 1748ms eval 15 cdp 435 reviews 461
+rep 4/5 combine-scroll-count wall 9752ms cpu 3270ms task 2168ms eval 13 cdp 433 reviews 461
+rep 4/5 combine-waitfor wall 7444ms cpu 2580ms task 1708ms eval 14 cdp 434 reviews 461
+rep 4/5 combine-tail wall 7452ms cpu 2610ms task 1706ms eval 14 cdp 435 reviews 461
+rep 4/5 install-helpers wall 7467ms cpu 2670ms task 1745ms eval 15 cdp 436 reviews 461
+rep 4/5 combines-all wall 7458ms cpu 2690ms task 1775ms eval 13 cdp 434 reviews 461
+rep 4/5 exists-shortcircuit wall 7434ms cpu 2570ms task 1672ms eval 15 cdp 435 reviews 461
+rep 4/5 native-count wall 7451ms cpu 2640ms task 1725ms eval 15 cdp 436 reviews 461
+rep 4/5 monitor wall 7432ms cpu 2600ms task 1700ms eval 15 cdp 436 reviews 461
+rep 4/5 tall-viewport wall 3373ms cpu 1610ms task 906ms eval 15 cdp 435 reviews 461
+rep 4/5 viewport-2000 wall 4633ms cpu 1950ms task 1075ms eval 15 cdp 435 reviews 461
+rep 4/5 viewport-10000 wall 2952ms cpu 1740ms task 881ms eval 15 cdp 435 reviews 461
+rep 4/5 viewport-taller-than-page wall 2765ms cpu 1720ms task 780ms eval 15 cdp 435 reviews 461
+rep 4/5 step-20ms wall 4174ms cpu 1880ms task 1092ms eval 15 cdp 435 reviews 461
+rep 4/5 step-raf wall 3137ms cpu 1580ms task 935ms eval 15 cdp 435 reviews 461
+rep 4/5 idle-tight wall 6545ms cpu 2450ms task 1576ms eval 15 cdp 435 reviews 461
+rep 4/5 poll-60ms wall 7452ms cpu 2610ms task 1696ms eval 15 cdp 435 reviews 461
+rep 4/5 shared-context wall 7439ms cpu 2580ms task 1665ms eval 15 cdp 432 reviews 461
+rep 4/5 images-off wall 7379ms cpu 2160ms task 1472ms eval 15 cdp 50 reviews 461
+rep 4/5 stub-off wall 7418ms cpu 2470ms task 1610ms eval 15 cdp 435 reviews 461
+rep 4/5 best-of wall 1754ms cpu 1110ms task 622ms eval 14 cdp 434 reviews 461
+rep 4/5 reduced-motion wall 7441ms cpu 2440ms task 1559ms eval 15 cdp 436 reviews 461
+rep 5/5 baseline wall 7451ms cpu 2550ms task 1668ms eval 15 cdp 435 reviews 461
+rep 5/5 combine-scroll-count wall 9740ms cpu 2970ms task 1959ms eval 13 cdp 433 reviews 461
+rep 5/5 combine-waitfor wall 7450ms cpu 2530ms task 1626ms eval 14 cdp 434 reviews 461
+rep 5/5 combine-tail wall 7449ms cpu 2520ms task 1615ms eval 14 cdp 435 reviews 461
+rep 5/5 install-helpers wall 7446ms cpu 2500ms task 1608ms eval 15 cdp 436 reviews 461
+rep 5/5 combines-all wall 7460ms cpu 2480ms task 1613ms eval 13 cdp 434 reviews 461
+rep 5/5 exists-shortcircuit wall 7456ms cpu 2470ms task 1578ms eval 15 cdp 435 reviews 461
+rep 5/5 native-count wall 7445ms cpu 2560ms task 1654ms eval 15 cdp 436 reviews 461
+rep 5/5 monitor wall 7464ms cpu 2570ms task 1689ms eval 15 cdp 436 reviews 461
+rep 5/5 tall-viewport wall 2761ms cpu 1370ms task 811ms eval 14 cdp 434 reviews 461
+rep 5/5 viewport-2000 wall 4624ms cpu 1840ms task 1017ms eval 15 cdp 435 reviews 461
+rep 5/5 viewport-10000 wall 2938ms cpu 1750ms task 891ms eval 15 cdp 435 reviews 461
+rep 5/5 viewport-taller-than-page wall 2769ms cpu 1670ms task 762ms eval 15 cdp 435 reviews 461
+rep 5/5 step-20ms wall 4119ms cpu 1720ms task 1002ms eval 15 cdp 435 reviews 461
+rep 5/5 step-raf wall 3124ms cpu 1530ms task 919ms eval 15 cdp 435 reviews 461
+rep 5/5 idle-tight wall 6543ms cpu 2320ms task 1479ms eval 15 cdp 435 reviews 461
+rep 5/5 poll-60ms wall 7438ms cpu 2460ms task 1572ms eval 15 cdp 435 reviews 461
+rep 5/5 shared-context wall 7475ms cpu 2520ms task 1636ms eval 15 cdp 432 reviews 461
+rep 5/5 images-off wall 7374ms cpu 2270ms task 1548ms eval 15 cdp 50 reviews 461
+rep 5/5 stub-off wall 7421ms cpu 2480ms task 1611ms eval 15 cdp 435 reviews 461
+rep 5/5 best-of wall 1717ms cpu 1140ms task 635ms eval 14 cdp 434 reviews 461
+rep 5/5 reduced-motion wall 7443ms cpu 2390ms task 1515ms eval 15 cdp 436 reviews 461
+
+## cost — median of 5 reps (delta vs baseline)
+
+variant wall cpu(tree) task task-other proc-cpu script v8compile devtools layout style settle scroll idle count gate postProc evals cdp-sent passes height elements
+------------------------- ----------- ----------- ----------- ----------- ----------- --------- --------- --------- --------- --------- ----------- ----------- ----------- --------- ------------ ---------- --------- --------- -------- ---------- ----------
+baseline 7451 2650 1732 1421 2356 21 0 214 47 33 7122 4968 1503 20 24 215 15 435 3 18358 21864
+combine-scroll-count 9750 (+31%) 3210 (+21%) 2137 (+23%) 1807 (+27%) 2880 (+22%) 24 (+14%) 0 219 (+2%) 53 (+13%) 33 (0%) 9419 (+32%) 0 (-100%) 2005 (+33%) 0 (-100%) 23 (-4%) 221 (+3%) 13 (-13%) 433 (0%) 4 (+33%) 18358 (0%) 21864 (0%)
+combine-waitfor 7444 (0%) 2610 (-2%) 1697 (-2%) 1383 (-3%) 2310 (-2%) 21 (0%) 0 214 (0%) 46 (-2%) 33 (0%) 7113 (0%) 4966 (0%) 1504 (0%) 18 (-10%) 12 (-50%) 216 (0%) 14 (-7%) 434 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+combine-tail 7460 (0%) 2610 (-2%) 1711 (-1%) 1386 (-2%) 2323 (-1%) 33 (+57%) 0 217 (+1%) 47 (0%) 34 (+3%) 7117 (0%) 4968 (0%) 1503 (0%) 16 (-20%) 20 (-17%) 221 (+3%) 14 (-7%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+install-helpers 7465 (0%) 2670 (+1%) 1739 (0%) 1412 (-1%) 2371 (+1%) 33 (+57%) 0 217 (+1%) 47 (0%) 33 (0%) 7112 (0%) 4967 (0%) 1503 (0%) 18 (-10%) 23 (-4%) 223 (+4%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+combines-all 7460 (0%) 2630 (-1%) 1730 (0%) 1395 (-2%) 2353 (0%) 33 (+57%) 0 217 (+1%) 46 (-2%) 34 (+3%) 7109 (0%) 4967 (0%) 1503 (0%) 17 (-15%) 12 (-50%) 224 (+4%) 13 (-13%) 434 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+exists-shortcircuit 7444 (0%) 2570 (-3%) 1672 (-3%) 1363 (-4%) 2275 (-3%) 21 (0%) 0 214 (0%) 45 (-4%) 33 (0%) 7113 (0%) 4968 (0%) 1503 (0%) 17 (-15%) 24 (0%) 216 (0%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+native-count 7448 (0%) 2630 (-1%) 1724 (0%) 1404 (-1%) 2339 (-1%) 33 (+57%) 0 204 (-5%) 47 (0%) 33 (0%) 7092 (0%) 4968 (0%) 1504 (0%) 6 (-70%) 12 (-50%) 227 (+6%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+monitor 7451 (0%) 2600 (-2%) 1700 (-2%) 1391 (-2%) 2311 (-2%) 33 (+57%) 0 203 (-5%) 47 (0%) 34 (+3%) 7095 (0%) 4968 (0%) 1503 (0%) 3 (-85%) 14 (-42%) 227 (+6%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+tall-viewport 3390 (-55%) 1630 (-38%) 925 (-47%) 637 (-55%) 1294 (-45%) 11 (-48%) 0 211 (-1%) 35 (-26%) 31 (-6%) 3058 (-57%) 904 (-82%) 1504 (0%) 15 (-25%) 24 (0%) 220 (+2%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+viewport-2000 4628 (-38%) 1930 (-27%) 1062 (-39%) 762 (-46%) 1542 (-35%) 14 (-33%) 0 215 (0%) 37 (-21%) 31 (-6%) 4289 (-40%) 2147 (-57%) 1504 (0%) 17 (-15%) 19 (-21%) 224 (+4%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+viewport-10000 2952 (-60%) 1750 (-34%) 890 (-49%) 607 (-57%) 1290 (-45%) 9 (-57%) 0 204 (-5%) 35 (-26%) 29 (-12%) 2619 (-63%) 472 (-90%) 1503 (0%) 12 (-40%) 18 (-25%) 212 (-1%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+viewport-taller-than-page 2766 (-63%) 1710 (-35%) 780 (-55%) 511 (-64%) 1092 (-54%) 9 (-57%) 0 197 (-8%) 35 (-26%) 30 (-9%) 2435 (-66%) 294 (-94%) 1503 (0%) 26 (+30%) 6 (-75%) 213 (-1%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+step-20ms 4162 (-44%) 1880 (-29%) 1092 (-37%) 796 (-44%) 1561 (-34%) 14 (-33%) 0 210 (-2%) 36 (-23%) 30 (-9%) 3834 (-46%) 1684 (-66%) 1504 (0%) 16 (-20%) 15 (-37%) 219 (+2%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+step-raf 3130 (-58%) 1590 (-40%) 955 (-45%) 660 (-54%) 1355 (-42%) 12 (-43%) 0 213 (0%) 37 (-21%) 30 (-9%) 2795 (-61%) 645 (-87%) 1504 (0%) 21 (+5%) 24 (0%) 223 (+4%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+idle-tight 6544 (-12%) 2400 (-9%) 1528 (-12%) 1218 (-14%) 2111 (-10%) 20 (-5%) 0 211 (-1%) 44 (-6%) 32 (-3%) 6208 (-13%) 4967 (0%) 603 (-60%) 18 (-10%) 17 (-29%) 217 (+1%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+poll-60ms 7439 (0%) 2540 (-4%) 1659 (-4%) 1352 (-5%) 2257 (-4%) 21 (0%) 0 209 (-2%) 45 (-4%) 33 (0%) 7109 (0%) 4967 (0%) 1503 (0%) 16 (-20%) 20 (-17%) 216 (0%) 15 (0%) 435 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+shared-context 7474 (0%) 2580 (-3%) 1665 (-4%) 1352 (-5%) 2274 (-3%) 21 (0%) 0 213 (0%) 45 (-4%) 33 (0%) 7083 (-1%) 4936 (-1%) 1503 (0%) 17 (-15%) 17 (-29%) 224 (+4%) 15 (0%) 432 (-1%) 3 (0%) 18358 (0%) 21864 (0%)
+images-off 7383 (-1%) 2270 (-14%) 1577 (-9%) 1264 (-11%) 2119 (-10%) 21 (0%) 0 204 (-5%) 46 (-2%) 33 (0%) 7125 (0%) 4972 (0%) 1503 (0%) 16 (-20%) 22 (-8%) 190 (-12%) 15 (0%) 50 (-89%) 3 (0%) 18341 (0%) 21863 (0%)
+stub-off 7418 (0%) 2480 (-6%) 1611 (-7%) 1301 (-8%) 2207 (-6%) 21 (0%) 0 209 (-2%) 47 (0%) 34 (+3%) 7104 (0%) 4966 (0%) 1504 (0%) 15 (-25%) 17 (-29%) 217 (+1%) 15 (0%) 435 (0%) 3 (0%) 18341 (0%) 21863 (0%)
+best-of 1751 (-76%) 1140 (-57%) 635 (-63%) 364 (-74%) 916 (-61%) 9 (-57%) 0 201 (-6%) 32 (-32%) 29 (-12%) 1420 (-80%) 118 (-98%) 402 (-73%) 6 (-70%) 287 (+1096%) 216 (0%) 14 (-7%) 434 (0%) 2 (-33%) 18358 (0%) 21864 (0%)
+reduced-motion 7441 (0%) 2440 (-8%) 1559 (-10%) 1254 (-12%) 2162 (-8%) 21 (0%) 0 209 (-2%) 43 (-9%) 33 (0%) 7105 (0%) 4967 (0%) 1504 (0%) 15 (-25%) 14 (-42%) 218 (+1%) 15 (0%) 436 (0%) 3 (0%) 18358 (0%) 21864 (0%)
+
+## fidelity — must match baseline, or the cost win is not a win
+
+variant reviews revealed lazyTiles imgGated offers shadow bytes
+------------------------- ------- -------- --------- -------- ------ ------ ------
+baseline 461 1 288 1 6 1788 945084
+combine-scroll-count 461 1 288 1 6 1788 945084
+combine-waitfor 461 1 288 1 6 1788 945084
+combine-tail 461 1 288 1 6 1788 945084
+install-helpers 461 1 288 1 6 1788 945084
+combines-all 461 1 288 1 6 1788 945084
+exists-shortcircuit 461 1 288 1 6 1788 945084
+native-count 461 1 288 1 6 1788 945084
+monitor 461 1 288 1 6 1788 945084
+tall-viewport 461 1 288 1 6 1788 945084
+viewport-2000 461 1 288 1 6 1788 945084
+viewport-10000 461 1 288 1 6 1788 945084
+viewport-taller-than-page 461 1 288 1 6 1788 945084
+step-20ms 461 1 288 1 6 1788 945084
+step-raf 461 1 288 1 6 1788 945084
+idle-tight 461 1 288 1 6 1788 945084
+poll-60ms 461 1 288 1 6 1788 945084
+shared-context 461 1 288 1 6 1788 945084
+images-off 461 1 288 0 6 1788 945038
+stub-off 461 1 288 0 6 1788 945057
+best-of 461 1 288 1 6 1788 945084
+reduced-motion 461 1 288 1 6 1788 945084
+
+FIDELITY DIVERGENCE images-off: imgGated 1 -> 0
+FIDELITY DIVERGENCE stub-off: imgGated 1 -> 0
+
+## baseline CDP methods sent (one representative run, 435 total)
+
+ 385 Fetch.fulfillRequest
+ 15 Runtime.callFunctionOn
+ 4 Fetch.continueRequest
+ 2 Target.setAutoAttach
+ 2 Runtime.runIfWaitingForDebugger
+ 2 Page.addScriptToEvaluateOnNewDocument
+ 2 Emulation.setDeviceMetricsOverride
+ 2 Emulation.setTouchEmulationEnabled
+ 1 Target.createBrowserContext
+ 1 Browser.setDownloadBehavior
+ 1 Target.createTarget
+ 1 Network.enable
+ 1 Page.enable
+ 1 Page.getFrameTree
+
+wrote /private/tmp/prerender-plugin/perf-cdp/bench/render-cpu/results/mobile.json
diff --git a/bench/render-cpu/README.md b/bench/render-cpu/README.md
new file mode 100644
index 0000000..17ad523
--- /dev/null
+++ b/bench/render-cpu/README.md
@@ -0,0 +1,485 @@
+# `bench/render-cpu` — where a render's time and CPU actually go
+
+> ## READ THIS BEFORE QUOTING ANY NUMBER BELOW
+>
+> **The absolute milliseconds are this fixture's, on one laptop. The SPLIT is the result.** What
+> travels between this bench and production is the shape: which part of a render is wall-clock
+> waiting that we impose, which part is Chrome working, and which parts are rounding errors. The
+> per-render totals do not travel — a real PDP runs a framework, hydrates, and executes orders of
+> magnitude more of its own script than this fixture does (**40 ms** of `ScriptDuration` here
+> whole-render, ~150 ms at `?chunks=69`, against 800–2,465 ms measured on real customer pages).
+>
+> Two numbers that DO travel, because they are products of our own code and the page height rather
+> than of the site's JavaScript:
+>
+> - a scroll pass is `ceil(height / viewportHeight)` steps, each waiting `scroll.stepMs`. At the
+> deployed mobile profile (390×844, `stepMs: 60`) over an 18,358 px page that is 22 steps × 60 ms ×
+> 3 passes = **3,915 ms of pure `setInterval` waiting**, and the measured scroll total was 4,968 ms.
+> - the per-pass `waitForNetworkIdle` window is `networkIdleMs` per pass, floor, and 3 passes cost
+> **1,503 ms**.
+>
+> Together: **91% of settle, and 87% of the whole render, is our own two waits.** Chrome's own
+> accounting agrees it is not busy — 2.65 s of process-tree CPU across a 7.45 s render.
+>
+> ### Two more that travel, and they are what you size a pod from
+>
+> Measured under load against an external fixture, 3 reps, medians (see [§Under load](#under-load)).
+> Neither is about this fixture's JavaScript, so neither is fixture-specific:
+>
+> - **RSS is ~265 MB per concurrent render slot, and it is linear** — 535 MB at c=1, 2,611 MB at
+> c=8, 5,023 MB at c=16, 7,422 MB at c=24 (R²≈1). **Nothing measured in this bench moves it**:
+> every cache candidate landed inside 2,588–2,631 MB at c=8. On a memory-sensitive fleet this,
+> not cores, is what sizes a pod.
+> - **Splitting slots across worker PROCESSES costs 7–14% more RSS for no throughput.** At equal
+> total slots: 1×12 = 1.55/s at 3,788 MB, 2×6 = 1.56/s at 4,057 MB, 3×4 = 1.55/s at 4,305 MB. At
+> 24 slots, 1×24 = 2.49/s at 6,833 MB vs 4×6 = 2.61/s at 7,572 MB — and that +5% is inside the
+> rep spread ([2.50, 2.49, 2.18] vs [2.67, 2.61, 2.44]). Each extra process is another Node heap
+> and another Chrome browser. **The multi-process worker shape is not a throughput optimisation
+> on this workload.** Whatever justifies it — blast radius, a wedged browser taking down fewer
+> slots, per-process memory ceilings — is not a throughput argument, and at a memory-bound sizing
+> it is a small throughput COST.
+>
+> ### The concurrency knee below is a LAPTOP ARTIFACT — do not carry it to the fleet
+>
+> Throughput is linear to c=12, then 86% of linear at c=16 and 83% at c=24, and per-render
+> CPU-SECONDS rise 11–14% over the same range. It is tempting to read that as a concurrency limit.
+> It is not one:
+>
+> - total CPU utilisation at the knee is **25%** (c=12) rising to 49% (c=24) — nothing is saturated;
+> - the driver's event loop is **10% busy** with a p99 lateness of 31 ms in an 8.9 s render, so it is
+> not our thread either;
+> - splitting the same slots across processes does not fix it.
+>
+> This machine is an **Apple M4 Pro: 10 performance cores + 4 much slower efficiency cores**, not 14
+> equal ones. The knee sits exactly where concurrent renderers start landing on E-cores, which
+> inflates CPU-seconds for identical work without saturating anything. A pod with homogeneous cores
+> should not have this knee. **"Concurrency 12" is this laptop's core topology, not a fleet limit.**
+> (Not proven with per-core sampling — it is the explanation that fits every other measurement.)
+>
+> **The thing this bench was built to evaluate turned out to be noise.** It was commissioned to test
+> "can we make fewer CDP calls, maybe combine some `evaluate` calls". Measured: `Runtime.callFunctionOn`
+> is **15 calls of 435** CDP messages per render, and every in-page count and gate check TOGETHER is
+> **44 ms — 0.59% of the render**. Six separate candidate optimisations of that code (combining calls,
+> installing helpers once, native counting, a MutationObserver monitor, existence short-circuits,
+> finer polling) all measured within noise of baseline, several slightly WORSE. They are recorded
+> below as dead ends so nobody spends another day on them.
+
+## What it measures
+
+One deterministic, self-contained page ([fixture.js](fixture.js)) shaped like a commerce PDP —
+~21,900 light-DOM elements, 18,358 px tall, 12 open shadow roots with their own stylesheets, 3,000
+utility CSS rules of which most never match, IntersectionObserver-lazy grid sections, a review
+widget that appears only after its anchor intersects, a three-state reveal wrapper, `astro-island`
+`props` attributes, and perpetual cosmetic churn. It also serves **real script bundles** — 9 of
+~113 KB, each a few hundred real functions that V8 has to fetch, compile and run — because without
+them no caching or compile-cost candidate is measurable at all. `?chunks=N` raises the sub-resource
+count to production shape (~70 per render) **without touching the DOM, the CSS or the page height**,
+so every other number stays comparable. Served from 127.0.0.1, so there is no origin, no CDN and no
+network jitter in any number — which also means this bench **cannot price a network round trip**,
+and that is the one thing a resource cache exists to save.
+
+Each variant renders it through the real `renderOnce` → `defaultRenderer` path with a config
+mirroring the deployed render-service one, and is measured on three instruments at once
+([instrument.js](instrument.js)) plus a wall-clock split of the settle phase.
+
+```bash
+node bench/render-cpu/bench.js # all variants, mobile, 5 reps
+node bench/render-cpu/bench.js --device desktop --reps 7
+node bench/render-cpu/bench.js --only baseline,best-of --reps 9
+node bench/render-cpu/bench.js --json bench/render-cpu/results/mobile.json
+```
+
+Variants are **interleaved** (rep 1 of every variant, then rep 2) because a laptop's clock drifts
+with thermals over the ~15 minutes a full sweep takes, and a block layout would hand the whole
+drift to whichever variant ran last. Reported as **medians**. One warm-up rep per variant is
+discarded.
+
+### The four instruments, and how each one lies
+
+| instrument | what it sees | how it lies |
+| ------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------- |
+| `cdpCounter` (patched `Connection` prototype) | every CDP message, by method | a message count is not a cost |
+| `Performance.getMetrics` via puppeteer's own session | main-thread time split into script / layout / style / **DevToolsCommand** / other | main thread only — misses compositor, raster, network threads |
+| `processTreeCpu` (`ps` over the Chrome process tree) | CPU seconds and RSS of every Chrome process, bucketed by Chrome's own `--type=` — renderer / gpu-process / utility:network / browser — and the number of distinct `--renderer-client-id`s | a bucket is not a call stack; and `ps TIME` has 10 ms per-process resolution, so at c=1 the small buckets carry ±10 ms |
+| `process.cpuUsage()` + `monitorEventLoopDelay` ([load.js](load.js)) | the DRIVER's own CPU and event-loop lateness — the bench process is a process too | loop delay is ABSOLUTE lateness, so an idle process already reads about its own resolution (11 ms here); read it as a delta from that floor |
+
+Read together they cross-check: if main-thread task time falls but tree CPU does not, the work moved
+threads rather than disappearing.
+
+
+
+### Under load — [load.js](load.js), [fleet.js](fleet.js), [uvsweep.js](uvsweep.js)
+
+`bench.js` answers "how long is one render and where does its wall-clock go". `load.js` answers
+"renders/second and CPU-seconds/render at a given concurrency", which is what decides how much fleet
+a traffic level needs.
+
+```bash
+node bench/render-cpu/fixture-server.js --port 58200 & # REQUIRED above c≈8 — see below
+F=http://127.0.0.1:58200/product/prd-bench
+node bench/render-cpu/load.js --concurrency 8 --reps 3 --fixture $F
+node bench/render-cpu/load.js --concurrency 1,2,4,8,12,16,24 --only baseline --reps 3 --fixture $F
+node bench/render-cpu/load.js --only context-pool --trace --batches 6 # the warm-up curve
+node bench/render-cpu/fleet.js --shapes 1x12,2x6,3x4,1x24,4x6 --reps 3 # process shape
+node bench/render-cpu/uvsweep.js --sizes 4,16,32 --reps 3 # one child per setting
+node bench/render-cpu/aging.js --batches 60 --concurrency 4 # does an aged browser slow down?
+```
+
+[aging.js](aging.js) answers a different question from the other two and is documented in its own
+header: it keeps one browser alive for hundreds of pages and records every render in order, to test
+whether `browserExpirationThreshold` (retire at 200 pages) is buying anything. **It was started and
+stopped at 60 of 240 renders per arm — its result is INDETERMINATE, not negative.** See
+[§Does an aged browser get slower](#does-an-aged-browser-get-slower-unfinished).
+
+A UNIT is `(concurrency, variant, rep)`: its own browser, `--warmups` discarded batches, then one
+measured batch. Units are **interleaved rep-major**, for the same thermal reason as `bench.js`. A
+browser per unit rather than per variant is what makes the cache candidates honest — every variant
+starts from a cold Chrome and gets exactly `--warmups` batches to warm whatever it can.
+
+**Run the fixture in its OWN process above c≈8.** `startFixture()` serves HTTP on the CALLER's event
+loop, so in-process at c=24 one Node thread was serving ~9,600 responses (including the ~113 KB
+bundles) while also driving 24 renders' CDP traffic and every interception callback. Moving it out
+changed the c=24 row by **−22% batch / +27% renders/s** and left c≤16 untouched. Production's origin
+is a different machine; an in-process fixture is not, and a ladder run that way finds the harness.
+
+### Four hazards, each of which produced a wrong number before it was found
+
+- `page.metrics()` filters the result to a fixed allowlist and **drops `TaskOtherDuration`,
+ `V8CompileDuration`, `DevToolsCommandDuration` and `ProcessTime`** — the four most useful
+ counters. Opening a CDP session of your own does not help either: the Performance agent starts
+ accumulating at `enable()`, so a session opened after the render reports idle microseconds. Read
+ it through `page.mainFrame().client`, the session puppeteer enabled at page creation.
+- Process-tree CPU must be sampled **while the page is still open**, AND the "before" sample must
+ wait for the process count to stop moving. `after − before` is only a CPU measurement while every
+ process alive at `before` is still alive at `after`; a batch that just closed its pages leaves
+ renderer processes exiting asynchronously, and sampled too early they are counted in `before` and
+ gone by `after`. That flatters whichever variant tears down the MOST processes — which is exactly
+ the baseline-vs-pooled-context comparison this file exists for.
+- **`V8CompileDuration` cannot see compile work; `processTreeCpu` is the only instrument that can.**
+ It reads **0.22 ms whole-render against ~1 MB of script bundles**. That is not "nothing was
+ compiled": V8 compiles lazily and streams/compiles on background threads inside the renderer, and
+ this counter is main-thread compile only. Any inference of the form "compile is free because
+ `V8CompileDuration` is 0" is wrong — including the one that used to be in the dead-end table
+ below. Use `ScriptDuration` (compile + execute, main thread) and tree CPU.
+- **A config that changes between renders in one process may silently not take effect.**
+ `resolveConfigForJob` caches resolved configs by a signature of the device plus the ordered names
+ of the matching overrides, and drops that cache only when the **identity of the
+ `config.overrides` ARRAY** changes. A variant that spreads a base config and adds a field — which is exactly what a
+ bench variant does — keeps the same `overrides` array, so every URL that MATCHES an override
+ resolves to the PREVIOUS render's cached config. Measured cost: a live contract sweep reported "no
+ contract matched" on four of five pages; the fifth worked only because it matches no override at
+ all (`applied.length === 0` returns the base config and bypasses the cache), which is also why the
+ failure looks like a page-specific problem rather than a caching one. **In a bench, give each
+ variant a fresh `overrides` array so the identity check fires.** It is also worth fixing upstream:
+ production happens to be safe only because a config reload reparses JSON and so produces a new
+ array, and nothing enforces that — any path that changes the base config while reusing the same
+ overrides array serves stale resolved configs with no signal.
+
+## Results
+
+Full tables: `results/*.json` and the run logs. The summary that matters:
+
+### On the fixture (controlled A/B, single render, idle machine)
+
+| variant | wall | tree CPU | fidelity |
+| ------------------------------------- | -------------- | -------------- | --------- |
+| baseline (390x844, settleUntilStable) | 7,450ms | 2,520ms | — |
+| tall viewport (390x5000) | 3,381ms (-55%) | 1,560ms (-38%) | identical |
+| frame-paced scroll pass | 3,137ms (-58%) | 1,580ms (-37%) | identical |
+| all winners stacked | 1,733ms (-77%) | 1,170ms (-54%) | identical |
+
+Settle split at baseline: **scroll 4,968ms + network-idle 1,503ms + counting 20ms + gating 24ms.**
+91% of settle is two waits we impose; all in-page counting and gating together is **0.59%** of the
+render.
+
+### Under load, round 1 (concurrency 8) — the settle candidates
+
+| variant | renders/s | vs baseline | CPU/render |
+| ------------------------------ | --------- | ----------- | -------------- |
+| baseline | 1.05 | 1.00x | 2,206ms |
+| stacked winners | 4.07 | 3.88x | 1,358ms |
+| images off in Blink | 1.07 | 1.02x | 1,973ms (-11%) |
+| MutationObserver count monitor | 1.06 | 1.01x | 2,171ms |
+
+At c=8 the machine is nowhere near CPU-saturated — 1.05/s × 2.2 s is 2.3 of 14 cores — so CPU-only
+wins understate here; a saturated pod converts them 1:1 into throughput.
+
+### Slot-scoped contexts, the two HTTP caches, and the V8 code cache
+
+Every render has always taken a fresh incognito context, whose HTTP cache is in-memory and dies with
+it. `load.js` leases a long-lived context per SLOT instead and wipes it between renders with
+`resetForNextVariant` — the same routine production already uses between a job's device variants,
+which fails closed. c=8, 3 reps, 2 warm-up batches, medians; rep spread in brackets.
+
+| variant | batch | CPU/render | renders/s | `ScriptDuration` | origin req/render | rssMb |
+| -------------------------------------------------- | ------- | -------------------------- | --------- | ---------------- | ----------------- | ----- |
+| baseline | 7,580ms | 2,261ms [2200, 2261, 2313] | 1.06 | 40ms | 12 | 2,631 |
+| slot-scoped context (incognito) | 7,560ms | 2,215ms [2180, 2215, 2273] | 1.06 | 40ms | **0** | 2,623 |
+| resource cache on | 7,615ms | 2,149ms [2148, 2149, 2233] | 1.05 | 40ms | 1 | 2,623 |
+| both caches | 7,576ms | 2,136ms [2134, 2136, 2139] | 1.06 | 41ms | 0 | 2,601 |
+| **default context + persistent `--user-data-dir`** | 7,569ms | 2,130ms [2093, 2130, 2153] | 1.06 | **24ms** | 0 | 2,588 |
+| default context, temp profile | 7,588ms | 2,156ms [2150, 2156, 2209] | 1.05 | **24ms** | 0 | 2,598 |
+
+Four findings, and only the second is worth acting on:
+
+1. **The pooling mechanism works and costs nothing.** Origin fetches fall **12/render → 0/render**
+ after one warm-up batch; the wipe never failed in ~400 renders; RSS is unchanged. But the CPU
+ difference against baseline is **inside noise** — 2,215 vs 2,261 is −2% against a baseline rep
+ spread of 113 ms. Report it as "no measurable CPU difference", not as −2%.
+2. **The V8 code cache needs the DEFAULT context, not a pooled incognito one.** `ScriptDuration`
+ stays at 40 ms with a pooled incognito context and drops to **24 ms (−40%)** with the default
+ context — with or without a persistent profile. The code cache lives in Chrome's disk-cache
+ backend, and an incognito context's cache is in-memory by construction, so it never gets one.
+3. **A persistent `--user-data-dir` therefore REPLACES the context pool rather than composing with
+ it.** It buys nothing for pooled incognito contexts; the −40% comes from `incognitoPages: false`,
+ and the persistent profile only adds survival across browser RESTARTS (a temp-profile default
+ context already reached 24 ms within one browser lifetime).
+4. **This fixture cannot size the win.** Its whole script budget is 40 ms of a 7,500 ms render, so
+ even a perfect code cache is worth 0.2% here.
+
+**On real pages it is worth having.** Measured on a live product page against incognito-per-render:
+**−11% wall, −9% CPU, −28% `ScriptDuration`** — and the isolation wipe is FREE, because
+`Storage.clearDataForOrigin` does not clear the HTTP disk cache, so the wiped arm matched the
+unwiped one. **The open blocker is isolation at CONCURRENCY, not performance**: the default context
+shares one cookie jar across every concurrent render, which is what `documentReuse.cookies.pin` and
+[variantContext.ts](../../packages/browser/src/variantContext.ts) exist to prevent. That routine was
+designed for a job's device variants rendered in SEQUENCE; proving a per-render wipe is sufficient
+when renders OVERLAP is a different argument and has not been made.
+
+### Is the resource cache a net win? At production resource counts, no
+
+`?chunks=69` — 70 cacheable sub-resources per render, production shape. c=8, external fixture, 3
+reps, 2 warm-ups.
+
+| variant | batch | Chrome CPU/render | Node CPU/render | Node busy | origin req/render | rssMb |
+| ------------------- | ------------------------------ | ----------------- | --------------- | --------- | ----------------- | ----- |
+| no cache (control) | **7,845ms** [7833, 7845, 8027] | 2,429ms | **44ms** | 4% | 73 | 2,909 |
+| resource cache on | 8,021ms [8001, 8021, 8076] | **2,359ms** | **93ms** | 9% | 1 | 2,850 |
+| slot-scoped context | 7,923ms [7767, 7923, 7946] | 2,521ms | **43ms** | 4% | **0** | 2,953 |
+| both | 8,005ms [8005, 7986, 8071] | 2,364ms | 92ms | 9% | 0 | 2,845 |
+
+- **Our resource cache does not remove CPU — it MOVES CPU out of Chrome and onto the single Node
+ thread.** Chrome-tree CPU falls 2,429 → 2,359 (−2.9%, mostly out of `utility:network`, 63 → 35 ms)
+ while Node CPU per render **doubles, 44 → 93 ms**, and the loop goes 4% → 9% busy. Net per render:
+ 2,473 ms vs 2,452 ms — a wash. That is the cost of replaying bodies base64-encoded through
+ `Fetch.fulfillRequest`, and it lands on the scarcest thread in the process.
+- **It also makes the render 2.2% SLOWER here** (7,845 → 8,021 ms, rep ranges barely overlapping).
+- **Chrome's own cache achieves the same origin offload for free** — 0 requests with Node CPU
+ unchanged at 43 ms — and with both on they are redundant (2,364 vs 2,359): ours answers first.
+
+**The caveat is load-bearing and this bench cannot lift it: the origin is localhost.** The thing a
+resource cache exists to save is a network round trip, and here that round trip is ~0 ms. At a real
+origin's 20–100 ms RTT × 70 resources the sign flips. What IS settled is the part that is not about
+latency: our cache costs ~50 ms of extra Node main-thread CPU per render to give back ~70 ms of
+Chrome CPU, and Chrome's own cache does the offload at zero Node cost — but only within one slot's
+browser lifetime, where ours is shared across slots, processes and restarts.
+
+### Does an aged browser get slower? (UNFINISHED — 25% of the intended run)
+
+The fleet retires a browser after 200 opened pages (`browserExpirationThreshold`, checked as
+`browser.totalOpenedPages > BROWSER_MAX_TOTAL_PAGES` in `Worker.ts`). The reason on record is a
+general "prevent memory leaks" recommendation plus an operator impression; it has never been
+measured. [aging.js](aging.js) was built to measure it and the run was stopped early, at **60 of the
+intended 240 renders per arm**. What follows is three sample points per arm and **cannot support a
+verdict either way** — it is recorded so a later attempt starts from data rather than from nothing.
+
+Four arms, interleaved batch-by-batch against simultaneously live browsers (so thermal drift, which
+looks exactly like aging, hits every arm equally), c=4, external fixture. Each cell is one batch's
+median, not a band median:
+
+| arm | page 20 | page 40 | page 60 |
+| ------------------------------------ | ------------------------------- | --------------------- | --------------------- |
+| fresh-incognito (today's shape) | 7,460ms / 2,187ms CPU / 1,430MB | 7,456 / 2,160 / 1,434 | 7,450 / 2,383 / 1,447 |
+| pooled-incognito | 7,467 / 2,260 / 1,430 | 7,451 / 2,173 / 1,436 | 7,462 / 2,270 / 1,439 |
+| pooled-recycle | 7,463 / 2,202 / 1,426 | 7,461 / 2,220 / 1,434 | 7,462 / 2,495 / 1,436 |
+| persist-default (persistent profile) | 7,478 / 2,262 / 1,418 | 7,472 / 2,395 / 1,425 | 7,466 / 2,327 / 1,426 |
+
+- **Wall is flat to 0.2% in every arm** over the first 60 pages — including the two that accumulate
+ state. Whatever aging is, it is not visible in wall-clock this early.
+- **CPU is non-monotonic in three of four arms** and every value sits inside the batch-to-batch
+ spread measured independently at this concurrency (~±5%). No signal.
+- **RSS rises ~1% per 40 pages in ALL FOUR arms, including the control** that creates and destroys a
+ browser context every render. A drift the control shares is not context accumulation, and at this
+ slope it is nowhere near a reason to retire at 200. Three points is not a curve, and this is the
+ one number worth re-measuring properly.
+
+**One real finding did come out of it, and it is about the wipe, not about aging.**
+`resetForNextVariant` with an EMPTY cookie jar costs **1–2 ms on an incognito context and ~20 ms on
+the default context of a persistent profile** — a 10–20× gap before a single cookie exists, so it is
+`Storage.clearDataForOrigin` touching a disk-backed profile rather than an in-memory one. At 20 ms
+that is 0.27% of a render here and would not decide anything on its own, but it is a cost that
+belongs to the persistent-profile candidate specifically, and it is the floor: puppeteer issues one
+`Network.deleteCookies` per cookie, and a real storefront handed over 125. (Measured over a 2-batch
+smoke run — small n, large effect.) The fixture's `?cookies=N` knob and the `wipe-0` / `wipe-25` /
+`wipe-125` arms exist to price that scaling and **were never run**.
+
+### The concurrency ladder, and what process shape buys
+
+Baseline, external fixture, 3 reps. **Read the knee caveat at the top of this file before quoting
+any of it.**
+
+| conc | batch | per-render | CPU/render | renders/s | vs linear | Node CPU/render | Node busy | loop p99 | rssMb |
+| ---- | ------- | ---------- | ---------- | --------- | --------- | --------------- | --------- | -------- | ----- |
+| 1 | 7,420ms | 7,420ms | 2,500ms | 0.13 | 100% | 60ms | 1% | 11.1ms | 535 |
+| 2 | 7,449ms | 7,448ms | 2,430ms | 0.27 | 100% | 37ms | 1% | 11.2ms | 841 |
+| 4 | 7,475ms | 7,468ms | 2,175ms | 0.54 | 100% | 24ms | 1% | 11.2ms | 1,430 |
+| 8 | 7,573ms | 7,553ms | 2,225ms | 1.06 | 98% | 21ms | 2% | 11.4ms | 2,611 |
+| 12 | 7,701ms | 7,663ms | 2,221ms | 1.56 | 96% | 18ms | 3% | 15.2ms | 3,786 |
+| 16 | 8,606ms | 7,802ms | 2,475ms | 1.86 | **86%** | 51ms | 10% | 15.0ms | 5,023 |
+| 24 | 8,908ms | 8,125ms | 2,530ms | 2.69 | **83%** | 38ms | 10% | 31.3ms | 7,422 |
+
+CPU-seconds/render is **flat from c=2 to c=12** (2,175–2,430 against a within-variant rep spread of
+~±120 ms), then rises 11% at c=16 and 14% at c=24. Throughput does not hard-plateau, it rolls off —
+doubling 12→24 still buys +72%.
+
+Per-process split, the same run: **the renderer is 90% of tree CPU at c=1 and 93% at c=8**
+(c=8: renderer 2,091 ms, gpu-process 70 ms, browser 53 ms, utility:network 23 ms, per render). No
+out-of-process-iframe blow-up — the process count is exactly `concurrency + 4` at every level from 1
+to 24, and the distinct `--renderer-client-id` count tracks it. The GPU process still burns
+70–130 ms/render **despite `--disable-gpu` and `--disable-software-rasterizer`**: small, but not
+zero, and pure overhead for a headless snapshot.
+
+Process shape at equal total slots (3 reps) is in the block at the top of this file: **1×12, 2×6 and
+3×4 are indistinguishable at 1.55–1.56/s, and splitting costs 7–14% more RSS.**
+
+### On real pages — and the correction that matters most
+
+**The fixture numbers above were measured against a baseline the fleet had already left behind.**
+Tall viewports on every device and per-page-type settle profiles had already shipped on the consumer
+branch. Re-based on the live deployed config, a real mobile PDP costs **4,050ms / 4,600ms CPU**, not
+the 13,621ms an out-of-date config measured — and the remaining headroom is correspondingly smaller.
+
+What is left, measured on four real pages (two PDPs, a populated catalog page, an empty-facet
+catalog page), median of 4 reps, against the deployed config:
+
+| variant | PDP wall | catalog wall | PDP CPU | catalog CPU |
+| ----------------------------------- | -------- | ------------ | -------- | ----------- |
+| skip the pre-gate plateau | -14% | -18% | -9% | -14% |
+| skip `topSettleMs` before a plateau | -4% | -14% | -5% | -11% |
+| skip the network-idle sleep alone | **+10%** | -7% | +6% | -4% |
+| all four blind waits removed | **-23%** | **-39%** | **-13%** | **-30%** |
+
+The settle budget on the deployed profile closes to within ~50ms as: scroll 72 + networkIdle 500 +
+`topSettleMs` 300 + plateau ~850 + gate ~2 + `topSettleMs` 300 + plateau ~850. **Every one of those
+except the gate is a blind Node-side sleep**, and `networkIdleMs` equals `networkIdleTimeoutMs` on
+this profile, so the idle call can never observe its window — it is structurally a fixed sleep.
+
+**Waits are fungible, which is why the single-lever rows mislead.** Removing the idle sleep ALONE
+made the PDP 10% slower: the plateau that follows simply waited longer (+75%), because the page was
+still changing. Removing a blind wait only pays when a later, content-conditional wait does not
+absorb it. That is why the stacked row is not the sum of its parts, and why the honest metric is
+`settle - (gate + plateau time actually spent waiting for a change)`.
+
+## The parity caveat on the real-page results
+
+Structural markers held identically across every variant and page: review-node count (1,430 / 1,661),
+JSON-LD blocks, `Offer` count, `
`, island count, and — the hydration tripwire — zero
+`ssr`-marked islands.
+
+**But two markers are too noisy to support a parity claim: `` count and product-link count.**
+They moved +-5% between variants AND between reps of the SAME variant, non-monotonically (one page's
+cheapest variant matched baseline exactly while a more conservative one was 18 links short, and a
+variant that changes nothing about timing came back with MORE links than baseline). That is the site
+serving different recommendation content per request, not us losing it. A parity verdict on those
+needs a per-page churn floor measured from repeated same-variant renders — which is exactly what
+`audit/reuseParity.ts` already implements. **Do not ship a settle change on the strength of the
+table above alone.**
+
+## What this cannot answer
+
+**Whether any of these changes is faithful on a real site.** The fixture carries three deliberate
+tripwires — content gated on an image's `load` event, content gated on a `scroll` EVENT (not an
+observer), and the three-state review reveal — and the fidelity table above is checked on every
+run. That catches the failure modes we thought of. It cannot catch:
+
+- **Viewport-height-dependent layout and script.** A 5,000 px-tall mobile viewport makes `100vh`
+ 5,000 px, and any script that computes slide counts, sticky offsets or "is this in view" from
+ `window.innerHeight` sees a different page. Desktop already renders at 1920×5000, so the site
+ tolerates it there; mobile CSS is not the same CSS.
+- **Fleet memory.** A taller viewport composites more area per step. This bench reports JS heap, not
+ layer memory, and the fleet is already memory-sensitive — watch pod RSS after any viewport change.
+- **Real hydration cost.** The fixture's own JS is 21 ms. Anything that changes WHEN we snapshot
+ relative to a real framework's hydration has to be judged on a real render.
+
+The gate for all of that already exists in this package and should be the next step for any change
+here: `renderAudit` / `paintParity` / `reuseParity` against staging, per device.
+
+## Tested again under CPU starvation — the answer does not change
+
+Every CDP-reduction result below was first measured on an IDLE machine, where an evaluate costs
+1-3ms and nothing else wants the core, so it reads as free. That is a fair objection: under
+starvation the same work competes with the page's own hydration on the same main thread, and
+`DevToolsCommandDuration` is main-thread time by definition (210ms idle, 300-480ms at 4x throttle).
+So the question was re-run at 4x CPU throttle against a live page.
+
+**Our polling is not the culprit.** No contract (no polling at all) vs polling every 250ms vs every
+1000ms, 3 reps interleaved: polling adds ~50-80ms of main-thread CDP time, about **1% of task
+time**, and quartering the calls did not recover it. Content was identical in all three arms (364
+product links, 1,430 review nodes, zero un-hydrated islands). Note 302ms of `devtools` time exists
+with NO contract at all — the residue is request interception and postProcess, not the poll loop.
+
+**Nor is the interception plane, once measured with enough reps.** At 3 reps, blocking images in
+Blink (`--blink-settings=imagesEnabled=false`, removing ~1,000 Fetch round-trips) looked like -15%
+CPU and -18% wall. At **5 reps it is within noise**: wall -3%, CPU 0%, `devtools` slightly WORSE.
+The 3-rep result was drift, and it was one reporting decision away from being filed as a finding.
+
+**And one arm shows why "faster under starvation" must never be read alone.** Aborting blocked
+images instead of stubbing them halved CDP servicing (486ms -> 215ms) and cut wall 39% — while
+storing **0 of 364 product links**. It was fastest because the page never finished. `images-off`
+does the same thing intermittently: content held on 4 of 5 runs and collapsed to 0 links on the
+fifth.
+
+So the conclusion below stands in both regimes, and the reason it stands is worth keeping: the CDP
+work that competes for the starved main thread is not ours to remove. It is the requests the page
+itself makes, and removing them removes the page.
+
+## Dead ends — measured, do not re-propose
+
+Each of these was implemented and measured against the same fixture in the same process. All of them
+preserved fidelity exactly, and none of them bought any CPU or throughput — a couple do change
+something else (origin fetches, memory), which the rows say explicitly.
+
+| candidate | result | why it cannot work |
+| --------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| fold the DOM count into the scroll-pass call | **+31% wall** | the loop then decides a pass late, so it runs a 4th scroll pass — which costs ~1.6 s to save one ~1 ms round trip |
+| install the in-page helpers once at document start | 0% wall, `ScriptDuration` 21→33 ms | the source shipped per call is a few hundred bytes against a page that compiles ~1 MB of its own bundles; there is no per-call compile cost of that size to recover |
+| count with native `querySelectorAll` + a shadow-root registry | count 20→6 ms, 0% wall | a 14 ms saving inside a 7,450 ms render |
+| MutationObserver monitor maintaining the count incrementally | count 20→3 ms, 0% wall | same: the walk was never expensive at this DOM size |
+| existence short-circuit for `minCount: 1` gates | 0% wall | the gate is 24 ms total, and it satisfies on its first tick |
+| combine the two waitFor calls per tick | gate 24→12 ms, 0% wall | the anchor exists in the served HTML, so the pre-scroll tick happens once |
+| `domStablePollMs` 250 → 60 | 0% wall | nothing in this config polls in a loop long enough to care |
+| `prefers-reduced-motion` | 0% wall, −8% CPU | the animation work was never on the critical path |
+| abort blocked images instead of stubbing them | **fidelity loss** | the 1×1 stub is load-bearing: it fires `load`, and content gated on an image's load event disappears without it |
+| **slot-scoped INCOGNITO context pool** (c=8, 3 reps) | 2,215 ms vs 2,261 ms CPU/render (baseline spread 2200–2313), identical renders/s and RSS — but origin fetches 12→0 | it gets Chrome's HTTP cache and NOT the V8 code cache, because an incognito context's cache is in-memory and the code cache lives in the disk-cache backend. Ship it for origin offload if you want it; there is no CPU case. The CPU case belongs to the DEFAULT context — see [§Slot-scoped contexts](#slot-scoped-contexts-the-two-http-caches-and-the-v8-code-cache) |
+| **`UV_THREADPOOL_SIZE` 4 → 16 → 32** (c=8 and c=16, resource cache on and warm, 3 reps) | within rep spread in both directions: c=8 renders/s 0.96 / 0.95 / 0.99, c=16 1.64 / 1.63 / 1.60 | the mechanism is real — `cache.get()` is an `fs.readFile` on that pool while the request is PAUSED in Chrome — but it is nowhere near loaded: 70 reads/render at c=16 is ~114 reads/s of page-cached files, and `ResourceCache` does no gzip, so the pool serves only `readFile`/`writeFile`/`rename`/`mkdir`. The signature to watch for on the fleet is Node loop lag rising with cache-hit count, which `load.js` now reports |
+| **more worker PROCESSES at equal total slots** (3 reps) | 1×12 / 2×6 / 3×4 all 1.55–1.56/s; 1×24 2.49/s vs 4×6 2.61/s with overlapping rep ranges | the driver's event loop is only 15% busy at 24 slots in ONE process, so there is no per-loop bottleneck to relieve — and each extra process adds a Node heap and a Chrome browser, for 7–14% more RSS |
+
+### Corrected — these rows used to say something stronger than the evidence
+
+- **"share one browser context across renders — 0% wall, −3% CPU (noise), per-render context cost is
+ not measurable"** was measured in a frame that could not contain its own mechanism: the resource
+ cache was hard-disabled, the fixture served no real script bundles, and the runner disposed the
+ shared context between reps. Re-measured with all three fixed, the mechanism is real (origin
+ fetches 12→0) even though the CPU verdict on the incognito form survives. The claim that replaced
+ it is in [§Slot-scoped contexts](#slot-scoped-contexts-the-two-http-caches-and-the-v8-code-cache),
+ and it is about which KIND of context, not whether
+ sharing works.
+- **"`V8CompileDuration` is 0 ms across every variant"** was true and the inference from it was
+ wrong. See the hazard list above: V8 compiles lazily on background threads and this counter is
+ main-thread compile only, so 0 means the instrument cannot see it, not that no compilation
+ happened.
+- Any claim that `--disk-cache-size=1073741824` (in `DEFAULT_CHROME_ARGS`) does anything for a
+ render is wrong, and always was: every render takes a fresh **incognito** context, whose cache is
+ in-memory and capped independently, so that flag has never applied to a single production render.
+ It starts applying the moment renders move to the default context.
+
+## Scaffolding to remove before this ships
+
+`packages/browser/src/experiments.ts` and the `experiments.*` branches in `renderer.ts` exist ONLY
+so a candidate and the code it replaces can be measured in one process. Whatever wins becomes
+unconditional and the flag is deleted; `src/inPage.ts` goes with it unless the helper install turns
+out to be worth keeping for another reason. The `window.__passes` counter in the scroll functions is
+bench scaffolding too. If any of it is still in the tree when the work lands, the work is not done.
+
+`fixture-server.js`, `fleet.js` and `uvsweep.js` are NOT in that category — they are part of the
+harness, and the first of them is required for any measurement above c≈8.
diff --git a/bench/render-cpu/aging.js b/bench/render-cpu/aging.js
new file mode 100644
index 0000000..1af0897
--- /dev/null
+++ b/bench/render-cpu/aging.js
@@ -0,0 +1,460 @@
+/**
+ * `bench/render-cpu/aging.js` — DOES AN AGED BROWSER ACTUALLY GET SLOWER?
+ *
+ * The fleet retires a browser after `browserExpirationThreshold` (200) opened pages. The reason on
+ * record is a general "prevent memory leaks" recommendation plus an operator impression that aged
+ * browsers felt slower. Nobody has measured it — and it now decides a policy question, because
+ * retiring every ~13 minutes is exactly what throws away the warm persistent profile behind the
+ * V8-code-cache candidate. If aging is not real, retirement costs us that cache for nothing.
+ *
+ * ## Why this file is not a variant of load.js
+ *
+ * load.js measures a STEADY STATE: a browser per unit, warm up, take one batch, throw the browser
+ * away. Aging is the opposite question — what a browser's 200th page costs relative to its 1st — so
+ * the browser has to live for the whole run and every render has to be recorded in order.
+ *
+ * ## The confound this is built around: thermal drift looks exactly like aging
+ *
+ * A curve that takes eight minutes of continuous rendering to draw will drift with the laptop's
+ * temperature, and a monotone slowdown over eight minutes is indistinguishable from "the browser got
+ * slower" if the arm is run as one block. So the arms are INTERLEAVED at batch granularity against
+ * SIMULTANEOUSLY LIVE browsers: all arms are launched up front, and one batch is taken from each in
+ * round-robin. Every arm then sees the same thermal history, while each arm's own browser still sees
+ * its pages strictly in order. Drift that is a property of time hits every arm; drift that is a
+ * property of browser age does not.
+ *
+ * ## The arms
+ *
+ * - `fresh-incognito` — today's production shape: a fresh incognito context per render. The control.
+ * - `pooled-incognito` — one long-lived incognito context per slot, wiped between renders.
+ * - `pooled-recycle` — `pooled-incognito`, but its contexts are disposed and recreated halfway
+ * through WITHOUT closing the browser. This is the attribution arm: if drift
+ * exists and resets at the recycle point, the unit that ages is the CONTEXT
+ * and the fix is to retire contexts; if it continues through, the unit is the
+ * browser or the profile.
+ * - `persist-default` — the code-cache candidate: the default (non-incognito) context on a
+ * persistent `--user-data-dir`, wiped between renders.
+ *
+ * node bench/render-cpu/fixture-server.js --port 58200 &
+ * node bench/render-cpu/aging.js --batches 50 --concurrency 4
+ * node bench/render-cpu/aging.js --arms wipe-0,wipe-25,wipe-125 --batches 8 # the wipe's price
+ */
+
+import { mkdirSync, writeFileSync, rmSync } from 'node:fs';
+import { dirname, resolve as resolvePath } from 'node:path';
+import ManagedBrowser from '../../packages/browser/dist/ManagedBrowser.js';
+import RenderJob from '../../packages/browser/dist/RenderJob.js';
+import defaultRenderer from '../../packages/browser/dist/renderer.js';
+import { resolveSettings, settings, defaultLaunchOptions } from '../../packages/browser/dist/settings.js';
+import { initResourceCache } from '../../packages/browser/dist/ResourceCache.js';
+import { resetExperiments } from '../../packages/browser/dist/experiments.js';
+import { resetForNextVariant } from '../../packages/browser/dist/variantContext.js';
+import { BASE_CONFIG } from './variants.js';
+import { median, processTreeCpu, readMetrics, pageMetrics } from './instrument.js';
+
+const args = process.argv.slice(2);
+const flag = (n, d) => {
+ const i = args.indexOf(`--${n}`);
+ return i >= 0 && args[i + 1] ? args[i + 1] : d;
+};
+const DEVICE = flag('device', 'mobile');
+const CONCURRENCY = Number(flag('concurrency', 4));
+const BATCHES = Number(flag('batches', 50));
+const FIXTURE = flag('fixture', 'http://127.0.0.1:58200/product/prd-bench');
+const JSON_OUT = flag('json', '');
+const ONLY = flag('arms', '').split(',').filter(Boolean);
+const UDD_ROOT = flag('udd', '/tmp/prerender-bench-aging');
+
+const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
+
+/**
+ * An arm is a browser plus a context policy. `cookies` is the fixture's `?cookies=N`, which prices
+ * the wipe: `resetForNextVariant` deletes the jar one `Network.deleteCookies` per cookie.
+ */
+const ARMS = [
+ { name: 'fresh-incognito', incognito: true, mode: 'fresh', udd: false, cookies: 0 },
+ { name: 'pooled-incognito', incognito: true, mode: 'pool', udd: false, cookies: 0 },
+ // The attribution arm (item 4). It has to be a POOLED arm, because the default context is the one
+ // context a browser cannot be made to let go of — so "dispose the context, keep the browser" is
+ // only an experiment you can run on contexts you created.
+ { name: 'pooled-recycle', incognito: true, mode: 'pool', udd: false, cookies: 0, recycleAtBatch: 0.5 },
+ { name: 'persist-default', incognito: false, mode: 'shared', udd: true, cookies: 0 },
+ // The wipe-cost sweep: same context policy, three jar sizes. Run with a small --batches, and at
+ // --concurrency 1, because on a shared default context N concurrent wipes delete each other's
+ // cookies and the timing stops being one wipe's cost.
+ { name: 'wipe-0', incognito: false, mode: 'shared', udd: true, cookies: 0 },
+ { name: 'wipe-25', incognito: false, mode: 'shared', udd: true, cookies: 25 },
+ { name: 'wipe-125', incognito: false, mode: 'shared', udd: true, cookies: 125 },
+];
+
+async function main() {
+ const selected = ARMS.filter((a) => (ONLY.length ? ONLY.includes(a.name) : a.name.startsWith('wipe-') === false));
+ console.log(`fixture: ${FIXTURE}`);
+ console.log(
+ `device: ${DEVICE} concurrency: ${CONCURRENCY} batches/arm: ${BATCHES} renders/arm: ${BATCHES * CONCURRENCY}`
+ );
+ console.log(
+ `arms (interleaved batch-by-batch against simultaneously live browsers): ${selected.map((a) => a.name).join(', ')}\n`
+ );
+
+ // Settings are process-global and the arms disagree about `incognitoPages`, so each arm re-resolves
+ // them immediately before its own batch. Safe only because exactly one arm renders at a time —
+ // which is also what makes the interleaving thermally fair.
+ resetExperiments();
+
+ for (const arm of selected) {
+ if (arm.udd) {
+ arm.uddPath = `${UDD_ROOT}-${arm.name}`;
+ rmSync(arm.uddPath, { recursive: true, force: true });
+ }
+ applySettings(arm);
+ await initResourceCache(settings.resourceCache);
+ arm.browser = await ManagedBrowser.launch({
+ maxActivePages: CONCURRENCY + 1,
+ puppeteerLaunchOptions: {
+ ...defaultLaunchOptions(),
+ ...(arm.uddPath ? { userDataDir: arm.uddPath } : {}),
+ },
+ });
+ arm.pool = arm.mode === 'pool' ? [] : null;
+ // The default context is not something `createContext()` will hand back (it returns null when
+ // `incognitoPages` is off), and without a handle to it `resetForNextVariant` never runs — which
+ // would quietly measure the persistent-profile arm WITHOUT the wipe that makes it shippable.
+ arm.sharedContext = arm.mode === 'shared' ? arm.browser.browser.defaultBrowserContext() : null;
+ arm.renders = [];
+ arm.batchRows = [];
+ arm.url = arm.cookies ? `${FIXTURE}?cookies=${arm.cookies}` : FIXTURE;
+ arm.recycleAt = arm.recycleAtBatch ? Math.floor(BATCHES * arm.recycleAtBatch) : null;
+ }
+
+ const started = Date.now();
+ for (let batch = 0; batch < BATCHES; batch++) {
+ for (const arm of selected) {
+ if (arm.recycleAt === batch && arm.pool) {
+ // Dispose the contexts, keep the browser. Item 4's attribution question.
+ for (const context of arm.pool.splice(0)) await arm.browser.disposeContext(context).catch(() => {});
+ console.log(` [${arm.name}] contexts recycled before batch ${batch + 1} (browser kept)`);
+ }
+ await runBatch(arm, batch);
+ }
+ if (batch % 5 === 4 || batch === BATCHES - 1) {
+ const line = selected
+ .map((a) => {
+ const r = a.batchRows[a.batchRows.length - 1];
+ return `${a.name} ${String(r.perRenderWall).padStart(5)}ms/${String(r.cpuMsPerRender).padStart(4)}cpu/${String(r.rssMb).padStart(4)}MB`;
+ })
+ .join(' | ');
+ console.log(
+ `batch ${String(batch + 1).padStart(3)}/${BATCHES} (page ${String((batch + 1) * CONCURRENCY).padStart(4)}, ` +
+ `${Math.round((Date.now() - started) / 1000)}s) ${line}`
+ );
+ }
+ }
+
+ for (const arm of selected) {
+ for (const context of arm.pool?.splice(0) ?? []) await arm.browser.disposeContext(context).catch(() => {});
+ await arm.browser.close().catch(() => {});
+ }
+
+ report(selected);
+ if (JSON_OUT) {
+ const path = resolvePath(process.cwd(), JSON_OUT);
+ mkdirSync(dirname(path), { recursive: true });
+ writeFileSync(
+ path,
+ JSON.stringify(
+ {
+ device: DEVICE,
+ concurrency: CONCURRENCY,
+ batches: BATCHES,
+ arms: selected.map((a) => ({
+ name: a.name,
+ cookies: a.cookies,
+ recycleAt: a.recycleAt,
+ renders: a.renders,
+ batches: a.batchRows,
+ })),
+ },
+ null,
+ 2
+ )
+ );
+ console.log(`\nwrote ${path}`);
+ }
+}
+
+function applySettings(arm) {
+ resolveSettings(
+ {
+ config: BASE_CONFIG,
+ resourceCache: { enabled: false },
+ harper: {},
+ incognitoPages: arm.incognito,
+ },
+ { requireHarper: false }
+ );
+}
+
+async function runBatch(arm, batch) {
+ applySettings(arm);
+ const before = await quiescedCpu(arm.browser.pid);
+ const open = [];
+ const startedAt = Date.now();
+ const runs = await Promise.all(Array.from({ length: CONCURRENCY }, (_, slot) => renderOne(arm, batch, slot, open)));
+ const batchMs = Date.now() - startedAt;
+ const after = processTreeCpu(arm.browser.pid);
+
+ const metrics = [];
+ for (const page of open) {
+ try {
+ metrics.push(pageMetrics(await readMetrics(page)));
+ } catch {
+ // A page that will not answer Performance.getMetrics still counted for wall and CPU.
+ }
+ }
+ for (const page of open) await arm.browser.closePage(page);
+
+ const m = (fn) => median(metrics.map(fn)) ?? 0;
+ for (let i = 0; i < runs.length; i++) {
+ arm.renders.push({
+ page: batch * CONCURRENCY + i + 1,
+ batch: batch + 1,
+ wallMs: runs[i].wallMs,
+ wipeMs: runs[i].wipeMs,
+ jar: runs[i].jar,
+ reviews: runs[i].reviews,
+ bytes: runs[i].bytes,
+ scriptMs: metrics[i]?.scriptMs ?? null,
+ taskMs: metrics[i]?.taskMs ?? null,
+ jsHeapMb: metrics[i]?.jsHeapMb ?? null,
+ });
+ }
+ arm.batchRows.push({
+ batch: batch + 1,
+ pagesBefore: batch * CONCURRENCY,
+ batchMs,
+ perRenderWall: median(runs.map((r) => r.wallMs)),
+ cpuMsPerRender: Math.round((((after?.cpuSeconds ?? 0) - (before?.cpuSeconds ?? 0)) * 1000) / CONCURRENCY),
+ rssMb: after?.rssMb ?? null,
+ processes: after?.processes ?? null,
+ byType: after?.byType ?? null,
+ scriptMs: m((x) => x.scriptMs),
+ taskMs: m((x) => x.taskMs),
+ jsHeapMb: m((x) => x.jsHeapMb),
+ wipeMs: median(runs.map((r) => r.wipeMs)),
+ jar: median(runs.map((r) => r.jar).filter((j) => j !== null)),
+ totalOpenedPages: arm.browser.totalOpenedPages,
+ });
+}
+
+async function renderOne(arm, batch, slot, keep) {
+ const t0 = Date.now();
+ let context = arm.pool ? (arm.pool.pop() ?? (await arm.browser.createContext())) : arm.sharedContext;
+ let page = await arm.browser.getPage(context);
+ let wipeMs = null;
+ let jar = null;
+ if (context) {
+ // Jar size on a sample of renders only: `context.cookies()` is itself a CDP round trip, and
+ // paying it on every render would put it inside the very number this is trying to price.
+ if (batch % 5 === 0 && slot === 0) {
+ try {
+ jar = (await context.cookies()).length;
+ } catch {
+ /* a jar we cannot read is not a measurement we need */
+ }
+ }
+ const w0 = Date.now();
+ const wiped = await resetForNextVariant(context, page, arm.url, []);
+ wipeMs = Date.now() - w0;
+ if (!wiped && arm.pool) {
+ // Fail closed, as production does — but only for contexts we own. The default context
+ // cannot be replaced, so a failed wipe there is recorded and nothing else (it never
+ // happened in any run; if it starts, the shared-context arm is the one to distrust).
+ await arm.browser.closePage(page);
+ await arm.browser.disposeContext(context).catch(() => {});
+ context = null;
+ page = await arm.browser.getPage(null);
+ } else if (!wiped) {
+ arm.wipeFailures = (arm.wipeFailures ?? 0) + 1;
+ }
+ }
+ const job = new RenderJob({
+ id: `aging-${arm.name}-${batch}-${slot}`,
+ url: arm.url,
+ expiresAt: Date.now() + 3600_000,
+ deviceType: DEVICE,
+ callbackOrigin: 'http://localhost',
+ isFromSitemap: true,
+ });
+ job.attemptStarted();
+ let html;
+ try {
+ html = await defaultRenderer(page, job);
+ } finally {
+ job.attemptEnded(undefined, html);
+ keep.push(page);
+ if (context && arm.pool) arm.pool.push(context);
+ }
+ return {
+ wallMs: Date.now() - t0,
+ wipeMs,
+ jar,
+ bytes: html?.length ?? 0,
+ reviews: (html ?? '').split('rv-item').length - 1,
+ };
+}
+
+/** See load.js: `after − before` is only a CPU measurement while the process set is not shrinking. */
+async function quiescedCpu(pid) {
+ let previous = processTreeCpu(pid);
+ for (let i = 0; i < 20; i++) {
+ await sleep(120);
+ const next = processTreeCpu(pid);
+ if (next && previous && next.processes === previous.processes) return next;
+ previous = next;
+ }
+ return previous;
+}
+
+// ---------------------------------------------------------------------------- reporting
+
+/** Least-squares slope of y against batch index, reported per 100 pages. */
+function trendPer100Pages(rows, key, concurrency) {
+ const points = rows.map((r, i) => [i * concurrency, r[key]]).filter(([, y]) => typeof y === 'number');
+ if (points.length < 4) return null;
+ const n = points.length;
+ const mx = points.reduce((a, [x]) => a + x, 0) / n;
+ const my = points.reduce((a, [, y]) => a + y, 0) / n;
+ let num = 0;
+ let den = 0;
+ for (const [x, y] of points) {
+ num += (x - mx) * (y - my);
+ den += (x - mx) ** 2;
+ }
+ return den ? Number(((num / den) * 100).toFixed(1)) : null;
+}
+
+function report(arms) {
+ const FIRST = 20;
+ console.log('\n## aging: first 20 renders vs last 20 (medians), and the trend across the whole run\n');
+ const head = [
+ 'arm',
+ 'pages',
+ 'wall 1st20',
+ 'wall last20',
+ 'Δ',
+ 'cpu 1st20',
+ 'cpu last20',
+ 'Δ',
+ 'script 1st',
+ 'script last',
+ 'rss 1st',
+ 'rss last',
+ 'wall /100pg',
+ 'cpu /100pg',
+ 'rss /100pg',
+ ];
+ const rows = arms.map((arm) => {
+ const r = arm.renders;
+ const b = arm.batchRows;
+ const firstR = r.slice(0, FIRST);
+ const lastR = r.slice(-FIRST);
+ const nb = Math.max(1, Math.round(FIRST / CONCURRENCY));
+ const firstB = b.slice(0, nb);
+ const lastB = b.slice(-nb);
+ const w1 = median(firstR.map((x) => x.wallMs));
+ const w2 = median(lastR.map((x) => x.wallMs));
+ const c1 = median(firstB.map((x) => x.cpuMsPerRender));
+ const c2 = median(lastB.map((x) => x.cpuMsPerRender));
+ const s1 = median(firstR.map((x) => x.scriptMs));
+ const s2 = median(lastR.map((x) => x.scriptMs));
+ const m1 = median(firstB.map((x) => x.rssMb));
+ const m2 = median(lastB.map((x) => x.rssMb));
+ const pct = (a, z) => (a ? `${z >= a ? '+' : ''}${(((z - a) / a) * 100).toFixed(1)}%` : '-');
+ return [
+ arm.name,
+ String(r.length),
+ `${w1}ms`,
+ `${w2}ms`,
+ pct(w1, w2),
+ `${c1}ms`,
+ `${c2}ms`,
+ pct(c1, c2),
+ `${s1}ms`,
+ `${s2}ms`,
+ `${m1}`,
+ `${m2}`,
+ String(trendPer100Pages(b, 'perRenderWall', CONCURRENCY)),
+ String(trendPer100Pages(b, 'cpuMsPerRender', CONCURRENCY)),
+ String(trendPer100Pages(b, 'rssMb', CONCURRENCY)),
+ ];
+ });
+ printTable(head, rows);
+
+ console.log('\n## the series — per-batch medians in decile bands (so a cliff cannot hide in a mean)\n');
+ for (const arm of arms) {
+ const bands = 10;
+ const size = Math.ceil(arm.batchRows.length / bands);
+ const cells = [];
+ for (let i = 0; i < arm.batchRows.length; i += size) {
+ const slice = arm.batchRows.slice(i, i + size);
+ cells.push(
+ `${String(i * CONCURRENCY + 1).padStart(3)}-${String(Math.min(arm.batchRows.length, i + size) * CONCURRENCY).padStart(3)}: ` +
+ `${String(median(slice.map((x) => x.perRenderWall))).padStart(5)}ms ` +
+ `${String(median(slice.map((x) => x.cpuMsPerRender))).padStart(4)}cpu ` +
+ `${String(median(slice.map((x) => x.rssMb))).padStart(4)}MB`
+ );
+ }
+ console.log(`${arm.name}${arm.recycleAt ? ` (contexts recycled at page ${arm.recycleAt * CONCURRENCY})` : ''}`);
+ for (const c of cells) console.log(` ${c}`);
+ console.log('');
+ }
+
+ console.log('## batch-to-batch spread, so a first-vs-last delta can be read against it\n');
+ for (const arm of arms) {
+ const walls = arm.batchRows.map((x) => x.perRenderWall).sort((a, b) => a - b);
+ const cpus = arm.batchRows.map((x) => x.cpuMsPerRender).sort((a, b) => a - b);
+ const q = (list, p) => list[Math.min(list.length - 1, Math.floor(list.length * p))];
+ console.log(
+ `${arm.name.padEnd(18)} wall p10-p50-p90 ${q(walls, 0.1)}-${q(walls, 0.5)}-${q(walls, 0.9)}ms ` +
+ `cpu p10-p50-p90 ${q(cpus, 0.1)}-${q(cpus, 0.5)}-${q(cpus, 0.9)}ms ` +
+ `fidelity (reviews) ${median(arm.renders.map((r) => r.reviews))}` +
+ (arm.wipeFailures ? ` WIPE-FAILURES ${arm.wipeFailures}` : '')
+ );
+ }
+
+ const wiping = arms.filter((a) => a.renders.some((r) => r.wipeMs !== null));
+ if (wiping.length) {
+ console.log('\n## what the wipe costs (resetForNextVariant, pin=[])\n');
+ const h = ['arm', 'jar', 'wipe p50', 'wipe p90', 'wipe max', 'render wall', 'wipe as % of render'];
+ const rs = wiping.map((arm) => {
+ const w = arm.renders
+ .map((r) => r.wipeMs)
+ .filter((x) => x !== null)
+ .sort((a, b) => a - b);
+ const q = (p) => w[Math.min(w.length - 1, Math.floor(w.length * p))];
+ const wall = median(arm.renders.map((r) => r.wallMs));
+ const jars = arm.renders.map((r) => r.jar).filter((j) => j !== null);
+ return [
+ arm.name,
+ jars.length ? `${Math.min(...jars)}-${Math.max(...jars)}` : '?',
+ `${q(0.5)}ms`,
+ `${q(0.9)}ms`,
+ `${w[w.length - 1]}ms`,
+ `${wall}ms`,
+ `${((q(0.5) / wall) * 100).toFixed(2)}%`,
+ ];
+ });
+ printTable(h, rs);
+ }
+}
+
+function printTable(head, rows) {
+ const widths = head.map((h, i) => Math.max(h.length, ...rows.map((r) => r[i].length)));
+ const line = (cells) => cells.map((c, i) => c.padEnd(widths[i])).join(' ');
+ console.log(line(head));
+ console.log(widths.map((w) => '-'.repeat(w)).join(' '));
+ for (const row of rows) console.log(line(row));
+}
+
+await main();
diff --git a/bench/render-cpu/bench.js b/bench/render-cpu/bench.js
new file mode 100644
index 0000000..184bc92
--- /dev/null
+++ b/bench/render-cpu/bench.js
@@ -0,0 +1,299 @@
+/**
+ * `bench/render-cpu` — what a render actually costs, and what each candidate change does to it.
+ *
+ * Method, and why it is this way:
+ *
+ * - ONE fixture, served from 127.0.0.1, deterministic (see fixture.js). No origin, no network
+ * jitter, no CDN. The numbers are about our in-page code, not about a site.
+ * - Variants are INTERLEAVED, not run in blocks: rep 1 of every variant, then rep 2, and so on. A
+ * laptop's clock speed drifts with thermals over minutes, and a block layout hands the whole
+ * drift to whichever variant ran last. Interleaving spreads it across all of them.
+ * - MEDIAN, not mean, over reps: one GC pause or one OS scheduling hiccup moves a mean and does
+ * not move a median.
+ * - Every variant renders the same URL with the same device profile, through the real
+ * `renderOnce` -> `defaultRenderer` path, so a variant cannot accidentally measure a different
+ * code path than production runs.
+ * - FIDELITY IS MEASURED ALONGSIDE COST, on every run. A cheaper render that loses the review
+ * widget is not a win, and the only way that stays honest is for the same table to carry both.
+ * `reviews`, `revealed`, `lazyTiles`, `imgGated` and `offers` are the fixture's content markers;
+ * a variant that changes any of them against baseline is flagged.
+ *
+ * Usage:
+ * node bench/render-cpu/bench.js # all variants, mobile, 5 reps
+ * node bench/render-cpu/bench.js --device desktop --reps 7
+ * node bench/render-cpu/bench.js --only baseline,monitor
+ * node bench/render-cpu/bench.js --json results/mobile.json
+ */
+
+import { writeFileSync, mkdirSync } from 'node:fs';
+import { dirname, resolve as resolvePath } from 'node:path';
+import { renderOnce } from '../../packages/browser/dist/index.js';
+import ManagedBrowser from '../../packages/browser/dist/ManagedBrowser.js';
+import { experiments, resetExperiments, splits } from '../../packages/browser/dist/experiments.js';
+import { defaultLaunchOptions } from '../../packages/browser/dist/settings.js';
+import { startFixture } from './fixture.js';
+import { cdpCounter, processTreeCpu, pageMetrics, readMetrics, median } from './instrument.js';
+import { VARIANTS, BASE_CONFIG, merge } from './variants.js';
+
+const args = process.argv.slice(2);
+const flag = (name, fallback) => {
+ const i = args.indexOf(`--${name}`);
+ return i >= 0 && args[i + 1] ? args[i + 1] : fallback;
+};
+const DEVICE = flag('device', 'mobile');
+const REPS = Number(flag('reps', 5));
+const ONLY = flag('only', '').split(',').filter(Boolean);
+const JSON_OUT = flag('json', '');
+const WARMUP = Number(flag('warmup', 1));
+
+/** Content markers in the returned HTML — the fidelity half of every row. */
+const fidelity = (html) => {
+ const count = (needle) => (html ? html.split(needle).length - 1 : 0);
+ return {
+ bytes: html?.length ?? 0,
+ reviews: count('rv-item'),
+ lazyTiles: count('data-tile="lz-'),
+ revealed: count('id="reviewWrap" class="transition-all opacity-100"') > 0 ? 1 : 0,
+ imgGated: count('id="onloadOk"'),
+ scrollGated: count('id="scrollGatedOk"'),
+ shadowFlattened: count('data-sh='),
+ propsStripped: count(' props=') === 0 ? 1 : 0,
+ };
+};
+
+async function main() {
+ const selected = VARIANTS.filter((v) => (ONLY.length ? ONLY.includes(v.name) : true)).filter(
+ (v) => !v.devices || v.devices.includes(DEVICE)
+ );
+ if (!selected.length) throw new Error(`no variants selected for device=${DEVICE}`);
+
+ const fixture = await startFixture();
+ console.log(`fixture: ${fixture.url}`);
+ console.log(`device: ${DEVICE} reps: ${REPS} variants: ${selected.map((v) => v.name).join(', ')}\n`);
+
+ // One browser per distinct launch-args set, all launched up front so a variant never pays a
+ // cold-launch cost another variant does not. Renders stay single-flight regardless.
+ const browsers = new Map();
+ const launchKeyOf = (v) => JSON.stringify(v.launch ?? null);
+ for (const variant of selected) {
+ const key = launchKeyOf(variant);
+ if (browsers.has(key)) continue;
+ const base = defaultLaunchOptions();
+ const options = variant.launch?.chromeArgs
+ ? { ...base, args: [...(base.args ?? []), ...variant.launch.chromeArgs] }
+ : base;
+ browsers.set(key, await ManagedBrowser.launch({ maxActivePages: 1, puppeteerLaunchOptions: options }));
+ }
+
+ const runs = new Map(selected.map((v) => [v.name, []]));
+
+ const runOnce = async (variant) => {
+ resetExperiments();
+ Object.assign(experiments, variant.experiments ?? {});
+ // Always on: the split is four Date.now() deltas per settle step, and without it the biggest
+ // line in the results ("settle") is a single undifferentiated number.
+ experiments.instrument = true;
+ const managed = browsers.get(launchKeyOf(variant));
+ const pid = managed.pid;
+ const cpuBefore = processTreeCpu(pid);
+ const counter = cdpCounter();
+ const started = Date.now();
+ let result;
+ let cdp;
+ try {
+ result = await renderOnce({
+ url: variant.urlQuery ? `${fixture.url}?${variant.urlQuery}` : fixture.url,
+ device: DEVICE,
+ browser: managed,
+ config: merge(BASE_CONFIG, variant.config),
+ captureNonIndexable: variant.captureNonIndexable ?? true,
+ ...(variant.browserOptions ?? {}),
+ probes: {
+ // Both sampled from the LIVE page, before teardown: the renderer process that did
+ // this render exits with its context and takes its CPU accounting with it.
+ metrics: async ({ page }) => pageMetrics(await readMetrics(page)),
+ cpu: () => processTreeCpu(pid),
+ passes: ({ page }) => page.evaluate(() => window.__passes ?? 0),
+ // Fixture geometry, reported so a run can be checked against PDP scale rather than
+ // assumed to be at it: page height drives the scroll-pass count, element count
+ // drives every walk.
+ shape: ({ page }) =>
+ page.evaluate(() => ({
+ height: document.body.scrollHeight,
+ elements: document.getElementsByTagName('*').length,
+ // The viewport Chrome actually gave us. An override can be clamped, and a
+ // variant that thinks it asked for 24,000px and got less would otherwise
+ // publish a conclusion about a height that never existed.
+ innerHeight: window.innerHeight,
+ })),
+ },
+ });
+ } finally {
+ cdp = counter.stop();
+ }
+ const wallMs = Date.now() - started;
+ const cpuAfter = result.probes?.cpu;
+ return {
+ wallMs,
+ renderMs: result.renderTimeMs ?? null,
+ ...result.timings,
+ cpuMs: cpuBefore && cpuAfter ? Math.round((cpuAfter.cpuSeconds - cpuBefore.cpuSeconds) * 1000) : null,
+ ...(result.probes?.metrics ?? {}),
+ cdpSent: cdp.sent,
+ cdpEvents: cdp.events,
+ cdpBytesInMb: Math.round(cdp.bytesIn / (1 << 20)),
+ evaluates: cdp.bySent['Runtime.callFunctionOn'] ?? 0,
+ fetchContinue: cdp.bySent['Fetch.continueRequest'] ?? 0,
+ fetchFulfill: cdp.bySent['Fetch.fulfillRequest'] ?? 0,
+ fetchFail: cdp.bySent['Fetch.failRequest'] ?? 0,
+ getResponseBody: cdp.bySent['Network.getResponseBody'] ?? 0,
+ outcome: result.outcome,
+ scrollMs: splits.scrollMs,
+ idleMs: splits.idleMs,
+ countMs: splits.countMs,
+ gateMs: splits.gateMs,
+ gateTicks: splits.gateTicks,
+ scrollPasses: result.probes?.passes ?? null,
+ pageHeight: result.probes?.shape?.height ?? null,
+ lightElements: result.probes?.shape?.elements ?? null,
+ innerHeight: result.probes?.shape?.innerHeight ?? null,
+ waitForMs: (result.waitForResults ?? []).reduce((a, r) => a + r.waitedMs, 0),
+ offers: result.job?.structuredOffers?.length ?? 0,
+ ...fidelity(result.html),
+ _bySent: cdp.bySent,
+ };
+ };
+
+ // Warm-up reps, discarded: the first render in a fresh Chrome pays one-time costs (V8 code cache
+ // for our in-page functions, the fixture's first compile, OS page faults) that no steady-state
+ // render pays, and they land entirely on whichever variant happens to go first.
+ for (let w = 0; w < WARMUP; w++) {
+ for (const variant of selected) await runOnce(variant);
+ }
+
+ for (let rep = 0; rep < REPS; rep++) {
+ for (const variant of selected) {
+ const run = await runOnce(variant);
+ runs.get(variant.name).push(run);
+ process.stdout.write(
+ `rep ${rep + 1}/${REPS} ${variant.name.padEnd(22)} ` +
+ `wall ${String(run.wallMs).padStart(6)}ms cpu ${String(run.cpuMs).padStart(6)}ms ` +
+ `task ${String(run.taskMs).padStart(5)}ms eval ${String(run.evaluates).padStart(3)} ` +
+ `cdp ${String(run.cdpSent).padStart(4)} reviews ${run.reviews}\n`
+ );
+ }
+ }
+
+ await fixture.close();
+ for (const managed of browsers.values()) await managed.close().catch(() => {});
+
+ report(selected, runs);
+ if (JSON_OUT) {
+ const path = resolvePath(process.cwd(), JSON_OUT);
+ mkdirSync(dirname(path), { recursive: true });
+ writeFileSync(
+ path,
+ JSON.stringify(
+ {
+ device: DEVICE,
+ reps: REPS,
+ node: process.version,
+ platform: `${process.platform}-${process.arch}`,
+ when: new Date().toISOString(),
+ runs: Object.fromEntries(runs),
+ },
+ null,
+ 2
+ )
+ );
+ console.log(`\nwrote ${path}`);
+ }
+}
+
+const COST_COLUMNS = [
+ ['wallMs', 'wall'],
+ ['cpuMs', 'cpu(tree)'],
+ ['taskMs', 'task'],
+ ['taskOtherMs', 'task-other'],
+ ['processTimeMs', 'proc-cpu'],
+ ['scriptMs', 'script'],
+ ['v8CompileMs', 'v8compile'],
+ ['devtoolsMs', 'devtools'],
+ ['layoutMs', 'layout'],
+ ['recalcStyleMs', 'style'],
+ ['settle', 'settle'],
+ ['scrollMs', 'scroll'],
+ ['idleMs', 'idle'],
+ ['countMs', 'count'],
+ ['gateMs', 'gate'],
+ ['postProcess', 'postProc'],
+ ['evaluates', 'evals'],
+ ['cdpSent', 'cdp-sent'],
+ ['scrollPasses', 'passes'],
+ ['pageHeight', 'height'],
+ ['lightElements', 'elements'],
+ ['innerHeight', 'innerH'],
+];
+
+const FIDELITY_COLUMNS = [
+ ['reviews', 'reviews'],
+ ['revealed', 'revealed'],
+ ['lazyTiles', 'lazyTiles'],
+ ['imgGated', 'imgGated'],
+ ['scrollGated', 'scrollGated'],
+ ['offers', 'offers'],
+ ['shadowFlattened', 'shadow'],
+ ['bytes', 'bytes'],
+];
+
+function report(selected, runs) {
+ const med = (name, key) => median(runs.get(name).map((r) => r[key]));
+ const baseName = selected[0].name;
+
+ const table = (columns, title, withDelta) => {
+ console.log(`\n## ${title}\n`);
+ const head = ['variant', ...columns.map(([, label]) => label)];
+ const rows = selected.map((v) => {
+ const cells = columns.map(([key]) => {
+ const value = med(v.name, key);
+ if (value === null) return '-';
+ if (!withDelta || v.name === baseName) return String(value);
+ const base = med(baseName, key);
+ if (base === null || base === 0) return String(value);
+ const pct = Math.round(((value - base) / base) * 100);
+ return `${value} (${pct > 0 ? '+' : ''}${pct}%)`;
+ });
+ return [v.name, ...cells];
+ });
+ const widths = head.map((h, i) => Math.max(h.length, ...rows.map((r) => r[i].length)));
+ const line = (cells) => cells.map((c, i) => c.padEnd(widths[i])).join(' ');
+ console.log(line(head));
+ console.log(widths.map((w) => '-'.repeat(w)).join(' '));
+ for (const row of rows) console.log(line(row));
+ };
+
+ table(COST_COLUMNS, `cost — median of ${REPS} reps (delta vs ${baseName})`, true);
+ table(FIDELITY_COLUMNS, 'fidelity — must match baseline, or the cost win is not a win', false);
+
+ // Flag any variant whose content markers diverge: the single most important line of output.
+ console.log('');
+ for (const v of selected.slice(1)) {
+ const diffs = FIDELITY_COLUMNS.filter(([key]) => key !== 'bytes')
+ .map(([key, label]) => [label, med(baseName, key), med(v.name, key)])
+ .filter(([, base, got]) => base !== got);
+ if (diffs.length) {
+ console.log(
+ `FIDELITY DIVERGENCE ${v.name}: ` + diffs.map(([label, base, got]) => `${label} ${base} -> ${got}`).join(', ')
+ );
+ }
+ }
+
+ // The CDP method breakdown for baseline: where the message count actually is.
+ const baseRun = runs.get(baseName)[0];
+ console.log(`\n## baseline CDP methods sent (one representative run, ${baseRun.cdpSent} total)\n`);
+ for (const [method, count] of Object.entries(baseRun._bySent).slice(0, 14)) {
+ console.log(`${String(count).padStart(5)} ${method}`);
+ }
+}
+
+await main();
diff --git a/bench/render-cpu/fixture-server.js b/bench/render-cpu/fixture-server.js
new file mode 100644
index 0000000..37056d5
--- /dev/null
+++ b/bench/render-cpu/fixture-server.js
@@ -0,0 +1,20 @@
+/**
+ * The fixture, in its OWN process.
+ *
+ * `startFixture()` runs an HTTP server on the caller's event loop. That is fine for a single render
+ * and actively misleading under load: at concurrency 24 one Node thread was serving ~9,600 HTTP
+ * responses (including ~113KB script bundles) WHILE handling CDP traffic and every
+ * request-interception callback for 24 renders. A concurrency ladder measured that way finds the
+ * harness's event loop, not the renderer's ceiling — and production does not have that shape at all,
+ * because the origin is a different machine.
+ *
+ * node bench/render-cpu/fixture-server.js --port 58200
+ */
+import { startFixture } from './fixture.js';
+
+const args = process.argv.slice(2);
+const i = args.indexOf('--port');
+const port = i >= 0 ? Number(args[i + 1]) : 58200;
+const fixture = await startFixture({ port });
+console.log(fixture.url);
+process.on('SIGTERM', () => fixture.close().then(() => process.exit(0)));
diff --git a/bench/render-cpu/fixture.js b/bench/render-cpu/fixture.js
new file mode 100644
index 0000000..2072662
--- /dev/null
+++ b/bench/render-cpu/fixture.js
@@ -0,0 +1,370 @@
+/**
+ * The page under test: a deterministic, self-contained commerce-PDP-shaped document.
+ *
+ * WHY SYNTHETIC. The measurements this harness exists for are of OUR OWN in-page code — full-tree
+ * element walks, scroll passes, shadow flattening, CSS pruning. Their cost is driven by the shape
+ * of the DOM (element count, shadow-root count, rule count, page height), not by whose site it is.
+ * A synthetic page fixes that shape, removes network jitter, and makes an A/B of a 3% change
+ * legible. What it CANNOT tell you is whether a change is FAITHFUL on a real site — that needs a
+ * real render (see README, "What this cannot answer").
+ *
+ * Shape is modelled on a production commerce PDP as measured in the deployment notes: ~18k light-DOM elements, a shadow-DOM
+ * review widget that appears only after its anchor intersects (0 -> ~1,380 nodes), a reveal wrapper
+ * with three class states, astro-island `props` attributes that dominate byte size, and a utility
+ * stylesheet where most selectors never match.
+ *
+ * Deterministic: one seeded PRNG, no Date.now() in the markup, no randomness in the server. Two
+ * runs produce byte-identical HTML, so a diff in a measurement is a diff in the code under test.
+ */
+
+import { createServer } from 'node:http';
+
+/** mulberry32 — small, fast, deterministic. */
+const rng = (seed) => () => {
+ seed = (seed + 0x6d2b79f5) | 0;
+ let t = seed;
+ t = Math.imul(t ^ (t >>> 15), t | 1);
+ t ^= t + Math.imul(t ^ (t >>> 7), t | 61);
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
+};
+
+// 1x1 transparent PNG — every on the page resolves to this, so image bytes are constant and
+// the only thing an image costs is the request + decode we are trying to measure.
+const PNG = Buffer.from(
+ 'iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8AAAwAB/gEBQ4kAAAAAAElFTkSuQmCC',
+ 'base64'
+);
+
+const SHAPE = {
+ sections: 24, // lazy grid sections, injected on intersect
+ tilesPerSection: 16,
+ lazyTilesPerSection: 12,
+ elementsPerTile: 18,
+ shadowWidgets: 12,
+ shadowRulesPerWidget: 140,
+ shadowElementsPerWidget: 160,
+ utilityRules: 3000, // most never match — what pruneUnmatchedCss is for
+ reviewBlocks: 460, // x3 elements each = the ~1,380 matched review elements measured on a real PDP
+ islands: 18, // astro-island elements carrying large props attributes
+ islandPropsBytes: 24000,
+};
+
+const CLASSES = ['flex', 'grid', 'px-4', 'py-2', 'text-sm', 'font-bold', 'rounded', 'border', 'shadow-sm', 'mt-2'];
+
+/** A utility stylesheet in the Tailwind idiom: thousands of rules, a handful of which match. */
+function utilityCss(random) {
+ let css =
+ ':root{--c:#111}body{margin:0;font:14px system-ui}.hdr{position:sticky;top:0;height:64px;background:#fff}\n';
+ css += '.hdr.hide{transform:translateY(-100%)}\n';
+ // Height is a measured property of the fixture, not an accident of the markup: a 4-column grid
+ // of 64px cells puts the page at ~20,000px on a 390px viewport, which is PDP scale. A single
+ // column of 280px cells (the obvious markup) is 270,000px and makes every scroll pass a
+ // 300-step, 19-second crawl that no real page would charge.
+ css += '.grid{display:grid;grid-template-columns:repeat(4,1fr);gap:2px}\n';
+ css += '.tile{display:block;height:64px;overflow:hidden;border:1px solid #eee;font-size:8px}\n';
+ css += '.tile svg{width:6px;height:6px}\n';
+ css += '#reviewList{display:grid;grid-template-columns:repeat(3,1fr);gap:2px}\n';
+ css += '.rv-item{height:40px;overflow:hidden;font-size:8px}\n';
+ css += '.transition-all{transition:all .7s}.max-h-0{max-height:0;opacity:0}.hidden{display:none}\n';
+ for (let i = 0; i < SHAPE.utilityRules; i++) {
+ const n = Math.floor(random() * 9999);
+ // Deliberately a mix: pseudo-classes, attribute selectors with quotes, media groups, and a
+ // small fraction that genuinely match — the same mix the prune pass has to judge.
+ if (i % 17 === 0) css += `@media (min-width:${640 + (i % 5) * 160}px){.u-${n}:hover .u-${n}-x{gap:${i % 9}px}}\n`;
+ else if (i % 11 === 0) css += `[data-u="${n}"]::after{content:"${n}"}\n`;
+ else if (i % 7 === 0) css += `.u-${n}:not(.off) > .u-${n}-y{margin:${i % 5}px}\n`;
+ else css += `.u-${n}{padding:${i % 13}px;color:var(--c)}\n`;
+ }
+ return css;
+}
+
+/** One product tile: the element-count workhorse, with lazy image attributes to resolve. */
+function tile(i, random) {
+ const cls = () => CLASSES[Math.floor(random() * CLASSES.length)];
+ let html = `
`;
+ html += ``;
+ html += `
Brand ${i % 40}
Product name ${i}
`;
+ html += `
$${20 + (i % 80)}.99$${60 + (i % 90)}.00
`;
+ // A star row inside plain light DOM plus one in a shadow widget elsewhere.
+ html += `
`;
+ for (let s = 0; s < 5; s++)
+ html += ``;
+ html += '
';
+ html += `
`;
+ return html;
+}
+
+/** A custom element with an open shadow root and its own (unscoped) stylesheet. */
+function shadowWidgetScript() {
+ return `
+class BenchWidget extends HTMLElement {
+ connectedCallback() {
+ if (this.shadowRoot) return;
+ const sr = this.attachShadow({ mode: 'open' });
+ const n = ${SHAPE.shadowElementsPerWidget};
+ const rules = ${SHAPE.shadowRulesPerWidget};
+ let css = ':host{display:block}button{background:#06c;color:#fff}svg{width:8px;height:8px}'
+ + '.w-0{display:grid;grid-template-columns:repeat(8,1fr)}.w-0>div{height:12px;overflow:hidden;font-size:7px}';
+ for (let i = 0; i < rules; i++) css += '.w-' + i + '{padding:' + (i % 7) + 'px}';
+ css += '@media (min-width:600px){.w-0{gap:2px}}';
+ let html = '
';
+ for (let i = 0; i < n; i++) {
+ html += '
cell ' + i + ''
+ + '
';
+ }
+ html += '
';
+ sr.innerHTML = html;
+ }
+}
+customElements.define('bench-widget', BenchWidget);`;
+}
+
+/**
+ * The page's own behaviour: lazy sections that fill on intersect, a review widget that only starts
+ * loading once its anchor intersects and then fills after a delay, a reveal wrapper that passes
+ * through three class states, an onload-gated block, and perpetual cosmetic churn.
+ */
+function pageScript() {
+ return `
+const mkTiles = (host, from, count) => {
+ let html = '';
+ for (let i = from; i < from + count; i++) {
+ html += '
lazy ' + i + '';
+ for (let k = 0; k < 14; k++) html += 'c' + k + '';
+ html += '
';
+ }
+ host.insertAdjacentHTML('beforeend', html);
+};
+
+// Lazy grid sections: fill on first intersection (what the scroll passes exist to trip).
+const io = new IntersectionObserver((entries) => {
+ for (const e of entries) {
+ if (!e.isIntersecting) continue;
+ const host = e.target;
+ io.unobserve(host);
+ if (host.dataset.filled) continue;
+ host.dataset.filled = '1';
+ setTimeout(() => mkTiles(host, Number(host.dataset.idx) * 1000, ${SHAPE.lazyTilesPerSection}), 40);
+ }
+}, { rootMargin: '0px' });
+for (const s of document.querySelectorAll('[data-lazy-section]')) io.observe(s);
+
+// The review widget: nothing happens until #reviewsAnchor intersects. Then a fetch-shaped delay, then
+// ~1,380 nodes land, then the reveal wrapper walks hidden -> max-h-0 -> revealed.
+const reviewObserver = new IntersectionObserver((entries) => {
+ if (!entries.some((e) => e.isIntersecting)) return;
+ reviewObserver.disconnect();
+ const list = document.getElementById('reviewList');
+ const wrap = document.getElementById('reviewWrap');
+ setTimeout(() => {
+ let html = '';
+ for (let i = 0; i < ${SHAPE.reviewBlocks}; i++) {
+ html += '
user' + i + ''
+ + '
Review body number ' + i + ' with a sentence of text.
';
+ }
+ list.innerHTML = html;
+ wrap.className = 'transition-all max-h-0';
+ setTimeout(() => { wrap.className = 'transition-all opacity-100'; }, 120);
+ }, Number(document.body.dataset.reviewDelay || 900));
+}, { rootMargin: '0px' });
+reviewObserver.observe(document.getElementById('reviewsAnchor'));
+
+// A block that loads only after a SCROLL EVENT has fired — the fidelity tripwire for any variant
+// that stops scrolling (a taller viewport, in particular). IntersectionObserver fires for content
+// that is in view without ever being scrolled to; a scroll LISTENER does not, and plenty of real
+// carousels and infinite lists hang off one. Without this marker, "the whole page in one viewport"
+// looks free.
+let scrollFired = false;
+addEventListener(
+ 'scroll',
+ () => {
+ if (scrollFired) return;
+ scrollFired = true;
+ const host = document.getElementById('scrollGated');
+ if (host) host.innerHTML = 'scroll-gated content';
+ },
+ { passive: true, once: false }
+);
+
+// A block revealed only by an image's load event — the fidelity tripwire for turning images off.
+const probe = document.getElementById('onloadProbe');
+if (probe) {
+ const img = new Image();
+ img.onload = () => { probe.innerHTML = 'image-gated content'; };
+ img.onerror = () => { probe.dataset.imgError = '1'; };
+ img.src = '/asset/img/gate.png';
+}
+
+// Sticky header that hides on scroll-down and re-reveals at the top, on a throttled handler —
+// the reason scroll.topSettleMs exists.
+let lastY = 0, ticking = false;
+addEventListener('scroll', () => {
+ if (ticking) return;
+ ticking = true;
+ setTimeout(() => {
+ const y = scrollY;
+ document.querySelector('.hdr').classList.toggle('hide', y > lastY && y > 100);
+ lastY = y;
+ ticking = false;
+ }, 30);
+}, { passive: true });
+
+// Cosmetic churn: text-only mutations forever. A DOM-stability check must NOT be reset by these.
+let tick = 0;
+setInterval(() => { document.getElementById('ticker').textContent = 'offer ends in ' + (tick++) + 's'; }, 100);
+`;
+}
+
+function buildHtml(query) {
+ const random = rng(1337);
+ const reviewDelay = Number(query.get('reviewDelay') ?? 900);
+ let html = '';
+ html += 'Bench PDP — deterministic fixture';
+ html += '';
+ html += ``;
+ html += ``;
+ html += '';
+ html += '';
+ html += ``;
+ html += '';
+
+ // Above-the-fold product block.
+ html += '
Bench Product
';
+ html += '';
+ for (let i = 0; i < SHAPE.islands; i++) {
+ // astro-island-shaped: a huge props attribute, the byte-dominant thing removeAttributes strips.
+ const props = JSON.stringify({ blob: 'x'.repeat(SHAPE.islandPropsBytes / 2), i });
+ html += `
island ${i}
`;
+ }
+ html += '';
+
+ // The element-count body: static tiles + lazy sections + shadow widgets interleaved.
+ for (let s = 0; s < SHAPE.sections; s++) {
+ html += ``;
+ for (let t = 0; t < SHAPE.tilesPerSection; t++) html += tile(s * 100 + t, random);
+ html += '';
+ if (s % Math.ceil(SHAPE.sections / SHAPE.shadowWidgets) === 0) {
+ html += `
Widget ${s}
light child
`;
+ }
+ }
+
+ // The review block, at the bottom — below the fold on a short viewport, in view on a tall one.
+ html += '