meat/app/renderer.ts
2026-08-07 12:49:13 +02:00

213 lines
8 KiB
TypeScript

import { Framebuffer } from "../engine/render/Framebuffer"
import type { RenderConfig } from "../engine/render/RenderConfig"
import type { Mat4 } from "../engine/math/Mat4"
import type { Camera } from "../engine/scene/Camera"
import { renderBand, MOB_FLOATS, type MobDraw, type Scene } from "./renderScene"
/** Sky is drawn at 1/SKY_STEP resolution; band splits align to it. */
const SKY_STEP = 2
/** Use worker threads when the page can share memory (else single-thread). */
const ENABLE_WORKERS = true
/** Cap on render worker threads. Deliberately low: a browser game shares the
* machine with the browser itself, the compositor, and whatever else is open, so
* grabbing every core backfires -- the frame-time *tail* blows up even while the
* median improves. Measured (`bun run bench:browser`) on a busy 16-core desktop:
* 3 workers beat single-thread on both median (~1.5x) and p95 (~1.2x); 6 were far
* worse. Raise only if the target is a dedicated/idle machine; retune via the
* bench. */
const MAX_WORKERS = 3
// Indices into the shared control Int32Array.
const FRAME = 0 // bumped by main to dispatch a frame
const DONE = 1 // workers add 1 when their band is finished
const VIS = 2 // number of visible chunk indices this frame
const MOBVIS = 3 // number of visible mobs this frame
/**
* Render driver. When the page is cross-origin-isolated it runs a pool of worker
* threads, each owning a disjoint row band of a `SharedArrayBuffer` framebuffer;
* otherwise it renders inline on the main thread. Same output either way.
*
* The worker barrier is lock-free: workers `Atomics.wait` on a frame counter, so
* there are **no per-frame messages** (the old postMessage barrier was the jitter
* source). Per-frame inputs (camera, matrix, visible list) live in shared arrays.
* `dispatch` starts a frame without blocking; the caller polls `done()` and
* presents when ready, so pacing stays on the caller's `requestAnimationFrame`.
*/
export type Renderer = {
readonly fb: Framebuffer
readonly parallel: boolean
reconfigure: (config: RenderConfig) => void
/** Start rendering one frame (non-blocking in the worker path). */
dispatch: (camera: Camera, viewProj: Mat4, visible: number[], time: number, mobDraws: MobDraw[]) => void
/** Has the dispatched frame finished? (always true single-threaded.) */
done: () => boolean
/** Critical-path render time of the last frame, ms (max band time / inline time). */
workMs: () => number
}
export function createRenderer(scene: Scene, initial: RenderConfig, forceWorkers?: boolean): Renderer {
const hw = (globalThis.navigator as Navigator | undefined)?.hardwareConcurrency ?? 4
const workerCount = Math.max(1, Math.min(MAX_WORKERS, hw - 1))
const maxVis = Math.max(1, scene.chunks.length)
const maxMobs = Math.max(1, scene.mobCount)
let config = initial
const want = forceWorkers ?? ENABLE_WORKERS
let parallel = want && canShare()
let fb = Framebuffer.create(1, 1)
let workers: Worker[] = []
let ctrl: Int32Array<ArrayBufferLike> = new Int32Array(0)
let cam: Float64Array<ArrayBufferLike> = new Float64Array(0) // pos x/y/z, yaw, pitch, fov, time
let vp: Float32Array<ArrayBufferLike> = new Float32Array(0) // the view-projection matrix
let vis: Int32Array<ArrayBufferLike> = new Int32Array(0) // visible chunk indices
let mob: Float32Array<ArrayBufferLike> = new Float32Array(0) // visible mob transforms (MOB_FLOATS each)
let times: Float64Array<ArrayBufferLike> = new Float64Array(0) // per-worker band render ms
let lastWork = 0
function setup(): void {
for (const w of workers) {
w.terminate()
}
workers = []
const width = config.internalWidth
const height = config.internalHeight
if (parallel) {
const n = width * height
fb = { width, height, color: new Uint32Array(new SharedArrayBuffer(n * 4)), depth: new Float32Array(new SharedArrayBuffer(n * 4)) }
const bands = splitBands(height, workerCount, SKY_STEP)
ctrl = new Int32Array(new SharedArrayBuffer(4 * 4))
cam = new Float64Array(new SharedArrayBuffer(7 * 8))
vp = new Float32Array(new SharedArrayBuffer(16 * 4))
vis = new Int32Array(new SharedArrayBuffer(maxVis * 4))
mob = new Float32Array(new SharedArrayBuffer(maxMobs * MOB_FLOATS * 4))
times = new Float64Array(new SharedArrayBuffer(bands.length * 8))
try {
bands.forEach((band, index) => {
const worker = new Worker(new URL("./render-worker.ts", import.meta.url), { type: "module" })
worker.addEventListener("error", () => {
parallel = false
})
worker.postMessage({
colorSAB: fb.color.buffer,
depthSAB: fb.depth.buffer,
width,
height,
scene,
band,
config,
skyStep: SKY_STEP,
ctrlSAB: ctrl.buffer,
camSAB: cam.buffer,
vpSAB: vp.buffer,
visSAB: vis.buffer,
mobSAB: mob.buffer,
timesSAB: times.buffer,
index,
})
workers.push(worker)
})
} catch {
parallel = false
for (const w of workers) {
w.terminate()
}
workers = []
}
}
if (!parallel || workers.length === 0) {
fb = Framebuffer.create(width, height)
}
}
setup()
return {
get fb() {
return fb
},
get parallel() {
return parallel && workers.length > 0
},
reconfigure(next) {
config = next
setup()
},
dispatch(camera, viewProj, visible, time, mobDraws) {
if (parallel && workers.length > 0) {
cam[0] = camera.position.x
cam[1] = camera.position.y
cam[2] = camera.position.z
cam[3] = camera.yaw
cam[4] = camera.pitch
cam[5] = camera.fov
cam[6] = time
vp.set(viewProj)
const count = Math.min(visible.length, vis.length)
for (let i = 0; i < count; i++) {
vis[i] = visible[i]
}
const mobCount = Math.min(mobDraws.length, maxMobs)
for (let i = 0; i < mobCount; i++) {
const d = mobDraws[i]
const o = i * MOB_FLOATS
mob[o] = d.kind === "bee" ? 1 : 0
mob[o + 1] = d.x
mob[o + 2] = d.y
mob[o + 3] = d.z
mob[o + 4] = d.heading
mob[o + 5] = d.scale
}
Atomics.store(ctrl, VIS, count)
Atomics.store(ctrl, MOBVIS, mobCount)
Atomics.store(ctrl, DONE, 0)
Atomics.add(ctrl, FRAME, 1)
Atomics.notify(ctrl, FRAME, workers.length)
return
}
const t0 = performance.now()
renderBand(fb, scene, camera, viewProj, visible, mobDraws, config, SKY_STEP, time, 0, fb.height)
lastWork = performance.now() - t0
},
done() {
return !(parallel && workers.length > 0) || Atomics.load(ctrl, DONE) >= workers.length
},
workMs() {
if (parallel && workers.length > 0) {
let m = 0
for (let i = 0; i < workers.length; i++) {
if (times[i] > m) {
m = times[i]
}
}
return m
}
return lastWork
},
}
}
/** Shared memory needs SharedArrayBuffer + a cross-origin-isolated page (the
* COOP/COEP headers Vite serves). Without it, render on the main thread. */
function canShare(): boolean {
return (
typeof SharedArrayBuffer !== "undefined" &&
typeof Worker !== "undefined" &&
(globalThis as { crossOriginIsolated?: boolean }).crossOriginIsolated === true
)
}
/** Split `height` rows into ~`count` bands. Interior boundaries snap up to a
* multiple of `step` so the sky's block grid stays aligned (no seam), while the
* bands stay disjoint so no two workers write the same pixel. */
function splitBands(height: number, count: number, step: number): [number, number][] {
const bands: [number, number][] = []
const per = Math.ceil(height / count)
let y = 0
while (y < height) {
const raw = y + per
const y1 = raw >= height ? height : Math.min(height, Math.ceil(raw / step) * step)
bands.push([y, y1])
y = y1
}
return bands
}