From 5cf5e0fdfaadd36e677cc4bba8a090dc721d9d71 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 00:44:35 +0200 Subject: [PATCH 01/13] fix(shadows): a light cut asks for each caster once a frame, whatever its views and batches (#525) --- packages/sdk-browser/src/gpu/dag/lightCut.ts | 4 +- .../src/gpu/dag/lightCutCapacity.ts | 2 +- .../src/gpu/dag/lightCutFrame.test.ts | 72 +++++++++++++++---- .../src/gpu/dag/shader/floorWgsl.ts | 14 ++-- .../src/gpu/dag/shader/snapshotWgsl.ts | 18 ++++- 5 files changed, 89 insertions(+), 21 deletions(-) diff --git a/packages/sdk-browser/src/gpu/dag/lightCut.ts b/packages/sdk-browser/src/gpu/dag/lightCut.ts index b92a281f46..62b77ec8b7 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCut.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCut.ts @@ -43,7 +43,7 @@ export function createDagLightCut(resources: DagResources) { const { worldCount, blockCount, buffers } = resources; const capacity = lightCutCapacity(device.limits, resources), queueCap = lightQueueCap(resources, capacity), - layout = dagWorkLayout(blockCount, capacity); + layout = dagWorkLayout(blockCount, capacity, pageCount); const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST; const own = (descriptor: GPUBufferDescriptor) => { const buffer = device.createBuffer(descriptor); @@ -141,6 +141,8 @@ export function createDagLightCut(resources: DagResources) { count: number, ) { if (count > capacity) throw new Error(`${count} light views, at most ${capacity}`); + // The frame's first cut starts its list, and forgets what the last frame asked for. + if (!listed) encoder.clearBuffer(work, layout.asked * 4, layout.askedWords * 4); cutViews.count = light.views = count; cutViews.append = listed; listed = true; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts b/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts index 17ec2c9751..0ddef4a8a7 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts @@ -42,7 +42,7 @@ export function lightCutCapacity(limits: LightCutLimits, shape: LightCutShape) { if (Math.min(shape.levelSizes[level] * views, queueCap) > threads) return false; const frames = views * shape.worldCount * FRAME_VEC4 * 16, flags = (queueCap * LEVEL_QUEUES + shape.pageCount * 4) * 4, - work = dagWorkLayout(shape.blockCount, views).words * 4; + work = dagWorkLayout(shape.blockCount, views, shape.pageCount).words * 4; return Math.max(frames, flags, work) <= bytes; }; let views = DAG_MAX_VIEWS; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts index 0e63aa1ca0..65474f0542 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts @@ -2,8 +2,9 @@ // Every batch's requests must be read back, whatever the batch count: the cuts append to one list // the frame copies once (`VIEW_APPEND`), so no batch's coarse view is redrawn for want of a report // slot, and its missing casters are asked for. The GPU side here is the shader's contract, run on -// the buffers the host wrote: a cut resets the list unless its uniform says append, and a view -// whose caster is not resident draws coarser and requests it. +// the buffers the host wrote: a cut resets the list unless its uniform says append, a view asks for +// each caster it wants the first time the frame's cuts do (`firstAsk`), and a view whose caster is +// not resident draws coarser. import test from 'node:test'; import assert from 'node:assert/strict'; import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; @@ -15,6 +16,8 @@ import { packRequest } from './request.ts'; import { COARSER_VIEWS, DAG_UNIFORM_BYTES, LIST_FULL } from './shader/viewsWgsl.ts'; import { VIEW_FLAGS_WORD } from './uniforms.ts'; import { VIEW_APPEND } from './shader/pagesWgsl.ts'; +import { dagWorkLayout } from './shader/floorWgsl.ts'; +import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; const CASTERS = 16; @@ -32,6 +35,9 @@ function lightCutFrame() { copyBufferToBuffer(from: GPUBuffer, at: number, to: GPUBuffer, toAt: number, size: number) { new Uint8Array(bytes(to)).set(new Uint8Array(bytes(from), at, size), toAt); }, + clearBuffer(buffer: GPUBuffer, at: number, size: number) { + new Uint8Array(bytes(buffer), at, size).fill(0); + }, beginComputePass: () => ({ setBindGroup() {}, setPipeline() {}, @@ -59,29 +65,41 @@ function lightCutFrame() { buffers: [], } as unknown as Parameters[0]); const output = buffers.find(({ label }) => label === 'Trillion3D light cut output')!; - /** The GPU running the cut just encoded, over the view of `caster`, against `resident`. */ - const run = (caster: number, resident: Set) => { + const work = buffers.find(({ label }) => label === 'Trillion3D light cut work')!; + const { asked } = dagWorkLayout(1, cut.capacity, CASTERS); + /** The GPU running the cut just encoded, over a view that wants `casters`, against `resident`. */ + const run = (casters: number[], resident: Set) => { const uniform = writes.findLast(({ buffer }) => buffer.size === DAG_UNIFORM_BYTES)!; const viewFlags = new Uint32Array(uniform.data.slice().buffer)[VIEW_FLAGS_WORD]; - const out = new Uint32Array(output.getMappedRange()); + const out = new Uint32Array(output.getMappedRange()), + bits = new Uint32Array(work.getMappedRange()); if (viewFlags & VIEW_APPEND) out[OUT_FLAGS] &= LIST_FULL; else out[OUT_COUNT] = out[OUT_FLAGS] = 0; - if (resident.has(caster)) return viewFlags; - out[OUT_FLAGS] |= 1 << COARSER_VIEWS; - const slot = out[OUT_COUNT]++; - if (slot < CASTERS) out[SELECTION_HEADER_WORDS + slot] = packRequest(caster, 1); - else out[OUT_FLAGS] |= LIST_FULL; + for (const caster of casters) { + if (!resident.has(caster)) out[OUT_FLAGS] |= 1 << COARSER_VIEWS; + const word = asked + (caster >> 5), + bit = 1 << (caster & 31); + if (bits[word] & bit) continue; + bits[word] |= bit; + const slot = out[OUT_COUNT]++; + if (slot < CASTERS) out[SELECTION_HEADER_WORDS + slot] = packRequest(caster, 1); + else out[OUT_FLAGS] |= LIST_FULL; + } return viewFlags; }; const views = [{ uniforms: sunRun(64).uniforms }]; - /** One frame drawing `pages` in one batch each: the page is its caster. Returns the batches' - * view flags, whether each batch's requests will be read, and how many report copies ran. */ - const frame = async (pages: number[], resident: Set) => { + /** One frame drawing `pages` in one batch each, whose view wants `wants(page)`: the page itself + * by default. Returns the batches' view flags and how many report copies ran. */ + const frame = async ( + pages: number[], + resident: Set, + wants = (page: number) => [page], + ) => { const settles: Array<(submitted: boolean) => void> = [], flags: number[] = []; for (const page of pages) { cut.encode(encoder, views, 1); - flags.push(run(page, resident)); + flags.push(run(wants(page), resident)); const settle = cut.redraws.encode(encoder, [page], [0], 1); if (settle) settles.push(settle); } @@ -130,3 +148,29 @@ test('a sweep of many batches a frame converges to full detail', async () => { assert.equal(resident.size, 12, 'every caster was asked for and arrived'); assert.deepEqual(drawn, [], 'every page is drawn at full detail, nothing left to redraw'); }); + +test('a light cut asks for a caster once a frame, its bit read before the atomic', () => { + for (const line of [ + 'if(isLightCut()&&!firstAsk(page)){return;}', + 'let word=drawnGroupsMax()+1u+(page>>5u);let bit=1u<<(page&31u);', + 'if((atomicLoad(&work[word])&bit)!=0u){return false;}', + 'return (atomicOr(&work[word],bit)&bit)==0u;', + ]) + assert.ok(DAG_RELEVE_WGSL.includes(line), line); +}); + +// Every sun level of every batch wants the casters that span the scene. Asked once a frame, they +// leave the list, as long as the catalogue, room for each batch's own casters: asked for, and never +// drawn again at once for want of a place in it (#525). +test("a caster every batch wants is asked for once a frame, and each batch's own casters too", async () => { + const { cut, frame } = lightCutFrame(); + const pages = [4, 5, 6, 7, 8, 9, 10, 11]; + for (const run of ['a frame', 'the next frame']) { + await frame(pages, new Set(), (page) => [0, 1, 2, 3, page]); + const asked = cut.reports.takeRequests()?.sort((a, b) => a - b); + assert.deepEqual(asked, [0, 1, 2, 3, ...pages], `${run}: every caster, each once`); + const now: number[] = []; + cut.redraws.takeRedraw((page) => now.push(page)); + assert.deepEqual(now, [], `${run}: no coarse page drawn again at once`); + } +}); diff --git a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts index a4e1076a5d..b8a4009ddd 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts @@ -23,10 +23,13 @@ * top-down pruning would then drop everything. * * `views` is the view capacity the buffer serves: one for a camera, one row each for a light cut. + * `asked` is the catalogue a light cut asks for: one bit per page behind the rest (`firstAsk`, + * `snapshotWgsl.ts`); a camera, which asks for a page once, has none. */ -export function dagWorkLayout(blockCount: number, views = 1) { +export function dagWorkLayout(blockCount: number, views = 1, asked = 0) { const base = blockCount * 2, - viewWords = base + 9; + viewWords = base + 9, + drawnGroupsMax = viewWords + VIEW_WORD_ROWS * views; return { base, /** The nine frame counters, in the order `levelWgsl.ts` names them. */ @@ -39,8 +42,11 @@ export function dagWorkLayout(blockCount: number, views = 1) { /** First per-view word: row `r` (`VIEW_WORD_ROWS`) of view `v` is `viewWords + r * views + v`. */ viewWords, /** The most sixty-four-wide groups any view drew, behind the per-view rows. */ - drawnGroupsMax: viewWords + VIEW_WORD_ROWS * views, - words: viewWords + VIEW_WORD_ROWS * views + 1, + drawnGroupsMax, + /** The frame's asked bits, behind it: `askedWords` words, cleared at the frame's first cut. */ + asked: drawnGroupsMax + 1, + askedWords: Math.ceil(asked / 32), + words: drawnGroupsMax + 1 + Math.ceil(asked / 32), }; } diff --git a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts index b1ca978331..17a26e5ee5 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts @@ -13,8 +13,24 @@ * refused rank sets bit 0: the snapshot is then TRUNCATED, and the frame refuses it whole * rather than adopt it amputated. Frame totals lose nothing — they describe the cut, not the * list that reports it (`totalsWgsl.ts`). + * + * A light cut's frame asks for a page ONCE, however many of its views and batches want it: every + * batch appends to one list (`VIEW_APPEND`) as long as the catalogue, and the same caster asked by + * each sun level and each batch filled it with repeats, so a late batch's own casters fell past it + * (`LIST_FULL`) and its coarse pages were drawn again every frame without ever being asked for. + * The test reads the page's bit before the atomic (`firstAsk`), as the shading's page requests do. */ -export const DAG_RELEVE_WGSL = `fn emitOne(page:u32,pixels:f32){emitWord(page,quantizePriority(pixels),true);} +export const DAG_RELEVE_WGSL = `fn emitOne(page:u32,pixels:f32){ + if(isLightCut()&&!firstAsk(page)){return;} + emitWord(page,quantizePriority(pixels),true); +} +/** True for the first ask of \`page\` in the frame's light cuts: its bit behind the per-view words + * (\`dagWorkLayout\`, \`asked\`), cleared by the host at the frame's first cut. */ +fn firstAsk(page:u32)->bool{ + let word=drawnGroupsMax()+1u+(page>>5u);let bit=1u<<(page&31u); + if((atomicLoad(&work[word])&bit)!=0u){return false;} + return (atomicOr(&work[word],bit)&bit)==0u; +} /** One request word in the sample; past the cap it is dropped, and \`declare\` says truncated. */ fn emitWord(page:u32,priority:u32,declare:bool){ let slot=atomicAdd(&out.count,1u); From d4483859a8a659c644f7d247f01593ec75a88e29 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 00:44:35 +0200 Subject: [PATCH 02/13] fix(shadows): a batch that dropped work keeps its view limit until the camera rests (#525) --- .../src/gpu/dag/lightCutRedraws.test.ts | 27 ++++++++++++++++++- .../src/gpu/dag/lightCutRedraws.ts | 13 ++++++--- .../src/gpu/dag/lightCutViewLimit.ts | 2 +- 3 files changed, 36 insertions(+), 6 deletions(-) diff --git a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts index d2cb84fee6..418e65bbe0 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts @@ -60,12 +60,37 @@ test('the pages of a frame that dropped work are drawn again, in fewer views unt await frame([4, 9, 12, 20, 21], true, [0, 0, 0, 1, 1]); assert.equal(redraws.viewLimit, 2, 'five pages in two views fit: no swing back to three'); redraws.residencyChanged(); - assert.equal(redraws.viewLimit, 24, 'residency moved: the drop is forgotten'); + assert.equal(redraws.viewLimit, 2, 'residency moved, the camera moves: the drop holds'); + redraws.rest(); + assert.equal(redraws.viewLimit, 24, 'residency moved, the camera rests: the drop is forgotten'); flag.value = WORK_DROPPED; for (let i = 0; i < 6; i++) await frame([1, 2], true, [0, 1]); assert.equal(redraws.viewLimit, 1, 'drops floor the limit at one view'); }); +// Under a moving camera residency changes every frame. A batch whose lists hold two views' casters +// drops past them; the limit settles there and stays, and no batch drops again, until the camera +// rests and the views may grow back (#525). +test('while the camera moves and residency changes, a batch that dropped work does not drop again', async () => { + const flag = { value: 0 }; + const { redraws, frame } = redrawsWith(flag); + const drops: number[] = []; + for (let at = 0; at < 8; at++) { + const views = Array.from({ length: Math.min(redraws.viewLimit, 8) }, (_, view) => view); + flag.value = views.length > 2 ? WORK_DROPPED : 0; + if ((await frame(views, true, views)).length) drops.push(at); + redraws.residencyChanged(); + } + assert.deepEqual( + drops.filter((at) => at >= 4), + [], + 'the last four frames draw whole', + ); + assert.equal(redraws.viewLimit, 2, 'two views, what the lists hold'); + redraws.rest(); + assert.equal(redraws.viewLimit, 24, 'at rest the views may grow back'); +}); + // A view drew a placement coarser than it wanted: every page it drew waits for residency to move // and the camera to rest, and is then drawn again — a cluster that never comes costs nothing, and a // camera that only moves redraws none of them. diff --git a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts index 396855a73e..3c430b229f 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts @@ -141,17 +141,22 @@ export function createLightCutRedraws( reported(copied: boolean) { if (open) open.reported = copied; }, - /** Residency the light cuts see changed: the drop is forgotten, and the pages that waited on - * it are drawn again once the camera rests (`rest`). */ + /** Residency the light cuts see changed: the pages that waited on it are drawn again, and the + * drop forgotten, once the camera rests (`rest`). */ residencyChanged() { moved = true; - limit.residencyChanged(); }, - /** The camera rests: what residency changed meanwhile is drawn again. */ + /** + * The camera rests: what residency changed meanwhile is drawn again, and the views a batch + * draws in may grow back. Never before: under a moving camera residency changes every frame, + * and a limit reset each time made the batch that dropped drop again, its pages drawn short, + * withdrawn and drawn again frame after frame. + */ rest() { if (!moved) return; moved = false; epoch++; + limit.residencyChanged(); for (const page of waiting) again(page, false); waiting.clear(); }, diff --git a/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts b/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts index e3bd7293cc..4444a36612 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts @@ -3,7 +3,7 @@ * fewest a batch dropped work with: a drop at `L` views never swings the limit between `L` and * `L / 2`, it settles on the largest count that fits, never below one. What dropped depends on the * clusters the views kept from the resident catalogue: a residency change forgets the drop, never - * what fitted. + * what fitted — once the camera rests (`lightCutRedraws.ts`). */ export function createViewLimit(viewCap: number) { let fits = 0, From f1cf1fe5496d40e74125b0327a861ebc81d2d984 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 00:49:28 +0200 Subject: [PATCH 03/13] refactor(shadows): a light cut keeps each page's best request of the frame, the harness a fixture (#525) --- packages/sdk-browser/src/gpu/dag/encode.ts | 13 ++ packages/sdk-browser/src/gpu/dag/lightCut.ts | 5 +- .../src/gpu/dag/lightCutAsks.test.ts | 46 ++++++ .../src/gpu/dag/lightCutFrame.fixture.ts | 124 ++++++++++++++++ .../src/gpu/dag/lightCutFrame.test.ts | 134 +----------------- .../src/gpu/dag/lightCutRedraws.ts | 2 +- .../src/gpu/dag/lightCutViewLimit.ts | 6 +- packages/sdk-browser/src/gpu/dag/pipeline.ts | 4 +- .../src/gpu/dag/shader/floorWgsl.ts | 18 +-- .../src/gpu/dag/shader/snapshotWgsl.ts | 22 +-- .../src/gpu/dag/shader/viewsWgsl.ts | 2 + 11 files changed, 219 insertions(+), 157 deletions(-) create mode 100644 packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts create mode 100644 packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts diff --git a/packages/sdk-browser/src/gpu/dag/encode.ts b/packages/sdk-browser/src/gpu/dag/encode.ts index f882a9b243..cacbdf19f7 100644 --- a/packages/sdk-browser/src/gpu/dag/encode.ts +++ b/packages/sdk-browser/src/gpu/dag/encode.ts @@ -145,3 +145,16 @@ function encodeOnce( } live.end(); } + +/** + * Once a frame's light cuts are all encoded: each page their list names takes the best request its + * views made of it (`dagAskedBest`, `shader/snapshotWgsl.ts`), before the list is copied. `pages` + * is the catalogue, which bounds the list. + */ +export function encodeAskedBest(encoder: GPUCommandEncoder, view: DagView, pages: number) { + const pass = encoder.beginComputePass({ label: LIGHT_CUT_PASS }); + pass.setBindGroup(0, view.bindGroup); + pass.setPipeline(view.askedBestPipeline); + pass.dispatchWorkgroups(Math.max(1, Math.ceil(pages / WORKGROUP))); + pass.end(); +} diff --git a/packages/sdk-browser/src/gpu/dag/lightCut.ts b/packages/sdk-browser/src/gpu/dag/lightCut.ts index 62b77ec8b7..b3e76517f4 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCut.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCut.ts @@ -2,7 +2,7 @@ import { FRAME_VEC4, type DagViewUniforms, type DrawnLog } from './types.ts'; import { writeDagUniforms, type DagCutViews } from './uniforms.ts'; import { createLightCutReports } from './lightCutReports.ts'; import { createLightCutRedraws } from './lightCutRedraws.ts'; -import { encodeDagKernels, type DagView } from './encode.ts'; +import { encodeAskedBest, encodeDagKernels, type DagView } from './encode.ts'; import { dagWorkLayout } from './shader/floorWgsl.ts'; import { LEVEL_QUEUES } from './shader/levelWgsl.ts'; import { lightCutCapacity, lightQueueCap } from './lightCutCapacity.ts'; @@ -142,7 +142,7 @@ export function createDagLightCut(resources: DagResources) { ) { if (count > capacity) throw new Error(`${count} light views, at most ${capacity}`); // The frame's first cut starts its list, and forgets what the last frame asked for. - if (!listed) encoder.clearBuffer(work, layout.asked * 4, layout.askedWords * 4); + if (!listed) encoder.clearBuffer(work, layout.askedAt * 4, layout.askedWords * 4); cutViews.count = light.views = count; cutViews.append = listed; listed = true; @@ -163,6 +163,7 @@ export function createDagLightCut(resources: DagResources) { encodeReports(encoder: GPUCommandEncoder) { if (!listed) return undefined; listed = false; + encodeAskedBest(encoder, view, pageCount); const settle = reports.encodeReadback(encoder); redraws.reported(settle !== undefined); return settle; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts new file mode 100644 index 0000000000..c671dc2069 --- /dev/null +++ b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts @@ -0,0 +1,46 @@ +// A light cut's frame lists each caster once, however many views and batches want it, at the best +// request any of them made (#525): the list, as long as the catalogue, never fills with repeats. +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { lightCutFrame } from './lightCutFrame.fixture.ts'; +import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; + +test('a light cut lists a caster once a frame, at its best request: the contract restated here', () => { + for (const line of [ + 'if(isLightCut()&&atomicMax(&work[askedWord(page)],packRequest(page,priority))!=0u){return;}', + 'let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;}', + 'out.pages[s]=atomicLoad(&work[askedWord(out.pages[s]&((1u< { + const { cut, frame } = lightCutFrame(); + const pages = [4, 5, 6, 7, 8, 9, 10, 11]; + for (const run of ['a frame', 'the next frame']) { + await frame(pages, new Set(), (page) => [0, 1, 2, 3, page].map((caster) => [caster, 1])); + const asked = cut.reports.takeRequests()?.sort((a, b) => a - b); + assert.deepEqual(asked, [0, 1, 2, 3, ...pages], `${run}: every caster, each once`); + const now: number[] = []; + cut.redraws.takeRedraw((page) => now.push(page)); + assert.deepEqual(now, [], `${run}: no coarse page drawn again at once`); + } +}); + +// Two captures of one pose must ask in the same order (`lightCutReports.ts`): a caster listed by a +// coarse view that a later, finer view needs more is asked at the finer view's priority. +test('a caster several views ask for is asked at the highest priority any of them gives it', async () => { + const { cut, frame } = lightCutFrame(); + await frame([4, 5], new Set(), (page) => + page === 4 + ? [ + [3, 2], + [9, 5], + ] + : [[3, 8]], + ); + assert.deepEqual(cut.reports.takeRequests(), [3, 9], 'caster 3 at 8, above caster 9 at 5'); +}); diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts new file mode 100644 index 0000000000..4ce7714343 --- /dev/null +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts @@ -0,0 +1,124 @@ +// A light cut over a small catalogue, on a device whose copies run as they are encoded. The GPU side +// is the shader's contract, run on the buffers the host wrote: a cut resets the list unless its +// uniform says append, the first view to want a caster in the frame lists it and every view raises +// its best request (`askedWord`), the frame's list then takes those best requests (`dagAskedBest`), +// and a view whose caster is not resident draws coarser. +import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; +import { sunRun } from '../../webgpu/shadow/runs.fixture.ts'; +import { createDagLightCut } from './lightCut.ts'; +import { FRAME_VEC4 } from './types.ts'; +import { OUT_COUNT, OUT_FLAGS, SELECTION_HEADER_WORDS } from './layout.ts'; +import { packRequest, requestPage } from './request.ts'; +import { COARSER_VIEWS, DAG_UNIFORM_BYTES, LIST_FULL } from './shader/viewsWgsl.ts'; +import { VIEW_FLAGS_WORD } from './uniforms.ts'; +import { VIEW_APPEND } from './shader/pagesWgsl.ts'; +import { dagWorkLayout } from './shader/floorWgsl.ts'; + +const CASTERS = 16; +/** The pipeline of `dagAskedBest`: a dispatch under it runs the kernel's contract. */ +const ASKED_BEST = {}; + +/** A light cut over `CASTERS` catalogue pages, on a device whose copies run as they are encoded. */ +export function lightCutFrame() { + const { device, writes, buffers } = fakeDevice({ + limits: { + maxComputeWorkgroupsPerDimension: 65535, + maxStorageBufferBindingSize: 1 << 27, + maxBufferSize: 1 << 28, + }, + }); + const bytes = (buffer: GPUBuffer) => (buffer as unknown as FakeBuffer).getMappedRange(); + const encoder = { + copyBufferToBuffer(from: GPUBuffer, at: number, to: GPUBuffer, toAt: number, size: number) { + new Uint8Array(bytes(to)).set(new Uint8Array(bytes(from), at, size), toAt); + }, + clearBuffer(buffer: GPUBuffer, at: number, size: number) { + new Uint8Array(bytes(buffer), at, size).fill(0); + }, + beginComputePass: () => { + let pipeline: unknown; + return { + setBindGroup() {}, + setPipeline: (next: unknown) => (pipeline = next), + dispatchWorkgroups: () => pipeline === ASKED_BEST && askedBest(), + dispatchWorkgroupsIndirect() {}, + end() {}, + }; + }, + } as unknown as GPUCommandEncoder; + const outputBytes = (SELECTION_HEADER_WORDS + CASTERS) * 4; + const packed = { pageCount: CASTERS, nodeCount: CASTERS, worldCount: 1 }; + const cut = createDagLightCut({ + device, + packed, + residentCut: false, + pageCount: CASTERS, + nodeCount: CASTERS, + worldCount: 1, + blockCount: 1, + levelSizes: [1], + levelPipelines: [{}], + askedBestPipeline: ASKED_BEST, + outputBytes, + readbackBytes: outputBytes, + frameData: new Float32Array(FRAME_VEC4 * 4), + frameWrites: { count: 0 }, + buffers: [], + } as unknown as Parameters[0]); + const output = buffers.find(({ label }) => label === 'Trillion3D light cut output')!; + const work = buffers.find(({ label }) => label === 'Trillion3D light cut work')!; + const { askedAt } = dagWorkLayout(1, cut.capacity, CASTERS); + const out = () => new Uint32Array(output.getMappedRange()), + best = () => new Uint32Array(work.getMappedRange()); + /** `dagAskedBest`: each listed page takes its best request of the frame. */ + const askedBest = () => { + const list = out(); + for (let s = 0; s < Math.min(list[OUT_COUNT], CASTERS); s++) { + const at = SELECTION_HEADER_WORDS + s; + list[at] = best()[askedAt + requestPage(list[at])]; + } + }; + /** The GPU running the cut just encoded, over a view that wants each `[caster, priority]` of + * `asks`, against `resident`. */ + const run = (asks: number[][], resident: Set) => { + const uniform = writes.findLast(({ buffer }) => buffer.size === DAG_UNIFORM_BYTES)!; + const viewFlags = new Uint32Array(uniform.data.slice().buffer)[VIEW_FLAGS_WORD]; + const list = out(), + words = best(); + if (viewFlags & VIEW_APPEND) list[OUT_FLAGS] &= LIST_FULL; + else list[OUT_COUNT] = list[OUT_FLAGS] = 0; + for (const [caster, priority] of asks) { + if (!resident.has(caster)) list[OUT_FLAGS] |= 1 << COARSER_VIEWS; + const word = packRequest(caster, priority), + before = words[askedAt + caster]; + words[askedAt + caster] = Math.max(before, word); + if (before) continue; + const slot = list[OUT_COUNT]++; + if (slot < CASTERS) list[SELECTION_HEADER_WORDS + slot] = word; + else list[OUT_FLAGS] |= LIST_FULL; + } + return viewFlags; + }; + const views = [{ uniforms: sunRun(64).uniforms }]; + /** One frame drawing `pages` in one batch each, whose view asks `wants(page)`: the page itself + * by default. Returns the batches' view flags and how many report copies ran. */ + const frame = async ( + pages: number[], + resident: Set, + wants = (page: number) => [[page, 1]], + ) => { + const settles: Array<(submitted: boolean) => void> = [], + flags: number[] = []; + for (const page of pages) { + cut.encode(encoder, views, 1); + flags.push(run(wants(page), resident)); + const settle = cut.redraws.encode(encoder, [page], [0], 1); + if (settle) settles.push(settle); + } + const report = cut.encodeReports(encoder); + for (const settle of [report, ...settles]) settle?.(true); + await cut.settled(); + return { flags, copies: report ? 1 : 0 }; + }; + return { cut, frame }; +} diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts index 65474f0542..6996c9d046 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts @@ -1,115 +1,11 @@ // A frame draws its shadow pages in as many batches as they take (#489), each batch a light cut. // Every batch's requests must be read back, whatever the batch count: the cuts append to one list // the frame copies once (`VIEW_APPEND`), so no batch's coarse view is redrawn for want of a report -// slot, and its missing casters are asked for. The GPU side here is the shader's contract, run on -// the buffers the host wrote: a cut resets the list unless its uniform says append, a view asks for -// each caster it wants the first time the frame's cuts do (`firstAsk`), and a view whose caster is -// not resident draws coarser. +// slot, and its missing casters are asked for. import test from 'node:test'; import assert from 'node:assert/strict'; -import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; -import { sunRun } from '../../webgpu/shadow/runs.fixture.ts'; -import { createDagLightCut } from './lightCut.ts'; -import { FRAME_VEC4 } from './types.ts'; -import { OUT_COUNT, OUT_FLAGS, SELECTION_HEADER_WORDS } from './layout.ts'; -import { packRequest } from './request.ts'; -import { COARSER_VIEWS, DAG_UNIFORM_BYTES, LIST_FULL } from './shader/viewsWgsl.ts'; -import { VIEW_FLAGS_WORD } from './uniforms.ts'; +import { lightCutFrame } from './lightCutFrame.fixture.ts'; import { VIEW_APPEND } from './shader/pagesWgsl.ts'; -import { dagWorkLayout } from './shader/floorWgsl.ts'; -import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; - -const CASTERS = 16; - -/** A light cut over `CASTERS` catalogue pages, on a device whose copies run as they are encoded. */ -function lightCutFrame() { - const { device, writes, buffers } = fakeDevice({ - limits: { - maxComputeWorkgroupsPerDimension: 65535, - maxStorageBufferBindingSize: 1 << 27, - maxBufferSize: 1 << 28, - }, - }); - const bytes = (buffer: GPUBuffer) => (buffer as unknown as FakeBuffer).getMappedRange(); - const encoder = { - copyBufferToBuffer(from: GPUBuffer, at: number, to: GPUBuffer, toAt: number, size: number) { - new Uint8Array(bytes(to)).set(new Uint8Array(bytes(from), at, size), toAt); - }, - clearBuffer(buffer: GPUBuffer, at: number, size: number) { - new Uint8Array(bytes(buffer), at, size).fill(0); - }, - beginComputePass: () => ({ - setBindGroup() {}, - setPipeline() {}, - dispatchWorkgroups() {}, - dispatchWorkgroupsIndirect() {}, - end() {}, - }), - } as unknown as GPUCommandEncoder; - const outputBytes = (SELECTION_HEADER_WORDS + CASTERS) * 4; - const packed = { pageCount: CASTERS, nodeCount: CASTERS, worldCount: 1 }; - const cut = createDagLightCut({ - device, - packed, - residentCut: false, - pageCount: CASTERS, - nodeCount: CASTERS, - worldCount: 1, - blockCount: 1, - levelSizes: [1], - levelPipelines: [{}], - outputBytes, - readbackBytes: outputBytes, - frameData: new Float32Array(FRAME_VEC4 * 4), - frameWrites: { count: 0 }, - buffers: [], - } as unknown as Parameters[0]); - const output = buffers.find(({ label }) => label === 'Trillion3D light cut output')!; - const work = buffers.find(({ label }) => label === 'Trillion3D light cut work')!; - const { asked } = dagWorkLayout(1, cut.capacity, CASTERS); - /** The GPU running the cut just encoded, over a view that wants `casters`, against `resident`. */ - const run = (casters: number[], resident: Set) => { - const uniform = writes.findLast(({ buffer }) => buffer.size === DAG_UNIFORM_BYTES)!; - const viewFlags = new Uint32Array(uniform.data.slice().buffer)[VIEW_FLAGS_WORD]; - const out = new Uint32Array(output.getMappedRange()), - bits = new Uint32Array(work.getMappedRange()); - if (viewFlags & VIEW_APPEND) out[OUT_FLAGS] &= LIST_FULL; - else out[OUT_COUNT] = out[OUT_FLAGS] = 0; - for (const caster of casters) { - if (!resident.has(caster)) out[OUT_FLAGS] |= 1 << COARSER_VIEWS; - const word = asked + (caster >> 5), - bit = 1 << (caster & 31); - if (bits[word] & bit) continue; - bits[word] |= bit; - const slot = out[OUT_COUNT]++; - if (slot < CASTERS) out[SELECTION_HEADER_WORDS + slot] = packRequest(caster, 1); - else out[OUT_FLAGS] |= LIST_FULL; - } - return viewFlags; - }; - const views = [{ uniforms: sunRun(64).uniforms }]; - /** One frame drawing `pages` in one batch each, whose view wants `wants(page)`: the page itself - * by default. Returns the batches' view flags and how many report copies ran. */ - const frame = async ( - pages: number[], - resident: Set, - wants = (page: number) => [page], - ) => { - const settles: Array<(submitted: boolean) => void> = [], - flags: number[] = []; - for (const page of pages) { - cut.encode(encoder, views, 1); - flags.push(run(wants(page), resident)); - const settle = cut.redraws.encode(encoder, [page], [0], 1); - if (settle) settles.push(settle); - } - const report = cut.encodeReports(encoder); - for (const settle of [report, ...settles]) settle?.(true); - await cut.settled(); - return { flags, copies: report ? 1 : 0 }; - }; - return { cut, frame }; -} test("a frame of many batches reads back every batch's requests in one copy", async () => { const { cut, frame } = lightCutFrame(); @@ -148,29 +44,3 @@ test('a sweep of many batches a frame converges to full detail', async () => { assert.equal(resident.size, 12, 'every caster was asked for and arrived'); assert.deepEqual(drawn, [], 'every page is drawn at full detail, nothing left to redraw'); }); - -test('a light cut asks for a caster once a frame, its bit read before the atomic', () => { - for (const line of [ - 'if(isLightCut()&&!firstAsk(page)){return;}', - 'let word=drawnGroupsMax()+1u+(page>>5u);let bit=1u<<(page&31u);', - 'if((atomicLoad(&work[word])&bit)!=0u){return false;}', - 'return (atomicOr(&work[word],bit)&bit)==0u;', - ]) - assert.ok(DAG_RELEVE_WGSL.includes(line), line); -}); - -// Every sun level of every batch wants the casters that span the scene. Asked once a frame, they -// leave the list, as long as the catalogue, room for each batch's own casters: asked for, and never -// drawn again at once for want of a place in it (#525). -test("a caster every batch wants is asked for once a frame, and each batch's own casters too", async () => { - const { cut, frame } = lightCutFrame(); - const pages = [4, 5, 6, 7, 8, 9, 10, 11]; - for (const run of ['a frame', 'the next frame']) { - await frame(pages, new Set(), (page) => [0, 1, 2, 3, page]); - const asked = cut.reports.takeRequests()?.sort((a, b) => a - b); - assert.deepEqual(asked, [0, 1, 2, 3, ...pages], `${run}: every caster, each once`); - const now: number[] = []; - cut.redraws.takeRedraw((page) => now.push(page)); - assert.deepEqual(now, [], `${run}: no coarse page drawn again at once`); - } -}); diff --git a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts index 3c430b229f..91981da53b 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts @@ -156,7 +156,7 @@ export function createLightCutRedraws( if (!moved) return; moved = false; epoch++; - limit.residencyChanged(); + limit.forgetDrop(); for (const page of waiting) again(page, false); waiting.clear(); }, diff --git a/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts b/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts index 4444a36612..d3c3266674 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts @@ -2,8 +2,8 @@ * The light views one batch draws in, bisected between the most views a batch drew whole and the * fewest a batch dropped work with: a drop at `L` views never swings the limit between `L` and * `L / 2`, it settles on the largest count that fits, never below one. What dropped depends on the - * clusters the views kept from the resident catalogue: a residency change forgets the drop, never - * what fitted — once the camera rests (`lightCutRedraws.ts`). + * clusters the views kept from the resident catalogue: `forgetDrop` lets the limit grow back, and + * keeps what fitted. */ export function createViewLimit(viewCap: number) { let fits = 0, @@ -18,7 +18,7 @@ export function createViewLimit(viewCap: number) { if (fits >= drops) drops = viewCap + 1; } }, - residencyChanged() { + forgetDrop() { drops = viewCap + 1; }, get value() { diff --git a/packages/sdk-browser/src/gpu/dag/pipeline.ts b/packages/sdk-browser/src/gpu/dag/pipeline.ts index 173bbf5dd5..39adf58513 100644 --- a/packages/sdk-browser/src/gpu/dag/pipeline.ts +++ b/packages/sdk-browser/src/gpu/dag/pipeline.ts @@ -41,7 +41,8 @@ export function createDagPipeline(device: GPUDevice, buffers: DagBuffers) { maskPipeline = stage('dagMask'); const drawPrefixPipeline = stage('dagDrawPrefix'), drawScatterPipeline = stage('dagDrawScatter'), - viewOffsetsPipeline = stage('dagViewOffsets'); + viewOffsetsPipeline = stage('dagViewOffsets'), + askedBestPipeline = stage('dagAskedBest'); const bindGroup = device.createBindGroup({ layout, entries: namedBufferEntries(DAG_BINDING, { @@ -68,6 +69,7 @@ export function createDagPipeline(device: GPUDevice, buffers: DagBuffers) { drawPrefixPipeline, drawScatterPipeline, viewOffsetsPipeline, + askedBestPipeline, bindGroup, }; }); diff --git a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts index b8a4009ddd..ca069edcb9 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts @@ -23,13 +23,14 @@ * top-down pruning would then drop everything. * * `views` is the view capacity the buffer serves: one for a camera, one row each for a light cut. - * `asked` is the catalogue a light cut asks for: one bit per page behind the rest (`firstAsk`, - * `snapshotWgsl.ts`); a camera, which asks for a page once, has none. + * `pages` is the catalogue a light cut asks for: one word per page behind the rest, its best request + * of the frame (`askedWord`, `snapshotWgsl.ts`); a camera, which asks for a page once, has none. */ -export function dagWorkLayout(blockCount: number, views = 1, asked = 0) { +export function dagWorkLayout(blockCount: number, views = 1, pages = 0) { const base = blockCount * 2, viewWords = base + 9, - drawnGroupsMax = viewWords + VIEW_WORD_ROWS * views; + drawnGroupsMax = viewWords + VIEW_WORD_ROWS * views, + askedAt = drawnGroupsMax + 1; return { base, /** The nine frame counters, in the order `levelWgsl.ts` names them. */ @@ -43,10 +44,11 @@ export function dagWorkLayout(blockCount: number, views = 1, asked = 0) { viewWords, /** The most sixty-four-wide groups any view drew, behind the per-view rows. */ drawnGroupsMax, - /** The frame's asked bits, behind it: `askedWords` words, cleared at the frame's first cut. */ - asked: drawnGroupsMax + 1, - askedWords: Math.ceil(asked / 32), - words: drawnGroupsMax + 1 + Math.ceil(asked / 32), + /** The frame's best request of each page, behind it: `askedWords` words, cleared at the + * frame's first cut. */ + askedAt, + askedWords: pages, + words: askedAt + pages, }; } diff --git a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts index 17a26e5ee5..daff1461bf 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts @@ -14,22 +14,24 @@ * rather than adopt it amputated. Frame totals lose nothing — they describe the cut, not the * list that reports it (`totalsWgsl.ts`). * - * A light cut's frame asks for a page ONCE, however many of its views and batches want it: every + * A light cut's frame lists a page ONCE, however many of its views and batches want it: every * batch appends to one list (`VIEW_APPEND`) as long as the catalogue, and the same caster asked by * each sun level and each batch filled it with repeats, so a late batch's own casters fell past it * (`LIST_FULL`) and its coarse pages were drawn again every frame without ever being asked for. - * The test reads the page's bit before the atomic (`firstAsk`), as the shading's page requests do. + * Each page keeps its best request of the frame (`askedWord`): the first view to raise it from zero + * lists the page, and once the frame's cuts are done `dagAskedBest` writes that best word over its + * entry — the highest priority any view gave it, whatever view won the race. */ export const DAG_RELEVE_WGSL = `fn emitOne(page:u32,pixels:f32){ - if(isLightCut()&&!firstAsk(page)){return;} - emitWord(page,quantizePriority(pixels),true); + let priority=quantizePriority(pixels); + if(isLightCut()&&atomicMax(&work[askedWord(page)],packRequest(page,priority))!=0u){return;} + emitWord(page,priority,true); } -/** True for the first ask of \`page\` in the frame's light cuts: its bit behind the per-view words - * (\`dagWorkLayout\`, \`asked\`), cleared by the host at the frame's first cut. */ -fn firstAsk(page:u32)->bool{ - let word=drawnGroupsMax()+1u+(page>>5u);let bit=1u<<(page&31u); - if((atomicLoad(&work[word])&bit)!=0u){return false;} - return (atomicOr(&work[word],bit)&bit)==0u; +/** After a frame's last light cut: each listed page at the best request its views made of it. */ +@compute @workgroup_size(64) +fn dagAskedBest(@builtin(global_invocation_id) id:vec3u){ + let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;} + out.pages[s]=atomicLoad(&work[askedWord(out.pages[s]&((1u<u32{return vi*views[0u].worldCount+w;} fn viewWord(row:u32,v:u32)->u32{return extraBase()+row*views[0u].viewCapacity+v;} /** The word behind the per-view rows: the most sixty-four-wide groups any view drew. */ fn drawnGroupsMax()->u32{return viewWord(${VIEW_WORD_ROWS}u,0u);} +/** A light cut's best request of \`page\` this frame, behind it (\`dagWorkLayout\`, \`askedAt\`). */ +fn askedWord(page:u32)->u32{return drawnGroupsMax()+1u+page;} fn dropWork(){atomicOr(&out.overflow,${WORK_DROPPED}u);} fn noteCoarser(){if(isLightCut()){atomicOr(&out.overflow,1u<<(${COARSER_VIEWS}u+vi));}} fn isLightCut()->bool{return (views[0u].viewFlags&VIEW_LIGHT)!=0u;} From 1732759c0de3e846bb53200dfef535aa8a10a048 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 00:53:37 +0200 Subject: [PATCH 04/13] fix(shadows): page zero at priority zero is listed once too; the best-request pass spans the list's cap (#525) --- packages/sdk-browser/src/gpu/dag/encode.ts | 8 ++++---- packages/sdk-browser/src/gpu/dag/lightCut.ts | 3 ++- .../sdk-browser/src/gpu/dag/lightCutAsks.test.ts | 12 ++++++++++-- .../sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts | 12 ++++++------ packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts | 6 +++--- .../sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts | 12 +++++++----- packages/sdk-browser/src/gpu/dag/shader/viewsWgsl.ts | 3 ++- 7 files changed, 34 insertions(+), 22 deletions(-) diff --git a/packages/sdk-browser/src/gpu/dag/encode.ts b/packages/sdk-browser/src/gpu/dag/encode.ts index cacbdf19f7..5665e810ba 100644 --- a/packages/sdk-browser/src/gpu/dag/encode.ts +++ b/packages/sdk-browser/src/gpu/dag/encode.ts @@ -148,13 +148,13 @@ function encodeOnce( /** * Once a frame's light cuts are all encoded: each page their list names takes the best request its - * views made of it (`dagAskedBest`, `shader/snapshotWgsl.ts`), before the list is copied. `pages` - * is the catalogue, which bounds the list. + * views made of it (`dagAskedBest`, `shader/snapshotWgsl.ts`), before the list is copied. `entries` + * is the list's cap: one thread per entry it can hold. */ -export function encodeAskedBest(encoder: GPUCommandEncoder, view: DagView, pages: number) { +export function encodeAskedBest(encoder: GPUCommandEncoder, view: DagView, entries: number) { const pass = encoder.beginComputePass({ label: LIGHT_CUT_PASS }); pass.setBindGroup(0, view.bindGroup); pass.setPipeline(view.askedBestPipeline); - pass.dispatchWorkgroups(Math.max(1, Math.ceil(pages / WORKGROUP))); + pass.dispatchWorkgroups(Math.max(1, Math.ceil(entries / WORKGROUP))); pass.end(); } diff --git a/packages/sdk-browser/src/gpu/dag/lightCut.ts b/packages/sdk-browser/src/gpu/dag/lightCut.ts index b3e76517f4..fb1cd2f00e 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCut.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCut.ts @@ -4,6 +4,7 @@ import { createLightCutReports } from './lightCutReports.ts'; import { createLightCutRedraws } from './lightCutRedraws.ts'; import { encodeAskedBest, encodeDagKernels, type DagView } from './encode.ts'; import { dagWorkLayout } from './shader/floorWgsl.ts'; +import { selectionListCap } from './layout.ts'; import { LEVEL_QUEUES } from './shader/levelWgsl.ts'; import { lightCutCapacity, lightQueueCap } from './lightCutCapacity.ts'; import { DAG_UNIFORM_BYTES, DAG_VIEW_WORDS } from './shader/viewsWgsl.ts'; @@ -163,7 +164,7 @@ export function createDagLightCut(resources: DagResources) { encodeReports(encoder: GPUCommandEncoder) { if (!listed) return undefined; listed = false; - encodeAskedBest(encoder, view, pageCount); + encodeAskedBest(encoder, view, selectionListCap(pageCount)); const settle = reports.encodeReadback(encoder); redraws.reported(settle !== undefined); return settle; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts index c671dc2069..9dc9a4de64 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts @@ -7,9 +7,9 @@ import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; test('a light cut lists a caster once a frame, at its best request: the contract restated here', () => { for (const line of [ - 'if(isLightCut()&&atomicMax(&work[askedWord(page)],packRequest(page,priority))!=0u){return;}', + 'if(isLightCut()&&atomicMax(&work[askedWord(page)],priority+1u)!=0u){return;}', 'let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;}', - 'out.pages[s]=atomicLoad(&work[askedWord(out.pages[s]&((1u< { + const { cut, frame } = lightCutFrame(); + await frame([4, 5, 6], new Set(), () => [[0, 0]]); + assert.deepEqual(cut.reports.takeRequests(), [0], 'page zero, once'); +}); diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts index 4ce7714343..72627cb6ac 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts @@ -1,7 +1,7 @@ // A light cut over a small catalogue, on a device whose copies run as they are encoded. The GPU side // is the shader's contract, run on the buffers the host wrote: a cut resets the list unless its // uniform says append, the first view to want a caster in the frame lists it and every view raises -// its best request (`askedWord`), the frame's list then takes those best requests (`dagAskedBest`), +// its best priority (`askedWord`), the frame's list then takes those best requests (`dagAskedBest`), // and a view whose caster is not resident draws coarser. import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; import { sunRun } from '../../webgpu/shadow/runs.fixture.ts'; @@ -75,7 +75,8 @@ export function lightCutFrame() { const list = out(); for (let s = 0; s < Math.min(list[OUT_COUNT], CASTERS); s++) { const at = SELECTION_HEADER_WORDS + s; - list[at] = best()[askedAt + requestPage(list[at])]; + const page = requestPage(list[at]); + list[at] = packRequest(page, best()[askedAt + page] - 1); } }; /** The GPU running the cut just encoded, over a view that wants each `[caster, priority]` of @@ -89,12 +90,11 @@ export function lightCutFrame() { else list[OUT_COUNT] = list[OUT_FLAGS] = 0; for (const [caster, priority] of asks) { if (!resident.has(caster)) list[OUT_FLAGS] |= 1 << COARSER_VIEWS; - const word = packRequest(caster, priority), - before = words[askedAt + caster]; - words[askedAt + caster] = Math.max(before, word); + const before = words[askedAt + caster]; + words[askedAt + caster] = Math.max(before, priority + 1); if (before) continue; const slot = list[OUT_COUNT]++; - if (slot < CASTERS) list[SELECTION_HEADER_WORDS + slot] = word; + if (slot < CASTERS) list[SELECTION_HEADER_WORDS + slot] = packRequest(caster, priority); else list[OUT_FLAGS] |= LIST_FULL; } return viewFlags; diff --git a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts index ca069edcb9..f18404bd14 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts @@ -23,8 +23,8 @@ * top-down pruning would then drop everything. * * `views` is the view capacity the buffer serves: one for a camera, one row each for a light cut. - * `pages` is the catalogue a light cut asks for: one word per page behind the rest, its best request - * of the frame (`askedWord`, `snapshotWgsl.ts`); a camera, which asks for a page once, has none. + * `pages` is the catalogue a light cut asks for: one word per page behind the rest, its best priority + * of the frame plus one (`askedWord`, `snapshotWgsl.ts`); a camera, which asks for a page once, has none. */ export function dagWorkLayout(blockCount: number, views = 1, pages = 0) { const base = blockCount * 2, @@ -44,7 +44,7 @@ export function dagWorkLayout(blockCount: number, views = 1, pages = 0) { viewWords, /** The most sixty-four-wide groups any view drew, behind the per-view rows. */ drawnGroupsMax, - /** The frame's best request of each page, behind it: `askedWords` words, cleared at the + /** The frame's best priority of each page plus one, behind it: `askedWords` words, cleared at the * frame's first cut. */ askedAt, askedWords: pages, diff --git a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts index daff1461bf..757c0fa459 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts @@ -18,20 +18,22 @@ * batch appends to one list (`VIEW_APPEND`) as long as the catalogue, and the same caster asked by * each sun level and each batch filled it with repeats, so a late batch's own casters fell past it * (`LIST_FULL`) and its coarse pages were drawn again every frame without ever being asked for. - * Each page keeps its best request of the frame (`askedWord`): the first view to raise it from zero - * lists the page, and once the frame's cuts are done `dagAskedBest` writes that best word over its - * entry — the highest priority any view gave it, whatever view won the race. + * Each page keeps its best priority of the frame, plus one (`askedWord`): zero is "not asked yet", + * even for page zero at priority zero, whose request word is zero. The first view to raise it from + * zero lists the page, and once the frame's cuts are done `dagAskedBest` writes the request at that + * best priority over its entry — the highest any view gave it, whatever view won the race. */ export const DAG_RELEVE_WGSL = `fn emitOne(page:u32,pixels:f32){ let priority=quantizePriority(pixels); - if(isLightCut()&&atomicMax(&work[askedWord(page)],packRequest(page,priority))!=0u){return;} + if(isLightCut()&&atomicMax(&work[askedWord(page)],priority+1u)!=0u){return;} emitWord(page,priority,true); } /** After a frame's last light cut: each listed page at the best request its views made of it. */ @compute @workgroup_size(64) fn dagAskedBest(@builtin(global_invocation_id) id:vec3u){ let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;} - out.pages[s]=atomicLoad(&work[askedWord(out.pages[s]&((1u<u32{return vi*views[0u].worldCount+w;} fn viewWord(row:u32,v:u32)->u32{return extraBase()+row*views[0u].viewCapacity+v;} /** The word behind the per-view rows: the most sixty-four-wide groups any view drew. */ fn drawnGroupsMax()->u32{return viewWord(${VIEW_WORD_ROWS}u,0u);} -/** A light cut's best request of \`page\` this frame, behind it (\`dagWorkLayout\`, \`askedAt\`). */ +/** A light cut's best priority of \`page\` this frame plus one, zero if unasked, behind it + * (\`dagWorkLayout\`, \`askedAt\`). */ fn askedWord(page:u32)->u32{return drawnGroupsMax()+1u+page;} fn dropWork(){atomicOr(&out.overflow,${WORK_DROPPED}u);} fn noteCoarser(){if(isLightCut()){atomicOr(&out.overflow,1u<<(${COARSER_VIEWS}u+vi));}} From 3e60674241f51f4f5e584dfbc2226d0d28a64371 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 01:11:00 +0200 Subject: [PATCH 05/13] refactor(shadows): the best-request pass reads the page with the WGSL requestPage, and its word offset is pinned to the layout (#525) --- packages/sdk-browser/src/gpu/dag/encode.ts | 6 ++++-- packages/sdk-browser/src/gpu/dag/lightCut.ts | 2 +- packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts | 9 +++++++++ .../sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts | 5 +++-- packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts | 3 +-- packages/sdk-browser/src/gpu/dag/request.ts | 1 + packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts | 5 ++--- packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts | 2 +- 8 files changed, 22 insertions(+), 11 deletions(-) diff --git a/packages/sdk-browser/src/gpu/dag/encode.ts b/packages/sdk-browser/src/gpu/dag/encode.ts index 5665e810ba..047fb6dcc9 100644 --- a/packages/sdk-browser/src/gpu/dag/encode.ts +++ b/packages/sdk-browser/src/gpu/dag/encode.ts @@ -12,6 +12,9 @@ export type DagView = NonNullable> light?: { views: number; queueCap: number }; }; +/** Workgroups for `count` threads, never none. */ +const groups = (count: number) => Math.max(1, Math.ceil(count / WORKGROUP)); + /** * Cut kernels, encoded in order. Each dispatch waits for the previous — the GPU empties its queue * and caches between two —, and that wait is attributed to no kernel: it is the number of @@ -75,7 +78,6 @@ function encodeOnce( const views = light?.views ?? 1, queueCap = light?.queueCap ?? resources.nodeCount; const label = light ? LIGHT_CUT_PASS : 'Trillion3D DAG selection'; - const groups = (count: number) => Math.max(1, Math.ceil(count / WORKGROUP)); // Head word of the dispatch argument, copied outside a pass: the other two have been one since // the buffer was created. That is the only reason for cuts between passes. const arm = (offset: number) => encoder.copyBufferToBuffer(work, offset, dispatchArgs, 0, 4); @@ -155,6 +157,6 @@ export function encodeAskedBest(encoder: GPUCommandEncoder, view: DagView, entri const pass = encoder.beginComputePass({ label: LIGHT_CUT_PASS }); pass.setBindGroup(0, view.bindGroup); pass.setPipeline(view.askedBestPipeline); - pass.dispatchWorkgroups(Math.max(1, Math.ceil(entries / WORKGROUP))); + pass.dispatchWorkgroups(groups(entries)); pass.end(); } diff --git a/packages/sdk-browser/src/gpu/dag/lightCut.ts b/packages/sdk-browser/src/gpu/dag/lightCut.ts index fb1cd2f00e..fcf49dde50 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCut.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCut.ts @@ -143,7 +143,7 @@ export function createDagLightCut(resources: DagResources) { ) { if (count > capacity) throw new Error(`${count} light views, at most ${capacity}`); // The frame's first cut starts its list, and forgets what the last frame asked for. - if (!listed) encoder.clearBuffer(work, layout.askedAt * 4, layout.askedWords * 4); + if (!listed) encoder.clearBuffer(work, layout.askedAt * 4, pageCount * 4); cutViews.count = light.views = count; cutViews.append = listed; listed = true; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts index 9dc9a4de64..fbbe6c368f 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts @@ -4,6 +4,8 @@ import test from 'node:test'; import assert from 'node:assert/strict'; import { lightCutFrame } from './lightCutFrame.fixture.ts'; import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; +import { DAG_VIEWS_WGSL } from './shader/viewsWgsl.ts'; +import { dagWorkLayout } from './shader/floorWgsl.ts'; test('a light cut lists a caster once a frame, at its best request: the contract restated here', () => { for (const line of [ @@ -12,6 +14,13 @@ test('a light cut lists a caster once a frame, at its best request: the contract 'out.pages[s]=packRequest(page,atomicLoad(&work[askedWord(page)])-1u);', ]) assert.ok(DAG_RELEVE_WGSL.includes(line), line); + // The words the frame's first cut clears are the words the kernel marks: behind `drawnGroupsMax`. + assert.ok( + DAG_VIEWS_WGSL.includes('fn askedWord(page:u32)->u32{return drawnGroupsMax()+1u+page;}'), + ); + const layout = dagWorkLayout(3, 5, 7); + assert.equal(layout.askedAt, layout.drawnGroupsMax + 1); + assert.equal(layout.words, layout.askedAt + 7); }); // Every sun level of every batch wants the casters that span the scene. Asked once a frame, they diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts index 72627cb6ac..6accfb34c1 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts @@ -72,11 +72,12 @@ export function lightCutFrame() { best = () => new Uint32Array(work.getMappedRange()); /** `dagAskedBest`: each listed page takes its best request of the frame. */ const askedBest = () => { - const list = out(); + const list = out(), + words = best(); for (let s = 0; s < Math.min(list[OUT_COUNT], CASTERS); s++) { const at = SELECTION_HEADER_WORDS + s; const page = requestPage(list[at]); - list[at] = packRequest(page, best()[askedAt + page] - 1); + list[at] = packRequest(page, words[askedAt + page] - 1); } }; /** The GPU running the cut just encoded, over a view that wants each `[caster, priority]` of diff --git a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts index 91981da53b..cbbbdb08ed 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts @@ -149,8 +149,7 @@ export function createLightCutRedraws( /** * The camera rests: what residency changed meanwhile is drawn again, and the views a batch * draws in may grow back. Never before: under a moving camera residency changes every frame, - * and a limit reset each time made the batch that dropped drop again, its pages drawn short, - * withdrawn and drawn again frame after frame. + * and the batch that dropped would drop again each time. */ rest() { if (!moved) return; diff --git a/packages/sdk-browser/src/gpu/dag/request.ts b/packages/sdk-browser/src/gpu/dag/request.ts index cdcbd0cbf1..a5d5ae4f60 100644 --- a/packages/sdk-browser/src/gpu/dag/request.ts +++ b/packages/sdk-browser/src/gpu/dag/request.ts @@ -66,4 +66,5 @@ fn quantizePriority(pixels:f32)->u32{ return u32(clamp(pas,0,${REQUEST_STEP_MAX})); } fn packRequest(page:u32,priority:u32)->u32{return (priority<u32{return word&((1u<=min(atomicLoad(&out.count),views[0u].listCap)){return;} - let page=out.pages[s]&((1u< Date: Sat, 26 Sep 2026 01:23:40 +0200 Subject: [PATCH 06/13] fix(shadows): a light cut's asked words carry the frame's stamp, cleared only when it wraps (#525) --- .../sdk-browser/src/gpu/dag/askedStamp.ts | 29 +++++++++++ packages/sdk-browser/src/gpu/dag/lightCut.ts | 13 ++++- .../src/gpu/dag/lightCutAsks.test.ts | 48 ++++++++++++++++--- .../src/gpu/dag/lightCutFrame.fixture.ts | 29 +++++++---- .../src/gpu/dag/shader/floorWgsl.ts | 10 ++-- .../src/gpu/dag/shader/snapshotWgsl.ts | 22 ++++++--- .../src/gpu/dag/shader/viewsWgsl.ts | 7 +-- 7 files changed, 124 insertions(+), 34 deletions(-) create mode 100644 packages/sdk-browser/src/gpu/dag/askedStamp.ts diff --git a/packages/sdk-browser/src/gpu/dag/askedStamp.ts b/packages/sdk-browser/src/gpu/dag/askedStamp.ts new file mode 100644 index 0000000000..1bd8a91a81 --- /dev/null +++ b/packages/sdk-browser/src/gpu/dag/askedStamp.ts @@ -0,0 +1,29 @@ +import { REQUEST_PRIORITY_MAX } from './request.ts'; + +/** + * THE FRAME STAMP OF A LIGHT CUT'S ASKED WORDS (`askedWord`, `shader/snapshotWgsl.ts`). A page's word + * holds the frame's stamp in its high bits and its best priority plus one in the low ones, so a + * word left by an earlier frame always loses the `atomicMax` to this frame's first ask, and no + * word needs clearing between frames: a frame costs what its views ask for, never the catalogue. + * + * The stamp counts frames from 1 to `ASKED_STAMP_MAX`. When it wraps back to 1, a word may still + * hold a larger stamp from before: that frame alone clears every word once, one word per catalogue + * page, once every `ASKED_STAMP_MAX` frames — about 9.7 hours at 60 frames a second. + */ +/** Low bits of an asked word: the priority plus one, so that zero is never asked. */ +export const ASKED_PRIORITY_BITS = Math.ceil(Math.log2(REQUEST_PRIORITY_MAX + 2)); +/** The largest stamp: every bit above the priority's. */ +export const ASKED_STAMP_MAX = 2 ** (32 - ASKED_PRIORITY_BITS) - 1; + +/** The stamps of a light cut's frames, and the frame whose words must be cleared first. */ +export function createAskedStamp() { + let stamp = 0; + return { + /** The next frame's stamp, and whether its words are cleared before it asks. */ + next() { + const wraps = stamp >= ASKED_STAMP_MAX; + stamp = wraps ? 1 : stamp + 1; + return { stamp, clear: wraps }; + }, + }; +} diff --git a/packages/sdk-browser/src/gpu/dag/lightCut.ts b/packages/sdk-browser/src/gpu/dag/lightCut.ts index fcf49dde50..236bce7fa0 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCut.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCut.ts @@ -5,6 +5,7 @@ import { createLightCutRedraws } from './lightCutRedraws.ts'; import { encodeAskedBest, encodeDagKernels, type DagView } from './encode.ts'; import { dagWorkLayout } from './shader/floorWgsl.ts'; import { selectionListCap } from './layout.ts'; +import { createAskedStamp } from './askedStamp.ts'; import { LEVEL_QUEUES } from './shader/levelWgsl.ts'; import { lightCutCapacity, lightQueueCap } from './lightCutCapacity.ts'; import { DAG_UNIFORM_BYTES, DAG_VIEW_WORDS } from './shader/viewsWgsl.ts'; @@ -124,6 +125,8 @@ export function createDagLightCut(resources: DagResources) { const cutViews: DagCutViews = { count: 0, capacity, queueCap }; /** A cut ran since the requests were last copied: the next one appends to its list. */ let listed = false; + const asked = createAskedStamp(), + stampWord = new Uint32Array(1); const reports = createLightCutReports(own, output, outputBytes); const redraws = createLightCutRedraws(own, output, capacity); return { @@ -142,8 +145,14 @@ export function createDagLightCut(resources: DagResources) { count: number, ) { if (count > capacity) throw new Error(`${count} light views, at most ${capacity}`); - // The frame's first cut starts its list, and forgets what the last frame asked for. - if (!listed) encoder.clearBuffer(work, layout.askedAt * 4, pageCount * 4); + // The frame's first cut starts its list under a new stamp: what earlier frames asked for loses + // to it, and only a wrapped stamp clears the words (`askedStamp.ts`). + if (!listed) { + const { stamp, clear } = asked.next(); + if (clear) encoder.clearBuffer(work, (layout.askedAt + 1) * 4, pageCount * 4); + stampWord[0] = stamp; + device.queue.writeBuffer(work, layout.askedAt * 4, stampWord); + } cutViews.count = light.views = count; cutViews.append = listed; listed = true; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts index fbbe6c368f..5d9119369e 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts @@ -6,21 +6,26 @@ import { lightCutFrame } from './lightCutFrame.fixture.ts'; import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; import { DAG_VIEWS_WGSL } from './shader/viewsWgsl.ts'; import { dagWorkLayout } from './shader/floorWgsl.ts'; +import { ASKED_PRIORITY_BITS, ASKED_STAMP_MAX, createAskedStamp } from './askedStamp.ts'; test('a light cut lists a caster once a frame, at its best request: the contract restated here', () => { for (const line of [ - 'if(isLightCut()&&atomicMax(&work[askedWord(page)],priority+1u)!=0u){return;}', + 'let stamp=atomicLoad(&work[askedStamp()]);', + 'if((atomicMax(&work[askedWord(page)],(stamp<>ASKED_BITS)==stamp){return;}', 'let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;}', - 'out.pages[s]=packRequest(page,atomicLoad(&work[askedWord(page)])-1u);', + 'out.pages[s]=packRequest(page,(atomicLoad(&work[askedWord(page)])&((1u<u32{return drawnGroupsMax()+1u+page;}'), - ); + // The stamp the host writes and the words a wrap clears are the kernel's: behind `drawnGroupsMax`. + for (const line of [ + 'fn askedStamp()->u32{return drawnGroupsMax()+1u;}', + 'fn askedWord(page:u32)->u32{return drawnGroupsMax()+2u+page;}', + ]) + assert.ok(DAG_VIEWS_WGSL.includes(line), line); const layout = dagWorkLayout(3, 5, 7); assert.equal(layout.askedAt, layout.drawnGroupsMax + 1); - assert.equal(layout.words, layout.askedAt + 7); + assert.equal(layout.words, layout.askedAt + 1 + 7); }); // Every sun level of every batch wants the casters that span the scene. Asked once a frame, they @@ -61,3 +66,32 @@ test('page zero at priority zero is listed once a frame too', async () => { await frame([4, 5, 6], new Set(), () => [[0, 0]]); assert.deepEqual(cut.reports.takeRequests(), [0], 'page zero, once'); }); + +// A frame's asks are told from the last frame's by the stamp alone: nothing clears the words between +// two frames, and a word the last frame left, even at a higher priority, loses to this frame's (#525). +test("a frame's asks do not see the previous frame's, with no clear between them", async () => { + const { cut, frame, words, askedAt } = lightCutFrame(); + await frame([4], new Set(), () => [[3, 9]]); + cut.reports.takeRequests(); + const left = words()[askedAt + 1 + 3]; + assert.ok(left !== 0, 'the first frame left its word'); + await frame([4], new Set(), () => [[3, 2]]); + assert.deepEqual(cut.reports.takeRequests(), [3], 'asked again, the next frame'); + assert.equal(words()[askedAt + 1 + 3] & ((1 << ASKED_PRIORITY_BITS) - 1), 3, 'at its own best'); +}); + +// The stamp counts frames up to its last value, then starts again: that frame alone clears the +// words, once, since a word may still hold a larger stamp from before (`askedStamp.ts`). +test('the stamp wraps once every ASKED_STAMP_MAX frames, and only the wrap clears the words', () => { + const asked = createAskedStamp(); + let clears = 0, + last = 0; + for (let frame = 0; frame < ASKED_STAMP_MAX + 3; frame++) { + const { stamp, clear } = asked.next(); + if (clear) clears++; + assert.ok(clear ? stamp === 1 && last === ASKED_STAMP_MAX : stamp === last + 1); + last = stamp; + } + assert.equal(clears, 1); + assert.ok((ASKED_STAMP_MAX << ASKED_PRIORITY_BITS) >>> 0 > 0, 'the largest stamp fits the word'); +}); diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts index 6accfb34c1..1c7f1a8d40 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts @@ -1,8 +1,9 @@ // A light cut over a small catalogue, on a device whose copies run as they are encoded. The GPU side // is the shader's contract, run on the buffers the host wrote: a cut resets the list unless its -// uniform says append, the first view to want a caster in the frame lists it and every view raises -// its best priority (`askedWord`), the frame's list then takes those best requests (`dagAskedBest`), -// and a view whose caster is not resident draws coarser. +// uniform says append, the first view to want a caster under the frame's stamp lists it and every +// view raises its best priority (`askedWord`), the frame's list then takes those best requests +// (`dagAskedBest`), and a view whose caster is not resident draws coarser. The encoder applies its +// clears; the device's writes are applied by `run`, as they land before the batch. import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; import { sunRun } from '../../webgpu/shadow/runs.fixture.ts'; import { createDagLightCut } from './lightCut.ts'; @@ -13,6 +14,7 @@ import { COARSER_VIEWS, DAG_UNIFORM_BYTES, LIST_FULL } from './shader/viewsWgsl. import { VIEW_FLAGS_WORD } from './uniforms.ts'; import { VIEW_APPEND } from './shader/pagesWgsl.ts'; import { dagWorkLayout } from './shader/floorWgsl.ts'; +import { ASKED_PRIORITY_BITS } from './askedStamp.ts'; const CASTERS = 16; /** The pipeline of `dagAskedBest`: a dispatch under it runs the kernel's contract. */ @@ -69,7 +71,8 @@ export function lightCutFrame() { const work = buffers.find(({ label }) => label === 'Trillion3D light cut work')!; const { askedAt } = dagWorkLayout(1, cut.capacity, CASTERS); const out = () => new Uint32Array(output.getMappedRange()), - best = () => new Uint32Array(work.getMappedRange()); + best = () => new Uint32Array(work.getMappedRange()), + priorityMask = (1 << ASKED_PRIORITY_BITS) - 1; /** `dagAskedBest`: each listed page takes its best request of the frame. */ const askedBest = () => { const list = out(), @@ -77,7 +80,7 @@ export function lightCutFrame() { for (let s = 0; s < Math.min(list[OUT_COUNT], CASTERS); s++) { const at = SELECTION_HEADER_WORDS + s; const page = requestPage(list[at]); - list[at] = packRequest(page, words[askedAt + page] - 1); + list[at] = packRequest(page, (words[askedAt + 1 + page] & priorityMask) - 1); } }; /** The GPU running the cut just encoded, over a view that wants each `[caster, priority]` of @@ -86,14 +89,20 @@ export function lightCutFrame() { const uniform = writes.findLast(({ buffer }) => buffer.size === DAG_UNIFORM_BYTES)!; const viewFlags = new Uint32Array(uniform.data.slice().buffer)[VIEW_FLAGS_WORD]; const list = out(), - words = best(); + words = best(), + stamp = writes.findLast(({ buffer }) => buffer === work); + if (stamp) words[askedAt] = new Uint32Array(stamp.data.slice().buffer)[0]; if (viewFlags & VIEW_APPEND) list[OUT_FLAGS] &= LIST_FULL; else list[OUT_COUNT] = list[OUT_FLAGS] = 0; for (const [caster, priority] of asks) { if (!resident.has(caster)) list[OUT_FLAGS] |= 1 << COARSER_VIEWS; - const before = words[askedAt + caster]; - words[askedAt + caster] = Math.max(before, priority + 1); - if (before) continue; + const at = askedAt + 1 + caster, + before = words[at]; + words[at] = Math.max( + before, + ((words[askedAt] << ASKED_PRIORITY_BITS) | (priority + 1)) >>> 0, + ); + if (before >>> ASKED_PRIORITY_BITS === words[askedAt]) continue; const slot = list[OUT_COUNT]++; if (slot < CASTERS) list[SELECTION_HEADER_WORDS + slot] = packRequest(caster, priority); else list[OUT_FLAGS] |= LIST_FULL; @@ -121,5 +130,5 @@ export function lightCutFrame() { await cut.settled(); return { flags, copies: report ? 1 : 0 }; }; - return { cut, frame }; + return { cut, frame, words: best, askedAt }; } diff --git a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts index 0ebdaf7351..564ad4fa07 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts @@ -23,8 +23,9 @@ * top-down pruning would then drop everything. * * `views` is the view capacity the buffer serves: one for a camera, one row each for a light cut. - * `pages` is the catalogue a light cut asks for: one word per page behind the rest, its best priority - * of the frame plus one (`askedWord`, `snapshotWgsl.ts`); a camera, which asks for a page once, has none. + * `pages` is the catalogue a light cut asks for: the frame's stamp, then one word per page, its + * stamp and best priority (`askedWord`, `snapshotWgsl.ts`, `../askedStamp.ts`); a camera, which asks + * for a page once, has none. */ export function dagWorkLayout(blockCount: number, views = 1, pages = 0) { const base = blockCount * 2, @@ -44,10 +45,9 @@ export function dagWorkLayout(blockCount: number, views = 1, pages = 0) { viewWords, /** The most sixty-four-wide groups any view drew, behind the per-view rows. */ drawnGroupsMax, - /** The frame's best priority of each page plus one, behind it: one word per page, cleared at - * the frame's first cut. */ + /** The frame's stamp, behind it, then each page's asked word: never cleared between frames. */ askedAt, - words: askedAt + pages, + words: askedAt + (pages && pages + 1), }; } diff --git a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts index 5d41239307..d2769780d6 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/snapshotWgsl.ts @@ -1,3 +1,5 @@ +import { ASKED_PRIORITY_BITS } from '../askedStamp.ts'; + /** * SNAPSHOT write: what the GPU reports to the CPU, and the ceiling that bounds it. * @@ -18,14 +20,20 @@ * batch appends to one list (`VIEW_APPEND`) as long as the catalogue, and the same caster asked by * each sun level and each batch filled it with repeats, so a late batch's own casters fell past it * (`LIST_FULL`) and its coarse pages were drawn again every frame without ever being asked for. - * Each page keeps its best priority of the frame, plus one (`askedWord`): zero is "not asked yet", - * even for page zero at priority zero, whose request word is zero. The first view to raise it from - * zero lists the page, and once the frame's cuts are done `dagAskedBest` writes the request at that - * best priority over its entry — the highest any view gave it, whatever view won the race. + * Each page keeps the frame's stamp and its best priority plus one (`askedWord`, `../askedStamp.ts`): + * the view whose `atomicMax` finds an earlier frame's stamp there lists the page — an earlier stamp + * is smaller, so it always loses, and nothing is cleared between frames —, and once the frame's cuts + * are done `dagAskedBest` writes the request at that best priority over its entry: the highest any + * view gave it, whatever view won the race. Page zero at priority zero, whose request word is zero, + * is marked too: the mark is the priority plus one. */ -export const DAG_RELEVE_WGSL = `fn emitOne(page:u32,pixels:f32){ +export const DAG_RELEVE_WGSL = `const ASKED_BITS:u32=${ASKED_PRIORITY_BITS}u; +fn emitOne(page:u32,pixels:f32){ let priority=quantizePriority(pixels); - if(isLightCut()&&atomicMax(&work[askedWord(page)],priority+1u)!=0u){return;} + if(isLightCut()){ + let stamp=atomicLoad(&work[askedStamp()]); + if((atomicMax(&work[askedWord(page)],(stamp<>ASKED_BITS)==stamp){return;} + } emitWord(page,priority,true); } /** After a frame's last light cut: each listed page at the best request its views made of it. */ @@ -33,7 +41,7 @@ export const DAG_RELEVE_WGSL = `fn emitOne(page:u32,pixels:f32){ fn dagAskedBest(@builtin(global_invocation_id) id:vec3u){ let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;} let page=requestPage(out.pages[s]); - out.pages[s]=packRequest(page,atomicLoad(&work[askedWord(page)])-1u); + out.pages[s]=packRequest(page,(atomicLoad(&work[askedWord(page)])&((1u<u32{return vi*views[0u].worldCount+w;} fn viewWord(row:u32,v:u32)->u32{return extraBase()+row*views[0u].viewCapacity+v;} /** The word behind the per-view rows: the most sixty-four-wide groups any view drew. */ fn drawnGroupsMax()->u32{return viewWord(${VIEW_WORD_ROWS}u,0u);} -/** A light cut's best priority of \`page\` this frame plus one, zero if unasked, behind it - * (\`dagWorkLayout\`, \`askedAt\`). */ -fn askedWord(page:u32)->u32{return drawnGroupsMax()+1u+page;} +/** A light cut's frame stamp, behind it, then each page's asked word: the stamp of the last frame + * that asked for \`page\` and its best priority there (\`dagWorkLayout\`, \`askedAt\`). */ +fn askedStamp()->u32{return drawnGroupsMax()+1u;} +fn askedWord(page:u32)->u32{return drawnGroupsMax()+2u+page;} fn dropWork(){atomicOr(&out.overflow,${WORK_DROPPED}u);} fn noteCoarser(){if(isLightCut()){atomicOr(&out.overflow,1u<<(${COARSER_VIEWS}u+vi));}} fn isLightCut()->bool{return (views[0u].viewFlags&VIEW_LIGHT)!=0u;} From ffac647267bf7d6128d68ebf967d3d45333eb938 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 01:30:08 +0200 Subject: [PATCH 07/13] test(shadows): the largest stamp fills the asked word; the layout says a wrap clears it (#525) --- packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts | 5 ++++- packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts | 2 +- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts index 5d9119369e..147fd3836c 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts @@ -7,6 +7,7 @@ import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; import { DAG_VIEWS_WGSL } from './shader/viewsWgsl.ts'; import { dagWorkLayout } from './shader/floorWgsl.ts'; import { ASKED_PRIORITY_BITS, ASKED_STAMP_MAX, createAskedStamp } from './askedStamp.ts'; +import { REQUEST_PRIORITY_MAX } from './request.ts'; test('a light cut lists a caster once a frame, at its best request: the contract restated here', () => { for (const line of [ @@ -93,5 +94,7 @@ test('the stamp wraps once every ASKED_STAMP_MAX frames, and only the wrap clear last = stamp; } assert.equal(clears, 1); - assert.ok((ASKED_STAMP_MAX << ASKED_PRIORITY_BITS) >>> 0 > 0, 'the largest stamp fits the word'); + const marks = 2 ** ASKED_PRIORITY_BITS; + assert.equal(ASKED_STAMP_MAX * marks + marks - 1, 0xffffffff, 'the largest stamp fills the word'); + assert.ok(REQUEST_PRIORITY_MAX + 1 < marks, 'every priority plus one fits below the stamp'); }); diff --git a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts index 564ad4fa07..f5ea0b0809 100644 --- a/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts +++ b/packages/sdk-browser/src/gpu/dag/shader/floorWgsl.ts @@ -45,7 +45,7 @@ export function dagWorkLayout(blockCount: number, views = 1, pages = 0) { viewWords, /** The most sixty-four-wide groups any view drew, behind the per-view rows. */ drawnGroupsMax, - /** The frame's stamp, behind it, then each page's asked word: never cleared between frames. */ + /** The frame's stamp, behind it, then each page's asked word: cleared only when the stamp wraps. */ askedAt, words: askedAt + (pages && pages + 1), }; From df3d25093d427d36ae48fdaa3cc991025cbfc0b0 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 01:32:31 +0200 Subject: [PATCH 08/13] test(shadows): the fixture reads the stamp write of the work buffer it made (#525) --- packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts index 1c7f1a8d40..772edbb8ea 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts @@ -90,7 +90,7 @@ export function lightCutFrame() { const viewFlags = new Uint32Array(uniform.data.slice().buffer)[VIEW_FLAGS_WORD]; const list = out(), words = best(), - stamp = writes.findLast(({ buffer }) => buffer === work); + stamp = writes.findLast(({ buffer }) => (buffer as unknown) === work); if (stamp) words[askedAt] = new Uint32Array(stamp.data.slice().buffer)[0]; if (viewFlags & VIEW_APPEND) list[OUT_FLAGS] &= LIST_FULL; else list[OUT_COUNT] = list[OUT_FLAGS] = 0; From 6f16bddf8a87753d59a81971662f8e81493ea0e5 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 01:41:26 +0200 Subject: [PATCH 09/13] fix(shadows): the stamp's wrap clear is submitted before the stamp, so a dropped frame cannot lose it; an empty catalogue writes no stamp (#525) --- packages/sdk-browser/src/gpu/dag/lightCut.ts | 13 ++++++++++--- .../src/gpu/dag/lightCutFrame.fixture.ts | 7 ++----- 2 files changed, 12 insertions(+), 8 deletions(-) diff --git a/packages/sdk-browser/src/gpu/dag/lightCut.ts b/packages/sdk-browser/src/gpu/dag/lightCut.ts index 236bce7fa0..173eca2a98 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCut.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCut.ts @@ -146,10 +146,17 @@ export function createDagLightCut(resources: DagResources) { ) { if (count > capacity) throw new Error(`${count} light views, at most ${capacity}`); // The frame's first cut starts its list under a new stamp: what earlier frames asked for loses - // to it, and only a wrapped stamp clears the words (`askedStamp.ts`). - if (!listed) { + // to it, and only a wrapped stamp clears the words (`askedStamp.ts`). Both go on the queue, in + // order, before this frame's command buffer: a frame whose encoder is dropped still leaves its + // stamp, so its wrap clear must land too, or the older, larger stamps would outlive it. An + // empty catalogue asks for nothing and has no stamp word (`dagWorkLayout`). + if (!listed && pageCount) { const { stamp, clear } = asked.next(); - if (clear) encoder.clearBuffer(work, (layout.askedAt + 1) * 4, pageCount * 4); + if (clear) { + const wrap = device.createCommandEncoder(); + wrap.clearBuffer(work, (layout.askedAt + 1) * 4, pageCount * 4); + device.queue.submit([wrap.finish()]); + } stampWord[0] = stamp; device.queue.writeBuffer(work, layout.askedAt * 4, stampWord); } diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts index 772edbb8ea..738a786eb7 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts @@ -2,8 +2,8 @@ // is the shader's contract, run on the buffers the host wrote: a cut resets the list unless its // uniform says append, the first view to want a caster under the frame's stamp lists it and every // view raises its best priority (`askedWord`), the frame's list then takes those best requests -// (`dagAskedBest`), and a view whose caster is not resident draws coarser. The encoder applies its -// clears; the device's writes are applied by `run`, as they land before the batch. +// (`dagAskedBest`), and a view whose caster is not resident draws coarser. The stamp the device +// writes is applied by `run`, as it lands before the batch. import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; import { sunRun } from '../../webgpu/shadow/runs.fixture.ts'; import { createDagLightCut } from './lightCut.ts'; @@ -34,9 +34,6 @@ export function lightCutFrame() { copyBufferToBuffer(from: GPUBuffer, at: number, to: GPUBuffer, toAt: number, size: number) { new Uint8Array(bytes(to)).set(new Uint8Array(bytes(from), at, size), toAt); }, - clearBuffer(buffer: GPUBuffer, at: number, size: number) { - new Uint8Array(bytes(buffer), at, size).fill(0); - }, beginComputePass: () => { let pipeline: unknown; return { From ab5d18846c1c3cd77dc2b825be27b8aa810e5243 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 02:59:11 +0200 Subject: [PATCH 10/13] fix(shadows): a sun's floor holds the pages over the scene, not the view's whole reach; loops A and B reverted (#525) --- .../sdk-browser/src/gpu/dag/askedStamp.ts | 29 ---- packages/sdk-browser/src/gpu/dag/encode.ts | 17 +-- packages/sdk-browser/src/gpu/dag/lightCut.ts | 24 +--- .../src/gpu/dag/lightCutAsks.test.ts | 100 ------------- .../src/gpu/dag/lightCutCapacity.ts | 2 +- .../src/gpu/dag/lightCutFrame.fixture.ts | 131 ------------------ .../src/gpu/dag/lightCutFrame.test.ts | 90 +++++++++++- .../src/gpu/dag/lightCutRedraws.test.ts | 27 +--- .../src/gpu/dag/lightCutRedraws.ts | 12 +- .../src/gpu/dag/lightCutViewLimit.ts | 6 +- packages/sdk-browser/src/gpu/dag/pipeline.ts | 4 +- packages/sdk-browser/src/gpu/dag/request.ts | 1 - .../src/gpu/dag/shader/floorWgsl.ts | 15 +- .../src/gpu/dag/shader/snapshotWgsl.ts | 30 +--- .../src/gpu/dag/shader/viewsWgsl.ts | 4 - .../scene/light-shadow/dollyRefetch.test.ts | 53 +++++++ .../src/scene/light-shadow/requests.ts | 7 +- .../src/scene/light-shadow/sunLevels.ts | 53 +++++-- 18 files changed, 204 insertions(+), 401 deletions(-) delete mode 100644 packages/sdk-browser/src/gpu/dag/askedStamp.ts delete mode 100644 packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts delete mode 100644 packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts create mode 100644 packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts diff --git a/packages/sdk-browser/src/gpu/dag/askedStamp.ts b/packages/sdk-browser/src/gpu/dag/askedStamp.ts deleted file mode 100644 index 1bd8a91a81..0000000000 --- a/packages/sdk-browser/src/gpu/dag/askedStamp.ts +++ /dev/null @@ -1,29 +0,0 @@ -import { REQUEST_PRIORITY_MAX } from './request.ts'; - -/** - * THE FRAME STAMP OF A LIGHT CUT'S ASKED WORDS (`askedWord`, `shader/snapshotWgsl.ts`). A page's word - * holds the frame's stamp in its high bits and its best priority plus one in the low ones, so a - * word left by an earlier frame always loses the `atomicMax` to this frame's first ask, and no - * word needs clearing between frames: a frame costs what its views ask for, never the catalogue. - * - * The stamp counts frames from 1 to `ASKED_STAMP_MAX`. When it wraps back to 1, a word may still - * hold a larger stamp from before: that frame alone clears every word once, one word per catalogue - * page, once every `ASKED_STAMP_MAX` frames — about 9.7 hours at 60 frames a second. - */ -/** Low bits of an asked word: the priority plus one, so that zero is never asked. */ -export const ASKED_PRIORITY_BITS = Math.ceil(Math.log2(REQUEST_PRIORITY_MAX + 2)); -/** The largest stamp: every bit above the priority's. */ -export const ASKED_STAMP_MAX = 2 ** (32 - ASKED_PRIORITY_BITS) - 1; - -/** The stamps of a light cut's frames, and the frame whose words must be cleared first. */ -export function createAskedStamp() { - let stamp = 0; - return { - /** The next frame's stamp, and whether its words are cleared before it asks. */ - next() { - const wraps = stamp >= ASKED_STAMP_MAX; - stamp = wraps ? 1 : stamp + 1; - return { stamp, clear: wraps }; - }, - }; -} diff --git a/packages/sdk-browser/src/gpu/dag/encode.ts b/packages/sdk-browser/src/gpu/dag/encode.ts index 047fb6dcc9..f882a9b243 100644 --- a/packages/sdk-browser/src/gpu/dag/encode.ts +++ b/packages/sdk-browser/src/gpu/dag/encode.ts @@ -12,9 +12,6 @@ export type DagView = NonNullable> light?: { views: number; queueCap: number }; }; -/** Workgroups for `count` threads, never none. */ -const groups = (count: number) => Math.max(1, Math.ceil(count / WORKGROUP)); - /** * Cut kernels, encoded in order. Each dispatch waits for the previous — the GPU empties its queue * and caches between two —, and that wait is attributed to no kernel: it is the number of @@ -78,6 +75,7 @@ function encodeOnce( const views = light?.views ?? 1, queueCap = light?.queueCap ?? resources.nodeCount; const label = light ? LIGHT_CUT_PASS : 'Trillion3D DAG selection'; + const groups = (count: number) => Math.max(1, Math.ceil(count / WORKGROUP)); // Head word of the dispatch argument, copied outside a pass: the other two have been one since // the buffer was created. That is the only reason for cuts between passes. const arm = (offset: number) => encoder.copyBufferToBuffer(work, offset, dispatchArgs, 0, 4); @@ -147,16 +145,3 @@ function encodeOnce( } live.end(); } - -/** - * Once a frame's light cuts are all encoded: each page their list names takes the best request its - * views made of it (`dagAskedBest`, `shader/snapshotWgsl.ts`), before the list is copied. `entries` - * is the list's cap: one thread per entry it can hold. - */ -export function encodeAskedBest(encoder: GPUCommandEncoder, view: DagView, entries: number) { - const pass = encoder.beginComputePass({ label: LIGHT_CUT_PASS }); - pass.setBindGroup(0, view.bindGroup); - pass.setPipeline(view.askedBestPipeline); - pass.dispatchWorkgroups(groups(entries)); - pass.end(); -} diff --git a/packages/sdk-browser/src/gpu/dag/lightCut.ts b/packages/sdk-browser/src/gpu/dag/lightCut.ts index 173eca2a98..b92a281f46 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCut.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCut.ts @@ -2,10 +2,8 @@ import { FRAME_VEC4, type DagViewUniforms, type DrawnLog } from './types.ts'; import { writeDagUniforms, type DagCutViews } from './uniforms.ts'; import { createLightCutReports } from './lightCutReports.ts'; import { createLightCutRedraws } from './lightCutRedraws.ts'; -import { encodeAskedBest, encodeDagKernels, type DagView } from './encode.ts'; +import { encodeDagKernels, type DagView } from './encode.ts'; import { dagWorkLayout } from './shader/floorWgsl.ts'; -import { selectionListCap } from './layout.ts'; -import { createAskedStamp } from './askedStamp.ts'; import { LEVEL_QUEUES } from './shader/levelWgsl.ts'; import { lightCutCapacity, lightQueueCap } from './lightCutCapacity.ts'; import { DAG_UNIFORM_BYTES, DAG_VIEW_WORDS } from './shader/viewsWgsl.ts'; @@ -45,7 +43,7 @@ export function createDagLightCut(resources: DagResources) { const { worldCount, blockCount, buffers } = resources; const capacity = lightCutCapacity(device.limits, resources), queueCap = lightQueueCap(resources, capacity), - layout = dagWorkLayout(blockCount, capacity, pageCount); + layout = dagWorkLayout(blockCount, capacity); const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST; const own = (descriptor: GPUBufferDescriptor) => { const buffer = device.createBuffer(descriptor); @@ -125,8 +123,6 @@ export function createDagLightCut(resources: DagResources) { const cutViews: DagCutViews = { count: 0, capacity, queueCap }; /** A cut ran since the requests were last copied: the next one appends to its list. */ let listed = false; - const asked = createAskedStamp(), - stampWord = new Uint32Array(1); const reports = createLightCutReports(own, output, outputBytes); const redraws = createLightCutRedraws(own, output, capacity); return { @@ -145,21 +141,6 @@ export function createDagLightCut(resources: DagResources) { count: number, ) { if (count > capacity) throw new Error(`${count} light views, at most ${capacity}`); - // The frame's first cut starts its list under a new stamp: what earlier frames asked for loses - // to it, and only a wrapped stamp clears the words (`askedStamp.ts`). Both go on the queue, in - // order, before this frame's command buffer: a frame whose encoder is dropped still leaves its - // stamp, so its wrap clear must land too, or the older, larger stamps would outlive it. An - // empty catalogue asks for nothing and has no stamp word (`dagWorkLayout`). - if (!listed && pageCount) { - const { stamp, clear } = asked.next(); - if (clear) { - const wrap = device.createCommandEncoder(); - wrap.clearBuffer(work, (layout.askedAt + 1) * 4, pageCount * 4); - device.queue.submit([wrap.finish()]); - } - stampWord[0] = stamp; - device.queue.writeBuffer(work, layout.askedAt * 4, stampWord); - } cutViews.count = light.views = count; cutViews.append = listed; listed = true; @@ -180,7 +161,6 @@ export function createDagLightCut(resources: DagResources) { encodeReports(encoder: GPUCommandEncoder) { if (!listed) return undefined; listed = false; - encodeAskedBest(encoder, view, selectionListCap(pageCount)); const settle = reports.encodeReadback(encoder); redraws.reported(settle !== undefined); return settle; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts deleted file mode 100644 index 147fd3836c..0000000000 --- a/packages/sdk-browser/src/gpu/dag/lightCutAsks.test.ts +++ /dev/null @@ -1,100 +0,0 @@ -// A light cut's frame lists each caster once, however many views and batches want it, at the best -// request any of them made (#525): the list, as long as the catalogue, never fills with repeats. -import test from 'node:test'; -import assert from 'node:assert/strict'; -import { lightCutFrame } from './lightCutFrame.fixture.ts'; -import { DAG_RELEVE_WGSL } from './shader/snapshotWgsl.ts'; -import { DAG_VIEWS_WGSL } from './shader/viewsWgsl.ts'; -import { dagWorkLayout } from './shader/floorWgsl.ts'; -import { ASKED_PRIORITY_BITS, ASKED_STAMP_MAX, createAskedStamp } from './askedStamp.ts'; -import { REQUEST_PRIORITY_MAX } from './request.ts'; - -test('a light cut lists a caster once a frame, at its best request: the contract restated here', () => { - for (const line of [ - 'let stamp=atomicLoad(&work[askedStamp()]);', - 'if((atomicMax(&work[askedWord(page)],(stamp<>ASKED_BITS)==stamp){return;}', - 'let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;}', - 'out.pages[s]=packRequest(page,(atomicLoad(&work[askedWord(page)])&((1u<u32{return drawnGroupsMax()+1u;}', - 'fn askedWord(page:u32)->u32{return drawnGroupsMax()+2u+page;}', - ]) - assert.ok(DAG_VIEWS_WGSL.includes(line), line); - const layout = dagWorkLayout(3, 5, 7); - assert.equal(layout.askedAt, layout.drawnGroupsMax + 1); - assert.equal(layout.words, layout.askedAt + 1 + 7); -}); - -// Every sun level of every batch wants the casters that span the scene. Asked once a frame, they -// leave the list, as long as the catalogue, room for each batch's own casters: asked for, and never -// drawn again at once for want of a place in it (#525). -test("a caster every batch wants is asked for once a frame, and each batch's own casters too", async () => { - const { cut, frame } = lightCutFrame(); - const pages = [4, 5, 6, 7, 8, 9, 10, 11]; - for (const run of ['a frame', 'the next frame']) { - await frame(pages, new Set(), (page) => [0, 1, 2, 3, page].map((caster) => [caster, 1])); - const asked = cut.reports.takeRequests()?.sort((a, b) => a - b); - assert.deepEqual(asked, [0, 1, 2, 3, ...pages], `${run}: every caster, each once`); - const now: number[] = []; - cut.redraws.takeRedraw((page) => now.push(page)); - assert.deepEqual(now, [], `${run}: no coarse page drawn again at once`); - } -}); - -// Two captures of one pose must ask in the same order (`lightCutReports.ts`): a caster listed by a -// coarse view that a later, finer view needs more is asked at the finer view's priority. -test('a caster several views ask for is asked at the highest priority any of them gives it', async () => { - const { cut, frame } = lightCutFrame(); - await frame([4, 5], new Set(), (page) => - page === 4 - ? [ - [3, 2], - [9, 5], - ] - : [[3, 8]], - ); - assert.deepEqual(cut.reports.takeRequests(), [3, 9], 'caster 3 at 8, above caster 9 at 5'); -}); - -// Page zero at priority zero packs to the word zero: the frame's mark is its priority plus one, so -// that request too is listed once, not once per view that wants it. -test('page zero at priority zero is listed once a frame too', async () => { - const { cut, frame } = lightCutFrame(); - await frame([4, 5, 6], new Set(), () => [[0, 0]]); - assert.deepEqual(cut.reports.takeRequests(), [0], 'page zero, once'); -}); - -// A frame's asks are told from the last frame's by the stamp alone: nothing clears the words between -// two frames, and a word the last frame left, even at a higher priority, loses to this frame's (#525). -test("a frame's asks do not see the previous frame's, with no clear between them", async () => { - const { cut, frame, words, askedAt } = lightCutFrame(); - await frame([4], new Set(), () => [[3, 9]]); - cut.reports.takeRequests(); - const left = words()[askedAt + 1 + 3]; - assert.ok(left !== 0, 'the first frame left its word'); - await frame([4], new Set(), () => [[3, 2]]); - assert.deepEqual(cut.reports.takeRequests(), [3], 'asked again, the next frame'); - assert.equal(words()[askedAt + 1 + 3] & ((1 << ASKED_PRIORITY_BITS) - 1), 3, 'at its own best'); -}); - -// The stamp counts frames up to its last value, then starts again: that frame alone clears the -// words, once, since a word may still hold a larger stamp from before (`askedStamp.ts`). -test('the stamp wraps once every ASKED_STAMP_MAX frames, and only the wrap clears the words', () => { - const asked = createAskedStamp(); - let clears = 0, - last = 0; - for (let frame = 0; frame < ASKED_STAMP_MAX + 3; frame++) { - const { stamp, clear } = asked.next(); - if (clear) clears++; - assert.ok(clear ? stamp === 1 && last === ASKED_STAMP_MAX : stamp === last + 1); - last = stamp; - } - assert.equal(clears, 1); - const marks = 2 ** ASKED_PRIORITY_BITS; - assert.equal(ASKED_STAMP_MAX * marks + marks - 1, 0xffffffff, 'the largest stamp fills the word'); - assert.ok(REQUEST_PRIORITY_MAX + 1 < marks, 'every priority plus one fits below the stamp'); -}); diff --git a/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts b/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts index 0ddef4a8a7..17ec2c9751 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutCapacity.ts @@ -42,7 +42,7 @@ export function lightCutCapacity(limits: LightCutLimits, shape: LightCutShape) { if (Math.min(shape.levelSizes[level] * views, queueCap) > threads) return false; const frames = views * shape.worldCount * FRAME_VEC4 * 16, flags = (queueCap * LEVEL_QUEUES + shape.pageCount * 4) * 4, - work = dagWorkLayout(shape.blockCount, views, shape.pageCount).words * 4; + work = dagWorkLayout(shape.blockCount, views).words * 4; return Math.max(frames, flags, work) <= bytes; }; let views = DAG_MAX_VIEWS; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts deleted file mode 100644 index 738a786eb7..0000000000 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.fixture.ts +++ /dev/null @@ -1,131 +0,0 @@ -// A light cut over a small catalogue, on a device whose copies run as they are encoded. The GPU side -// is the shader's contract, run on the buffers the host wrote: a cut resets the list unless its -// uniform says append, the first view to want a caster under the frame's stamp lists it and every -// view raises its best priority (`askedWord`), the frame's list then takes those best requests -// (`dagAskedBest`), and a view whose caster is not resident draws coarser. The stamp the device -// writes is applied by `run`, as it lands before the batch. -import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; -import { sunRun } from '../../webgpu/shadow/runs.fixture.ts'; -import { createDagLightCut } from './lightCut.ts'; -import { FRAME_VEC4 } from './types.ts'; -import { OUT_COUNT, OUT_FLAGS, SELECTION_HEADER_WORDS } from './layout.ts'; -import { packRequest, requestPage } from './request.ts'; -import { COARSER_VIEWS, DAG_UNIFORM_BYTES, LIST_FULL } from './shader/viewsWgsl.ts'; -import { VIEW_FLAGS_WORD } from './uniforms.ts'; -import { VIEW_APPEND } from './shader/pagesWgsl.ts'; -import { dagWorkLayout } from './shader/floorWgsl.ts'; -import { ASKED_PRIORITY_BITS } from './askedStamp.ts'; - -const CASTERS = 16; -/** The pipeline of `dagAskedBest`: a dispatch under it runs the kernel's contract. */ -const ASKED_BEST = {}; - -/** A light cut over `CASTERS` catalogue pages, on a device whose copies run as they are encoded. */ -export function lightCutFrame() { - const { device, writes, buffers } = fakeDevice({ - limits: { - maxComputeWorkgroupsPerDimension: 65535, - maxStorageBufferBindingSize: 1 << 27, - maxBufferSize: 1 << 28, - }, - }); - const bytes = (buffer: GPUBuffer) => (buffer as unknown as FakeBuffer).getMappedRange(); - const encoder = { - copyBufferToBuffer(from: GPUBuffer, at: number, to: GPUBuffer, toAt: number, size: number) { - new Uint8Array(bytes(to)).set(new Uint8Array(bytes(from), at, size), toAt); - }, - beginComputePass: () => { - let pipeline: unknown; - return { - setBindGroup() {}, - setPipeline: (next: unknown) => (pipeline = next), - dispatchWorkgroups: () => pipeline === ASKED_BEST && askedBest(), - dispatchWorkgroupsIndirect() {}, - end() {}, - }; - }, - } as unknown as GPUCommandEncoder; - const outputBytes = (SELECTION_HEADER_WORDS + CASTERS) * 4; - const packed = { pageCount: CASTERS, nodeCount: CASTERS, worldCount: 1 }; - const cut = createDagLightCut({ - device, - packed, - residentCut: false, - pageCount: CASTERS, - nodeCount: CASTERS, - worldCount: 1, - blockCount: 1, - levelSizes: [1], - levelPipelines: [{}], - askedBestPipeline: ASKED_BEST, - outputBytes, - readbackBytes: outputBytes, - frameData: new Float32Array(FRAME_VEC4 * 4), - frameWrites: { count: 0 }, - buffers: [], - } as unknown as Parameters[0]); - const output = buffers.find(({ label }) => label === 'Trillion3D light cut output')!; - const work = buffers.find(({ label }) => label === 'Trillion3D light cut work')!; - const { askedAt } = dagWorkLayout(1, cut.capacity, CASTERS); - const out = () => new Uint32Array(output.getMappedRange()), - best = () => new Uint32Array(work.getMappedRange()), - priorityMask = (1 << ASKED_PRIORITY_BITS) - 1; - /** `dagAskedBest`: each listed page takes its best request of the frame. */ - const askedBest = () => { - const list = out(), - words = best(); - for (let s = 0; s < Math.min(list[OUT_COUNT], CASTERS); s++) { - const at = SELECTION_HEADER_WORDS + s; - const page = requestPage(list[at]); - list[at] = packRequest(page, (words[askedAt + 1 + page] & priorityMask) - 1); - } - }; - /** The GPU running the cut just encoded, over a view that wants each `[caster, priority]` of - * `asks`, against `resident`. */ - const run = (asks: number[][], resident: Set) => { - const uniform = writes.findLast(({ buffer }) => buffer.size === DAG_UNIFORM_BYTES)!; - const viewFlags = new Uint32Array(uniform.data.slice().buffer)[VIEW_FLAGS_WORD]; - const list = out(), - words = best(), - stamp = writes.findLast(({ buffer }) => (buffer as unknown) === work); - if (stamp) words[askedAt] = new Uint32Array(stamp.data.slice().buffer)[0]; - if (viewFlags & VIEW_APPEND) list[OUT_FLAGS] &= LIST_FULL; - else list[OUT_COUNT] = list[OUT_FLAGS] = 0; - for (const [caster, priority] of asks) { - if (!resident.has(caster)) list[OUT_FLAGS] |= 1 << COARSER_VIEWS; - const at = askedAt + 1 + caster, - before = words[at]; - words[at] = Math.max( - before, - ((words[askedAt] << ASKED_PRIORITY_BITS) | (priority + 1)) >>> 0, - ); - if (before >>> ASKED_PRIORITY_BITS === words[askedAt]) continue; - const slot = list[OUT_COUNT]++; - if (slot < CASTERS) list[SELECTION_HEADER_WORDS + slot] = packRequest(caster, priority); - else list[OUT_FLAGS] |= LIST_FULL; - } - return viewFlags; - }; - const views = [{ uniforms: sunRun(64).uniforms }]; - /** One frame drawing `pages` in one batch each, whose view asks `wants(page)`: the page itself - * by default. Returns the batches' view flags and how many report copies ran. */ - const frame = async ( - pages: number[], - resident: Set, - wants = (page: number) => [[page, 1]], - ) => { - const settles: Array<(submitted: boolean) => void> = [], - flags: number[] = []; - for (const page of pages) { - cut.encode(encoder, views, 1); - flags.push(run(wants(page), resident)); - const settle = cut.redraws.encode(encoder, [page], [0], 1); - if (settle) settles.push(settle); - } - const report = cut.encodeReports(encoder); - for (const settle of [report, ...settles]) settle?.(true); - await cut.settled(); - return { flags, copies: report ? 1 : 0 }; - }; - return { cut, frame, words: best, askedAt }; -} diff --git a/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts index 6996c9d046..0e63aa1ca0 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutFrame.test.ts @@ -1,12 +1,98 @@ // A frame draws its shadow pages in as many batches as they take (#489), each batch a light cut. // Every batch's requests must be read back, whatever the batch count: the cuts append to one list // the frame copies once (`VIEW_APPEND`), so no batch's coarse view is redrawn for want of a report -// slot, and its missing casters are asked for. +// slot, and its missing casters are asked for. The GPU side here is the shader's contract, run on +// the buffers the host wrote: a cut resets the list unless its uniform says append, and a view +// whose caster is not resident draws coarser and requests it. import test from 'node:test'; import assert from 'node:assert/strict'; -import { lightCutFrame } from './lightCutFrame.fixture.ts'; +import { fakeDevice, type FakeBuffer } from '../../../../../tests/kit/gpu/fakeDevice.ts'; +import { sunRun } from '../../webgpu/shadow/runs.fixture.ts'; +import { createDagLightCut } from './lightCut.ts'; +import { FRAME_VEC4 } from './types.ts'; +import { OUT_COUNT, OUT_FLAGS, SELECTION_HEADER_WORDS } from './layout.ts'; +import { packRequest } from './request.ts'; +import { COARSER_VIEWS, DAG_UNIFORM_BYTES, LIST_FULL } from './shader/viewsWgsl.ts'; +import { VIEW_FLAGS_WORD } from './uniforms.ts'; import { VIEW_APPEND } from './shader/pagesWgsl.ts'; +const CASTERS = 16; + +/** A light cut over `CASTERS` catalogue pages, on a device whose copies run as they are encoded. */ +function lightCutFrame() { + const { device, writes, buffers } = fakeDevice({ + limits: { + maxComputeWorkgroupsPerDimension: 65535, + maxStorageBufferBindingSize: 1 << 27, + maxBufferSize: 1 << 28, + }, + }); + const bytes = (buffer: GPUBuffer) => (buffer as unknown as FakeBuffer).getMappedRange(); + const encoder = { + copyBufferToBuffer(from: GPUBuffer, at: number, to: GPUBuffer, toAt: number, size: number) { + new Uint8Array(bytes(to)).set(new Uint8Array(bytes(from), at, size), toAt); + }, + beginComputePass: () => ({ + setBindGroup() {}, + setPipeline() {}, + dispatchWorkgroups() {}, + dispatchWorkgroupsIndirect() {}, + end() {}, + }), + } as unknown as GPUCommandEncoder; + const outputBytes = (SELECTION_HEADER_WORDS + CASTERS) * 4; + const packed = { pageCount: CASTERS, nodeCount: CASTERS, worldCount: 1 }; + const cut = createDagLightCut({ + device, + packed, + residentCut: false, + pageCount: CASTERS, + nodeCount: CASTERS, + worldCount: 1, + blockCount: 1, + levelSizes: [1], + levelPipelines: [{}], + outputBytes, + readbackBytes: outputBytes, + frameData: new Float32Array(FRAME_VEC4 * 4), + frameWrites: { count: 0 }, + buffers: [], + } as unknown as Parameters[0]); + const output = buffers.find(({ label }) => label === 'Trillion3D light cut output')!; + /** The GPU running the cut just encoded, over the view of `caster`, against `resident`. */ + const run = (caster: number, resident: Set) => { + const uniform = writes.findLast(({ buffer }) => buffer.size === DAG_UNIFORM_BYTES)!; + const viewFlags = new Uint32Array(uniform.data.slice().buffer)[VIEW_FLAGS_WORD]; + const out = new Uint32Array(output.getMappedRange()); + if (viewFlags & VIEW_APPEND) out[OUT_FLAGS] &= LIST_FULL; + else out[OUT_COUNT] = out[OUT_FLAGS] = 0; + if (resident.has(caster)) return viewFlags; + out[OUT_FLAGS] |= 1 << COARSER_VIEWS; + const slot = out[OUT_COUNT]++; + if (slot < CASTERS) out[SELECTION_HEADER_WORDS + slot] = packRequest(caster, 1); + else out[OUT_FLAGS] |= LIST_FULL; + return viewFlags; + }; + const views = [{ uniforms: sunRun(64).uniforms }]; + /** One frame drawing `pages` in one batch each: the page is its caster. Returns the batches' + * view flags, whether each batch's requests will be read, and how many report copies ran. */ + const frame = async (pages: number[], resident: Set) => { + const settles: Array<(submitted: boolean) => void> = [], + flags: number[] = []; + for (const page of pages) { + cut.encode(encoder, views, 1); + flags.push(run(page, resident)); + const settle = cut.redraws.encode(encoder, [page], [0], 1); + if (settle) settles.push(settle); + } + const report = cut.encodeReports(encoder); + for (const settle of [report, ...settles]) settle?.(true); + await cut.settled(); + return { flags, copies: report ? 1 : 0 }; + }; + return { cut, frame }; +} + test("a frame of many batches reads back every batch's requests in one copy", async () => { const { cut, frame } = lightCutFrame(); const pages = [0, 1, 2, 3, 4, 5, 6, 7]; diff --git a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts index 418e65bbe0..d2cb84fee6 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.test.ts @@ -60,37 +60,12 @@ test('the pages of a frame that dropped work are drawn again, in fewer views unt await frame([4, 9, 12, 20, 21], true, [0, 0, 0, 1, 1]); assert.equal(redraws.viewLimit, 2, 'five pages in two views fit: no swing back to three'); redraws.residencyChanged(); - assert.equal(redraws.viewLimit, 2, 'residency moved, the camera moves: the drop holds'); - redraws.rest(); - assert.equal(redraws.viewLimit, 24, 'residency moved, the camera rests: the drop is forgotten'); + assert.equal(redraws.viewLimit, 24, 'residency moved: the drop is forgotten'); flag.value = WORK_DROPPED; for (let i = 0; i < 6; i++) await frame([1, 2], true, [0, 1]); assert.equal(redraws.viewLimit, 1, 'drops floor the limit at one view'); }); -// Under a moving camera residency changes every frame. A batch whose lists hold two views' casters -// drops past them; the limit settles there and stays, and no batch drops again, until the camera -// rests and the views may grow back (#525). -test('while the camera moves and residency changes, a batch that dropped work does not drop again', async () => { - const flag = { value: 0 }; - const { redraws, frame } = redrawsWith(flag); - const drops: number[] = []; - for (let at = 0; at < 8; at++) { - const views = Array.from({ length: Math.min(redraws.viewLimit, 8) }, (_, view) => view); - flag.value = views.length > 2 ? WORK_DROPPED : 0; - if ((await frame(views, true, views)).length) drops.push(at); - redraws.residencyChanged(); - } - assert.deepEqual( - drops.filter((at) => at >= 4), - [], - 'the last four frames draw whole', - ); - assert.equal(redraws.viewLimit, 2, 'two views, what the lists hold'); - redraws.rest(); - assert.equal(redraws.viewLimit, 24, 'at rest the views may grow back'); -}); - // A view drew a placement coarser than it wanted: every page it drew waits for residency to move // and the camera to rest, and is then drawn again — a cluster that never comes costs nothing, and a // camera that only moves redraws none of them. diff --git a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts index cbbbdb08ed..396855a73e 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutRedraws.ts @@ -141,21 +141,17 @@ export function createLightCutRedraws( reported(copied: boolean) { if (open) open.reported = copied; }, - /** Residency the light cuts see changed: the pages that waited on it are drawn again, and the - * drop forgotten, once the camera rests (`rest`). */ + /** Residency the light cuts see changed: the drop is forgotten, and the pages that waited on + * it are drawn again once the camera rests (`rest`). */ residencyChanged() { moved = true; + limit.residencyChanged(); }, - /** - * The camera rests: what residency changed meanwhile is drawn again, and the views a batch - * draws in may grow back. Never before: under a moving camera residency changes every frame, - * and the batch that dropped would drop again each time. - */ + /** The camera rests: what residency changed meanwhile is drawn again. */ rest() { if (!moved) return; moved = false; epoch++; - limit.forgetDrop(); for (const page of waiting) again(page, false); waiting.clear(); }, diff --git a/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts b/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts index d3c3266674..e3bd7293cc 100644 --- a/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts +++ b/packages/sdk-browser/src/gpu/dag/lightCutViewLimit.ts @@ -2,8 +2,8 @@ * The light views one batch draws in, bisected between the most views a batch drew whole and the * fewest a batch dropped work with: a drop at `L` views never swings the limit between `L` and * `L / 2`, it settles on the largest count that fits, never below one. What dropped depends on the - * clusters the views kept from the resident catalogue: `forgetDrop` lets the limit grow back, and - * keeps what fitted. + * clusters the views kept from the resident catalogue: a residency change forgets the drop, never + * what fitted. */ export function createViewLimit(viewCap: number) { let fits = 0, @@ -18,7 +18,7 @@ export function createViewLimit(viewCap: number) { if (fits >= drops) drops = viewCap + 1; } }, - forgetDrop() { + residencyChanged() { drops = viewCap + 1; }, get value() { diff --git a/packages/sdk-browser/src/gpu/dag/pipeline.ts b/packages/sdk-browser/src/gpu/dag/pipeline.ts index 39adf58513..173bbf5dd5 100644 --- a/packages/sdk-browser/src/gpu/dag/pipeline.ts +++ b/packages/sdk-browser/src/gpu/dag/pipeline.ts @@ -41,8 +41,7 @@ export function createDagPipeline(device: GPUDevice, buffers: DagBuffers) { maskPipeline = stage('dagMask'); const drawPrefixPipeline = stage('dagDrawPrefix'), drawScatterPipeline = stage('dagDrawScatter'), - viewOffsetsPipeline = stage('dagViewOffsets'), - askedBestPipeline = stage('dagAskedBest'); + viewOffsetsPipeline = stage('dagViewOffsets'); const bindGroup = device.createBindGroup({ layout, entries: namedBufferEntries(DAG_BINDING, { @@ -69,7 +68,6 @@ export function createDagPipeline(device: GPUDevice, buffers: DagBuffers) { drawPrefixPipeline, drawScatterPipeline, viewOffsetsPipeline, - askedBestPipeline, bindGroup, }; }); diff --git a/packages/sdk-browser/src/gpu/dag/request.ts b/packages/sdk-browser/src/gpu/dag/request.ts index a5d5ae4f60..cdcbd0cbf1 100644 --- a/packages/sdk-browser/src/gpu/dag/request.ts +++ b/packages/sdk-browser/src/gpu/dag/request.ts @@ -66,5 +66,4 @@ fn quantizePriority(pixels:f32)->u32{ return u32(clamp(pas,0,${REQUEST_STEP_MAX})); } fn packRequest(page:u32,priority:u32)->u32{return (priority<u32{return word&((1u<>ASKED_BITS)==stamp){return;} - } - emitWord(page,priority,true); -} -/** After a frame's last light cut: each listed page at the best request its views made of it. */ -@compute @workgroup_size(64) -fn dagAskedBest(@builtin(global_invocation_id) id:vec3u){ - let s=id.x;if(s>=min(atomicLoad(&out.count),views[0u].listCap)){return;} - let page=requestPage(out.pages[s]); - out.pages[s]=packRequest(page,(atomicLoad(&work[askedWord(page)])&((1u<u32{return vi*views[0u].worldCount+w;} fn viewWord(row:u32,v:u32)->u32{return extraBase()+row*views[0u].viewCapacity+v;} /** The word behind the per-view rows: the most sixty-four-wide groups any view drew. */ fn drawnGroupsMax()->u32{return viewWord(${VIEW_WORD_ROWS}u,0u);} -/** A light cut's frame stamp, behind it, then each page's asked word: the stamp of the last frame - * that asked for \`page\` and its best priority there (\`dagWorkLayout\`, \`askedAt\`). */ -fn askedStamp()->u32{return drawnGroupsMax()+1u;} -fn askedWord(page:u32)->u32{return drawnGroupsMax()+2u+page;} fn dropWork(){atomicOr(&out.overflow,${WORK_DROPPED}u);} fn noteCoarser(){if(isLightCut()){atomicOr(&out.overflow,1u<<(${COARSER_VIEWS}u+vi));}} fn isLightCut()->bool{return (views[0u].viewFlags&VIEW_LIGHT)!=0u;} diff --git a/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts b/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts new file mode 100644 index 0000000000..73ba2d89a8 --- /dev/null +++ b/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts @@ -0,0 +1,53 @@ +// #525: a camera over a small scene, with a far distance much wider than the scene, rocks back and +// forth over the pages it reads. What a frame reads fits the pool many times over, so no page may be +// evicted and asked for again. The sun's floor pages were every last-level page within the view's +// far distance, receivers or not: over empty ground they pinned most of the pool, and the pages the +// camera came back to had been evicted meanwhile. +import test from 'node:test'; +import assert from 'node:assert/strict'; +import { createSceneLightStore } from '../light/store.ts'; +import { createShadowPlan } from './plan.ts'; +import { SUN, VIEW, report, sunPages } from './lightShadow.fixture.ts'; +import { sunFloorLevel, sunPageMetres } from './virtual.ts'; + +/** A scene twenty metres wide, under a view that sees two hundred metres around. */ +const BOX_MIN = [-10, 0, -10], + BOX_MAX = [10, 5, 10]; +const near = 0.001; +const viewAt = (x: number) => ({ + ...VIEW, + position: [x, 5, 0] as [number, number, number], + near, + far: 120, + pixelNear: (near * 2 * Math.tan(VIEW.halfFovY)) / 720, +}); + +test('a dolly over a small scene whose reads fit the pool evicts nothing it reads again', () => { + const store = createSceneLightStore(); + const plan = createShadowPlan(32); + store.add(SUN); + let floors = 0; + for (let frame = 0; frame < 60; frame++) { + // The camera rocks over two metres, and reads a block of fine pages under it. + const x = Math.abs((frame % 20) - 10) * 0.2, + view = viewAt(x); + plan.plan(store, view, BOX_MIN, BOX_MAX, frame, frame * 16); + plan.commit(); + const slice = store.sliceOf(0), + level = plan.sun.finest[slice] + 9, + // The sun stands overhead: a page's light-plane column is `-x` over its metres. + first = Math.floor(-x / sunPageMetres(level)) - 6, + block: number[][] = []; + for (let ay = -6; ay < 6; ay++) + for (let ax = first; ax < first + 12; ax++) block.push([ax, ay]); + report(plan, store, frame, sunPages(plan, slice, level, block)); + const floor = sunFloorLevel(plan.sun.finest[slice]); + floors = 0; + for (let page = 0; page < plan.pool.pages; page++) + if (plan.pool.owner[page] >= 0 && plan.pool.view[page] === floor) floors++; + } + // The box's twenty metres in the floor's four-metre pages, one page around: 8 × 8, where the + // view's reach held 880. + assert.ok(floors <= 64, `the floor holds the scene's pages, not the view's reach: ${floors}`); + assert.equal(plan.pool.refetched, 0, 'no page evicted and asked for again'); +}); diff --git a/packages/sdk-core/src/scene/light-shadow/requests.ts b/packages/sdk-core/src/scene/light-shadow/requests.ts index 77b0266f85..be11b0f788 100644 --- a/packages/sdk-core/src/scene/light-shadow/requests.ts +++ b/packages/sdk-core/src/scene/light-shadow/requests.ts @@ -159,9 +159,10 @@ export function createShadowRequests( needs.allocate(reportFrame, nowMs, frame, counts); }, /** Asks, as if the latest report named them, for the floor pages a reader may need that no - * report names yet: every sun's within the view's far distance (`sun.floorReach`), whatever - * moved, and each face's of a lamp posed after that report — new, moved or reshaped: what it - * named was read at a past pose. Evicts only what it did not name; the next may evict it. */ + * report names yet: every sun's over the scene within the view's far distance + * (`sun.floorReach`), whatever moved, and each face's of a lamp posed after that report — new, + * moved or reshaped: what it named was read at a past pose. Evicts only what it did not name; + * the next may evict it. */ floors(posed: ArrayLike, view: ShadowViewpoint, nowMs: number, frame: number) { reportFrame = counts.latest; for (let slice = 0; slice < posed.length; slice++) { diff --git a/packages/sdk-core/src/scene/light-shadow/sunLevels.ts b/packages/sdk-core/src/scene/light-shadow/sunLevels.ts index 36828e4b50..ef90ad00c1 100644 --- a/packages/sdk-core/src/scene/light-shadow/sunLevels.ts +++ b/packages/sdk-core/src/scene/light-shadow/sunLevels.ts @@ -14,6 +14,9 @@ import { /** Frames of layout kept to read a request report back: deeper than any readback lag. */ const HISTORY = 8; const LEVEL_WORDS = SUN_LEVELS * 2; +/** A light-plane rectangle nothing is in yet, and one that bounds nothing (no scene box). */ +const NO_RECT = [Infinity, -Infinity, Infinity, -Infinity], + ALL_RECT = NO_RECT.map((x) => -x); /** * THE CLIPMAP OF EACH SUN: its light-plane frame, the depth range its maps span, its finest @@ -32,6 +35,8 @@ const LEVEL_WORDS = SUN_LEVELS * 2; export function createSunLevels() { const frame = new Float64Array(MAX_SHADOW_SLICES * 9), depth = new Float64Array(MAX_SHADOW_SLICES * 2), + /** The scene box's rectangle on the light plane, `u0, u1, v0, v1` in metres, as pages count. */ + extent = new Float64Array(MAX_SHADOW_SLICES * 4), finest = new Int32Array(MAX_SHADOW_SLICES), origins = new Int32Array(MAX_SHADOW_SLICES * LEVEL_WORDS), /** Clipmap slots whose extent moved at the last update, one bit per slot. */ @@ -40,7 +45,8 @@ export function createSunLevels() { pastFinest = new Int32Array(MAX_SHADOW_SLICES * HISTORY), pastOrigins = new Int32Array(MAX_SHADOW_SLICES * HISTORY * LEVEL_WORDS); const right = new Float64Array(3), - up = new Float64Array(3); + up = new Float64Array(3), + corner3 = new Float64Array(3); /** Level held in slot `slot` while the finest level is `low`. */ const levelIn = (low: number, slot: number) => low + ringOf(slot - low, SUN_LEVELS); return { @@ -75,14 +81,23 @@ export function createSunLevels() { } let low = Infinity, high = -Infinity; + const e = slice * 4; + extent.set(NO_RECT, e); for (let corner = 0; corner < 8; corner++) { - const z = - axis[0] * (corner & 1 ? boxMax[0] : boxMin[0]) + - axis[1] * (corner & 2 ? boxMax[1] : boxMin[1]) + - axis[2] * (corner & 4 ? boxMax[2] : boxMin[2]); + corner3[0] = corner & 1 ? boxMax[0] : boxMin[0]; + corner3[1] = corner & 2 ? boxMax[1] : boxMin[1]; + corner3[2] = corner & 4 ? boxMax[2] : boxMin[2]; + const z = dotVector3(corner3, axis), + cu = dotVector3(corner3, right), + cv = -dotVector3(corner3, up); low = Math.min(low, z); high = Math.max(high, z); + extent[e] = Math.min(extent[e], cu); + extent[e + 1] = Math.max(extent[e + 1], cu); + extent[e + 2] = Math.min(extent[e + 2], cv); + extent[e + 3] = Math.max(extent[e + 3], cv); } + if (!Number.isFinite(low)) extent.set(ALL_RECT, e); if (Number.isFinite(low) && Number.isFinite(high)) { const grid = 2 ** Math.ceil(Math.log2(Math.max(high - low, 1e-6))); const zNear = Math.floor(low / grid) * grid, @@ -127,9 +142,12 @@ export function createSunLevels() { ay - origins[at + 1] < SUN_WINDOW ); }, - /** The floor pages the view can read, whatever it looks at — every page of the last level - * within its far distance of the camera, in this frame's clipmap —: `x0, y0, x1, y1` into - * `out`, inclusive. */ + /** + * The floor pages the view can read, whatever it looks at — every last-level page within its + * far distance and over the scene's box, one page around it for the normal offset and the PCF, + * in this frame's clipmap —: `x0, y0, x1, y1` into `out`, inclusive. No receiver lies past the + * box: floor pages there held nothing, and half the pool over a small scene (#525). + */ floorReach(slice: number, view: ShadowViewpoint, out: Int32Array) { const level = sunFloorLevel(finest[slice]), page = sunPageMetres(level), @@ -137,10 +155,21 @@ export function createSunLevels() { at = slice * LEVEL_WORDS + ringOf(level, SUN_LEVELS) * 2; const u = dotVector3(view.position, frame, 0, f), v = -dotVector3(view.position, frame, 0, f + 3); - out[0] = Math.max(origins[at], Math.floor((u - view.far) / page)); - out[1] = Math.max(origins[at + 1], Math.floor((v - view.far) / page)); - out[2] = Math.min(origins[at] + SUN_WINDOW - 1, Math.floor((u + view.far) / page)); - out[3] = Math.min(origins[at + 1] + SUN_WINDOW - 1, Math.floor((v + view.far) / page)); + const e = slice * 4; + for (let k = 0; k < 2; k++) { + const c = k ? v : u, + o = origins[at + k]; + out[k] = Math.max( + o, + Math.floor((c - view.far) / page), + Math.floor(extent[e + 2 * k] / page) - 1, + ); + out[k + 2] = Math.min( + o + SUN_WINDOW - 1, + Math.floor((c + view.far) / page), + Math.floor(extent[e + 2 * k + 1] / page) + 1, + ); + } return out; }, /** From 0101f64c7b98fade31f48d302f55e75b08f4f5a9 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 03:02:44 +0200 Subject: [PATCH 11/13] refactor(shadows): one light-plane rectangle for the floor's reach and the staling boxes (#525) --- .../scene/light-shadow/dollyRefetch.test.ts | 9 ++-- .../src/scene/light-shadow/invalidate.ts | 26 ++--------- .../sdk-core/src/scene/light-shadow/math.ts | 28 ++++++++++++ .../src/scene/light-shadow/sunLevels.ts | 44 +++++++------------ 4 files changed, 51 insertions(+), 56 deletions(-) diff --git a/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts b/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts index 73ba2d89a8..d68fd296ea 100644 --- a/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts +++ b/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts @@ -26,7 +26,6 @@ test('a dolly over a small scene whose reads fit the pool evicts nothing it read const store = createSceneLightStore(); const plan = createShadowPlan(32); store.add(SUN); - let floors = 0; for (let frame = 0; frame < 60; frame++) { // The camera rocks over two metres, and reads a block of fine pages under it. const x = Math.abs((frame % 20) - 10) * 0.2, @@ -41,11 +40,11 @@ test('a dolly over a small scene whose reads fit the pool evicts nothing it read for (let ay = -6; ay < 6; ay++) for (let ax = first; ax < first + 12; ax++) block.push([ax, ay]); report(plan, store, frame, sunPages(plan, slice, level, block)); - const floor = sunFloorLevel(plan.sun.finest[slice]); - floors = 0; - for (let page = 0; page < plan.pool.pages; page++) - if (plan.pool.owner[page] >= 0 && plan.pool.view[page] === floor) floors++; } + const floor = sunFloorLevel(plan.sun.finest[store.sliceOf(0)]); + let floors = 0; + for (let page = 0; page < plan.pool.pages; page++) + if (plan.pool.owner[page] >= 0 && plan.pool.view[page] === floor) floors++; // The box's twenty metres in the floor's four-metre pages, one page around: 8 × 8, where the // view's reach held 880. assert.ok(floors <= 64, `the floor holds the scene's pages, not the view's reach: ${floors}`); diff --git a/packages/sdk-core/src/scene/light-shadow/invalidate.ts b/packages/sdk-core/src/scene/light-shadow/invalidate.ts index 9752e5dd2f..904b3a2e09 100644 --- a/packages/sdk-core/src/scene/light-shadow/invalidate.ts +++ b/packages/sdk-core/src/scene/light-shadow/invalidate.ts @@ -1,6 +1,7 @@ import { LIGHT_KIND, type SceneLight } from '../light/contracts.ts'; import type { createShadowChanges } from './changes.ts'; import { writeFace } from './faces.ts'; +import { sunBoxRect } from './math.ts'; import type { ShadowPool } from './pool.ts'; import type { ShadowTable } from './table.ts'; import type { SunLevels } from './sunLevels.ts'; @@ -55,33 +56,14 @@ function lampPageMeets(face: number, mip: number, x: number, y: number) { ); } -/** Writes the light-plane rectangle of a world box under the sun of `slice` into `rects[0..4)`. */ -function sunRect(sun: SunLevels, slice: number, min: ArrayLike, max: ArrayLike) { - const f = slice * 9, - frame = sun.frame; - rects[0] = rects[2] = Infinity; - rects[1] = rects[3] = -Infinity; - for (let corner = 0; corner < 8; corner++) { - const x = corner & 1 ? max[0] : min[0], - y = corner & 2 ? max[1] : min[1], - z = corner & 4 ? max[2] : min[2]; - const u = frame[f] * x + frame[f + 1] * y + frame[f + 2] * z, - v = frame[f + 3] * x + frame[f + 4] * y + frame[f + 5] * z; - rects[0] = Math.min(rects[0], u); - rects[1] = Math.max(rects[1], u); - rects[2] = Math.min(rects[2], v); - rects[3] = Math.max(rects[3], v); - } -} - /** True when sun page `(level, ax, ay)` — rows down the `up` axis — meets `rects[0..4)`. */ function sunPageMeets(level: number, ax: number, ay: number) { const page = sunPageMetres(level); return ( rects[1] >= ax * page && rects[0] <= (ax + 1) * page && - -rects[2] >= ay * page && - -rects[3] <= (ay + 1) * page + rects[3] >= ay * page && + rects[2] <= (ay + 1) * page ); } @@ -129,7 +111,7 @@ export function invalidateLightPages( const read = whole ? undefined : changes.read(box), moved = byPage ? read : undefined, wrong = !read || (!read.detail && !read.moving); - if (moved && isSun) sunRect(sun, slice, moved.min, moved.max); + if (moved && isSun) sunBoxRect(sun.frame, slice * 9, moved.min, moved.max, rects, 0); if (moved && !isSun) for (let face = 0; face < faces; face++) faceRect(face, moved.min, moved.max); for (let page = 0; page < pool.pages; page++) { diff --git a/packages/sdk-core/src/scene/light-shadow/math.ts b/packages/sdk-core/src/scene/light-shadow/math.ts index 78e2d69999..254acf9dd6 100644 --- a/packages/sdk-core/src/scene/light-shadow/math.ts +++ b/packages/sdk-core/src/scene/light-shadow/math.ts @@ -154,3 +154,31 @@ export function composeFace( multiplyMatrix4(faceScratch, faceProjection, faceView); copyMatrix4(out, faceScratch, base); } + +/** + * The rectangle a world box covers on a sun's light plane, `u0, u1, v0, v1` in metres from `out[o]`: + * `u` along `right` (`frame[f..f+3)`), `v` DOWN `up` (`frame[f+3..f+6)`), as sun pages count their + * rows (`sunLevels.ts`). + */ +export function sunBoxRect( + frame: ArrayLike, + f: number, + min: ArrayLike, + max: ArrayLike, + out: Float64Array, + o: number, +) { + out[o] = out[o + 2] = Infinity; + out[o + 1] = out[o + 3] = -Infinity; + for (let corner = 0; corner < 8; corner++) { + const x = corner & 1 ? max[0] : min[0], + y = corner & 2 ? max[1] : min[1], + z = corner & 4 ? max[2] : min[2]; + const u = frame[f] * x + frame[f + 1] * y + frame[f + 2] * z, + v = -(frame[f + 3] * x + frame[f + 4] * y + frame[f + 5] * z); + out[o] = Math.min(out[o], u); + out[o + 1] = Math.max(out[o + 1], u); + out[o + 2] = Math.min(out[o + 2], v); + out[o + 3] = Math.max(out[o + 3], v); + } +} diff --git a/packages/sdk-core/src/scene/light-shadow/sunLevels.ts b/packages/sdk-core/src/scene/light-shadow/sunLevels.ts index ef90ad00c1..686d47e799 100644 --- a/packages/sdk-core/src/scene/light-shadow/sunLevels.ts +++ b/packages/sdk-core/src/scene/light-shadow/sunLevels.ts @@ -1,6 +1,6 @@ import { dotVector3 } from '../../math/primitives/vector.ts'; import { MAX_SHADOW_SLICES, type ShadowViewpoint } from '../light/contracts.ts'; -import { faceFrame } from './math.ts'; +import { faceFrame, sunBoxRect } from './math.ts'; import { SUN_LEVELS, SUN_LEVEL_ENTRIES, @@ -14,9 +14,8 @@ import { /** Frames of layout kept to read a request report back: deeper than any readback lag. */ const HISTORY = 8; const LEVEL_WORDS = SUN_LEVELS * 2; -/** A light-plane rectangle nothing is in yet, and one that bounds nothing (no scene box). */ -const NO_RECT = [Infinity, -Infinity, Infinity, -Infinity], - ALL_RECT = NO_RECT.map((x) => -x); +/** The light-plane rectangle of no box: it bounds nothing. */ +const UNBOUNDED = [-Infinity, Infinity, -Infinity, Infinity]; /** * THE CLIPMAP OF EACH SUN: its light-plane frame, the depth range its maps span, its finest @@ -35,7 +34,7 @@ const NO_RECT = [Infinity, -Infinity, Infinity, -Infinity], export function createSunLevels() { const frame = new Float64Array(MAX_SHADOW_SLICES * 9), depth = new Float64Array(MAX_SHADOW_SLICES * 2), - /** The scene box's rectangle on the light plane, `u0, u1, v0, v1` in metres, as pages count. */ + /** The scene box's rectangle on the light plane, `u0, u1, v0, v1` in metres (`sunBoxRect`). */ extent = new Float64Array(MAX_SHADOW_SLICES * 4), finest = new Int32Array(MAX_SHADOW_SLICES), origins = new Int32Array(MAX_SHADOW_SLICES * LEVEL_WORDS), @@ -45,8 +44,7 @@ export function createSunLevels() { pastFinest = new Int32Array(MAX_SHADOW_SLICES * HISTORY), pastOrigins = new Int32Array(MAX_SHADOW_SLICES * HISTORY * LEVEL_WORDS); const right = new Float64Array(3), - up = new Float64Array(3), - corner3 = new Float64Array(3); + up = new Float64Array(3); /** Level held in slot `slot` while the finest level is `low`. */ const levelIn = (low: number, slot: number) => low + ringOf(slot - low, SUN_LEVELS); return { @@ -81,23 +79,16 @@ export function createSunLevels() { } let low = Infinity, high = -Infinity; - const e = slice * 4; - extent.set(NO_RECT, e); for (let corner = 0; corner < 8; corner++) { - corner3[0] = corner & 1 ? boxMax[0] : boxMin[0]; - corner3[1] = corner & 2 ? boxMax[1] : boxMin[1]; - corner3[2] = corner & 4 ? boxMax[2] : boxMin[2]; - const z = dotVector3(corner3, axis), - cu = dotVector3(corner3, right), - cv = -dotVector3(corner3, up); + const z = + axis[0] * (corner & 1 ? boxMax[0] : boxMin[0]) + + axis[1] * (corner & 2 ? boxMax[1] : boxMin[1]) + + axis[2] * (corner & 4 ? boxMax[2] : boxMin[2]); low = Math.min(low, z); high = Math.max(high, z); - extent[e] = Math.min(extent[e], cu); - extent[e + 1] = Math.max(extent[e + 1], cu); - extent[e + 2] = Math.min(extent[e + 2], cv); - extent[e + 3] = Math.max(extent[e + 3], cv); } - if (!Number.isFinite(low)) extent.set(ALL_RECT, e); + sunBoxRect(frame, f, boxMin, boxMax, extent, slice * 4); + if (!Number.isFinite(low)) extent.set(UNBOUNDED, slice * 4); if (Number.isFinite(low) && Number.isFinite(high)) { const grid = 2 ** Math.ceil(Math.log2(Math.max(high - low, 1e-6))); const zNear = Math.floor(low / grid) * grid, @@ -155,19 +146,14 @@ export function createSunLevels() { at = slice * LEVEL_WORDS + ringOf(level, SUN_LEVELS) * 2; const u = dotVector3(view.position, frame, 0, f), v = -dotVector3(view.position, frame, 0, f + 3); - const e = slice * 4; for (let k = 0; k < 2; k++) { const c = k ? v : u, - o = origins[at + k]; - out[k] = Math.max( - o, - Math.floor((c - view.far) / page), - Math.floor(extent[e + 2 * k] / page) - 1, - ); + o = origins[at + k], + e = slice * 4 + 2 * k; + out[k] = Math.max(o, Math.floor(Math.max(c - view.far, extent[e] - page) / page)); out[k + 2] = Math.min( o + SUN_WINDOW - 1, - Math.floor((c + view.far) / page), - Math.floor(extent[e + 2 * k + 1] / page) + 1, + Math.floor(Math.min(c + view.far, extent[e + 1] + page) / page), ); } return out; From 7da22029c1e92269d2dc743522b5e16c418be797 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 03:07:50 +0200 Subject: [PATCH 12/13] test(shadows): the floor holds every page over the box, and no box bounds nothing (#525) --- .../scene/light-shadow/dollyRefetch.test.ts | 11 ++++++++- .../src/scene/light-shadow/sunLevels.test.ts | 23 +++++++++++++++++++ 2 files changed, 33 insertions(+), 1 deletion(-) diff --git a/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts b/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts index d68fd296ea..f19444745a 100644 --- a/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts +++ b/packages/sdk-core/src/scene/light-shadow/dollyRefetch.test.ts @@ -8,7 +8,7 @@ import assert from 'node:assert/strict'; import { createSceneLightStore } from '../light/store.ts'; import { createShadowPlan } from './plan.ts'; import { SUN, VIEW, report, sunPages } from './lightShadow.fixture.ts'; -import { sunFloorLevel, sunPageMetres } from './virtual.ts'; +import { PAGE_VALID, sunEntry, sunFloorLevel, sunPageMetres } from './virtual.ts'; /** A scene twenty metres wide, under a view that sees two hundred metres around. */ const BOX_MIN = [-10, 0, -10], @@ -48,5 +48,14 @@ test('a dolly over a small scene whose reads fit the pool evicts nothing it read // The box's twenty metres in the floor's four-metre pages, one page around: 8 × 8, where the // view's reach held 880. assert.ok(floors <= 64, `the floor holds the scene's pages, not the view's reach: ${floors}`); + // No receiver loses its floor: the box is symmetric, so its pages are the same whichever way the + // plane's axes point. + const page = sunPageMetres(floor), + slice = store.sliceOf(0); + for (let ay = Math.floor(-10 / page); ay <= Math.floor(10 / page); ay++) + for (let ax = Math.floor(-10 / page); ax <= Math.floor(10 / page); ax++) { + const entry = plan.table.baseOf(slice) + sunEntry(floor, ax, ay); + assert.ok(plan.table.words[entry] & PAGE_VALID, `floor page ${ax},${ay} over the box`); + } assert.equal(plan.pool.refetched, 0, 'no page evicted and asked for again'); }); diff --git a/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts b/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts index 5e7717c2a7..6f4e5875ee 100644 --- a/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts +++ b/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts @@ -60,3 +60,26 @@ test('a step smaller than a page moves no extent; a step of one finest page move assert.equal(movedNow()[0], true); assert.equal(movedNow()[15], false, 'the coarsest extent holds'); }); + +test('a box that bounds nothing leaves the floor the whole view reach', () => { + const sun = createSunLevels(), + bounded = new Int32Array(4), + reach = new Int32Array(4); + // No box yet, and a plane unbounded along x: neither bounds the floor. + for (const [min, max] of [ + [ + [Infinity, Infinity, Infinity], + [-Infinity, -Infinity, -Infinity], + ], + [ + [-Infinity, 0, -10], + [Infinity, 0, 10], + ], + ]) { + sun.update(0, AXIS, at(0), min, max, 1); + sun.floorReach(0, at(0), reach); + sun.update(0, AXIS, at(0), [-1e9, -1, -1e9], [1e9, 1, 1e9], 1); + sun.floorReach(0, at(0), bounded); + assert.deepEqual(Array.from(reach), Array.from(bounded), `box ${min} … ${max}`); + } +}); From 381e4213bdf431adfebe20129d3c7eb6b06e7528 Mon Sep 17 00:00:00 2001 From: Pasquelin Alban Date: Sat, 26 Sep 2026 03:23:46 +0200 Subject: [PATCH 13/13] fix(shadows): the scene box follows engine pose moves, and a box unbounded along the sun's axis bounds no floor (#525) --- .../src/lighting/direct/shadowFactorWgsl.ts | 3 ++- .../visibility/shader/spriteShadowCut.test.ts | 11 +++++++++++ .../src/webgpu/pages/render/encodeShadows.ts | 2 +- .../src/webgpu/pages/state/lights.ts | 2 +- .../sdk-browser/src/webgpu/shadow/sceneBox.ts | 17 +++++++++++------ .../src/scene/light-shadow/requests.ts | 3 ++- .../src/scene/light-shadow/sunLevels.test.ts | 7 ++++++- .../src/scene/light-shadow/sunLevels.ts | 19 +++++++++++-------- 8 files changed, 45 insertions(+), 19 deletions(-) diff --git a/packages/sdk-browser/src/lighting/direct/shadowFactorWgsl.ts b/packages/sdk-browser/src/lighting/direct/shadowFactorWgsl.ts index 7b0cd666ea..9c03819740 100644 --- a/packages/sdk-browser/src/lighting/direct/shadowFactorWgsl.ts +++ b/packages/sdk-browser/src/lighting/direct/shadowFactorWgsl.ts @@ -25,7 +25,8 @@ export const SHADOW_DEPTH_ROUNDING = 2 * 2 ** -23; * as the texture streamer falls back to a coarser tile. The scheduler keeps the last level under * every page a receiver reads mapped and drawn in the frame (`admit.ts`), so the far-shadow ray * of a sun and the unshadowed answer of a lamp past their last level only answer before a light's - * first request report. + * first request report — and, for a sun, past the scene's box, where no caster lies and the floor + * is asked only by the report (`sunLevels.ts` floorReach). * * The receiver's plane must not shade itself over the PCF's reach: its depth's slope across the * map is covered, in texels of the level read, by a depth margin up to a slope of 1 diff --git a/packages/sdk-browser/src/visibility/shader/spriteShadowCut.test.ts b/packages/sdk-browser/src/visibility/shader/spriteShadowCut.test.ts index e000026bec..ecf96daa5c 100644 --- a/packages/sdk-browser/src/visibility/shader/spriteShadowCut.test.ts +++ b/packages/sdk-browser/src/visibility/shader/spriteShadowCut.test.ts @@ -73,3 +73,14 @@ test('the shadow scene box leaves every sprite root out', () => { }); assert.deepEqual([...min, ...max], [-1, -1, -1, 1, 1, 1]); }); + +test('the shadow scene box follows a pose the engine moved, with no table change', () => { + const sceneBox = createShadowSceneBox(), + worldBox = Float64Array.of(-1, -1, -1, 1, 1, 1), + layout = { selectionRoots: [{ worldBox }], rows: { tableEpoch: 0 } }; + sceneBox(layout, 0); + // A placement or engine pose moves the root's box and bumps the scene revision alone. + worldBox.set([9, -1, -1, 11, 1, 1]); + const { min, max } = sceneBox(layout, 1); + assert.deepEqual([...min, ...max], [9, -1, -1, 11, 1, 1]); +}); diff --git a/packages/sdk-browser/src/webgpu/pages/render/encodeShadows.ts b/packages/sdk-browser/src/webgpu/pages/render/encodeShadows.ts index 73c04507a9..e65fc50d34 100644 --- a/packages/sdk-browser/src/webgpu/pages/render/encodeShadows.ts +++ b/packages/sdk-browser/src/webgpu/pages/render/encodeShadows.ts @@ -89,7 +89,7 @@ export function planShadowRegions( // The light cuts measure their error at the camera's threshold. lights.shadowPixelError = followLightThreshold(lights, rt.run.gate.pixelError); const view = shadowViewpointOf(cam, rt.gpu.targetSize[1]); - const box = lights.sceneBox(rt.layout); + const box = lights.sceneBox(rt.layout, rt.run.gate.revisions.scene); ensureStaticLayer(rt); redrawShortPages(rt, frame, nowMs, residencyMoved); const count = plan.plan(store, view, box.min, box.max, frame, nowMs); diff --git a/packages/sdk-browser/src/webgpu/pages/state/lights.ts b/packages/sdk-browser/src/webgpu/pages/state/lights.ts index 35f61236cc..ce15db881a 100644 --- a/packages/sdk-browser/src/webgpu/pages/state/lights.ts +++ b/packages/sdk-browser/src/webgpu/pages/state/lights.ts @@ -49,7 +49,7 @@ export interface WebgpuLightState { /** The static layer's page pyramids and the test of the moving casters against them. */ pageHiz: ShadowPageHiz | undefined; occlusion: ShadowOcclusion | undefined; - /** The scene's world box, what a sun's depth range spans (`../../shadow/sceneBox.ts`). */ + /** The scene's world box, what a sun's depth range and floor span (`../../shadow/sceneBox.ts`). */ sceneBox: ReturnType; /** Per-page cull and the world spheres it reads; absent while the pool does not exist. */ cull: GpuShadowCull | undefined; diff --git a/packages/sdk-browser/src/webgpu/shadow/sceneBox.ts b/packages/sdk-browser/src/webgpu/shadow/sceneBox.ts index 5c650c9498..5950460862 100644 --- a/packages/sdk-browser/src/webgpu/shadow/sceneBox.ts +++ b/packages/sdk-browser/src/webgpu/shadow/sceneBox.ts @@ -7,10 +7,13 @@ interface SceneRoots { } /** - * The world box of every opaque primitive the scene draws: what a sun's clipmap spans along its - * axis, so every caster lies inside its depth range. A sprite root (`ClusterRoot.sprite`) is left - * out: a sprite casts no shadow, and its box would only spread the range. Rebuilt only when a pose moved or the scene - * changed — the row-table epoch and the root list say so —, from boxes the engine already holds. + * The world box of every primitive the scene draws: what a sun's clipmap spans along + * its axis, so every caster lies inside its depth range, and the rectangle its floor pages cover + * on its plane (`sunLevels.ts` floorReach). A sprite root (`ClusterRoot.sprite`) is left out: a + * sprite casts no shadow, and its box would only spread the range. Rebuilt only when a pose moved + * or the scene changed — the scene revision, the row-table epoch and the root list say so: an + * engine pose or placement move bumps the scene revision alone —, from boxes the engine already + * holds. */ export function createShadowSceneBox() { const box = new Float64Array(6), @@ -18,11 +21,13 @@ export function createShadowSceneBox() { max = box.subarray(3, 6), read = { min, max }; let epoch = -1, + revision = -1, roots: unknown = undefined; - return (layout: SceneRoots) => { + return (layout: SceneRoots, sceneRevision = 0) => { const { selectionRoots, rows } = layout; - if (rows.tableEpoch !== epoch || selectionRoots !== roots) { + if (rows.tableEpoch !== epoch || sceneRevision !== revision || selectionRoots !== roots) { epoch = rows.tableEpoch; + revision = sceneRevision; roots = selectionRoots; boxEmpty(box, 0); for (const root of selectionRoots) diff --git a/packages/sdk-core/src/scene/light-shadow/requests.ts b/packages/sdk-core/src/scene/light-shadow/requests.ts index be11b0f788..79ababd7b2 100644 --- a/packages/sdk-core/src/scene/light-shadow/requests.ts +++ b/packages/sdk-core/src/scene/light-shadow/requests.ts @@ -46,7 +46,8 @@ const CAP: number = LIGHT_SETTINGS.shadowRequestCap; * never evicted while anything above it is read; like every page named, it is drawn in the frame it * goes stale (`admit.ts`). The * floor covers all the light reaches, so it needs no report to know what the view will read: a - * sun asks every frame for the floor pages its view reaches (`floors`), and a new, moved or + * sun asks every frame for the floor pages its view reaches over the scene's box (`floors`) — past + * it no caster lies, and a receiver there asks through the report —, and a new, moved or * reshaped lamp for each face's until a report written at its pose is read — a report from a past * pose names only the pages that pose's receivers read. * diff --git a/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts b/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts index 6f4e5875ee..3b386a2056 100644 --- a/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts +++ b/packages/sdk-core/src/scene/light-shadow/sunLevels.test.ts @@ -65,7 +65,7 @@ test('a box that bounds nothing leaves the floor the whole view reach', () => { const sun = createSunLevels(), bounded = new Int32Array(4), reach = new Int32Array(4); - // No box yet, and a plane unbounded along x: neither bounds the floor. + // No box yet, a plane unbounded along x, a box unbounded along the axis: none bounds the floor. for (const [min, max] of [ [ [Infinity, Infinity, Infinity], @@ -75,6 +75,11 @@ test('a box that bounds nothing leaves the floor the whole view reach', () => { [-Infinity, 0, -10], [Infinity, 0, 10], ], + // Unbounded down the axis only: the near end is finite, the far one is not. + [ + [-10, -Infinity, -10], + [10, 0, 10], + ], ]) { sun.update(0, AXIS, at(0), min, max, 1); sun.floorReach(0, at(0), reach); diff --git a/packages/sdk-core/src/scene/light-shadow/sunLevels.ts b/packages/sdk-core/src/scene/light-shadow/sunLevels.ts index 686d47e799..1c17639dac 100644 --- a/packages/sdk-core/src/scene/light-shadow/sunLevels.ts +++ b/packages/sdk-core/src/scene/light-shadow/sunLevels.ts @@ -35,7 +35,7 @@ export function createSunLevels() { const frame = new Float64Array(MAX_SHADOW_SLICES * 9), depth = new Float64Array(MAX_SHADOW_SLICES * 2), /** The scene box's rectangle on the light plane, `u0, u1, v0, v1` in metres (`sunBoxRect`). */ - extent = new Float64Array(MAX_SHADOW_SLICES * 4), + boxRect = new Float64Array(MAX_SHADOW_SLICES * 4), finest = new Int32Array(MAX_SHADOW_SLICES), origins = new Int32Array(MAX_SHADOW_SLICES * LEVEL_WORDS), /** Clipmap slots whose extent moved at the last update, one bit per slot. */ @@ -87,9 +87,10 @@ export function createSunLevels() { low = Math.min(low, z); high = Math.max(high, z); } - sunBoxRect(frame, f, boxMin, boxMax, extent, slice * 4); - if (!Number.isFinite(low)) extent.set(UNBOUNDED, slice * 4); - if (Number.isFinite(low) && Number.isFinite(high)) { + // An empty or unbounded box bounds neither the depth range nor the floor. + if (!Number.isFinite(low) || !Number.isFinite(high)) boxRect.set(UNBOUNDED, slice * 4); + else { + sunBoxRect(frame, f, boxMin, boxMax, boxRect, slice * 4); const grid = 2 ** Math.ceil(Math.log2(Math.max(high - low, 1e-6))); const zNear = Math.floor(low / grid) * grid, zFar = Math.max(zNear + grid, Math.ceil(high / grid) * grid); @@ -136,8 +137,10 @@ export function createSunLevels() { /** * The floor pages the view can read, whatever it looks at — every last-level page within its * far distance and over the scene's box, one page around it for the normal offset and the PCF, - * in this frame's clipmap —: `x0, y0, x1, y1` into `out`, inclusive. No receiver lies past the - * box: floor pages there held nothing, and half the pool over a small scene (#525). + * in this frame's clipmap —: `x0, y0, x1, y1` into `out`, inclusive. No caster lies past the + * box: a receiver there is lit, and its floor pages held nothing but half the pool over a small + * scene (#525). A sprite, left out of the box, reads the far-shadow ray there until the page it + * asks for is drawn: lit too. */ floorReach(slice: number, view: ShadowViewpoint, out: Int32Array) { const level = sunFloorLevel(finest[slice]), @@ -150,10 +153,10 @@ export function createSunLevels() { const c = k ? v : u, o = origins[at + k], e = slice * 4 + 2 * k; - out[k] = Math.max(o, Math.floor(Math.max(c - view.far, extent[e] - page) / page)); + out[k] = Math.max(o, Math.floor(Math.max(c - view.far, boxRect[e] - page) / page)); out[k + 2] = Math.min( o + SUN_WINDOW - 1, - Math.floor(Math.min(c + view.far, extent[e + 1] + page) / page), + Math.floor(Math.min(c + view.far, boxRect[e + 1] + page) / page), ); } return out;