diff --git a/.github/workflows/gh-pages.yml b/.github/workflows/gh-pages.yml index 5bf51ed..2d3bf94 100644 --- a/.github/workflows/gh-pages.yml +++ b/.github/workflows/gh-pages.yml @@ -19,12 +19,10 @@ jobs: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 - - uses: actions/setup-node@v4 + - uses: denoland/setup-deno@v2 with: - node-version: '20' - cache: 'npm' - - run: npm ci - - run: npm run build:demo + deno-version: v2.x + - run: deno task build:demo - name: Stage site run: | mkdir -p _stage/demo diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml new file mode 100644 index 0000000..b80617f --- /dev/null +++ b/.github/workflows/test.yml @@ -0,0 +1,23 @@ +name: Test +on: + push: + branches: [main] + pull_request: + branches: [main] +permissions: + contents: read +jobs: + test: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v7 + - uses: denoland/setup-deno@v2 + with: + deno-version: v2.x + - uses: mlugg/setup-zig@v2 + with: + version: 0.16.0 + - run: deno lint + - run: deno task check + # GPU tests skip when no WebGPU adapter is available. + - run: deno task test:gpu diff --git a/README.md b/README.md index 7e0c74e..15da504 100644 --- a/README.md +++ b/README.md @@ -16,17 +16,14 @@ Compact sparse volumetric data format optimized for WebGPU real-time rendering. - **32-bit addressing** for better GPU compatibility - **Fast traversal** with hierarchical raymarching (HDDA) -This repository includes: -- `wgsl/picovdb.wgsl` - WGSL shader library -- `ts/picovdb.ts` - TypeScript loader -- `src/main.zig` - NanoVDB → PicoVDB converter -- `src/stl.zig`, `src/mesh_to_grid.zig` - STL mesh → PicoVDB level set voxelizer +This repository includes, WGSL shader library, Typescript loader, GPU Modelling +API, converter from NanoVDB and STL files. ## How It Works PicoVDB compresses NanoVDB files through: - **Rank query compression**: Bit masks + counts eliminate inactive voxel storage -- **32-bit offsets**: Replace 64-bit pointers with computed indices (limits to 4 billion active voxels) +- **32-bit offsets**: WebGPU compatible with 64-bit extensions - **GPU-aligned structs**: Minimize padding, maximize cache efficiency ## Usage @@ -59,6 +56,66 @@ fn main(@builtin(global_invocation_id) global_id: vec3u) { } ``` +## Modelling + +Grids can be edited with Constructive Solid Geometry (CSG). Modelling +`Op`'s apply to a `Solid` within a `Space`. Build solids from +primitives, then union, intersect, subtract, or offset them. See +`ts/model.ts` for the API. + +```ts +import { Space, box, cylinder, sphere } from '@emcfarlane/picovdb/model'; + +const space = new Space(device, { halfWidth: 3 }); + +// A bolt: a ball and a cylinder with a slot cut out. +using bolt = await space.solid(sphere([0, 0, 0], 20)) + .union(cylinder([0, -30, 0], [0, 30, 0], 6)) + .subtract(box([0, 0, 0], [30, 4, 4])); + +// A hollow bunny: grow by two voxels, subtract the original, and move it. +using bunny = space.fromPvdb(await (await fetch('bunny.pvdb')).arrayBuffer()); +using shell = await bunny.offset(2).subtract(bunny).translate([0, 0, -10]); + +const bytes = await shell.toPvdb(); +``` + +Shapes are WGSL distance functions. `sphere`, `box`, `capsule`, and +`cylinder` name the built-in ones. Add your own to a `Space` and use +them by name: + +```ts +const space = new Space(device, { + shapes: /* wgsl */ ` + fn torus(p: vec3) -> f32 { + let d = p - args[0].xyz; + let q = vec2(length(d.xz) - args[1].x, d.y); + return length(q) - args[1].y; + } + `, +}); +using ring = await space.solid({ + fn: 'torus', + args: [0, 0, 0, 0, 18, 6], // args[0] = center, args[1].xy = ring and tube radius + bounds: { min: [-24, -6, -24], max: [24, 6, 24] }, +}); +``` + +A function takes absolute voxel coordinates, reads its arguments from +`args`, an `array`, and returns the signed distance in voxels. +Adding needs bounds. Carving does not. + +**Try it in the demo.** The [live demo](https://emcfarlane.github.io/picovdb/demo/) +exposes `space`, `scene.solid`, and the shape functions in the browser +console. `scene.solid` is the loaded model. Assign a solid or an op to render it. The following +makes a half shell out of the model: + +```js +scene.solid = scene.solid.offset(2) + .subtract(scene.solid) + .subtract(box([4000, 0, 0], [4000, 4000, 4000])); +``` + ## Converting Files ```bash # Build converter diff --git a/demo/index.ts b/demo/index.ts index e052d6e..5ba0929 100644 --- a/demo/index.ts +++ b/demo/index.ts @@ -1,12 +1,14 @@ import { vec3, mat4 } from 'wgpu-matrix'; -import DisplayShader from "./blit.wgsl"; -import ComputeShader from "./compute.wgsl"; -import PicoVDBShader from "../wgsl/picovdb.wgsl"; -import { fetchPicoVDB } from '../ts/picovdb.ts'; -import { createOrbitCamera } from './lib/camera'; -import { createInputHandler } from "./lib/input"; -import { initGUI } from './lib/gui'; -import type { ModelConfig } from './lib/gui'; +import DisplayShader from "./blit.wgsl" with { type: "text" }; +import ComputeShader from "./compute.wgsl" with { type: "text" }; +import { picovdbWgsl as PicoVDBShader } from "picovdb/ts/shaders.ts"; +import { fetchPicoVDB, type PicoVDBFile, GRID_TYPE_SDF_FLOAT, PICOVDB_GRID_SIZE } from '../ts/picovdb.ts'; +import { gridLimits } from '../ts/gpu/device.ts'; +import { Space, Op, box, capsule, cylinder, sphere, type Solid, type PicoVDBTree } from '../ts/model.ts'; +import { createOrbitCamera } from './lib/camera.ts'; +import { createInputHandler } from "./lib/input.ts"; +import { initGUI } from './lib/gui.ts'; +import type { ModelConfig } from './lib/gui.ts'; const MODEL_BASE = './models/'; const models: ModelConfig[] = [ @@ -18,14 +20,14 @@ const models: ModelConfig[] = [ //{ name: 'Skeleton', url: `${MODEL_BASE}skeleton.pvdb.gz`, translation: [0, 240, 0], scale: 120 }, ]; -const modelParam = new URLSearchParams(window.location.search).get('model'); +const modelParam = new URLSearchParams(globalThis.location.search).get('model'); const initialModel = models.find(m => m.url.endsWith('/' + modelParam)) ?? models[0]; const { controls, modelController, pauseController, highDPIController, rotationController } = initGUI(models, initialModel.name); -import { createSkyState } from "./lib/hw_skymodel"; -import { computeSkyIrradianceSH } from "./lib/sky_irradiance"; -import { TimestampQueryManager } from './lib/TimestampQueryManager'; -import { Stats } from './lib/Stats'; +import { createSkyState } from "./lib/hw_skymodel.ts"; +import { computeSkyIrradianceSH } from "./lib/sky_irradiance.ts"; +import { TimestampQueryManager } from './lib/TimestampQueryManager.ts'; +import { Stats } from './lib/Stats.ts'; const canvas = document.getElementById("canvas") as HTMLCanvasElement; const infoTextElement = document.getElementById("info-text")!; @@ -34,7 +36,7 @@ if (!canvas) { throw new Error("No canvas found."); } if (!navigator.gpu) { - const isInsecure = window.isSecureContext === false; + const isInsecure = globalThis.isSecureContext === false; throw new Error( isInsecure ? "WebGPU requires a secure context (HTTPS or localhost). Current origin is not secure." @@ -43,9 +45,10 @@ if (!navigator.gpu) { } console.log("WebGPU is supported!"); +// featureLevel is newer than Deno's bundled WebGPU types. const adapter = await navigator.gpu.requestAdapter({ featureLevel: 'compatibility', -}); +} as unknown as GPURequestAdapterOptions); if (!adapter) { throw new Error("No appropriate GPUAdapter found."); } @@ -60,9 +63,9 @@ let passBindGroup: GPUBindGroup; // Set canvas to fullscreen size and recreate GPU resources function resizeCanvas() { - const pixelRatio = controls.highDPI ? window.devicePixelRatio : 1.0; - canvas.width = window.innerWidth * pixelRatio; - canvas.height = window.innerHeight * pixelRatio; + const pixelRatio = controls.highDPI ? globalThis.devicePixelRatio : 1.0; + canvas.width = globalThis.innerWidth * pixelRatio; + canvas.height = globalThis.innerHeight * pixelRatio; width = canvas.width; height = canvas.height; @@ -74,7 +77,7 @@ function resizeCanvas() { } resizeCanvas(); -window.addEventListener('resize', resizeCanvas); +globalThis.addEventListener('resize', resizeCanvas); // Update canvas size when High DPI setting changes highDPIController.onChange(() => { @@ -91,18 +94,19 @@ const supportsTimestampQueries = adapter?.features.has(timestampQueryFeature); const requiredFeatures: GPUFeatureName[] = []; if (supportsTimestampQueries) { requiredFeatures.push(timestampQueryFeature); } -const device = await adapter.requestDevice({ requiredFeatures: requiredFeatures }); +const device = await adapter.requestDevice({ requiredFeatures: requiredFeatures, requiredLimits: gridLimits(adapter) }); device.addEventListener('uncapturederror', event => { console.log(event.error); }); -const context = canvas.getContext("webgpu"); +// The DOM lib doesn't know the "webgpu" context id. +const context = canvas.getContext("webgpu") as unknown as GPUCanvasContext | null; if (!context) { throw new Error("No context found."); } -var stats = new Stats(); -var gpuPanel = stats.addPanel(new Stats.Panel('GPU', '#ff8', '#221')); +const stats = new Stats(); +const gpuPanel = stats.addPanel(new Stats.Panel('GPU', '#ff8', '#221')); document.body.appendChild(stats.dom); // GPU-side timer and the CPU-side counter where we accumulate statistics: @@ -154,6 +158,7 @@ let lowersBuffer: GPUBuffer; let leavesBuffer: GPUBuffer; let dataBuffer: GPUBuffer; let currentModelConfig: ModelConfig = models[0]; +let currentFile: PicoVDBFile | null = null; // Create size-dependent GPU resources function createGPUResources() { @@ -245,26 +250,14 @@ const displayPipeline = device.createRenderPipeline({ const inputHandler = createInputHandler(window, canvas); -// Load a PicoVDB model and create GPU buffers -async function loadModel(config: ModelConfig) { - infoTextElement.textContent = `Loading ${config.name}...`; - - const picoVDBFile = await fetchPicoVDB(config.url); - console.log('PicoVDB File loaded successfully:'); - console.log('PicoVDB File Header:'); - console.log(` Magic: [0x${picoVDBFile.header.magic[0].toString(16)}, 0x${picoVDBFile.header.magic[1].toString(16)}]`); - console.log(` Version: ${picoVDBFile.header.version}`); - console.log(` Grid Count: ${picoVDBFile.header.gridCount}`); - console.log(` Upper Count: ${picoVDBFile.header.upperCount}`); - console.log(` Lower Count: ${picoVDBFile.header.lowerCount}`); - console.log(` Leaf Count: ${picoVDBFile.header.leafCount}`); - console.log(` Data Count: ${picoVDBFile.header.dataCount} bytes`); - console.log(` Voxel Count: ${picoVDBFile.getVoxelCount()}`); - if (picoVDBFile.header.gridCount === 0) { - throw new Error('PicoVDB file contains no grids'); - } +function uploadBytes(label: string, bytes: Uint8Array): GPUBuffer { + const buffer = device.createBuffer({ label, size: bytes.byteLength, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(buffer, 0, bytes); + return buffer; +} - // Destroy old GPU buffers +// Swaps in a set of picovdb node buffers and rebuilds the data bind group. +function setGridBuffers(b: { grids: GPUBuffer; roots: GPUBuffer; uppers: GPUBuffer; lowers: GPUBuffer; leaves: GPUBuffer; data: GPUBuffer }) { if (gridsBuffer) { gridsBuffer.destroy(); rootsBuffer.destroy(); @@ -273,53 +266,12 @@ async function loadModel(config: ModelConfig) { leavesBuffer.destroy(); dataBuffer.destroy(); } - - // Create new GPU buffers - gridsBuffer = device.createBuffer({ - label: 'PicoVDB Grids', - size: picoVDBFile.gridsBuffer.byteLength, - usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST, - }); - device.queue.writeBuffer(gridsBuffer, 0, picoVDBFile.gridsBuffer); - - rootsBuffer = device.createBuffer({ - label: 'PicoVDB Roots', - size: picoVDBFile.rootsBuffer.byteLength, - usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST, - }); - device.queue.writeBuffer(rootsBuffer, 0, picoVDBFile.rootsBuffer); - - uppersBuffer = device.createBuffer({ - label: 'PicoVDB Uppers', - size: picoVDBFile.uppersBuffer.byteLength, - usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST, - }); - device.queue.writeBuffer(uppersBuffer, 0, picoVDBFile.uppersBuffer); - - lowersBuffer = device.createBuffer({ - label: 'PicoVDB Lowers', - size: picoVDBFile.lowersBuffer.byteLength, - usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST, - }); - device.queue.writeBuffer(lowersBuffer, 0, picoVDBFile.lowersBuffer); - - leavesBuffer = device.createBuffer({ - label: 'PicoVDB Leaves', - size: picoVDBFile.leavesBuffer.byteLength, - usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST, - }); - device.queue.writeBuffer(leavesBuffer, 0, picoVDBFile.leavesBuffer); - - dataBuffer = device.createBuffer({ - label: 'PicoVDB Data', - size: picoVDBFile.dataBuffer.byteLength, - usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST, - }); - device.queue.writeBuffer(dataBuffer, 0, picoVDBFile.dataBuffer); - - currentModelConfig = config; - - // Recreate data bind group with new buffers + gridsBuffer = b.grids; + rootsBuffer = b.roots; + uppersBuffer = b.uppers; + lowersBuffer = b.lowers; + leavesBuffer = b.leaves; + dataBuffer = b.data; dataBindGroup = device.createBindGroup({ label: 'Data bind group', layout: dataBindGroupLayout, @@ -332,6 +284,61 @@ async function loadModel(config: ModelConfig) { { binding: 5, resource: { buffer: dataBuffer } }, ] }); +} + +// Renders an emitted tree in place of the loaded model, keeping its transform. +function showTree(tree: PicoVDBTree, name: string) { + const grid = new ArrayBuffer(PICOVDB_GRID_SIZE); + new Uint32Array(grid, 0, 8).set([0, 0, 0, 0, 0, tree.dataElemCount, GRID_TYPE_SDF_FLOAT, 0]); + new Int32Array(grid, 32, 3).set(tree.indexBoundsMin); + new Int32Array(grid, 48, 3).set(tree.indexBoundsMax); + setGridBuffers({ + grids: uploadBytes('PicoVDB Grids', new Uint8Array(grid)), + roots: tree.roots, + uppers: tree.uppers, + lowers: tree.lowers, + leaves: tree.leaves, + data: tree.data, + }); + updateObjects(); + const size = tree.indexBoundsMax.map((v, a) => v - tree.indexBoundsMin[a]); + infoTextElement.textContent = `PicoVDB +${name} +Grid: ${size[0]} × ${size[1]} × ${size[2]} units +Voxels: ${tree.activeVoxels}`; +} + +// Load a PicoVDB model and create GPU buffers +async function loadModel(config: ModelConfig) { + infoTextElement.textContent = `Loading ${config.name}...`; + + const picoVDBFile = await fetchPicoVDB(config.url); + console.log('PicoVDB File loaded successfully:'); + console.log('PicoVDB File Header:'); + console.log(` Magic: [0x${picoVDBFile.header.magic[0].toString(16)}, 0x${picoVDBFile.header.magic[1].toString(16)}]`); + console.log(` Version: ${picoVDBFile.header.version}`); + console.log(` Grid Count: ${picoVDBFile.header.gridCount}`); + console.log(` Upper Count: ${picoVDBFile.header.upperCount}`); + console.log(` Lower Count: ${picoVDBFile.header.lowerCount}`); + console.log(` Leaf Count: ${picoVDBFile.header.leafCount}`); + console.log(` Data Count: ${picoVDBFile.header.dataCount} bytes`); + console.log(` Voxel Count: ${picoVDBFile.getVoxelCount()}`); + if (picoVDBFile.header.gridCount === 0) { + throw new Error('PicoVDB file contains no grids'); + } + + setGridBuffers({ + grids: uploadBytes('PicoVDB Grids', picoVDBFile.gridsBuffer), + roots: uploadBytes('PicoVDB Roots', picoVDBFile.rootsBuffer), + uppers: uploadBytes('PicoVDB Uppers', picoVDBFile.uppersBuffer), + lowers: uploadBytes('PicoVDB Lowers', picoVDBFile.lowersBuffer), + leaves: uploadBytes('PicoVDB Leaves', picoVDBFile.leavesBuffer), + data: uploadBytes('PicoVDB Data', picoVDBFile.dataBuffer), + }); + currentModelConfig = config; + currentFile = picoVDBFile; + sceneSolid?.destroy(); + sceneSolid = null; // Update transform for new model updateObjects(); @@ -594,11 +601,44 @@ createGPUResources(); // Load initial model (from URL param or first in list) await loadModel(initialModel); +// Modelling console. Try in devtools: +// scene.solid = scene.solid.offset(2).subtract(scene.solid).subtract(box([4000, 0, 0], [4000, 4000, 4000])); +globalThis.space = new Space(device); +Object.assign(globalThis, { sphere, box, capsule, cylinder }); +let sceneSolid: Solid | null = null; +globalThis.scene = { + /** The rendered model as a solid, loaded from the current file on first read. */ + get solid(): Solid { + if (!currentFile) throw new Error('no model loaded'); + return sceneSolid ??= space.fromPvdb(currentFile); + }, + /** Renders a solid, or the result of an op, in place of the loaded model. */ + set solid(value: Solid | Op) { + (async () => { + const t0 = performance.now(); + const solid = await value; + const tree = await solid.toTree(); + showTree(tree, value instanceof Op ? 'op' : 'solid'); + if (sceneSolid && sceneSolid !== solid) sceneSolid.destroy(); + sceneSolid = solid; + console.log(`scene: ${tree.leafCount} leaves, ${tree.activeVoxels} active, ${tree.surfaceVoxels} surface in ${(performance.now() - t0).toFixed(0)} ms`); + })().catch(console.error); + }, +}; +declare global { + var space: Space; + var scene: { solid: Solid | Op }; + var sphere: typeof import('../ts/model.ts').sphere; + var box: typeof import('../ts/model.ts').box; + var capsule: typeof import('../ts/model.ts').capsule; + var cylinder: typeof import('../ts/model.ts').cylinder; +} + // Wire up model switching — update URL and load modelController.onChange(async (name: string) => { const config = models.find(m => m.name === name)!; const filename = config.url.split('/').pop()!; - const url = new URL(window.location.href); + const url = new URL(globalThis.location.href); url.searchParams.set('model', filename); history.replaceState(null, '', url); await loadModel(config); diff --git a/demo/lib/Stats.ts b/demo/lib/Stats.ts index 7182924..8ac5170 100644 --- a/demo/lib/Stats.ts +++ b/demo/lib/Stats.ts @@ -1,3 +1,4 @@ +// deno-lint-ignore-file no-explicit-any // stats based on stats.js export class Panel { @@ -9,7 +10,7 @@ export class Panel { private min = Infinity; private max = 0; private round = Math.round; - private PR = Math.round(window.devicePixelRatio || 1); + private PR = Math.round(globalThis.devicePixelRatio || 1); private WIDTH = 80 * this.PR; private HEIGHT = 48 * this.PR; @@ -128,10 +129,7 @@ export class Stats { this.frames++; const time = (performance || Date).now(); - const frameTime = time - this.beginTime; - if (time >= this.prevTime + 1000) { - console.log(frameTime); this.fpsPanel.update((this.frames * 1000) / (time - this.prevTime), 100); this.prevTime = time; diff --git a/demo/lib/camera.ts b/demo/lib/camera.ts index 51326e7..a791c19 100644 --- a/demo/lib/camera.ts +++ b/demo/lib/camera.ts @@ -1,6 +1,6 @@ import type { Mat4, Vec3 } from 'wgpu-matrix'; import { mat4, vec3 } from 'wgpu-matrix'; -import type Input from './input.js'; +import type Input from './input.ts'; export interface OrbitCamera { update(dt: number, input: Input): Mat4; diff --git a/demo/lib/sky_irradiance.ts b/demo/lib/sky_irradiance.ts index b903d20..315b07d 100644 --- a/demo/lib/sky_irradiance.ts +++ b/demo/lib/sky_irradiance.ts @@ -6,7 +6,7 @@ // Environment Maps"). The shader then evaluates per-pixel irradiance with // 9 fused multiply-adds instead of sampling the sky model per pixel. -import { skyStateRadiance, Channel } from './hw_skymodel'; +import { skyStateRadiance, type Channel } from './hw_skymodel.ts'; const SOLAR_RADIUS_RADIANS = 0.004450589; // must match hw_skymodel.ts diff --git a/deno.json b/deno.json new file mode 100644 index 0000000..890096c --- /dev/null +++ b/deno.json @@ -0,0 +1,42 @@ +{ + "name": "@emcfarlane/picovdb", + "version": "0.0.1", + "exports": { + ".": "./ts/picovdb.ts", + "./stl": "./ts/stl.ts", + "./model": "./ts/model.ts", + "./shaders": "./ts/shaders.ts" + }, + "unstable": ["webgpu", "raw-imports"], + "nodeModulesDir": "none", + "imports": { + "picovdb/": "./", + "lil-gui": "npm:lil-gui@^0.21.0", + "wgpu-matrix": "npm:wgpu-matrix@^3.4.0" + }, + "compilerOptions": { + "lib": ["dom", "dom.iterable", "dom.asynciterable", "deno.webgpu", "deno.ns"], + "strict": true, + "noFallthroughCasesInSwitch": true + }, + "tasks": { + "check": "deno check ts/ demo/index.ts", + "build": "deno bundle --platform browser --minify -o dist/picovdb.js ts/picovdb.ts", + "build:demo": "deno bundle --platform browser -o demo/index.js demo/index.ts", + "watch:demo": "deno bundle --watch --platform browser -o demo/index.js demo/index.ts", + "serve": "deno run --allow-net --allow-read jsr:@std/http/file-server demo --port 8000", + "dev": "mkdir -p demo/models && cp data/*.pvdb.gz demo/models/ && (deno task watch:demo & deno task serve)", + "test:gpu": "deno test --allow-read=data,zig-out ts/" + }, + "exclude": [ + "node_modules", + "dist", + "zig-out", + "zig-pkg", + "chats", + "data", + "output", + "_stage", + "demo/index.js" + ] +} diff --git a/deno.lock b/deno.lock new file mode 100644 index 0000000..2516eac --- /dev/null +++ b/deno.lock @@ -0,0 +1,21 @@ +{ + "version": "5", + "specifiers": { + "npm:lil-gui@0.21": "0.21.0", + "npm:wgpu-matrix@^3.4.0": "3.4.2" + }, + "npm": { + "lil-gui@0.21.0": { + "integrity": "sha512-tpvxN7v1GvE/Tv+GRopfOp0W7fVEjF4PltkuX8vOCIfim22rD1ztvfkoEMcv9lzQeuNUSeIrUmUjBwmlW/oUew==" + }, + "wgpu-matrix@3.4.2": { + "integrity": "sha512-IfZFbG7olEYBD76VCZy2TzqM57GKy3LT++AZh37tMHPPByO6R2h61zXGBZFNPi2IzxNivNi8y5dvze5PjaczjA==" + } + }, + "workspace": { + "dependencies": [ + "npm:lil-gui@0.21", + "npm:wgpu-matrix@^3.4.0" + ] + } +} diff --git a/package-lock.json b/package-lock.json deleted file mode 100644 index d50639b..0000000 --- a/package-lock.json +++ /dev/null @@ -1,1184 +0,0 @@ -{ - "name": "picovdb", - "version": "0.1.0", - "lockfileVersion": 3, - "requires": true, - "packages": { - "": { - "name": "picovdb", - "version": "0.1.0", - "dependencies": { - "lil-gui": "^0.21.0", - "wgpu-matrix": "^3.4.0" - }, - "devDependencies": { - "@webgpu/types": "^0.1.49", - "esbuild": "0.27.0", - "http-server": "^14.1.1", - "prettier": "^3.3.3", - "typescript": "^5.0.0" - } - }, - "node_modules/@esbuild/aix-ppc64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.27.0.tgz", - "integrity": "sha512-KuZrd2hRjz01y5JK9mEBSD3Vj3mbCvemhT466rSuJYeE/hjuBrHfjjcjMdTm/sz7au+++sdbJZJmuBwQLuw68A==", - "cpu": [ - "ppc64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "aix" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/android-arm": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.27.0.tgz", - "integrity": "sha512-j67aezrPNYWJEOHUNLPj9maeJte7uSMM6gMoxfPC9hOg8N02JuQi/T7ewumf4tNvJadFkvLZMlAq73b9uwdMyQ==", - "cpu": [ - "arm" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "android" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/android-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.27.0.tgz", - "integrity": "sha512-CC3vt4+1xZrs97/PKDkl0yN7w8edvU2vZvAFGD16n9F0Cvniy5qvzRXjfO1l94efczkkQE6g1x0i73Qf5uthOQ==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "android" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/android-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.27.0.tgz", - "integrity": "sha512-wurMkF1nmQajBO1+0CJmcN17U4BP6GqNSROP8t0X/Jiw2ltYGLHpEksp9MpoBqkrFR3kv2/te6Sha26k3+yZ9Q==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "android" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/darwin-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.27.0.tgz", - "integrity": "sha512-uJOQKYCcHhg07DL7i8MzjvS2LaP7W7Pn/7uA0B5S1EnqAirJtbyw4yC5jQ5qcFjHK9l6o/MX9QisBg12kNkdHg==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "darwin" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/darwin-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.27.0.tgz", - "integrity": "sha512-8mG6arH3yB/4ZXiEnXof5MK72dE6zM9cDvUcPtxhUZsDjESl9JipZYW60C3JGreKCEP+p8P/72r69m4AZGJd5g==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "darwin" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/freebsd-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.27.0.tgz", - "integrity": "sha512-9FHtyO988CwNMMOE3YIeci+UV+x5Zy8fI2qHNpsEtSF83YPBmE8UWmfYAQg6Ux7Gsmd4FejZqnEUZCMGaNQHQw==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "freebsd" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/freebsd-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.27.0.tgz", - "integrity": "sha512-zCMeMXI4HS/tXvJz8vWGexpZj2YVtRAihHLk1imZj4efx1BQzN76YFeKqlDr3bUWI26wHwLWPd3rwh6pe4EV7g==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "freebsd" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-arm": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.27.0.tgz", - "integrity": "sha512-t76XLQDpxgmq2cNXKTVEB7O7YMb42atj2Re2Haf45HkaUpjM2J0UuJZDuaGbPbamzZ7bawyGFUkodL+zcE+jvQ==", - "cpu": [ - "arm" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.27.0.tgz", - "integrity": "sha512-AS18v0V+vZiLJyi/4LphvBE+OIX682Pu7ZYNsdUHyUKSoRwdnOsMf6FDekwoAFKej14WAkOef3zAORJgAtXnlQ==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-ia32": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.27.0.tgz", - "integrity": "sha512-Mz1jxqm/kfgKkc/KLHC5qIujMvnnarD9ra1cEcrs7qshTUSksPihGrWHVG5+osAIQ68577Zpww7SGapmzSt4Nw==", - "cpu": [ - "ia32" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-loong64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.27.0.tgz", - "integrity": "sha512-QbEREjdJeIreIAbdG2hLU1yXm1uu+LTdzoq1KCo4G4pFOLlvIspBm36QrQOar9LFduavoWX2msNFAAAY9j4BDg==", - "cpu": [ - "loong64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-mips64el": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.27.0.tgz", - "integrity": "sha512-sJz3zRNe4tO2wxvDpH/HYJilb6+2YJxo/ZNbVdtFiKDufzWq4JmKAiHy9iGoLjAV7r/W32VgaHGkk35cUXlNOg==", - "cpu": [ - "mips64el" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-ppc64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.27.0.tgz", - "integrity": "sha512-z9N10FBD0DCS2dmSABDBb5TLAyF1/ydVb+N4pi88T45efQ/w4ohr/F/QYCkxDPnkhkp6AIpIcQKQ8F0ANoA2JA==", - "cpu": [ - "ppc64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-riscv64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.27.0.tgz", - "integrity": "sha512-pQdyAIZ0BWIC5GyvVFn5awDiO14TkT/19FTmFcPdDec94KJ1uZcmFs21Fo8auMXzD4Tt+diXu1LW1gHus9fhFQ==", - "cpu": [ - "riscv64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-s390x": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.27.0.tgz", - "integrity": "sha512-hPlRWR4eIDDEci953RI1BLZitgi5uqcsjKMxwYfmi4LcwyWo2IcRP+lThVnKjNtk90pLS8nKdroXYOqW+QQH+w==", - "cpu": [ - "s390x" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/linux-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.27.0.tgz", - "integrity": "sha512-1hBWx4OUJE2cab++aVZ7pObD6s+DK4mPGpemtnAORBvb5l/g5xFGk0vc0PjSkrDs0XaXj9yyob3d14XqvnQ4gw==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "linux" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/netbsd-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.27.0.tgz", - "integrity": "sha512-6m0sfQfxfQfy1qRuecMkJlf1cIzTOgyaeXaiVaaki8/v+WB+U4hc6ik15ZW6TAllRlg/WuQXxWj1jx6C+dfy3w==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "netbsd" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/netbsd-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.27.0.tgz", - "integrity": "sha512-xbbOdfn06FtcJ9d0ShxxvSn2iUsGd/lgPIO2V3VZIPDbEaIj1/3nBBe1AwuEZKXVXkMmpr6LUAgMkLD/4D2PPA==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "netbsd" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/openbsd-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.27.0.tgz", - "integrity": "sha512-fWgqR8uNbCQ/GGv0yhzttj6sU/9Z5/Sv/VGU3F5OuXK6J6SlriONKrQ7tNlwBrJZXRYk5jUhuWvF7GYzGguBZQ==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "openbsd" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/openbsd-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.27.0.tgz", - "integrity": "sha512-aCwlRdSNMNxkGGqQajMUza6uXzR/U0dIl1QmLjPtRbLOx3Gy3otfFu/VjATy4yQzo9yFDGTxYDo1FfAD9oRD2A==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "openbsd" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/openharmony-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.27.0.tgz", - "integrity": "sha512-nyvsBccxNAsNYz2jVFYwEGuRRomqZ149A39SHWk4hV0jWxKM0hjBPm3AmdxcbHiFLbBSwG6SbpIcUbXjgyECfA==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "openharmony" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/sunos-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.27.0.tgz", - "integrity": "sha512-Q1KY1iJafM+UX6CFEL+F4HRTgygmEW568YMqDA5UV97AuZSm21b7SXIrRJDwXWPzr8MGr75fUZPV67FdtMHlHA==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "sunos" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/win32-arm64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.27.0.tgz", - "integrity": "sha512-W1eyGNi6d+8kOmZIwi/EDjrL9nxQIQ0MiGqe/AWc6+IaHloxHSGoeRgDRKHFISThLmsewZ5nHFvGFWdBYlgKPg==", - "cpu": [ - "arm64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "win32" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/win32-ia32": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.27.0.tgz", - "integrity": "sha512-30z1aKL9h22kQhilnYkORFYt+3wp7yZsHWus+wSKAJR8JtdfI76LJ4SBdMsCopTR3z/ORqVu5L1vtnHZWVj4cQ==", - "cpu": [ - "ia32" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "win32" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@esbuild/win32-x64": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.27.0.tgz", - "integrity": "sha512-aIitBcjQeyOhMTImhLZmtxfdOcuNRpwlPNmlFKPcHQYPhEssw75Cl1TSXJXpMkzaua9FUetx/4OQKq7eJul5Cg==", - "cpu": [ - "x64" - ], - "dev": true, - "license": "MIT", - "optional": true, - "os": [ - "win32" - ], - "engines": { - "node": ">=18" - } - }, - "node_modules/@webgpu/types": { - "version": "0.1.69", - "resolved": "https://registry.npmjs.org/@webgpu/types/-/types-0.1.69.tgz", - "integrity": "sha512-RPmm6kgRbI8e98zSD3RVACvnuktIja5+yLgDAkTmxLr90BEwdTXRQWNLF3ETTTyH/8mKhznZuN5AveXYFEsMGQ==", - "dev": true, - "license": "BSD-3-Clause" - }, - "node_modules/ansi-styles": { - "version": "4.3.0", - "resolved": "https://registry.npmjs.org/ansi-styles/-/ansi-styles-4.3.0.tgz", - "integrity": "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg==", - "dev": true, - "license": "MIT", - "dependencies": { - "color-convert": "^2.0.1" - }, - "engines": { - "node": ">=8" - }, - "funding": { - "url": "https://github.com/chalk/ansi-styles?sponsor=1" - } - }, - "node_modules/async": { - "version": "3.2.6", - "resolved": "https://registry.npmjs.org/async/-/async-3.2.6.tgz", - "integrity": "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA==", - "dev": true, - "license": "MIT" - }, - "node_modules/basic-auth": { - "version": "2.0.1", - "resolved": "https://registry.npmjs.org/basic-auth/-/basic-auth-2.0.1.tgz", - "integrity": "sha512-NF+epuEdnUYVlGuhaxbbq+dvJttwLnGY+YixlXlME5KpQ5W3CnXA5cVTneY3SPbPDRkcjMbifrwmFYcClgOZeg==", - "dev": true, - "license": "MIT", - "dependencies": { - "safe-buffer": "5.1.2" - }, - "engines": { - "node": ">= 0.8" - } - }, - "node_modules/call-bind-apply-helpers": { - "version": "1.0.2", - "resolved": "https://registry.npmjs.org/call-bind-apply-helpers/-/call-bind-apply-helpers-1.0.2.tgz", - "integrity": "sha512-Sp1ablJ0ivDkSzjcaJdxEunN5/XvksFJ2sMBFfq6x0ryhQV/2b/KwFe21cMpmHtPOSij8K99/wSfoEuTObmuMQ==", - "dev": true, - "license": "MIT", - "dependencies": { - "es-errors": "^1.3.0", - "function-bind": "^1.1.2" - }, - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/call-bound": { - "version": "1.0.4", - "resolved": "https://registry.npmjs.org/call-bound/-/call-bound-1.0.4.tgz", - "integrity": "sha512-+ys997U96po4Kx/ABpBCqhA9EuxJaQWDQg7295H4hBphv3IZg0boBKuwYpt4YXp6MZ5AmZQnU/tyMTlRpaSejg==", - "dev": true, - "license": "MIT", - "dependencies": { - "call-bind-apply-helpers": "^1.0.2", - "get-intrinsic": "^1.3.0" - }, - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/chalk": { - "version": "4.1.2", - "resolved": "https://registry.npmjs.org/chalk/-/chalk-4.1.2.tgz", - "integrity": "sha512-oKnbhFyRIXpUuez8iBMmyEa4nbj4IOQyuhc/wy9kY7/WVPcwIO9VA668Pu8RkO7+0G76SLROeyw9CpQ061i4mA==", - "dev": true, - "license": "MIT", - "dependencies": { - "ansi-styles": "^4.1.0", - "supports-color": "^7.1.0" - }, - "engines": { - "node": ">=10" - }, - "funding": { - "url": "https://github.com/chalk/chalk?sponsor=1" - } - }, - "node_modules/color-convert": { - "version": "2.0.1", - "resolved": "https://registry.npmjs.org/color-convert/-/color-convert-2.0.1.tgz", - "integrity": "sha512-RRECPsj7iu/xb5oKYcsFHSppFNnsj/52OVTRKb4zP5onXwVF3zVmmToNcOfGC+CRDpfK/U584fMg38ZHCaElKQ==", - "dev": true, - "license": "MIT", - "dependencies": { - "color-name": "~1.1.4" - }, - "engines": { - "node": ">=7.0.0" - } - }, - "node_modules/color-name": { - "version": "1.1.4", - "resolved": "https://registry.npmjs.org/color-name/-/color-name-1.1.4.tgz", - "integrity": "sha512-dOy+3AuW3a2wNbZHIuMZpTcgjGuLU/uBL/ubcZF9OXbDo8ff4O8yVp5Bf0efS8uEoYo5q4Fx7dY9OgQGXgAsQA==", - "dev": true, - "license": "MIT" - }, - "node_modules/corser": { - "version": "2.0.1", - "resolved": "https://registry.npmjs.org/corser/-/corser-2.0.1.tgz", - "integrity": "sha512-utCYNzRSQIZNPIcGZdQc92UVJYAhtGAteCFg0yRaFm8f0P+CPtyGyHXJcGXnffjCybUCEx3FQ2G7U3/o9eIkVQ==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">= 0.4.0" - } - }, - "node_modules/debug": { - "version": "4.4.3", - "resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz", - "integrity": "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA==", - "dev": true, - "license": "MIT", - "dependencies": { - "ms": "^2.1.3" - }, - "engines": { - "node": ">=6.0" - }, - "peerDependenciesMeta": { - "supports-color": { - "optional": true - } - } - }, - "node_modules/dunder-proto": { - "version": "1.0.1", - "resolved": "https://registry.npmjs.org/dunder-proto/-/dunder-proto-1.0.1.tgz", - "integrity": "sha512-KIN/nDJBQRcXw0MLVhZE9iQHmG68qAVIBg9CqmUYjmQIhgij9U5MFvrqkUL5FbtyyzZuOeOt0zdeRe4UY7ct+A==", - "dev": true, - "license": "MIT", - "dependencies": { - "call-bind-apply-helpers": "^1.0.1", - "es-errors": "^1.3.0", - "gopd": "^1.2.0" - }, - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/es-define-property": { - "version": "1.0.1", - "resolved": "https://registry.npmjs.org/es-define-property/-/es-define-property-1.0.1.tgz", - "integrity": "sha512-e3nRfgfUZ4rNGL232gUgX06QNyyez04KdjFrF+LTRoOXmrOgFKDg4BCdsjW8EnT69eqdYGmRpJwiPVYNrCaW3g==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/es-errors": { - "version": "1.3.0", - "resolved": "https://registry.npmjs.org/es-errors/-/es-errors-1.3.0.tgz", - "integrity": "sha512-Zf5H2Kxt2xjTvbJvP2ZWLEICxA6j+hAmMzIlypy4xcBg1vKVnx89Wy0GbS+kf5cwCVFFzdCFh2XSCFNULS6csw==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/es-object-atoms": { - "version": "1.1.1", - "resolved": "https://registry.npmjs.org/es-object-atoms/-/es-object-atoms-1.1.1.tgz", - "integrity": "sha512-FGgH2h8zKNim9ljj7dankFPcICIK9Cp5bm+c2gQSYePhpaG5+esrLODihIorn+Pe6FGJzWhXQotPv73jTaldXA==", - "dev": true, - "license": "MIT", - "dependencies": { - "es-errors": "^1.3.0" - }, - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/esbuild": { - "version": "0.27.0", - "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.27.0.tgz", - "integrity": "sha512-jd0f4NHbD6cALCyGElNpGAOtWxSq46l9X/sWB0Nzd5er4Kz2YTm+Vl0qKFT9KUJvD8+fiO8AvoHhFvEatfVixA==", - "dev": true, - "hasInstallScript": true, - "license": "MIT", - "bin": { - "esbuild": "bin/esbuild" - }, - "engines": { - "node": ">=18" - }, - "optionalDependencies": { - "@esbuild/aix-ppc64": "0.27.0", - "@esbuild/android-arm": "0.27.0", - "@esbuild/android-arm64": "0.27.0", - "@esbuild/android-x64": "0.27.0", - "@esbuild/darwin-arm64": "0.27.0", - "@esbuild/darwin-x64": "0.27.0", - "@esbuild/freebsd-arm64": "0.27.0", - "@esbuild/freebsd-x64": "0.27.0", - "@esbuild/linux-arm": "0.27.0", - "@esbuild/linux-arm64": "0.27.0", - "@esbuild/linux-ia32": "0.27.0", - "@esbuild/linux-loong64": "0.27.0", - "@esbuild/linux-mips64el": "0.27.0", - "@esbuild/linux-ppc64": "0.27.0", - "@esbuild/linux-riscv64": "0.27.0", - "@esbuild/linux-s390x": "0.27.0", - "@esbuild/linux-x64": "0.27.0", - "@esbuild/netbsd-arm64": "0.27.0", - "@esbuild/netbsd-x64": "0.27.0", - "@esbuild/openbsd-arm64": "0.27.0", - "@esbuild/openbsd-x64": "0.27.0", - "@esbuild/openharmony-arm64": "0.27.0", - "@esbuild/sunos-x64": "0.27.0", - "@esbuild/win32-arm64": "0.27.0", - "@esbuild/win32-ia32": "0.27.0", - "@esbuild/win32-x64": "0.27.0" - } - }, - "node_modules/eventemitter3": { - "version": "4.0.7", - "resolved": "https://registry.npmjs.org/eventemitter3/-/eventemitter3-4.0.7.tgz", - "integrity": "sha512-8guHBZCwKnFhYdHr2ysuRWErTwhoN2X8XELRlrRwpmfeY2jjuUN4taQMsULKUVo1K4DvZl+0pgfyoysHxvmvEw==", - "dev": true, - "license": "MIT" - }, - "node_modules/follow-redirects": { - "version": "1.15.11", - "resolved": "https://registry.npmjs.org/follow-redirects/-/follow-redirects-1.15.11.tgz", - "integrity": "sha512-deG2P0JfjrTxl50XGCDyfI97ZGVCxIpfKYmfyrQ54n5FO/0gfIES8C/Psl6kWVDolizcaaxZJnTS0QSMxvnsBQ==", - "dev": true, - "funding": [ - { - "type": "individual", - "url": "https://github.com/sponsors/RubenVerborgh" - } - ], - "license": "MIT", - "engines": { - "node": ">=4.0" - }, - "peerDependenciesMeta": { - "debug": { - "optional": true - } - } - }, - "node_modules/function-bind": { - "version": "1.1.2", - "resolved": "https://registry.npmjs.org/function-bind/-/function-bind-1.1.2.tgz", - "integrity": "sha512-7XHNxH7qX9xG5mIwxkhumTox/MIRNcOgDrxWsMt2pAr23WHp6MrRlN7FBSFpCpr+oVO0F744iUgR82nJMfG2SA==", - "dev": true, - "license": "MIT", - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/get-intrinsic": { - "version": "1.3.0", - "resolved": "https://registry.npmjs.org/get-intrinsic/-/get-intrinsic-1.3.0.tgz", - "integrity": "sha512-9fSjSaos/fRIVIp+xSJlE6lfwhES7LNtKaCBIamHsjr2na1BiABJPo0mOjjz8GJDURarmCPGqaiVg5mfjb98CQ==", - "dev": true, - "license": "MIT", - "dependencies": { - "call-bind-apply-helpers": "^1.0.2", - "es-define-property": "^1.0.1", - "es-errors": "^1.3.0", - "es-object-atoms": "^1.1.1", - "function-bind": "^1.1.2", - "get-proto": "^1.0.1", - "gopd": "^1.2.0", - "has-symbols": "^1.1.0", - "hasown": "^2.0.2", - "math-intrinsics": "^1.1.0" - }, - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/get-proto": { - "version": "1.0.1", - "resolved": "https://registry.npmjs.org/get-proto/-/get-proto-1.0.1.tgz", - "integrity": "sha512-sTSfBjoXBp89JvIKIefqw7U2CCebsc74kiY6awiGogKtoSGbgjYE/G/+l9sF3MWFPNc9IcoOC4ODfKHfxFmp0g==", - "dev": true, - "license": "MIT", - "dependencies": { - "dunder-proto": "^1.0.1", - "es-object-atoms": "^1.0.0" - }, - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/gopd": { - "version": "1.2.0", - "resolved": "https://registry.npmjs.org/gopd/-/gopd-1.2.0.tgz", - "integrity": "sha512-ZUKRh6/kUFoAiTAtTYPZJ3hw9wNxx+BIBOijnlG9PnrJsCcSjs1wyyD6vJpaYtgnzDrKYRSqf3OO6Rfa93xsRg==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/has-flag": { - "version": "4.0.0", - "resolved": "https://registry.npmjs.org/has-flag/-/has-flag-4.0.0.tgz", - "integrity": "sha512-EykJT/Q1KjTWctppgIAgfSO0tKVuZUjhgMr17kqTumMl6Afv3EISleU7qZUzoXDFTAHTDC4NOoG/ZxU3EvlMPQ==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">=8" - } - }, - "node_modules/has-symbols": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/has-symbols/-/has-symbols-1.1.0.tgz", - "integrity": "sha512-1cDNdwJ2Jaohmb3sg4OmKaMBwuC48sYni5HUw2DvsC8LjGTLK9h+eb1X6RyuOHe4hT0ULCW68iomhjUoKUqlPQ==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/hasown": { - "version": "2.0.2", - "resolved": "https://registry.npmjs.org/hasown/-/hasown-2.0.2.tgz", - "integrity": "sha512-0hJU9SCPvmMzIBdZFqNPXWa6dqh7WdH0cII9y+CyS8rG3nL48Bclra9HmKhVVUHyPWNH5Y7xDwAB7bfgSjkUMQ==", - "dev": true, - "license": "MIT", - "dependencies": { - "function-bind": "^1.1.2" - }, - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/he": { - "version": "1.2.0", - "resolved": "https://registry.npmjs.org/he/-/he-1.2.0.tgz", - "integrity": "sha512-F/1DnUGPopORZi0ni+CvrCgHQ5FyEAHRLSApuYWMmrbSwoN2Mn/7k+Gl38gJnR7yyDZk6WLXwiGod1JOWNDKGw==", - "dev": true, - "license": "MIT", - "bin": { - "he": "bin/he" - } - }, - "node_modules/html-encoding-sniffer": { - "version": "3.0.0", - "resolved": "https://registry.npmjs.org/html-encoding-sniffer/-/html-encoding-sniffer-3.0.0.tgz", - "integrity": "sha512-oWv4T4yJ52iKrufjnyZPkrN0CH3QnrUqdB6In1g5Fe1mia8GmF36gnfNySxoZtxD5+NmYw1EElVXiBk93UeskA==", - "dev": true, - "license": "MIT", - "dependencies": { - "whatwg-encoding": "^2.0.0" - }, - "engines": { - "node": ">=12" - } - }, - "node_modules/http-proxy": { - "version": "1.18.1", - "resolved": "https://registry.npmjs.org/http-proxy/-/http-proxy-1.18.1.tgz", - "integrity": "sha512-7mz/721AbnJwIVbnaSv1Cz3Am0ZLT/UBwkC92VlxhXv/k/BBQfM2fXElQNC27BVGr0uwUpplYPQM9LnaBMR5NQ==", - "dev": true, - "license": "MIT", - "dependencies": { - "eventemitter3": "^4.0.0", - "follow-redirects": "^1.0.0", - "requires-port": "^1.0.0" - }, - "engines": { - "node": ">=8.0.0" - } - }, - "node_modules/http-server": { - "version": "14.1.1", - "resolved": "https://registry.npmjs.org/http-server/-/http-server-14.1.1.tgz", - "integrity": "sha512-+cbxadF40UXd9T01zUHgA+rlo2Bg1Srer4+B4NwIHdaGxAGGv59nYRnGGDJ9LBk7alpS0US+J+bLLdQOOkJq4A==", - "dev": true, - "license": "MIT", - "dependencies": { - "basic-auth": "^2.0.1", - "chalk": "^4.1.2", - "corser": "^2.0.1", - "he": "^1.2.0", - "html-encoding-sniffer": "^3.0.0", - "http-proxy": "^1.18.1", - "mime": "^1.6.0", - "minimist": "^1.2.6", - "opener": "^1.5.1", - "portfinder": "^1.0.28", - "secure-compare": "3.0.1", - "union": "~0.5.0", - "url-join": "^4.0.1" - }, - "bin": { - "http-server": "bin/http-server" - }, - "engines": { - "node": ">=12" - } - }, - "node_modules/iconv-lite": { - "version": "0.6.3", - "resolved": "https://registry.npmjs.org/iconv-lite/-/iconv-lite-0.6.3.tgz", - "integrity": "sha512-4fCk79wshMdzMp2rH06qWrJE4iolqLhCUH+OiuIgU++RB0+94NlDL81atO7GX55uUKueo0txHNtvEyI6D7WdMw==", - "dev": true, - "license": "MIT", - "dependencies": { - "safer-buffer": ">= 2.1.2 < 3.0.0" - }, - "engines": { - "node": ">=0.10.0" - } - }, - "node_modules/lil-gui": { - "version": "0.21.0", - "resolved": "https://registry.npmjs.org/lil-gui/-/lil-gui-0.21.0.tgz", - "integrity": "sha512-tpvxN7v1GvE/Tv+GRopfOp0W7fVEjF4PltkuX8vOCIfim22rD1ztvfkoEMcv9lzQeuNUSeIrUmUjBwmlW/oUew==", - "license": "MIT" - }, - "node_modules/math-intrinsics": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/math-intrinsics/-/math-intrinsics-1.1.0.tgz", - "integrity": "sha512-/IXtbwEk5HTPyEwyKX6hGkYXxM9nbj64B+ilVJnC/R6B0pH5G4V3b0pVbL7DBj4tkhBAppbQUlf6F6Xl9LHu1g==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">= 0.4" - } - }, - "node_modules/mime": { - "version": "1.6.0", - "resolved": "https://registry.npmjs.org/mime/-/mime-1.6.0.tgz", - "integrity": "sha512-x0Vn8spI+wuJ1O6S7gnbaQg8Pxh4NNHb7KSINmEWKiPE4RKOplvijn+NkmYmmRgP68mc70j2EbeTFRsrswaQeg==", - "dev": true, - "license": "MIT", - "bin": { - "mime": "cli.js" - }, - "engines": { - "node": ">=4" - } - }, - "node_modules/minimist": { - "version": "1.2.8", - "resolved": "https://registry.npmjs.org/minimist/-/minimist-1.2.8.tgz", - "integrity": "sha512-2yyAR8qBkN3YuheJanUpWC5U3bb5osDywNB8RzDVlDwDHbocAJveqqj1u8+SVD7jkWT4yvsHCpWqqWqAxb0zCA==", - "dev": true, - "license": "MIT", - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/ms": { - "version": "2.1.3", - "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", - "integrity": "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==", - "dev": true, - "license": "MIT" - }, - "node_modules/object-inspect": { - "version": "1.13.4", - "resolved": "https://registry.npmjs.org/object-inspect/-/object-inspect-1.13.4.tgz", - "integrity": "sha512-W67iLl4J2EXEGTbfeHCffrjDfitvLANg0UlX3wFUUSTx92KXRFegMHUVgSqE+wvhAbi4WqjGg9czysTV2Epbew==", - "dev": true, - "license": "MIT", - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/opener": { - "version": "1.5.2", - "resolved": "https://registry.npmjs.org/opener/-/opener-1.5.2.tgz", - "integrity": "sha512-ur5UIdyw5Y7yEj9wLzhqXiy6GZ3Mwx0yGI+5sMn2r0N0v3cKJvUmFH5yPP+WXh9e0xfyzyJX95D8l088DNFj7A==", - "dev": true, - "license": "(WTFPL OR MIT)", - "bin": { - "opener": "bin/opener-bin.js" - } - }, - "node_modules/portfinder": { - "version": "1.0.38", - "resolved": "https://registry.npmjs.org/portfinder/-/portfinder-1.0.38.tgz", - "integrity": "sha512-rEwq/ZHlJIKw++XtLAO8PPuOQA/zaPJOZJ37BVuN97nLpMJeuDVLVGRwbFoBgLudgdTMP2hdRJP++H+8QOA3vg==", - "dev": true, - "license": "MIT", - "dependencies": { - "async": "^3.2.6", - "debug": "^4.3.6" - }, - "engines": { - "node": ">= 10.12" - } - }, - "node_modules/prettier": { - "version": "3.8.1", - "resolved": "https://registry.npmjs.org/prettier/-/prettier-3.8.1.tgz", - "integrity": "sha512-UOnG6LftzbdaHZcKoPFtOcCKztrQ57WkHDeRD9t/PTQtmT0NHSeWWepj6pS0z/N7+08BHFDQVUrfmfMRcZwbMg==", - "dev": true, - "license": "MIT", - "bin": { - "prettier": "bin/prettier.cjs" - }, - "engines": { - "node": ">=14" - }, - "funding": { - "url": "https://github.com/prettier/prettier?sponsor=1" - } - }, - "node_modules/qs": { - "version": "6.15.0", - "resolved": "https://registry.npmjs.org/qs/-/qs-6.15.0.tgz", - "integrity": "sha512-mAZTtNCeetKMH+pSjrb76NAM8V9a05I9aBZOHztWy/UqcJdQYNsf59vrRKWnojAT9Y+GbIvoTBC++CPHqpDBhQ==", - "dev": true, - "license": "BSD-3-Clause", - "dependencies": { - "side-channel": "^1.1.0" - }, - "engines": { - "node": ">=0.6" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/requires-port": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/requires-port/-/requires-port-1.0.0.tgz", - "integrity": "sha512-KigOCHcocU3XODJxsu8i/j8T9tzT4adHiecwORRQ0ZZFcp7ahwXuRU1m+yuO90C5ZUyGeGfocHDI14M3L3yDAQ==", - "dev": true, - "license": "MIT" - }, - "node_modules/safe-buffer": { - "version": "5.1.2", - "resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.1.2.tgz", - "integrity": "sha512-Gd2UZBJDkXlY7GbJxfsE8/nvKkUEU1G38c1siN6QP6a9PT9MmHB8GnpscSmMJSoF8LOIrt8ud/wPtojys4G6+g==", - "dev": true, - "license": "MIT" - }, - "node_modules/safer-buffer": { - "version": "2.1.2", - "resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz", - "integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg==", - "dev": true, - "license": "MIT" - }, - "node_modules/secure-compare": { - "version": "3.0.1", - "resolved": "https://registry.npmjs.org/secure-compare/-/secure-compare-3.0.1.tgz", - "integrity": "sha512-AckIIV90rPDcBcglUwXPF3kg0P0qmPsPXAj6BBEENQE1p5yA1xfmDJzfi1Tappj37Pv2mVbKpL3Z1T+Nn7k1Qw==", - "dev": true, - "license": "MIT" - }, - "node_modules/side-channel": { - "version": "1.1.0", - "resolved": "https://registry.npmjs.org/side-channel/-/side-channel-1.1.0.tgz", - "integrity": "sha512-ZX99e6tRweoUXqR+VBrslhda51Nh5MTQwou5tnUDgbtyM0dBgmhEDtWGP/xbKn6hqfPRHujUNwz5fy/wbbhnpw==", - "dev": true, - "license": "MIT", - "dependencies": { - "es-errors": "^1.3.0", - "object-inspect": "^1.13.3", - "side-channel-list": "^1.0.0", - "side-channel-map": "^1.0.1", - "side-channel-weakmap": "^1.0.2" - }, - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/side-channel-list": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/side-channel-list/-/side-channel-list-1.0.0.tgz", - "integrity": "sha512-FCLHtRD/gnpCiCHEiJLOwdmFP+wzCmDEkc9y7NsYxeF4u7Btsn1ZuwgwJGxImImHicJArLP4R0yX4c2KCrMrTA==", - "dev": true, - "license": "MIT", - "dependencies": { - "es-errors": "^1.3.0", - "object-inspect": "^1.13.3" - }, - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/side-channel-map": { - "version": "1.0.1", - "resolved": "https://registry.npmjs.org/side-channel-map/-/side-channel-map-1.0.1.tgz", - "integrity": "sha512-VCjCNfgMsby3tTdo02nbjtM/ewra6jPHmpThenkTYh8pG9ucZ/1P8So4u4FGBek/BjpOVsDCMoLA/iuBKIFXRA==", - "dev": true, - "license": "MIT", - "dependencies": { - "call-bound": "^1.0.2", - "es-errors": "^1.3.0", - "get-intrinsic": "^1.2.5", - "object-inspect": "^1.13.3" - }, - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/side-channel-weakmap": { - "version": "1.0.2", - "resolved": "https://registry.npmjs.org/side-channel-weakmap/-/side-channel-weakmap-1.0.2.tgz", - "integrity": "sha512-WPS/HvHQTYnHisLo9McqBHOJk2FkHO/tlpvldyrnem4aeQp4hai3gythswg6p01oSoTl58rcpiFAjF2br2Ak2A==", - "dev": true, - "license": "MIT", - "dependencies": { - "call-bound": "^1.0.2", - "es-errors": "^1.3.0", - "get-intrinsic": "^1.2.5", - "object-inspect": "^1.13.3", - "side-channel-map": "^1.0.1" - }, - "engines": { - "node": ">= 0.4" - }, - "funding": { - "url": "https://github.com/sponsors/ljharb" - } - }, - "node_modules/supports-color": { - "version": "7.2.0", - "resolved": "https://registry.npmjs.org/supports-color/-/supports-color-7.2.0.tgz", - "integrity": "sha512-qpCAvRl9stuOHveKsn7HncJRvv501qIacKzQlO/+Lwxc9+0q2wLyv4Dfvt80/DPn2pqOBsJdDiogXGR9+OvwRw==", - "dev": true, - "license": "MIT", - "dependencies": { - "has-flag": "^4.0.0" - }, - "engines": { - "node": ">=8" - } - }, - "node_modules/typescript": { - "version": "5.9.3", - "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz", - "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", - "dev": true, - "license": "Apache-2.0", - "bin": { - "tsc": "bin/tsc", - "tsserver": "bin/tsserver" - }, - "engines": { - "node": ">=14.17" - } - }, - "node_modules/union": { - "version": "0.5.0", - "resolved": "https://registry.npmjs.org/union/-/union-0.5.0.tgz", - "integrity": "sha512-N6uOhuW6zO95P3Mel2I2zMsbsanvvtgn6jVqJv4vbVcz/JN0OkL9suomjQGmWtxJQXOCqUJvquc1sMeNz/IwlA==", - "dev": true, - "dependencies": { - "qs": "^6.4.0" - }, - "engines": { - "node": ">= 0.8.0" - } - }, - "node_modules/url-join": { - "version": "4.0.1", - "resolved": "https://registry.npmjs.org/url-join/-/url-join-4.0.1.tgz", - "integrity": "sha512-jk1+QP6ZJqyOiuEI9AEWQfju/nB2Pw466kbA0LEZljHwKeMgd9WrAEgEGxjPDD2+TNbbb37rTyhEfrCXfuKXnA==", - "dev": true, - "license": "MIT" - }, - "node_modules/wgpu-matrix": { - "version": "3.4.2", - "resolved": "https://registry.npmjs.org/wgpu-matrix/-/wgpu-matrix-3.4.2.tgz", - "integrity": "sha512-IfZFbG7olEYBD76VCZy2TzqM57GKy3LT++AZh37tMHPPByO6R2h61zXGBZFNPi2IzxNivNi8y5dvze5PjaczjA==", - "license": "MIT" - }, - "node_modules/whatwg-encoding": { - "version": "2.0.0", - "resolved": "https://registry.npmjs.org/whatwg-encoding/-/whatwg-encoding-2.0.0.tgz", - "integrity": "sha512-p41ogyeMUrw3jWclHWTQg1k05DSVXPLcVxRTYsXUk+ZooOCZLcoYgPZ/HL/D/N+uQPOtcp1me1WhBEaX02mhWg==", - "deprecated": "Use @exodus/bytes instead for a more spec-conformant and faster implementation", - "dev": true, - "license": "MIT", - "dependencies": { - "iconv-lite": "0.6.3" - }, - "engines": { - "node": ">=12" - } - } - } -} diff --git a/package.json b/package.json deleted file mode 100644 index 3cea7a1..0000000 --- a/package.json +++ /dev/null @@ -1,29 +0,0 @@ -{ - "name": "picovdb", - "version": "0.0.1", - "type": "module", - "exports": { - ".": "./ts/picovdb.ts", - "./stl": "./ts/stl.ts", - "./picovdb.wgsl": "./wgsl/picovdb.wgsl" - }, - "scripts": { - "build": "esbuild ts/picovdb.ts --bundle --outfile=dist/picovdb.js --format=esm --loader:.wgsl=text", - "build:demo": "esbuild demo/index.ts --bundle --outfile=demo/index.js --format=esm --loader:.wgsl=text", - "watch:demo": "esbuild demo/index.ts --bundle --outfile=demo/index.js --format=esm --loader:.wgsl=text --watch=forever", - "serve": "http-server demo -p 8000 -o -c-1", - "dev": "mkdir -p demo/models && cp data/*.pvdb.gz demo/models/ && (npm run watch:demo & npm run serve)", - "check": "tsc" - }, - "devDependencies": { - "@webgpu/types": "^0.1.49", - "esbuild": "0.27.0", - "http-server": "^14.1.1", - "prettier": "^3.3.3", - "typescript": "^5.0.0" - }, - "dependencies": { - "lil-gui": "^0.21.0", - "wgpu-matrix": "^3.4.0" - } -} diff --git a/src/mesh_to_grid.zig b/src/mesh_to_grid.zig index 0740b52..d75e914 100644 --- a/src/mesh_to_grid.zig +++ b/src/mesh_to_grid.zig @@ -1,7 +1,4 @@ -//! Mesh -> narrow-band signed distance field -> PicoVDB conversion. -//! -//! Replaces the OpenVDB `meshToLevelSet` + NanoVDB conversion pipeline with a -//! pure Zig implementation that writes PicoVDB structures directly. +//! Mesh to Grid conversion. //! //! Pipeline (all in index space, distances in voxel units): //! 1. Rasterize each triangle's half-width-dilated bounding box, keeping the @@ -582,11 +579,15 @@ pub fn meshToGrid( defer arena_state.deinit(); const arena = arena_state.allocator(); - // Transform vertices to index space (voxel units). + // Transforms vertices to index space in voxel units. Multiplies by the + // reciprocal because WGSL guarantees correctly rounded multiplication + // but not division, so the GPU pipeline reproduces this transform + // exactly. + const inv_voxel_size = 1.0 / opts.voxel_size; const pts = try arena.alloc(f32, vertices.len); for (vertices, 0..) |v, i| { if (!std.math.isFinite(v)) return error.NonFiniteVertex; - pts[i] = v / opts.voxel_size; + pts[i] = v * inv_voxel_size; } var mesh_min = V3{ pts[0], pts[1], pts[2] }; diff --git a/ts/gpu/device.ts b/ts/gpu/device.ts new file mode 100644 index 0000000..ffaa0ad --- /dev/null +++ b/ts/gpu/device.ts @@ -0,0 +1,98 @@ +// WebGPU device and buffer helpers. + +/** Whether a GPU adapter is available. */ +export async function hasWebGPU(): Promise { + const gpu = (globalThis as { navigator?: Navigator }).navigator?.gpu; + if (!gpu) return false; + return (await gpu.requestAdapter()) !== null; +} + +/** + * Device limits the grid kernels need above the WebGPU defaults, raised to + * what the adapter supports: value slabs are 2 KB per leaf, workgroups are + * 256 wide, and kernels bind up to ten storage buffers, which the defaults + * of a compatibility mode adapter do not allow. + */ +export function gridLimits(adapter: GPUAdapter): Record { + const l = adapter.limits; + return { + maxStorageBufferBindingSize: l.maxStorageBufferBindingSize, + maxBufferSize: l.maxBufferSize, + maxStorageBuffersPerShaderStage: l.maxStorageBuffersPerShaderStage, + maxComputeWorkgroupSizeX: l.maxComputeWorkgroupSizeX, + maxComputeInvocationsPerWorkgroup: l.maxComputeInvocationsPerWorkgroup, + maxComputeWorkgroupStorageSize: l.maxComputeWorkgroupStorageSize, + maxComputeWorkgroupsPerDimension: l.maxComputeWorkgroupsPerDimension, + }; +} + +export async function requestDevice(): Promise { + const gpu = (globalThis as { navigator?: Navigator }).navigator?.gpu; + if (!gpu) throw new Error('WebGPU unavailable'); + const adapter = await gpu.requestAdapter(); + if (!adapter) throw new Error('WebGPU adapter unavailable'); + return adapter.requestDevice({ requiredLimits: gridLimits(adapter) }); +} + +export function createU32Buffer(device: GPUDevice, data: Uint32Array, extraUsage: GPUBufferUsageFlags = 0): GPUBuffer { + const buffer = device.createBuffer({ + size: Math.max(data.byteLength, 4), + usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC | extraUsage, + }); + device.queue.writeBuffer(buffer, 0, data); + return buffer; +} + +export const DISPATCH_STRIDE = 65535; + +/** + * Linearized 2D dispatch. The WGSL side derives the workgroup index from + * the stride, so when groups spill into y the x count must equal the + * stride. + */ +export function dispatch2D(pass: GPUComputePassEncoder, groups: number): void { + if (groups <= DISPATCH_STRIDE) { + pass.dispatchWorkgroups(groups, 1); + } else { + pass.dispatchWorkgroups(DISPATCH_STRIDE, Math.ceil(groups / DISPATCH_STRIDE)); + } +} + +/** Reads one u32 per request in a single copy and map. */ +export async function readBackTotals( + device: GPUDevice, + reads: Array<{ buffer: GPUBuffer; index: number }> +): Promise { + const staging = device.createBuffer({ + size: reads.length * 4, + usage: GPUBufferUsage.MAP_READ | GPUBufferUsage.COPY_DST, + }); + const encoder = device.createCommandEncoder(); + reads.forEach((r, i) => encoder.copyBufferToBuffer(r.buffer, r.index * 4, staging, i * 4, 4)); + device.queue.submit([encoder.finish()]); + await staging.mapAsync(GPUMapMode.READ); + const out = [...new Uint32Array(staging.getMappedRange().slice(0))]; + staging.destroy(); + return out; +} + +/** Throws when a storage binding would exceed the device limit. */ +export function checkBindingSize(device: GPUDevice, bytes: number, label: string): void { + if (bytes > device.limits.maxStorageBufferBindingSize) { + throw new Error(`${label} needs ${bytes} bytes, over the ${device.limits.maxStorageBufferBindingSize} byte storage binding limit`); + } +} + +export async function readBackU32(device: GPUDevice, src: GPUBuffer, count: number): Promise { + const staging = device.createBuffer({ + size: count * 4, + usage: GPUBufferUsage.MAP_READ | GPUBufferUsage.COPY_DST, + }); + const encoder = device.createCommandEncoder(); + encoder.copyBufferToBuffer(src, 0, staging, 0, count * 4); + device.queue.submit([encoder.finish()]); + await staging.mapAsync(GPUMapMode.READ); + const out = new Uint32Array(staging.getMappedRange().slice(0)); + staging.destroy(); + return out; +} diff --git a/ts/gpu/dilate.test.ts b/ts/gpu/dilate.test.ts new file mode 100644 index 0000000..979e44e --- /dev/null +++ b/ts/gpu/dilate.test.ts @@ -0,0 +1,94 @@ +import { hasWebGPU, requestDevice, createU32Buffer, readBackU32 } from './device.ts'; +import { mulberry32, assertU32ArrayEqual } from './test_util.ts'; +import { Dilator } from './dilate.ts'; + +const gpu = await hasWebGPU(); + +const pack = (x: number, y: number, z: number): number => ((x << 20) | (y << 10) | z) >>> 0; + +// Brute force reference. Expands every active voxel to its face neighbors +// in global voxel space and rebuilds leaves, keeping originals even when +// empty. +function refDilate(keys: Uint32Array, masks: Uint32Array): { keys: Uint32Array; masks: Uint32Array } { + const active = new Set(); + keys.forEach((key, li) => { + const lx = (key >>> 20) & 0x3ff; + const ly = (key >>> 10) & 0x3ff; + const lz = key & 0x3ff; + for (let n = 0; n < 512; n++) { + if ((masks[li * 16 + (n >> 5)] >>> (n & 31)) & 1) { + active.add(`${lx * 8 + (n >> 6)},${ly * 8 + ((n >> 3) & 7)},${lz * 8 + (n & 7)}`); + } + } + }); + const dilated = new Set(active); + for (const v of active) { + const [x, y, z] = v.split(',').map(Number); + for (const [dx, dy, dz] of [[1, 0, 0], [-1, 0, 0], [0, 1, 0], [0, -1, 0], [0, 0, 1], [0, 0, -1]]) { + const nx = x + dx, ny = y + dy, nz = z + dz; + if (nx >= 0 && ny >= 0 && nz >= 0 && nx < 8192 && ny < 8192 && nz < 8192) { + dilated.add(`${nx},${ny},${nz}`); + } + } + } + const leafSet = new Set(keys); + for (const v of dilated) { + const [x, y, z] = v.split(',').map(Number); + leafSet.add(pack(x >> 3, y >> 3, z >> 3)); + } + const outKeys = new Uint32Array([...leafSet].sort((a, b) => a - b)); + const slot = new Map(); + outKeys.forEach((k, i) => slot.set(k, i)); + const outMasks = new Uint32Array(outKeys.length * 16); + for (const v of dilated) { + const [x, y, z] = v.split(',').map(Number); + const li = slot.get(pack(x >> 3, y >> 3, z >> 3))!; + const n = ((x & 7) << 6) | ((y & 7) << 3) | (z & 7); + outMasks[li * 16 + (n >> 5)] |= 1 << (n & 31); + } + return { keys: outKeys, masks: outMasks }; +} + +async function checkDilate(dilator: Dilator, keys: Uint32Array, masks: Uint32Array, label: string) { + const keyBuf = createU32Buffer(dilator.device, keys); + const maskBuf = createU32Buffer(dilator.device, masks); + const result = await dilator.dilate(keyBuf, maskBuf, keys.length); + const ref = refDilate(keys, masks); + if (result.leafCount !== ref.keys.length) { + throw new Error(`${label}: leaf count ${result.leafCount} != ref ${ref.keys.length}`); + } + assertU32ArrayEqual(await readBackU32(dilator.device, result.leafKeys, result.leafCount), ref.keys, `${label} keys`); + assertU32ArrayEqual(await readBackU32(dilator.device, result.masks, result.leafCount * 16), ref.masks, `${label} masks`); +} + +Deno.test({ name: 'dilate matches brute-force reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const dilator = new Dilator(device); + const rand = mulberry32(6); + + // A single center voxel dilates within its leaf. + const single = new Uint32Array(16); + single[(((4 << 6) | (4 << 3) | 4) >> 5)] |= 1 << (((4 << 6) | (4 << 3) | 4) & 31); + await checkDilate(dilator, new Uint32Array([pack(10, 10, 10)]), single, 'center'); + + // A full leaf spills into all six neighbors. + await checkDilate(dilator, new Uint32Array([pack(10, 10, 10)]), new Uint32Array(16).fill(0xffffffff), 'full'); + + // A corner voxel spills across three faces. Face dilation must not + // spawn diagonal leaves. + const corner = new Uint32Array(16); + corner[0] |= 1; // voxel zero + await checkDilate(dilator, new Uint32Array([pack(10, 10, 10)]), corner, 'corner'); + + // Random cluster of adjacent leaves with random masks. + const keySet = new Set(); + while (keySet.size < 30) { + keySet.add(pack(20 + Math.floor(rand() * 4), 20 + Math.floor(rand() * 4), 20 + Math.floor(rand() * 4))); + } + const keys = new Uint32Array([...keySet].sort((a, b) => a - b)); + const masks = new Uint32Array(keys.length * 16); + for (let i = 0; i < masks.length; i++) { + if (rand() < 0.3) masks[i] = Math.floor(rand() * 4294967296); + } + await checkDilate(dilator, keys, masks, 'cluster'); +}); diff --git a/ts/gpu/dilate.ts b/ts/gpu/dilate.ts new file mode 100644 index 0000000..6273e4d --- /dev/null +++ b/ts/gpu/dilate.ts @@ -0,0 +1,147 @@ +// Host side of wgsl/dilate.wgsl. Dilates an active voxel mask set by one +// voxel across face neighbors, growing the leaf table where masks spill +// across leaf boundaries. + +import dilateWgsl from 'picovdb/wgsl/dilate.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { Sorter } from './radix_sort.ts'; +import { dispatch2D, readBackTotals } from './device.ts'; + +const WG_SIZE = 256; + +export interface DilateResult { + /** Sorted unique leaf keys including spawned face neighbors. */ + leafKeys: GPUBuffer; + /** Dilated masks with 16 words per leaf. */ + masks: GPUBuffer; + leafCount: number; +} + +export class Dilator { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly sorter: Sorter; + readonly layout: GPUBindGroupLayout; + private readonly pipelines: Record = {}; + + constructor(device: GPUDevice, scanner = new Scanner(device), sorter = new Sorter(device, scanner)) { + this.device = device; + this.scanner = scanner; + this.sorter = sorter; + const entry = (binding: number, type: GPUBufferBindingType): GPUBindGroupLayoutEntry => ({ + binding, + visibility: GPUShaderStage.COMPUTE, + buffer: { type }, + }); + this.layout = device.createBindGroupLayout({ + entries: [ + entry(0, 'uniform'), + entry(1, 'read-only-storage'), + entry(2, 'read-only-storage'), + entry(3, 'storage'), + entry(4, 'storage'), + entry(5, 'storage'), + entry(6, 'storage'), + entry(7, 'storage'), + entry(8, 'storage'), + ], + }); + const module = device.createShaderModule({ code: dilateWgsl }); + const layout = device.createPipelineLayout({ bindGroupLayouts: [this.layout] }); + for (const entryPoint of ['count_spawn', 'emit_spawn', 'mark_unique', 'compact_unique', 'dilate_masks']) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout, compute: { module, entryPoint } }); + } + } + + async dilate(leafKeys: GPUBuffer, masks: GPUBuffer, leafCount: number): Promise { + const device = this.device; + if (leafCount === 0) { + return { leafKeys, masks, leafCount: 0 }; + } + const params = device.createBuffer({ size: 16, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([leafCount])); + + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const counts = device.createBuffer({ size: (leafCount + 1) * 4, usage: storage }); + const clipped = device.createBuffer({ size: 4, usage: storage }); + // Distinct placeholders: writable bindings may not alias one buffer. + const placeholders = [0, 1, 2, 3].map(() => device.createBuffer({ size: 4, usage: GPUBufferUsage.STORAGE })); + + const bindGroup = (spawnKeys: GPUBuffer, flags: GPUBuffer, newKeys: GPUBuffer, newMasks: GPUBuffer) => + device.createBindGroup({ + layout: this.layout, + entries: [ + { binding: 0, resource: { buffer: params } }, + { binding: 1, resource: { buffer: leafKeys } }, + { binding: 2, resource: { buffer: masks } }, + { binding: 3, resource: { buffer: counts } }, + { binding: 4, resource: { buffer: spawnKeys } }, + { binding: 5, resource: { buffer: flags } }, + { binding: 6, resource: { buffer: newKeys } }, + { binding: 7, resource: { buffer: newMasks } }, + { binding: 8, resource: { buffer: clipped } }, + ], + }); + + // Count spawned leaves, scan, and read the total. + const countScan = this.scanner.plan(counts, leafCount + 1); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, bindGroup(placeholders[0], placeholders[1], placeholders[2], placeholders[3])); + pass.setPipeline(this.pipelines['count_spawn']); + dispatch2D(pass, Math.ceil((leafCount + 1) / WG_SIZE)); + countScan.encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [spawnCount, clippedCount] = await readBackTotals(device, [ + { buffer: counts, index: leafCount }, + { buffer: clipped, index: 0 }, + ]); + if (clippedCount > 0) { + throw new Error(`dilation spills past the leaf key space boundary at ${clippedCount} faces`); + } + + // Emit, sort, and dedupe. + device.queue.writeBuffer(params, 4, new Uint32Array([spawnCount])); + const spawnKeys = device.createBuffer({ size: spawnCount * 4, usage: storage }); + const sortVals = device.createBuffer({ size: spawnCount * 4, usage: GPUBufferUsage.STORAGE }); + const flags = device.createBuffer({ size: (spawnCount + 1) * 4, usage: storage }); + const newKeys = device.createBuffer({ size: spawnCount * 4, usage: storage }); + const flagScan = this.scanner.plan(flags, spawnCount + 1); + const group = bindGroup(spawnKeys, flags, newKeys, placeholders[3]); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, group); + pass.setPipeline(this.pipelines['emit_spawn']); + dispatch2D(pass, Math.ceil(leafCount / WG_SIZE)); + this.sorter.plan(spawnKeys, sortVals, spawnCount).encode(pass); + pass.setBindGroup(0, group); + pass.setPipeline(this.pipelines['mark_unique']); + dispatch2D(pass, Math.ceil((spawnCount + 1) / WG_SIZE)); + flagScan.encode(pass); + pass.setBindGroup(0, group); + pass.setPipeline(this.pipelines['compact_unique']); + dispatch2D(pass, Math.ceil(spawnCount / WG_SIZE)); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [newCount] = await readBackTotals(device, [{ buffer: flags, index: spawnCount }]); + + // Build the dilated masks. + device.queue.writeBuffer(params, 8, new Uint32Array([newCount])); + const newMasks = device.createBuffer({ size: newCount * 16 * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, bindGroup(spawnKeys, flags, newKeys, newMasks)); + pass.setPipeline(this.pipelines['dilate_masks']); + dispatch2D(pass, Math.ceil(newCount / WG_SIZE)); + pass.end(); + device.queue.submit([encoder.finish()]); + } + return { leafKeys: newKeys, masks: newMasks, leafCount: newCount }; + } +} diff --git a/ts/gpu/emit.test.ts b/ts/gpu/emit.test.ts new file mode 100644 index 0000000..bdaa0ec --- /dev/null +++ b/ts/gpu/emit.test.ts @@ -0,0 +1,42 @@ +import { hasWebGPU, requestDevice } from './device.ts'; +import { compareTreeToCpu } from './test_util.ts'; +import { parseBinarySTL } from './reference.ts'; +import { Binner } from './mesh_to_grid.ts'; +import { Rasterizer } from './rasterize.ts'; +import { Signer } from './sign.ts'; +import { Emitter } from './emit.ts'; +import { initSTL, importSTL } from '../stl.ts'; + +const gpu = await hasWebGPU(); + +// The CPU oracle is the wasm converter run on the same mesh. The test +// skips when either file is missing. +let stl: Uint8Array | null = null; +let wasm: Uint8Array | null = null; +try { + stl = Deno.readFileSync(new URL('../../data/bases/base_32mm.stl', import.meta.url)); + wasm = Deno.readFileSync(new URL('../../zig-out/wasm/picovdb.wasm', import.meta.url)); +} catch { + // skip +} + +Deno.test({ name: 'GPU tree emission matches CPU converter', ignore: !gpu || !stl || !wasm }, async () => { + const device = await requestDevice(); + const { points, triangles } = parseBinarySTL(stl!); + + initSTL({ wasmBinary: wasm! }); + const cpu = (await importSTL(stl!, { voxelsPerUnit: 4 })).file; + + const binner = new Binner(device); + const opts = { voxelSize: 0.25, halfWidth: 3 }; + const bin = await binner.bin(points, triangles, opts); + const leafValues = new Rasterizer(device).rasterize(bin, opts); + const sign = await new Signer(device).sign(bin); + const tree = await new Emitter(device).emit(bin, leafValues, sign, opts); + + const maxAbs = await compareTreeToCpu(device, tree, cpu); + console.log( + ` tree: ${tree.leafCount} leaves / ${tree.lowerCount} lowers / ${tree.upperCount} uppers, ` + + `${tree.activeVoxels} active, ${tree.surfaceVoxels} surface, max |Δv| ${maxAbs.toExponential(2)}` + ); +}); diff --git a/ts/gpu/emit.ts b/ts/gpu/emit.ts new file mode 100644 index 0000000..ab833b9 --- /dev/null +++ b/ts/gpu/emit.ts @@ -0,0 +1,238 @@ +// Host side of wgsl/emit.wgsl: builds the picovdb tree of an op layer +// grid, and turns the mesh converter's distance slabs into an op layer +// grid. + +import emitWgsl from 'picovdb/wgsl/emit.wgsl' with { type: 'text' }; +import { PICOVDB_LOWER_SIZE, PICOVDB_UPPER_SIZE } from '../picovdb.ts'; +import { Scanner } from './scan.ts'; +import { Sorter } from './radix_sort.ts'; +import { dispatch2D, readBackTotals, readBackU32 } from './device.ts'; +import { GridWriter, LEAF_U32, preludeWgsl, readerWgsl, type OpGrid } from './opgrid.ts'; +import type { BinResult } from './mesh_to_grid.ts'; +import type { SignResult } from './sign.ts'; + +export { LEAF_U32 }; +export type { OpGrid }; + +const WG_SIZE = 256; +export const LOWER_U32 = PICOVDB_LOWER_SIZE / 4; +export const UPPER_U32 = PICOVDB_UPPER_SIZE / 4; + +export interface EmitOptions { + /** Narrow band half width in voxels. Must match the earlier stages. */ + halfWidth: number; +} + +export interface EmitResult { + roots: GPUBuffer; // two u32 key words per upper, unpadded + uppers: GPUBuffer; + lowers: GPUBuffer; + leaves: GPUBuffer; + data: GPUBuffer; // f32 values, two implicit background slots then per voxel values + leafCount: number; + lowerCount: number; + upperCount: number; + /** Total value slots including the two implicit entries. */ + dataElemCount: number; + activeVoxels: number; + surfaceVoxels: number; + indexBoundsMin: [number, number, number]; + indexBoundsMax: [number, number, number]; +} + +export class Emitter { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly sorter: Sorter; + private readonly pipelines: Record = {}; + + constructor(device: GPUDevice, scanner: Scanner = new Scanner(device), sorter: Sorter = new Sorter(device, scanner)) { + this.device = device; + this.scanner = scanner; + this.sorter = sorter; + const code = preludeWgsl + readerWgsl('cand', 'params.cand_count') + emitWgsl; + // The shader hardcodes the node strides. Fail construction on drift + // from the picovdb sizes. + for (const [name, value] of [['LEAF_U32', LEAF_U32], ['LOWER_U32', LOWER_U32], ['UPPER_U32', UPPER_U32]] as const) { + if (!code.includes(`${name}: u32 = ${value}u`)) { + throw new Error(`wgsl/emit.wgsl ${name} does not match the picovdb node size`); + } + } + const module = device.createShaderModule({ code }); + for (const entryPoint of [ + 'classify_mark', 'classify_apply', 'opgrid_compact', 'leaf_stats', 'hier_lo', 'hier_hi', 'reorder_final', + 'leaf_value_counts', 'mark_lower', 'compact_lower', 'mark_upper', 'compact_upper', + 'surface', 'write_leaves', 'write_data', 'write_lowers', 'write_uppers', 'write_roots', + ]) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint } }); + } + } + + /** The tree of a converted mesh. */ + async emit(bin: BinResult, dist2: GPUBuffer, sign: SignResult, opts: EmitOptions): Promise { + const grid = await this.classifyOnly(bin, dist2, sign, opts); + const tree = await this.reEmit(grid, opts); + grid.leafKeys.destroy(); + grid.leaves.destroy(); + grid.data.destroy(); + return tree; + } + + /** An op layer grid from the converter's squared distance slabs and inside masks. */ + classifyOnly(bin: BinResult, dist2: GPUBuffer, sign: SignResult, opts: EmitOptions): Promise { + const params = this.params(bin.leafCount, bin.leafMin, opts.halfWidth); + const run = this.runner(params); + const writer = new GridWriter(this.device, bin.leafCount); + { + const encoder = this.device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'classify_mark', bin.leafCount + 1, { 4: dist2, 11: sign.inside, ...writer.markBindings }); + pass.end(); + this.device.queue.submit([encoder.finish()]); + } + return writer.finish(this.scanner, this.pipelines['opgrid_compact'], bin.leafKeys, opts.halfWidth, bin, (pass, out, count) => { + run(pass, 'classify_apply', count, { 1: bin.leafKeys, 4: dist2, 11: sign.inside, ...out }); + }); + } + + /** The tree of an op layer grid. */ + async reEmit(grid: OpGrid, opts: EmitOptions): Promise { + const device = this.device; + const cand = grid.leafCount; + if (cand === 0) throw new Error('no active voxels'); + const params = this.params(cand, grid.leafMin, opts.halfWidth); + const run = this.runner(params); + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const reader = { 1: grid.leafKeys, 2: grid.leaves, 3: grid.data }; + const bandCounts = device.createBuffer({ size: (cand + 1) * 4, usage: storage }); + const bounds = device.createBuffer({ size: 24, usage: storage }); + device.queue.writeBuffer(bounds, 0, new Int32Array([0x7fffffff, 0x7fffffff, 0x7fffffff, -0x80000000, -0x80000000, -0x80000000])); + const flags = device.createBuffer({ size: (cand + 1) * 4, usage: storage }); + const hier = device.createBuffer({ size: cand * 4, usage: storage }); + const idx = device.createBuffer({ size: cand * 4, usage: storage }); + const finalKeys = device.createBuffer({ size: cand * 4, usage: storage }); + const finalCand = device.createBuffer({ size: cand * 4, usage: storage }); + const valueCounts = device.createBuffer({ size: (cand + 1) * 4, usage: storage }); + { + // Sort the leaves into the CPU's order, then derive value offsets + // and lowers. + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'leaf_stats', cand + 1, { 1: grid.leafKeys, 2: grid.leaves, 5: bandCounts, 6: bounds }); + run(pass, 'hier_lo', cand, { 1: grid.leafKeys, 26: hier, 27: idx }); + this.sorter.plan(hier, idx, cand).encode(pass); + run(pass, 'hier_hi', cand, { 1: grid.leafKeys, 26: hier, 27: idx }); + this.sorter.plan(hier, idx, cand).encode(pass); + run(pass, 'reorder_final', cand, { 1: grid.leafKeys, 8: finalKeys, 9: finalCand, 27: idx }); + run(pass, 'leaf_value_counts', cand + 1, { 5: bandCounts, 9: finalCand, 10: valueCounts }); + this.scanner.plan(valueCounts, cand + 1).encode(pass); + run(pass, 'mark_lower', cand + 1, { 7: flags, 8: finalKeys }); + this.scanner.plan(flags, cand + 1).encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [dataValues, lowerCount] = await readBackTotals(device, [ + { buffer: valueCounts, index: cand }, + { buffer: flags, index: cand }, + ]); + + device.queue.writeBuffer(params, 4, new Uint32Array([lowerCount])); + const lowerKeys = device.createBuffer({ size: lowerCount * 4, usage: storage }); + const lowerFirst = device.createBuffer({ size: lowerCount * 4, usage: storage }); + const flatLower = device.createBuffer({ size: lowerCount * 4, usage: storage }); + const flatVals = device.createBuffer({ size: lowerCount * 4, usage: GPUBufferUsage.STORAGE }); + { + const encoder = device.createCommandEncoder(); + const passA = encoder.beginComputePass(); + run(passA, 'compact_lower', cand, { 7: flags, 8: finalKeys, 17: lowerKeys, 18: lowerFirst }); + passA.end(); + encoder.copyBufferToBuffer(lowerKeys, 0, flatLower, 0, lowerCount * 4); + const passB = encoder.beginComputePass(); + this.sorter.plan(flatLower, flatVals, lowerCount).encode(passB); + run(passB, 'mark_upper', lowerCount + 1, { 7: flags, 17: lowerKeys }); + this.scanner.plan(flags, lowerCount + 1).encode(passB); + passB.end(); + device.queue.submit([encoder.finish()]); + } + const [upperCount] = await readBackTotals(device, [{ buffer: flags, index: lowerCount }]); + + // Surface masks and all node and value outputs. + device.queue.writeBuffer(params, 8, new Uint32Array([upperCount])); + const upperKeys = device.createBuffer({ size: upperCount * 4, usage: storage }); + const upperFirst = device.createBuffer({ size: upperCount * 4, usage: storage }); + const surfMasks = device.createBuffer({ size: cand * 16 * 4, usage: storage }); + const surfCounts = device.createBuffer({ size: (cand + 1) * 4, usage: storage }); + const leaves = device.createBuffer({ size: cand * LEAF_U32 * 4, usage: storage }); + const dataBytes = Math.ceil(((2 + dataValues) * 4) / 16) * 16; + const data = device.createBuffer({ size: dataBytes, usage: storage }); + const lowers = device.createBuffer({ size: lowerCount * LOWER_U32 * 4, usage: storage }); + const uppers = device.createBuffer({ size: upperCount * UPPER_U32 * 4, usage: storage }); + const roots = device.createBuffer({ size: upperCount * 2 * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'compact_upper', lowerCount, { 7: flags, 17: lowerKeys, 20: upperKeys, 21: upperFirst }); + run(pass, 'surface', cand + 1, { ...reader, 9: finalCand, 13: surfMasks, 14: surfCounts }); + this.scanner.plan(surfCounts, cand + 1).encode(pass); + run(pass, 'write_leaves', cand, { 2: grid.leaves, 9: finalCand, 10: valueCounts, 13: surfMasks, 14: surfCounts, 15: leaves }); + run(pass, 'write_data', cand, { 2: grid.leaves, 3: grid.data, 5: bandCounts, 9: finalCand, 10: valueCounts, 16: data }); + run(pass, 'write_lowers', lowerCount, { ...reader, 10: valueCounts, 17: lowerKeys, 18: lowerFirst, 19: lowers }); + run(pass, 'write_uppers', upperCount, { ...reader, 19: lowers, 20: upperKeys, 21: upperFirst, 22: uppers, 28: flatLower }); + run(pass, 'write_roots', upperCount, { 20: upperKeys, 23: roots }); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [surfaceVoxels] = await readBackTotals(device, [{ buffer: surfCounts, index: cand }]); + const boundsOut = new Int32Array((await readBackU32(device, bounds, 6)).buffer); + for (const b of [bandCounts, bounds, flags, hier, idx, finalKeys, finalCand, valueCounts, lowerKeys, lowerFirst, flatLower, flatVals, upperKeys, upperFirst, surfMasks, surfCounts, params]) { + b.destroy(); + } + + return { + roots, + uppers, + lowers, + leaves, + data, + leafCount: cand, + lowerCount, + upperCount, + dataElemCount: 2 + dataValues, + activeVoxels: dataValues, + surfaceVoxels, + indexBoundsMin: [boundsOut[0], boundsOut[1], boundsOut[2]], + indexBoundsMax: [boundsOut[3], boundsOut[4], boundsOut[5]], + }; + } + + private params(cand: number, leafMin: [number, number, number], halfWidth: number): GPUBuffer { + const device = this.device; + const lowerMin = leafMin.map((v) => v >> 4); + const upperMin = lowerMin.map((v) => v >> 5); + const params = device.createBuffer({ size: 64, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([cand])); + device.queue.writeBuffer(params, 12, new Float32Array([halfWidth])); + device.queue.writeBuffer(params, 16, new Int32Array(leafMin)); + device.queue.writeBuffer(params, 32, new Int32Array(lowerMin)); + device.queue.writeBuffer(params, 48, new Int32Array(upperMin)); + return params; + } + + private runner(params: GPUBuffer) { + return (pass: GPUComputePassEncoder, name: string, threads: number, buffers: Record): void => { + if (threads === 0) return; + pass.setPipeline(this.pipelines[name]); + pass.setBindGroup( + 0, + this.device.createBindGroup({ + layout: this.pipelines[name].getBindGroupLayout(0), + entries: [ + { binding: 0, resource: { buffer: params } }, + ...Object.entries(buffers).map(([binding, buffer]) => ({ binding: Number(binding), resource: { buffer } })), + ], + }) + ); + dispatch2D(pass, Math.ceil(threads / WG_SIZE)); + }; + } +} diff --git a/ts/gpu/extract.test.ts b/ts/gpu/extract.test.ts new file mode 100644 index 0000000..502d847 --- /dev/null +++ b/ts/gpu/extract.test.ts @@ -0,0 +1,30 @@ +import { hasWebGPU, requestDevice, readBackU32 } from './device.ts'; +import { emptyGrid } from './test_util.ts'; +import { Stamper, sphere } from './stamp.ts'; +import { Extractor } from './extract.ts'; + +const gpu = await hasWebGPU(); + +Deno.test({ name: 'extracted sphere mesh sits on the level set with the right area', ignore: !gpu }, async () => { + const device = await requestDevice(); + const halfWidth = 3; + const center = [100.3, 97.2, 88.9]; + const r = 20; + const grid = await new Stamper(device).stamp(emptyGrid(device), { shape: sphere([center[0], center[1], center[2]], r), mode: 'add', halfWidth }); + const mesh = await new Extractor(device).extract(grid, halfWidth); + const pts = new Float32Array((await readBackU32(device, mesh.points, mesh.triangleCount * 9)).buffer); + let maxDev = 0; + let area = 0; + for (let t = 0; t < mesh.triangleCount; t++) { + const p = [0, 1, 2].map((v) => [pts[t * 9 + v * 3], pts[t * 9 + v * 3 + 1], pts[t * 9 + v * 3 + 2]]); + for (const q of p) maxDev = Math.max(maxDev, Math.abs(Math.hypot(q[0] - center[0], q[1] - center[1], q[2] - center[2]) - r)); + const u = p[1].map((x, i) => x - p[0][i]); + const w = p[2].map((x, i) => x - p[0][i]); + const c = [u[1] * w[2] - u[2] * w[1], u[2] * w[0] - u[0] * w[2], u[0] * w[1] - u[1] * w[0]]; + area += 0.5 * Math.hypot(c[0], c[1], c[2]); + } + const expected = 4 * Math.PI * r * r; + console.log(` extract: ${mesh.triangleCount} triangles, max deviation ${maxDev.toFixed(4)} voxels, area ${area.toFixed(0)} vs ${expected.toFixed(0)}`); + if (maxDev > 0.05) throw new Error(`vertices off the sphere by ${maxDev}`); + if (Math.abs(area - expected) / expected > 0.02) throw new Error(`area ${area} vs ${expected}`); +}); diff --git a/ts/gpu/extract.ts b/ts/gpu/extract.ts new file mode 100644 index 0000000..16be989 --- /dev/null +++ b/ts/gpu/extract.ts @@ -0,0 +1,92 @@ +// Host side of wgsl/extract.wgsl: a level set of an op layer grid as +// triangles, for redistancing and export. + +import extractWgsl from 'picovdb/wgsl/extract.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { createU32Buffer, dispatch2D, readBackTotals } from './device.ts'; +import { MC_TRI_COUNT, MC_TRI_TABLE } from './mc_table.ts'; +import { preludeWgsl, readerWgsl, type OpGrid } from './opgrid.ts'; + +export interface Mesh { + /** xyz f32 triples, three unshared vertices per triangle, in the grid's relative voxel coordinates. */ + points: GPUBuffer; + /** Vertex index triples 0..3n. */ + triangles: GPUBuffer; + triangleCount: number; +} + +export class Extractor { + readonly device: GPUDevice; + readonly scanner: Scanner; + private readonly pipelines: Record = {}; + private readonly table: GPUBuffer; + + constructor(device: GPUDevice, scanner = new Scanner(device)) { + this.device = device; + this.scanner = scanner; + const module = device.createShaderModule({ code: preludeWgsl + readerWgsl('old', 'params.leaf_count') + extractWgsl }); + for (const entryPoint of ['count', 'emit']) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint } }); + } + const table = new Int32Array(MC_TRI_TABLE.length + MC_TRI_COUNT.length); + table.set(MC_TRI_TABLE); + table.set(MC_TRI_COUNT, MC_TRI_TABLE.length); + this.table = createU32Buffer(device, new Uint32Array(table.buffer)); + } + + /** The level set at iso as triangles. iso must lie inside the band. */ + async extract(grid: OpGrid, halfWidth: number, iso = 0): Promise { + const device = this.device; + if (grid.leafCount === 0) throw new Error('empty grid'); + const params = device.createBuffer({ size: 16, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([grid.leafCount])); + device.queue.writeBuffer(params, 4, new Float32Array([halfWidth, iso])); + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const counts = device.createBuffer({ size: (grid.leafCount + 1) * 4, usage: storage }); + + // Auto layouts only hold the bindings an entry point uses. + const run = (pass: GPUComputePassEncoder, name: string, extra: GPUBindGroupEntry[]) => { + pass.setPipeline(this.pipelines[name]); + pass.setBindGroup( + 0, + device.createBindGroup({ + layout: this.pipelines[name].getBindGroupLayout(0), + entries: [ + { binding: 0, resource: { buffer: params } }, + { binding: 1, resource: { buffer: grid.leafKeys } }, + { binding: 2, resource: { buffer: grid.leaves } }, + { binding: 7, resource: { buffer: grid.data } }, + { binding: 3, resource: { buffer: this.table } }, + { binding: 4, resource: { buffer: counts } }, + ...extra, + ], + }) + ); + dispatch2D(pass, grid.leafCount); + }; + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'count', []); + this.scanner.plan(counts, grid.leafCount + 1).encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [triangleCount] = await readBackTotals(device, [{ buffer: counts, index: grid.leafCount }]); + if (triangleCount === 0) throw new Error('no surface'); + const points = device.createBuffer({ size: triangleCount * 9 * 4, usage: storage }); + const triangles = device.createBuffer({ size: triangleCount * 3 * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'emit', [ + { binding: 5, resource: { buffer: points } }, + { binding: 6, resource: { buffer: triangles } }, + ]); + pass.end(); + device.queue.submit([encoder.finish()]); + } + counts.destroy(); + return { points, triangles, triangleCount }; + } +} diff --git a/ts/gpu/grid_ops.test.ts b/ts/gpu/grid_ops.test.ts new file mode 100644 index 0000000..6ff7fad --- /dev/null +++ b/ts/gpu/grid_ops.test.ts @@ -0,0 +1,115 @@ +import { hasWebGPU, requestDevice, createU32Buffer, readBackU32 } from './device.ts'; +import { mulberry32, assertU32ArrayEqual, packGrid, unpackGrid } from './test_util.ts'; +import { refCsgMerge } from './reference.ts'; +import { Pruner } from './prune.ts'; +import { Merger } from './merge.ts'; + +const gpu = await hasWebGPU(); + +const pack = (x: number, y: number, z: number): number => ((x << 20) | (y << 10) | z) >>> 0; + +function randomGrid(rand: () => number, count: number, withValues: boolean) { + const keySet = new Set(); + while (keySet.size < count) { + keySet.add(pack(10 + Math.floor(rand() * 5), 10 + Math.floor(rand() * 5), 10 + Math.floor(rand() * 5))); + } + const keys = new Uint32Array([...keySet].sort((a, b) => a - b)); + const masks = new Uint32Array(keys.length * 16); + for (let i = 0; i < masks.length; i++) { + if (rand() < 0.4) masks[i] = Math.floor(rand() * 4294967296); + } + const values = withValues ? new Float32Array(keys.length * 512) : undefined; + if (values) for (let i = 0; i < values.length; i++) values[i] = Math.fround((rand() - 0.5) * 6); + return { keys, masks, values }; +} + +Deno.test({ name: 'prune matches reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const pruner = new Pruner(device); + const rand = mulberry32(7); + const { keys, masks } = randomGrid(rand, 40, false); + const retain = new Uint32Array(keys.length * 16); + for (let i = 0; i < retain.length; i++) { + if (rand() < 0.5) retain[i] = Math.floor(rand() * 4294967296); + } + + const result = await pruner.prune( + createU32Buffer(device, keys), + createU32Buffer(device, masks), + createU32Buffer(device, retain), + keys.length + ); + + // The reference ANDs and drops empty leaves. + const refKeys: number[] = []; + const refMasks: number[] = []; + for (let i = 0; i < keys.length; i++) { + const anded = []; + let any = 0; + for (let w = 0; w < 16; w++) { + const m = (masks[i * 16 + w] & retain[i * 16 + w]) >>> 0; + anded.push(m); + any |= m; + } + if (any) { + refKeys.push(keys[i]); + refMasks.push(...anded); + } + } + if (result.leafCount !== refKeys.length) throw new Error(`count ${result.leafCount} != ${refKeys.length}`); + assertU32ArrayEqual(await readBackU32(device, result.leafKeys, result.leafCount), new Uint32Array(refKeys), 'prune keys'); + assertU32ArrayEqual(await readBackU32(device, result.masks, result.leafCount * 16), new Uint32Array(refMasks), 'prune masks'); +}); + +Deno.test({ name: 'topology merge matches reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const merger = new Merger(device); + const rand = mulberry32(8); + const a = randomGrid(rand, 30, false); + const b = randomGrid(rand, 30, false); // overlapping coordinate range + + const result = await merger.merge( + { leafKeys: createU32Buffer(device, a.keys), masks: createU32Buffer(device, a.masks), leafCount: a.keys.length }, + { leafKeys: createU32Buffer(device, b.keys), masks: createU32Buffer(device, b.masks), leafCount: b.keys.length } + ); + + // The reference is the sorted key union with OR masks. + const union = new Uint32Array([...new Set([...a.keys, ...b.keys])].sort((x, y) => x - y)); + if (result.leafCount !== union.length) throw new Error(`count ${result.leafCount} != ${union.length}`); + assertU32ArrayEqual(await readBackU32(device, result.leafKeys, result.leafCount), union, 'merge keys'); + + const aIdx = new Map(); + a.keys.forEach((k, i) => aIdx.set(k, i)); + const bIdx = new Map(); + b.keys.forEach((k, i) => bIdx.set(k, i)); + const refMasks = new Uint32Array(union.length * 16); + union.forEach((key, i) => { + const ai = aIdx.get(key); + const bi = bIdx.get(key); + for (let w = 0; w < 16; w++) { + refMasks[i * 16 + w] = ((ai !== undefined ? a.masks[ai * 16 + w] : 0) | (bi !== undefined ? b.masks[bi * 16 + w] : 0)) >>> 0; + } + }); + assertU32ArrayEqual(await readBackU32(device, result.masks, result.leafCount * 16), refMasks, 'merge masks'); +}); + +Deno.test({ name: 'CSG merge matches reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const merger = new Merger(device); + const rand = mulberry32(9); + const halfWidth = 3; + const a = randomGrid(rand, 30, true); + const b = randomGrid(rand, 30, true); // overlapping coordinate range + const ga = packGrid(device, a.keys, a.values!, halfWidth); + const gb = packGrid(device, b.keys, b.values!, halfWidth); + + for (const op of ['union', 'intersect', 'subtract'] as const) { + const result = await merger.mergeCsg(ga, gb, { halfWidth, op }); + const got = await unpackGrid(device, result, halfWidth); + const ref = refCsgMerge(a.keys, a.values!, b.keys, b.values!, halfWidth, op); + if (result.leafCount !== ref.keys.length) throw new Error(`${op} count ${result.leafCount} != ${ref.keys.length}`); + assertU32ArrayEqual(got.keys, ref.keys, `${op} keys`); + assertU32ArrayEqual(got.masks, ref.masks, `${op} masks`); + assertU32ArrayEqual(new Uint32Array(got.values.buffer), new Uint32Array(ref.values.buffer), `${op} values`); + } +}); diff --git a/ts/gpu/load.test.ts b/ts/gpu/load.test.ts new file mode 100644 index 0000000..3a0390b --- /dev/null +++ b/ts/gpu/load.test.ts @@ -0,0 +1,95 @@ +import { hasWebGPU, requestDevice, readBackU32 } from './device.ts'; +import { compareTreeToCpu } from './test_util.ts'; +import { Emitter } from './emit.ts'; +import { Loader, leafOrigins } from './load.ts'; +import { initSTL, importSTL } from '../stl.ts'; +import { PicoVDBFile } from '../picovdb.ts'; + +const gpu = await hasWebGPU(); + +let stl: Uint8Array | null = null; +let wasm: Uint8Array | null = null; +let bunny: Uint8Array | null = null; +let bunnyU8: Uint8Array | null = null; +try { + stl = Deno.readFileSync(new URL('../../data/bases/base_32mm.stl', import.meta.url)); + wasm = Deno.readFileSync(new URL('../../zig-out/wasm/picovdb.wasm', import.meta.url)); +} catch { + // skip +} +try { + bunny = Deno.readFileSync(new URL('../../data/bunny.pvdb', import.meta.url)); +} catch { + // skip +} +try { + const gz = Deno.readFileSync(new URL('../../data/bunny.u8.pvdb.gz', import.meta.url)); + const raw = await new Response(new Blob([gz]).stream().pipeThrough(new DecompressionStream('gzip'))).arrayBuffer(); + bunnyU8 = new Uint8Array(raw); +} catch { + // skip +} + +Deno.test({ name: 'load then emit reproduces the CPU tree', ignore: !gpu || !stl || !wasm }, async () => { + const device = await requestDevice(); + initSTL({ wasmBinary: wasm! }); + const cpu = (await importSTL(stl!, { voxelsPerUnit: 4 })).file; + + const grid = new Loader(device).load(cpu, { halfWidth: 3 }); + const tree = await new Emitter(device).reEmit(grid, { halfWidth: 3 }); + const maxAbs = await compareTreeToCpu(device, tree, cpu); + console.log(` load round trip: ${tree.leafCount} leaves, ${tree.activeVoxels} active, max |Δv| ${maxAbs.toExponential(2)}`); +}); + +Deno.test({ name: 'leaf origins recover the leaf order of a converted file', ignore: !bunny }, () => { + const file = new PicoVDBFile(bunny!.buffer); + const origins = leafOrigins(file); + const grid = file.getGrid(0); + // Every origin sits inside the grid's index bounds and on a leaf boundary. + for (let i = 0; i < file.header.leafCount; i++) { + for (let a = 0; a < 3; a++) { + const o = origins[i * 3 + a]; + if (o & 7) throw new Error(`leaf ${i} origin ${o} not leaf aligned`); + if (o + 7 < grid.indexBoundsMin[a] || o > grid.indexBoundsMax[a]) throw new Error(`leaf ${i} origin ${o} outside bounds on axis ${a}`); + } + } +}); + +Deno.test({ name: 'bunny loads and re-emits with the same topology', ignore: !gpu || !bunny }, async () => { + const device = await requestDevice(); + const file = new PicoVDBFile(bunny!.buffer); + const grid = new Loader(device).load(file, { halfWidth: 3 }); + const tree = await new Emitter(device).reEmit(grid, { halfWidth: 3 }); + const h = file.header; + if (tree.leafCount !== h.leafCount || tree.lowerCount !== h.lowerCount || tree.upperCount !== h.upperCount) { + throw new Error(`counts: gpu ${tree.leafCount}/${tree.lowerCount}/${tree.upperCount} != file ${h.leafCount}/${h.lowerCount}/${h.upperCount}`); + } + if (tree.dataElemCount !== file.getGrid(0).dataElemCount) throw new Error('active voxel count differs'); + // Values scaled from the file's background to the half width. + const data = new Float32Array((await readBackU32(device, tree.data, tree.dataElemCount)).buffer); + const src = new Float32Array(file.dataBuffer.buffer, file.dataBuffer.byteOffset, tree.dataElemCount); + const scale = 3 / src[0]; + let maxAbs = 0; + for (let i = 2; i < tree.dataElemCount; i++) maxAbs = Math.max(maxAbs, Math.abs(data[i] - src[i] * scale)); + if (maxAbs > 1e-4) throw new Error(`value divergence ${maxAbs}`); + console.log(` bunny: ${tree.leafCount} leaves, ${tree.activeVoxels} active, ${tree.surfaceVoxels} surface, max |Δv| ${maxAbs.toExponential(2)}`); +}); + +Deno.test({ name: 'u8 bunny loads to the f32 bunny within quantization', ignore: !gpu || !bunny || !bunnyU8 }, async () => { + const device = await requestDevice(); + const loader = new Loader(device); + const f32 = loader.load(new PicoVDBFile(bunny!.buffer), { halfWidth: 3 }); + const u8 = loader.load(new PicoVDBFile(bunnyU8!.buffer), { halfWidth: 3 }); + if (u8.leafCount !== f32.leafCount) throw new Error(`leaf counts ${u8.leafCount} != ${f32.leafCount}`); + const keysA = await readBackU32(device, f32.leafKeys, f32.leafCount); + const keysB = await readBackU32(device, u8.leafKeys, u8.leafCount); + for (let i = 0; i < keysA.length; i++) if (keysA[i] !== keysB[i]) throw new Error(`leaf key ${i} differs`); + if (u8.activeVoxels !== f32.activeVoxels) throw new Error(`active counts ${u8.activeVoxels} != ${f32.activeVoxels}`); + const a = new Float32Array((await readBackU32(device, f32.data, 2 + f32.activeVoxels)).buffer); + const b = new Float32Array((await readBackU32(device, u8.data, 2 + u8.activeVoxels)).buffer); + let maxAbs = 0; + for (let i = 0; i < a.length; i++) maxAbs = Math.max(maxAbs, Math.abs(a[i] - b[i])); + // One u8 step is 6 / 255 voxels; the f32 file is 3 / 0.15 = 20x rescaled. + if (maxAbs > 6 / 255) throw new Error(`u8 values off by ${maxAbs}`); + console.log(` u8 bunny: ${u8.leafCount} leaves, max |Δv| ${maxAbs.toExponential(2)}`); +}); diff --git a/ts/gpu/load.ts b/ts/gpu/load.ts new file mode 100644 index 0000000..d23138e --- /dev/null +++ b/ts/gpu/load.ts @@ -0,0 +1,174 @@ +// Host side of wgsl/load.wgsl: a picovdb file as an op layer grid. The +// CPU walks the tree for the leaf origins; the GPU reorders the leaves +// and rescales the values. + +import loadWgsl from 'picovdb/wgsl/load.wgsl' with { type: 'text' }; +import { GRID_TYPE_SDF_FLOAT, GRID_TYPE_SDF_UINT8, PICOVDB_LOWER_SIZE, PICOVDB_UPPER_SIZE, type PicoVDBFile } from '../picovdb.ts'; +import { checkBindingSize, createU32Buffer, dispatch2D } from './device.ts'; +import { LEAF_U32, type OpGrid } from './opgrid.ts'; + +const WG_SIZE = 256; + +export interface LoadOptions { + /** Narrow band half width in voxels of the op layer. */ + halfWidth: number; + /** Grid index in the file. */ + grid?: number; +} + +/** + * Leaf origins in voxels for every leaf of one grid, in leaf index order. + * Walks roots, uppers, and lowers using the child counts packed per mask + * word. + */ +export function leafOrigins(file: PicoVDBFile, gridIndex = 0): Int32Array { + const u32 = (bytes: Uint8Array) => new Uint32Array(bytes.buffer, bytes.byteOffset, bytes.byteLength / 4); + const roots = u32(file.rootsBuffer); + const uppers = u32(file.uppersBuffer); + const lowers = u32(file.lowersBuffer); + const range = file.getGridRange(gridIndex); + const lowerOrigins = new Int32Array(range.lowerCount * 3); + const origins = new Int32Array(range.leafCount * 3); + // Child indices in a node are relative to the grid's first child node. + const walk = ( + node: Uint32Array, + base: number, + words: number, + bits: number, + origin: [number, number, number], + childSize: number, + out: Int32Array + ) => { + const first = node[base]; + for (let w = 0; w < words; w++) { + const children = node[base + 4 + w * 3] & node[base + 5 + w * 3]; + if (children === 0) continue; + let index = first + (node[base + 6 + w * 3] >>> 16); + for (let b = 0; b < 32; b++) { + if (((children >>> b) & 1) === 0) continue; + const n = w * 32 + b; + const mask = (1 << bits) - 1; + out[index * 3] = origin[0] + ((n >>> (2 * bits)) & mask) * childSize; + out[index * 3 + 1] = origin[1] + ((n >>> bits) & mask) * childSize; + out[index * 3 + 2] = origin[2] + (n & mask) * childSize; + index++; + } + } + }; + for (let u = 0; u < range.upperCount; u++) { + // Mirrors picovdb coordToKey. + const w0 = roots[(range.upperStart + u) * 2]; + const w1 = roots[(range.upperStart + u) * 2 + 1]; + const iu = w1 >>> 10; + const ju = (w0 >>> 21) | ((w1 & 0x3ff) << 11); + const ku = w0 & 0x1fffff; + const origin: [number, number, number] = [(iu << 12) | 0, (ju << 12) | 0, (ku << 12) | 0]; + walk(uppers, (range.upperStart + u) * (PICOVDB_UPPER_SIZE / 4), 1024, 5, origin, 128, lowerOrigins); + } + for (let l = 0; l < range.lowerCount; l++) { + const origin: [number, number, number] = [lowerOrigins[l * 3], lowerOrigins[l * 3 + 1], lowerOrigins[l * 3 + 2]]; + walk(lowers, (range.lowerStart + l) * (PICOVDB_LOWER_SIZE / 4), 128, 4, origin, 8, origins); + } + return origins; +} + +export class Loader { + readonly device: GPUDevice; + private readonly pipelines: Record = {}; + + constructor(device: GPUDevice) { + this.device = device; + const module = device.createShaderModule({ code: loadWgsl }); + for (const entryPoint of ['gather_leaves', 'convert_data']) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint } }); + } + } + + /** + * Loads one grid of an f32 or u8 SDF file. Values scale so the file's + * background equals the half width. That also brings world unit files + * into voxel units. u8 values map to [-3, 3], as the renderer reads + * them. + */ + load(file: PicoVDBFile, opts: LoadOptions): OpGrid { + const device = this.device; + const gridIndex = opts.grid ?? 0; + const grid = file.getGrid(gridIndex); + if (grid.gridType !== GRID_TYPE_SDF_FLOAT && grid.gridType !== GRID_TYPE_SDF_UINT8) { + throw new Error(`grid type ${grid.gridType} is not an SDF`); + } + const range = file.getGridRange(gridIndex); + const leafCount = range.leafCount; + if (leafCount === 0) throw new Error('empty tree'); + const u8 = grid.gridType === GRID_TYPE_SDF_UINT8; + // The grid's records and values. dataStart counts 16 byte units. + const leafBytes = file.leavesBuffer.subarray(range.leafStart * LEAF_U32 * 4, (range.leafStart + leafCount) * LEAF_U32 * 4); + const dataBytes = file.dataBuffer.subarray(range.dataStart * 16, range.dataStart * 16 + Math.ceil((grid.dataElemCount * (u8 ? 1 : 4)) / 4) * 4); + const background = u8 ? (dataBytes[0] / 127.5 - 1) * 3 : new Float32Array(dataBytes.buffer, dataBytes.byteOffset, 1)[0]; + if (!(background > 0)) throw new Error(`background ${background} is not positive`); + + const origins = leafOrigins(file, gridIndex); + const leafMin: [number, number, number] = [Infinity, Infinity, Infinity]; + const leafMax: [number, number, number] = [-Infinity, -Infinity, -Infinity]; + for (let i = 0; i < leafCount; i++) { + for (let a = 0; a < 3; a++) { + const c = origins[i * 3 + a] >> 3; + if (c < leafMin[a]) leafMin[a] = c; + if (c > leafMax[a]) leafMax[a] = c; + } + } + for (let a = 0; a < 3; a++) { + if (leafMax[a] - leafMin[a] >= 1024) throw new Error(`tree exceeds 1024 leaves on axis ${a}`); + } + const keys = new Uint32Array(leafCount); + for (let i = 0; i < leafCount; i++) { + keys[i] = + (((origins[i * 3] >> 3) - leafMin[0]) << 20) | + (((origins[i * 3 + 1] >> 3) - leafMin[1]) << 10) | + ((origins[i * 3 + 2] >> 3) - leafMin[2]); + } + const order = Array.from(keys.keys()).sort((a, b) => keys[a] - keys[b]); + const sortedKeys = new Uint32Array(leafCount); + for (let j = 0; j < leafCount; j++) sortedKeys[j] = keys[order[j]]; + + const dataCount = grid.dataElemCount; + checkBindingSize(device, dataCount * 4, 'loaded values'); + const params = device.createBuffer({ size: 16, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([leafCount, dataCount])); + device.queue.writeBuffer(params, 8, new Float32Array([opts.halfWidth / background])); + device.queue.writeBuffer(params, 12, new Uint32Array([grid.gridType])); + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const leafKeys = createU32Buffer(device, sortedKeys); + const orderBuffer = createU32Buffer(device, Uint32Array.from(order)); + const fileLeaves = createU32Buffer(device, new Uint32Array(leafBytes.buffer, leafBytes.byteOffset, leafCount * LEAF_U32)); + const fileData = createU32Buffer(device, new Uint32Array(dataBytes.buffer, dataBytes.byteOffset, dataBytes.byteLength / 4)); + const leaves = device.createBuffer({ size: leafCount * LEAF_U32 * 4, usage: storage }); + const data = device.createBuffer({ size: dataCount * 4, usage: storage }); + + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + const run = (name: string, threads: number, buffers: Record) => { + pass.setPipeline(this.pipelines[name]); + pass.setBindGroup( + 0, + device.createBindGroup({ + layout: this.pipelines[name].getBindGroupLayout(0), + entries: [ + { binding: 0, resource: { buffer: params } }, + ...Object.entries(buffers).map(([binding, buffer]) => ({ binding: Number(binding), resource: { buffer } })), + ], + }) + ); + dispatch2D(pass, Math.ceil(threads / WG_SIZE)); + }; + run('gather_leaves', leafCount, { 1: orderBuffer, 2: fileLeaves, 4: leaves }); + run('convert_data', dataCount, { 3: fileData, 5: data }); + pass.end(); + device.queue.submit([encoder.finish()]); + orderBuffer.destroy(); + fileLeaves.destroy(); + fileData.destroy(); + params.destroy(); + return { leafKeys, leaves, data, leafCount, activeVoxels: dataCount - 2, leafMin, leafMax }; + } +} diff --git a/ts/gpu/mc_table.ts b/ts/gpu/mc_table.ts new file mode 100644 index 0000000..368d911 --- /dev/null +++ b/ts/gpu/mc_table.ts @@ -0,0 +1,270 @@ +// Marching cubes triangle table (Paul Bourke's classic tables, as +// distributed with three.js under the MIT license). Row c lists up to five +// edge index triples for cube configuration c, terminated by -1. Corner +// bit i is set when corner i is inside (value < 0); corners order +// (0,0,0) (1,0,0) (1,1,0) (0,1,0) (0,0,1) (1,0,1) (1,1,1) (0,1,1) and +// edges 0..11 join corners 0-1 1-2 2-3 3-0 4-5 5-6 6-7 7-4 0-4 1-5 2-6 3-7. + +export const MC_TRI_TABLE = new Int32Array([ + -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 1, 9, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 8, 3, 9, 8, 1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 10, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 3, 1, 2, 10, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 9, 2, 10, 0, 2, 9, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 2, 8, 3, 2, 10, 8, 10, 9, 8, -1, -1, -1, -1, -1, -1, -1, + 3, 11, 2, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 11, 2, 8, 11, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 9, 0, 2, 3, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 11, 2, 1, 9, 11, 9, 8, 11, -1, -1, -1, -1, -1, -1, -1, + 3, 10, 1, 11, 10, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 10, 1, 0, 8, 10, 8, 11, 10, -1, -1, -1, -1, -1, -1, -1, + 3, 9, 0, 3, 11, 9, 11, 10, 9, -1, -1, -1, -1, -1, -1, -1, + 9, 8, 10, 10, 8, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 7, 8, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 3, 0, 7, 3, 4, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 1, 9, 8, 4, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 1, 9, 4, 7, 1, 7, 3, 1, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 10, 8, 4, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 3, 4, 7, 3, 0, 4, 1, 2, 10, -1, -1, -1, -1, -1, -1, -1, + 9, 2, 10, 9, 0, 2, 8, 4, 7, -1, -1, -1, -1, -1, -1, -1, + 2, 10, 9, 2, 9, 7, 2, 7, 3, 7, 9, 4, -1, -1, -1, -1, + 8, 4, 7, 3, 11, 2, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 11, 4, 7, 11, 2, 4, 2, 0, 4, -1, -1, -1, -1, -1, -1, -1, + 9, 0, 1, 8, 4, 7, 2, 3, 11, -1, -1, -1, -1, -1, -1, -1, + 4, 7, 11, 9, 4, 11, 9, 11, 2, 9, 2, 1, -1, -1, -1, -1, + 3, 10, 1, 3, 11, 10, 7, 8, 4, -1, -1, -1, -1, -1, -1, -1, + 1, 11, 10, 1, 4, 11, 1, 0, 4, 7, 11, 4, -1, -1, -1, -1, + 4, 7, 8, 9, 0, 11, 9, 11, 10, 11, 0, 3, -1, -1, -1, -1, + 4, 7, 11, 4, 11, 9, 9, 11, 10, -1, -1, -1, -1, -1, -1, -1, + 9, 5, 4, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 9, 5, 4, 0, 8, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 5, 4, 1, 5, 0, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 8, 5, 4, 8, 3, 5, 3, 1, 5, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 10, 9, 5, 4, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 3, 0, 8, 1, 2, 10, 4, 9, 5, -1, -1, -1, -1, -1, -1, -1, + 5, 2, 10, 5, 4, 2, 4, 0, 2, -1, -1, -1, -1, -1, -1, -1, + 2, 10, 5, 3, 2, 5, 3, 5, 4, 3, 4, 8, -1, -1, -1, -1, + 9, 5, 4, 2, 3, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 11, 2, 0, 8, 11, 4, 9, 5, -1, -1, -1, -1, -1, -1, -1, + 0, 5, 4, 0, 1, 5, 2, 3, 11, -1, -1, -1, -1, -1, -1, -1, + 2, 1, 5, 2, 5, 8, 2, 8, 11, 4, 8, 5, -1, -1, -1, -1, + 10, 3, 11, 10, 1, 3, 9, 5, 4, -1, -1, -1, -1, -1, -1, -1, + 4, 9, 5, 0, 8, 1, 8, 10, 1, 8, 11, 10, -1, -1, -1, -1, + 5, 4, 0, 5, 0, 11, 5, 11, 10, 11, 0, 3, -1, -1, -1, -1, + 5, 4, 8, 5, 8, 10, 10, 8, 11, -1, -1, -1, -1, -1, -1, -1, + 9, 7, 8, 5, 7, 9, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 9, 3, 0, 9, 5, 3, 5, 7, 3, -1, -1, -1, -1, -1, -1, -1, + 0, 7, 8, 0, 1, 7, 1, 5, 7, -1, -1, -1, -1, -1, -1, -1, + 1, 5, 3, 3, 5, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 9, 7, 8, 9, 5, 7, 10, 1, 2, -1, -1, -1, -1, -1, -1, -1, + 10, 1, 2, 9, 5, 0, 5, 3, 0, 5, 7, 3, -1, -1, -1, -1, + 8, 0, 2, 8, 2, 5, 8, 5, 7, 10, 5, 2, -1, -1, -1, -1, + 2, 10, 5, 2, 5, 3, 3, 5, 7, -1, -1, -1, -1, -1, -1, -1, + 7, 9, 5, 7, 8, 9, 3, 11, 2, -1, -1, -1, -1, -1, -1, -1, + 9, 5, 7, 9, 7, 2, 9, 2, 0, 2, 7, 11, -1, -1, -1, -1, + 2, 3, 11, 0, 1, 8, 1, 7, 8, 1, 5, 7, -1, -1, -1, -1, + 11, 2, 1, 11, 1, 7, 7, 1, 5, -1, -1, -1, -1, -1, -1, -1, + 9, 5, 8, 8, 5, 7, 10, 1, 3, 10, 3, 11, -1, -1, -1, -1, + 5, 7, 0, 5, 0, 9, 7, 11, 0, 1, 0, 10, 11, 10, 0, -1, + 11, 10, 0, 11, 0, 3, 10, 5, 0, 8, 0, 7, 5, 7, 0, -1, + 11, 10, 5, 7, 11, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 10, 6, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 3, 5, 10, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 9, 0, 1, 5, 10, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 8, 3, 1, 9, 8, 5, 10, 6, -1, -1, -1, -1, -1, -1, -1, + 1, 6, 5, 2, 6, 1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 6, 5, 1, 2, 6, 3, 0, 8, -1, -1, -1, -1, -1, -1, -1, + 9, 6, 5, 9, 0, 6, 0, 2, 6, -1, -1, -1, -1, -1, -1, -1, + 5, 9, 8, 5, 8, 2, 5, 2, 6, 3, 2, 8, -1, -1, -1, -1, + 2, 3, 11, 10, 6, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 11, 0, 8, 11, 2, 0, 10, 6, 5, -1, -1, -1, -1, -1, -1, -1, + 0, 1, 9, 2, 3, 11, 5, 10, 6, -1, -1, -1, -1, -1, -1, -1, + 5, 10, 6, 1, 9, 2, 9, 11, 2, 9, 8, 11, -1, -1, -1, -1, + 6, 3, 11, 6, 5, 3, 5, 1, 3, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 11, 0, 11, 5, 0, 5, 1, 5, 11, 6, -1, -1, -1, -1, + 3, 11, 6, 0, 3, 6, 0, 6, 5, 0, 5, 9, -1, -1, -1, -1, + 6, 5, 9, 6, 9, 11, 11, 9, 8, -1, -1, -1, -1, -1, -1, -1, + 5, 10, 6, 4, 7, 8, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 3, 0, 4, 7, 3, 6, 5, 10, -1, -1, -1, -1, -1, -1, -1, + 1, 9, 0, 5, 10, 6, 8, 4, 7, -1, -1, -1, -1, -1, -1, -1, + 10, 6, 5, 1, 9, 7, 1, 7, 3, 7, 9, 4, -1, -1, -1, -1, + 6, 1, 2, 6, 5, 1, 4, 7, 8, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 5, 5, 2, 6, 3, 0, 4, 3, 4, 7, -1, -1, -1, -1, + 8, 4, 7, 9, 0, 5, 0, 6, 5, 0, 2, 6, -1, -1, -1, -1, + 7, 3, 9, 7, 9, 4, 3, 2, 9, 5, 9, 6, 2, 6, 9, -1, + 3, 11, 2, 7, 8, 4, 10, 6, 5, -1, -1, -1, -1, -1, -1, -1, + 5, 10, 6, 4, 7, 2, 4, 2, 0, 2, 7, 11, -1, -1, -1, -1, + 0, 1, 9, 4, 7, 8, 2, 3, 11, 5, 10, 6, -1, -1, -1, -1, + 9, 2, 1, 9, 11, 2, 9, 4, 11, 7, 11, 4, 5, 10, 6, -1, + 8, 4, 7, 3, 11, 5, 3, 5, 1, 5, 11, 6, -1, -1, -1, -1, + 5, 1, 11, 5, 11, 6, 1, 0, 11, 7, 11, 4, 0, 4, 11, -1, + 0, 5, 9, 0, 6, 5, 0, 3, 6, 11, 6, 3, 8, 4, 7, -1, + 6, 5, 9, 6, 9, 11, 4, 7, 9, 7, 11, 9, -1, -1, -1, -1, + 10, 4, 9, 6, 4, 10, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 10, 6, 4, 9, 10, 0, 8, 3, -1, -1, -1, -1, -1, -1, -1, + 10, 0, 1, 10, 6, 0, 6, 4, 0, -1, -1, -1, -1, -1, -1, -1, + 8, 3, 1, 8, 1, 6, 8, 6, 4, 6, 1, 10, -1, -1, -1, -1, + 1, 4, 9, 1, 2, 4, 2, 6, 4, -1, -1, -1, -1, -1, -1, -1, + 3, 0, 8, 1, 2, 9, 2, 4, 9, 2, 6, 4, -1, -1, -1, -1, + 0, 2, 4, 4, 2, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 8, 3, 2, 8, 2, 4, 4, 2, 6, -1, -1, -1, -1, -1, -1, -1, + 10, 4, 9, 10, 6, 4, 11, 2, 3, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 2, 2, 8, 11, 4, 9, 10, 4, 10, 6, -1, -1, -1, -1, + 3, 11, 2, 0, 1, 6, 0, 6, 4, 6, 1, 10, -1, -1, -1, -1, + 6, 4, 1, 6, 1, 10, 4, 8, 1, 2, 1, 11, 8, 11, 1, -1, + 9, 6, 4, 9, 3, 6, 9, 1, 3, 11, 6, 3, -1, -1, -1, -1, + 8, 11, 1, 8, 1, 0, 11, 6, 1, 9, 1, 4, 6, 4, 1, -1, + 3, 11, 6, 3, 6, 0, 0, 6, 4, -1, -1, -1, -1, -1, -1, -1, + 6, 4, 8, 11, 6, 8, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 7, 10, 6, 7, 8, 10, 8, 9, 10, -1, -1, -1, -1, -1, -1, -1, + 0, 7, 3, 0, 10, 7, 0, 9, 10, 6, 7, 10, -1, -1, -1, -1, + 10, 6, 7, 1, 10, 7, 1, 7, 8, 1, 8, 0, -1, -1, -1, -1, + 10, 6, 7, 10, 7, 1, 1, 7, 3, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 6, 1, 6, 8, 1, 8, 9, 8, 6, 7, -1, -1, -1, -1, + 2, 6, 9, 2, 9, 1, 6, 7, 9, 0, 9, 3, 7, 3, 9, -1, + 7, 8, 0, 7, 0, 6, 6, 0, 2, -1, -1, -1, -1, -1, -1, -1, + 7, 3, 2, 6, 7, 2, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 2, 3, 11, 10, 6, 8, 10, 8, 9, 8, 6, 7, -1, -1, -1, -1, + 2, 0, 7, 2, 7, 11, 0, 9, 7, 6, 7, 10, 9, 10, 7, -1, + 1, 8, 0, 1, 7, 8, 1, 10, 7, 6, 7, 10, 2, 3, 11, -1, + 11, 2, 1, 11, 1, 7, 10, 6, 1, 6, 7, 1, -1, -1, -1, -1, + 8, 9, 6, 8, 6, 7, 9, 1, 6, 11, 6, 3, 1, 3, 6, -1, + 0, 9, 1, 11, 6, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 7, 8, 0, 7, 0, 6, 3, 11, 0, 11, 6, 0, -1, -1, -1, -1, + 7, 11, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 7, 6, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 3, 0, 8, 11, 7, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 1, 9, 11, 7, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 8, 1, 9, 8, 3, 1, 11, 7, 6, -1, -1, -1, -1, -1, -1, -1, + 10, 1, 2, 6, 11, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 10, 3, 0, 8, 6, 11, 7, -1, -1, -1, -1, -1, -1, -1, + 2, 9, 0, 2, 10, 9, 6, 11, 7, -1, -1, -1, -1, -1, -1, -1, + 6, 11, 7, 2, 10, 3, 10, 8, 3, 10, 9, 8, -1, -1, -1, -1, + 7, 2, 3, 6, 2, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 7, 0, 8, 7, 6, 0, 6, 2, 0, -1, -1, -1, -1, -1, -1, -1, + 2, 7, 6, 2, 3, 7, 0, 1, 9, -1, -1, -1, -1, -1, -1, -1, + 1, 6, 2, 1, 8, 6, 1, 9, 8, 8, 7, 6, -1, -1, -1, -1, + 10, 7, 6, 10, 1, 7, 1, 3, 7, -1, -1, -1, -1, -1, -1, -1, + 10, 7, 6, 1, 7, 10, 1, 8, 7, 1, 0, 8, -1, -1, -1, -1, + 0, 3, 7, 0, 7, 10, 0, 10, 9, 6, 10, 7, -1, -1, -1, -1, + 7, 6, 10, 7, 10, 8, 8, 10, 9, -1, -1, -1, -1, -1, -1, -1, + 6, 8, 4, 11, 8, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 3, 6, 11, 3, 0, 6, 0, 4, 6, -1, -1, -1, -1, -1, -1, -1, + 8, 6, 11, 8, 4, 6, 9, 0, 1, -1, -1, -1, -1, -1, -1, -1, + 9, 4, 6, 9, 6, 3, 9, 3, 1, 11, 3, 6, -1, -1, -1, -1, + 6, 8, 4, 6, 11, 8, 2, 10, 1, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 10, 3, 0, 11, 0, 6, 11, 0, 4, 6, -1, -1, -1, -1, + 4, 11, 8, 4, 6, 11, 0, 2, 9, 2, 10, 9, -1, -1, -1, -1, + 10, 9, 3, 10, 3, 2, 9, 4, 3, 11, 3, 6, 4, 6, 3, -1, + 8, 2, 3, 8, 4, 2, 4, 6, 2, -1, -1, -1, -1, -1, -1, -1, + 0, 4, 2, 4, 6, 2, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 9, 0, 2, 3, 4, 2, 4, 6, 4, 3, 8, -1, -1, -1, -1, + 1, 9, 4, 1, 4, 2, 2, 4, 6, -1, -1, -1, -1, -1, -1, -1, + 8, 1, 3, 8, 6, 1, 8, 4, 6, 6, 10, 1, -1, -1, -1, -1, + 10, 1, 0, 10, 0, 6, 6, 0, 4, -1, -1, -1, -1, -1, -1, -1, + 4, 6, 3, 4, 3, 8, 6, 10, 3, 0, 3, 9, 10, 9, 3, -1, + 10, 9, 4, 6, 10, 4, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 9, 5, 7, 6, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 3, 4, 9, 5, 11, 7, 6, -1, -1, -1, -1, -1, -1, -1, + 5, 0, 1, 5, 4, 0, 7, 6, 11, -1, -1, -1, -1, -1, -1, -1, + 11, 7, 6, 8, 3, 4, 3, 5, 4, 3, 1, 5, -1, -1, -1, -1, + 9, 5, 4, 10, 1, 2, 7, 6, 11, -1, -1, -1, -1, -1, -1, -1, + 6, 11, 7, 1, 2, 10, 0, 8, 3, 4, 9, 5, -1, -1, -1, -1, + 7, 6, 11, 5, 4, 10, 4, 2, 10, 4, 0, 2, -1, -1, -1, -1, + 3, 4, 8, 3, 5, 4, 3, 2, 5, 10, 5, 2, 11, 7, 6, -1, + 7, 2, 3, 7, 6, 2, 5, 4, 9, -1, -1, -1, -1, -1, -1, -1, + 9, 5, 4, 0, 8, 6, 0, 6, 2, 6, 8, 7, -1, -1, -1, -1, + 3, 6, 2, 3, 7, 6, 1, 5, 0, 5, 4, 0, -1, -1, -1, -1, + 6, 2, 8, 6, 8, 7, 2, 1, 8, 4, 8, 5, 1, 5, 8, -1, + 9, 5, 4, 10, 1, 6, 1, 7, 6, 1, 3, 7, -1, -1, -1, -1, + 1, 6, 10, 1, 7, 6, 1, 0, 7, 8, 7, 0, 9, 5, 4, -1, + 4, 0, 10, 4, 10, 5, 0, 3, 10, 6, 10, 7, 3, 7, 10, -1, + 7, 6, 10, 7, 10, 8, 5, 4, 10, 4, 8, 10, -1, -1, -1, -1, + 6, 9, 5, 6, 11, 9, 11, 8, 9, -1, -1, -1, -1, -1, -1, -1, + 3, 6, 11, 0, 6, 3, 0, 5, 6, 0, 9, 5, -1, -1, -1, -1, + 0, 11, 8, 0, 5, 11, 0, 1, 5, 5, 6, 11, -1, -1, -1, -1, + 6, 11, 3, 6, 3, 5, 5, 3, 1, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 10, 9, 5, 11, 9, 11, 8, 11, 5, 6, -1, -1, -1, -1, + 0, 11, 3, 0, 6, 11, 0, 9, 6, 5, 6, 9, 1, 2, 10, -1, + 11, 8, 5, 11, 5, 6, 8, 0, 5, 10, 5, 2, 0, 2, 5, -1, + 6, 11, 3, 6, 3, 5, 2, 10, 3, 10, 5, 3, -1, -1, -1, -1, + 5, 8, 9, 5, 2, 8, 5, 6, 2, 3, 8, 2, -1, -1, -1, -1, + 9, 5, 6, 9, 6, 0, 0, 6, 2, -1, -1, -1, -1, -1, -1, -1, + 1, 5, 8, 1, 8, 0, 5, 6, 8, 3, 8, 2, 6, 2, 8, -1, + 1, 5, 6, 2, 1, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 3, 6, 1, 6, 10, 3, 8, 6, 5, 6, 9, 8, 9, 6, -1, + 10, 1, 0, 10, 0, 6, 9, 5, 0, 5, 6, 0, -1, -1, -1, -1, + 0, 3, 8, 5, 6, 10, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 10, 5, 6, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 11, 5, 10, 7, 5, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 11, 5, 10, 11, 7, 5, 8, 3, 0, -1, -1, -1, -1, -1, -1, -1, + 5, 11, 7, 5, 10, 11, 1, 9, 0, -1, -1, -1, -1, -1, -1, -1, + 10, 7, 5, 10, 11, 7, 9, 8, 1, 8, 3, 1, -1, -1, -1, -1, + 11, 1, 2, 11, 7, 1, 7, 5, 1, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 3, 1, 2, 7, 1, 7, 5, 7, 2, 11, -1, -1, -1, -1, + 9, 7, 5, 9, 2, 7, 9, 0, 2, 2, 11, 7, -1, -1, -1, -1, + 7, 5, 2, 7, 2, 11, 5, 9, 2, 3, 2, 8, 9, 8, 2, -1, + 2, 5, 10, 2, 3, 5, 3, 7, 5, -1, -1, -1, -1, -1, -1, -1, + 8, 2, 0, 8, 5, 2, 8, 7, 5, 10, 2, 5, -1, -1, -1, -1, + 9, 0, 1, 5, 10, 3, 5, 3, 7, 3, 10, 2, -1, -1, -1, -1, + 9, 8, 2, 9, 2, 1, 8, 7, 2, 10, 2, 5, 7, 5, 2, -1, + 1, 3, 5, 3, 7, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 7, 0, 7, 1, 1, 7, 5, -1, -1, -1, -1, -1, -1, -1, + 9, 0, 3, 9, 3, 5, 5, 3, 7, -1, -1, -1, -1, -1, -1, -1, + 9, 8, 7, 5, 9, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 5, 8, 4, 5, 10, 8, 10, 11, 8, -1, -1, -1, -1, -1, -1, -1, + 5, 0, 4, 5, 11, 0, 5, 10, 11, 11, 3, 0, -1, -1, -1, -1, + 0, 1, 9, 8, 4, 10, 8, 10, 11, 10, 4, 5, -1, -1, -1, -1, + 10, 11, 4, 10, 4, 5, 11, 3, 4, 9, 4, 1, 3, 1, 4, -1, + 2, 5, 1, 2, 8, 5, 2, 11, 8, 4, 5, 8, -1, -1, -1, -1, + 0, 4, 11, 0, 11, 3, 4, 5, 11, 2, 11, 1, 5, 1, 11, -1, + 0, 2, 5, 0, 5, 9, 2, 11, 5, 4, 5, 8, 11, 8, 5, -1, + 9, 4, 5, 2, 11, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 2, 5, 10, 3, 5, 2, 3, 4, 5, 3, 8, 4, -1, -1, -1, -1, + 5, 10, 2, 5, 2, 4, 4, 2, 0, -1, -1, -1, -1, -1, -1, -1, + 3, 10, 2, 3, 5, 10, 3, 8, 5, 4, 5, 8, 0, 1, 9, -1, + 5, 10, 2, 5, 2, 4, 1, 9, 2, 9, 4, 2, -1, -1, -1, -1, + 8, 4, 5, 8, 5, 3, 3, 5, 1, -1, -1, -1, -1, -1, -1, -1, + 0, 4, 5, 1, 0, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 8, 4, 5, 8, 5, 3, 9, 0, 5, 0, 3, 5, -1, -1, -1, -1, + 9, 4, 5, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 11, 7, 4, 9, 11, 9, 10, 11, -1, -1, -1, -1, -1, -1, -1, + 0, 8, 3, 4, 9, 7, 9, 11, 7, 9, 10, 11, -1, -1, -1, -1, + 1, 10, 11, 1, 11, 4, 1, 4, 0, 7, 4, 11, -1, -1, -1, -1, + 3, 1, 4, 3, 4, 8, 1, 10, 4, 7, 4, 11, 10, 11, 4, -1, + 4, 11, 7, 9, 11, 4, 9, 2, 11, 9, 1, 2, -1, -1, -1, -1, + 9, 7, 4, 9, 11, 7, 9, 1, 11, 2, 11, 1, 0, 8, 3, -1, + 11, 7, 4, 11, 4, 2, 2, 4, 0, -1, -1, -1, -1, -1, -1, -1, + 11, 7, 4, 11, 4, 2, 8, 3, 4, 3, 2, 4, -1, -1, -1, -1, + 2, 9, 10, 2, 7, 9, 2, 3, 7, 7, 4, 9, -1, -1, -1, -1, + 9, 10, 7, 9, 7, 4, 10, 2, 7, 8, 7, 0, 2, 0, 7, -1, + 3, 7, 10, 3, 10, 2, 7, 4, 10, 1, 10, 0, 4, 0, 10, -1, + 1, 10, 2, 8, 7, 4, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 9, 1, 4, 1, 7, 7, 1, 3, -1, -1, -1, -1, -1, -1, -1, + 4, 9, 1, 4, 1, 7, 0, 8, 1, 8, 7, 1, -1, -1, -1, -1, + 4, 0, 3, 7, 4, 3, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 4, 8, 7, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 9, 10, 8, 10, 11, 8, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 3, 0, 9, 3, 9, 11, 11, 9, 10, -1, -1, -1, -1, -1, -1, -1, + 0, 1, 10, 0, 10, 8, 8, 10, 11, -1, -1, -1, -1, -1, -1, -1, + 3, 1, 10, 11, 3, 10, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 2, 11, 1, 11, 9, 9, 11, 8, -1, -1, -1, -1, -1, -1, -1, + 3, 0, 9, 3, 9, 11, 1, 2, 9, 2, 11, 9, -1, -1, -1, -1, + 0, 2, 11, 8, 0, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 3, 2, 11, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 2, 3, 8, 2, 8, 10, 10, 8, 9, -1, -1, -1, -1, -1, -1, -1, + 9, 10, 2, 0, 9, 2, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 2, 3, 8, 2, 8, 10, 0, 1, 8, 1, 10, 8, -1, -1, -1, -1, + 1, 10, 2, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 1, 3, 8, 9, 1, 8, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 9, 1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + 0, 3, 8, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, + -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, +]); + +/** Triangles per configuration, derived from MC_TRI_TABLE. */ +export const MC_TRI_COUNT = new Uint32Array([ + 0, 1, 1, 2, 1, 2, 2, 3, 1, 2, 2, 3, 2, 3, 3, 2, 1, 2, 2, 3, 2, 3, 3, 4, 2, 3, 3, 4, 3, 4, 4, 3, 1, 2, 2, 3, 2, 3, 3, 4, 2, 3, 3, 4, 3, 4, 4, 3, 2, 3, 3, 2, 3, 4, 4, 3, 3, 4, 4, 3, 4, 5, 5, 2, 1, 2, 2, 3, 2, 3, 3, 4, 2, 3, 3, 4, 3, 4, 4, 3, 2, 3, 3, 4, 3, 4, 4, 5, 3, 4, 4, 5, 4, 5, 5, 4, 2, 3, 3, 4, 3, 4, 2, 3, 3, 4, 4, 5, 4, 5, 3, 2, 3, 4, 4, 3, 4, 5, 3, 2, 4, 5, 5, 4, 5, 2, 4, 1, 1, 2, 2, 3, 2, 3, 3, 4, 2, 3, 3, 4, 3, 4, 4, 3, 2, 3, 3, 4, 3, 4, 4, 5, 3, 2, 4, 3, 4, 3, 5, 2, 2, 3, 3, 4, 3, 4, 4, 5, 3, 4, 4, 5, 4, 5, 5, 4, 3, 4, 4, 3, 4, 5, 5, 4, 4, 3, 5, 2, 5, 4, 2, 1, 2, 3, 3, 4, 3, 4, 4, 5, 3, 4, 4, 5, 2, 3, 3, 2, 3, 4, 4, 5, 4, 5, 5, 2, 4, 3, 5, 4, 3, 2, 4, 1, 3, 4, 4, 5, 4, 5, 3, 4, 4, 5, 5, 2, 3, 4, 2, 1, 2, 3, 3, 2, 3, 4, 2, 1, 3, 2, 4, 1, 2, 1, 1, 0, +]); diff --git a/ts/gpu/merge.ts b/ts/gpu/merge.ts new file mode 100644 index 0000000..b6fd929 --- /dev/null +++ b/ts/gpu/merge.ts @@ -0,0 +1,140 @@ +// Host side of wgsl/merge.wgsl. merge ORs two mask grids, for topology. +// mergeCsg is an SDF boolean of two op layer grids. + +import mergeWgsl from 'picovdb/wgsl/merge.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { Sorter } from './radix_sort.ts'; +import { dispatch2D, readBackTotals } from './device.ts'; +import { GridWriter, preludeWgsl, readerWgsl, type OpGrid } from './opgrid.ts'; + +const WG_SIZE = 256; + +export interface MergeInput { + leafKeys: GPUBuffer; + masks: GPUBuffer; + leafCount: number; +} + +export type CsgOp = 'union' | 'intersect' | 'subtract'; +const CSG_OPS: Record = { union: 0, intersect: 1, subtract: 2 }; + +export interface CsgOptions { + halfWidth: number; + op?: CsgOp; +} + +export interface MergeResult { + leafKeys: GPUBuffer; + masks: GPUBuffer; + leafCount: number; +} + +export class Merger { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly sorter: Sorter; + private readonly pipelines: Record = {}; + + constructor(device: GPUDevice, scanner: Scanner = new Scanner(device), sorter: Sorter = new Sorter(device, scanner)) { + this.device = device; + this.scanner = scanner; + this.sorter = sorter; + const code = preludeWgsl + readerWgsl('a', 'params.a_count') + readerWgsl('b', 'params.b_count') + mergeWgsl; + const module = device.createShaderModule({ code }); + for (const entryPoint of ['mark_unique', 'compact_unique', 'merge_masks', 'csg_mark', 'csg_apply', 'opgrid_compact']) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint } }); + } + } + + /** Topology merge: the union of the leaf tables with ORed masks. */ + async merge(a: MergeInput, b: MergeInput): Promise { + const device = this.device; + const { params, run, outKeys, outCount } = await this.unionKeys(a, b); + const outMasks = device.createBuffer({ size: outCount * 16 * 4, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC }); + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'merge_masks', outCount, { 3: outKeys, 4: a.leafKeys, 5: a.masks, 6: b.leafKeys, 7: b.masks, 8: outMasks }); + pass.end(); + device.queue.submit([encoder.finish()]); + params.destroy(); + return { leafKeys: outKeys, masks: outMasks, leafCount: outCount }; + } + + /** SDF boolean of two op layer grids that share a key origin. */ + async mergeCsg(a: OpGrid, b: OpGrid, opts: CsgOptions): Promise { + const device = this.device; + if (a.leafMin.some((v, i) => v !== b.leafMin[i])) throw new Error('grids have different key origins'); + const { params, run, outKeys, outCount } = await this.unionKeys(a, b); + device.queue.writeBuffer(params, 16, new Float32Array([opts.halfWidth])); + device.queue.writeBuffer(params, 20, new Uint32Array([CSG_OPS[opts.op ?? 'union']])); + const inputs = { 4: a.leafKeys, 10: a.leaves, 11: a.data, 6: b.leafKeys, 12: b.leaves, 13: b.data }; + const writer = new GridWriter(device, outCount); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'csg_mark', outCount + 1, { ...inputs, 3: outKeys, ...writer.markBindings }); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const leafMax = a.leafMax.map((v, i) => Math.max(v, b.leafMax[i])) as [number, number, number]; + const out = await writer.finish(this.scanner, this.pipelines['opgrid_compact'], outKeys, opts.halfWidth, { leafMin: a.leafMin, leafMax }, (pass, bindings, count) => { + run(pass, 'csg_apply', count, { ...inputs, ...bindings }); + }); + outKeys.destroy(); + params.destroy(); + return out; + } + + /** The sorted union of two leaf tables. */ + private async unionKeys(a: { leafKeys: GPUBuffer; leafCount: number }, b: { leafKeys: GPUBuffer; leafCount: number }) { + const device = this.device; + const concatCount = a.leafCount + b.leafCount; + const params = device.createBuffer({ size: 32, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([a.leafCount, b.leafCount, concatCount])); + const run = (pass: GPUComputePassEncoder, name: string, threads: number, buffers: Record) => { + if (threads === 0) return; + pass.setPipeline(this.pipelines[name]); + pass.setBindGroup( + 0, + device.createBindGroup({ + layout: this.pipelines[name].getBindGroupLayout(0), + entries: [ + { binding: 0, resource: { buffer: params } }, + ...Object.entries(buffers).map(([binding, buffer]) => ({ binding: Number(binding), resource: { buffer } })), + ], + }) + ); + dispatch2D(pass, Math.ceil(threads / WG_SIZE)); + }; + + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const concatKeys = device.createBuffer({ size: concatCount * 4, usage: storage }); + const sortVals = device.createBuffer({ size: concatCount * 4, usage: GPUBufferUsage.STORAGE }); + const flags = device.createBuffer({ size: (concatCount + 1) * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + encoder.copyBufferToBuffer(a.leafKeys, 0, concatKeys, 0, a.leafCount * 4); + encoder.copyBufferToBuffer(b.leafKeys, 0, concatKeys, a.leafCount * 4, b.leafCount * 4); + const pass = encoder.beginComputePass(); + this.sorter.plan(concatKeys, sortVals, concatCount).encode(pass); + run(pass, 'mark_unique', concatCount + 1, { 1: concatKeys, 2: flags }); + this.scanner.plan(flags, concatCount + 1).encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [outCount] = await readBackTotals(device, [{ buffer: flags, index: concatCount }]); + device.queue.writeBuffer(params, 12, new Uint32Array([outCount])); + const outKeys = device.createBuffer({ size: Math.max(outCount, 1) * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'compact_unique', concatCount, { 1: concatKeys, 2: flags, 3: outKeys }); + pass.end(); + device.queue.submit([encoder.finish()]); + } + concatKeys.destroy(); + sortVals.destroy(); + flags.destroy(); + return { params, run, outKeys, outCount }; + } +} diff --git a/ts/gpu/mesh_to_grid.test.ts b/ts/gpu/mesh_to_grid.test.ts new file mode 100644 index 0000000..a1e93dc --- /dev/null +++ b/ts/gpu/mesh_to_grid.test.ts @@ -0,0 +1,64 @@ +import { hasWebGPU, requestDevice, readBackU32 } from './device.ts'; +import { mulberry32, assertU32ArrayEqual } from './test_util.ts'; +import { refBin, parseBinarySTL } from './reference.ts'; +import { Binner, type BinResult } from './mesh_to_grid.ts'; + +const gpu = await hasWebGPU(); + +let stl: Uint8Array | null = null; +try { + stl = Deno.readFileSync(new URL('../../data/bases/base_32mm.stl', import.meta.url)); +} catch { + // Sample data not present, so the STL test skips. +} + +async function checkAgainstRef( + binner: Binner, + points: Float32Array, + triangles: Uint32Array, + voxelSize: number, + halfWidth: number, + label: string +): Promise { + const result = await binner.bin(points, triangles, { voxelSize, halfWidth }); + const ref = refBin(points, triangles, voxelSize, halfWidth, result.leafMin); + if (result.pairCount !== ref.pairKeys.length) { + throw new Error(`${label}: pair count ${result.pairCount} != ref ${ref.pairKeys.length}`); + } + assertU32ArrayEqual(await readBackU32(binner.device, result.pairKeys, result.pairCount), ref.pairKeys, `${label} pair keys`); + assertU32ArrayEqual(await readBackU32(binner.device, result.pairTris, result.pairCount), ref.pairTris, `${label} pair tris`); + if (result.leafCount !== ref.leafKeys.length) { + throw new Error(`${label}: leaf count ${result.leafCount} != ref ${ref.leafKeys.length}`); + } + assertU32ArrayEqual(await readBackU32(binner.device, result.leafKeys, result.leafCount), ref.leafKeys, `${label} leaf keys`); + return result; +} + +Deno.test({ name: 'triangle binning matches reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const binner = new Binner(device); + const rand = mulberry32(3); + + // Random triangle soup with negative coords included. + const triCount = 300; + const points = new Float32Array(triCount * 9); + for (let i = 0; i < points.length; i++) points[i] = (rand() - 0.5) * 60; + const triangles = new Uint32Array([...Array(triCount * 3).keys()]); + await checkAgainstRef(binner, points, triangles, 0.25, 3, 'soup'); + + // Thin sliver whose dilated bounds span no integer coordinate on some axes. + const sliver = new Float32Array([0.2, 0.21, 0.2, 0.4, 0.22, 0.2, 0.3, 0.23, 0.21]); + await checkAgainstRef(binner, sliver, new Uint32Array([0, 1, 2]), 1, 0.1, 'sliver'); + + // Degenerate point triangle on a leaf boundary. + const point = new Float32Array([8, -8, 16, 8, -8, 16, 8, -8, 16]); + await checkAgainstRef(binner, point, new Uint32Array([0, 1, 2]), 1, 3, 'point'); +}); + +Deno.test({ name: 'STL binning matches reference', ignore: !gpu || !stl }, async () => { + const device = await requestDevice(); + const binner = new Binner(device); + const { points, triangles } = parseBinarySTL(stl!); + const result = await checkAgainstRef(binner, points, triangles, 0.25, 3, 'stl'); + if (result.leafCount < 100) throw new Error(`implausibly few leaves: ${result.leafCount}`); +}); diff --git a/ts/gpu/mesh_to_grid.ts b/ts/gpu/mesh_to_grid.ts new file mode 100644 index 0000000..5a154e3 --- /dev/null +++ b/ts/gpu/mesh_to_grid.ts @@ -0,0 +1,207 @@ +// Host side of wgsl/mesh_to_grid.wgsl. Bins triangles into the leaf +// blocks their dilated bounds touch, producing the sorted pair list and +// the deduplicated leaf table, all kept on the GPU. + +import binWgsl from 'picovdb/wgsl/mesh_to_grid.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { Sorter } from './radix_sort.ts'; +import { dispatch2D, readBackTotals } from './device.ts'; + +const WG_SIZE = 256; + +export interface BinOptions { + /** World units per voxel. */ + voxelSize: number; + /** Narrow band half width in voxels. */ + halfWidth: number; + /** + * Leaf space bounds override so several grids share one key space, as + * merging requires. Compute with leafBounds over a combined mesh. + */ + bounds?: { leafMin: [number, number, number]; leafMax: [number, number, number] }; +} + +export interface BinResult { + /** Index space vertex positions as xyz triples. */ + pointsIndex: GPUBuffer; + /** Vertex index triples as uploaded. */ + triangles: GPUBuffer; + /** (leaf key, triangle index) pairs sorted by key. */ + pairKeys: GPUBuffer; + pairTris: GPUBuffer; + pairCount: number; + /** Sorted deduplicated leaf keys. */ + leafKeys: GPUBuffer; + leafCount: number; + /** Leaf space bias. A leaf coordinate is the unpacked key plus leafMin. */ + leafMin: [number, number, number]; + /** Maximum leaf coordinate inclusive. */ + leafMax: [number, number, number]; +} + +export class Binner { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly sorter: Sorter; + readonly layout: GPUBindGroupLayout; + private readonly pipelines: Record = {}; + + constructor(device: GPUDevice, scanner = new Scanner(device), sorter = new Sorter(device, scanner)) { + this.device = device; + this.scanner = scanner; + this.sorter = sorter; + const storage = (binding: number, type: GPUBufferBindingType): GPUBindGroupLayoutEntry => ({ + binding, + visibility: GPUShaderStage.COMPUTE, + buffer: { type }, + }); + this.layout = device.createBindGroupLayout({ + entries: [ + storage(0, 'uniform'), + storage(1, 'read-only-storage'), + storage(2, 'storage'), + storage(3, 'read-only-storage'), + storage(4, 'storage'), + storage(5, 'storage'), + storage(6, 'storage'), + storage(7, 'storage'), + storage(8, 'storage'), + ], + }); + const module = device.createShaderModule({ code: binWgsl }); + const layout = device.createPipelineLayout({ bindGroupLayouts: [this.layout] }); + for (const entryPoint of ['transform_points', 'count_pairs', 'emit_pairs', 'mark_unique', 'compact_unique']) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout, compute: { module, entryPoint } }); + } + } + + bin(points: Float32Array, triangles: Uint32Array, opts: BinOptions): Promise { + const device = this.device; + const invVoxelSize = Math.fround(1 / opts.voxelSize); + const bounds = opts.bounds ?? leafBounds(points, invVoxelSize, opts.halfWidth); + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const pointsWorld = device.createBuffer({ size: points.byteLength, usage: storage }); + device.queue.writeBuffer(pointsWorld, 0, points); + const trianglesBuf = device.createBuffer({ size: triangles.byteLength, usage: storage }); + device.queue.writeBuffer(trianglesBuf, 0, triangles); + return this.binBuffers(pointsWorld, points.length / 3, trianglesBuf, triangles.length / 3, { ...opts, bounds }); + } + + /** Bins a mesh already on the GPU. Bounds are required since the points are not readable. */ + async binBuffers( + pointsWorld: GPUBuffer, + pointCount: number, + trianglesBuf: GPUBuffer, + triangleCount: number, + opts: BinOptions & { bounds: NonNullable } + ): Promise { + const device = this.device; + if (triangleCount === 0) throw new Error('empty mesh'); + const invVoxelSize = Math.fround(1 / opts.voxelSize); + const { leafMin, leafMax } = opts.bounds; + + const params = device.createBuffer({ size: 32, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([pointCount, triangleCount])); + device.queue.writeBuffer(params, 8, new Float32Array([invVoxelSize, opts.halfWidth])); + device.queue.writeBuffer(params, 16, new Int32Array(leafMin)); + + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const pointsIndex = device.createBuffer({ size: pointCount * 12, usage: GPUBufferUsage.STORAGE }); + const counts = device.createBuffer({ size: (triangleCount + 1) * 4, usage: storage }); + // Distinct placeholders: writable bindings may not alias one buffer. + const placeholders = [0, 1, 2, 3].map(() => device.createBuffer({ size: 4, usage: GPUBufferUsage.STORAGE })); + + const bindGroup = (pairKeys: GPUBuffer, pairTris: GPUBuffer, flags: GPUBuffer, uniqueKeys: GPUBuffer) => + device.createBindGroup({ + layout: this.layout, + entries: [ + { binding: 0, resource: { buffer: params } }, + { binding: 1, resource: { buffer: pointsWorld } }, + { binding: 2, resource: { buffer: pointsIndex } }, + { binding: 3, resource: { buffer: trianglesBuf } }, + { binding: 4, resource: { buffer: counts } }, + { binding: 5, resource: { buffer: pairKeys } }, + { binding: 6, resource: { buffer: pairTris } }, + { binding: 7, resource: { buffer: flags } }, + { binding: 8, resource: { buffer: uniqueKeys } }, + ], + }); + + // Transform, count, and scan, then read the total pair count. + const countGroup = bindGroup(placeholders[0], placeholders[1], placeholders[2], placeholders[3]); + const countScan = this.scanner.plan(counts, triangleCount + 1); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, countGroup); + this.dispatch(pass, 'transform_points', pointCount * 3); + this.dispatch(pass, 'count_pairs', triangleCount + 1); + countScan.encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [pairCount] = await readBackTotals(device, [{ buffer: counts, index: triangleCount }]); + if (pairCount === 0) throw new Error('no leaves touched (degenerate mesh?)'); + + // Emit, sort by key, then mark and compact unique leaves. + device.queue.writeBuffer(params, 28, new Uint32Array([pairCount])); + const pairKeys = device.createBuffer({ size: pairCount * 4, usage: storage }); + const pairTris = device.createBuffer({ size: pairCount * 4, usage: storage }); + const flags = device.createBuffer({ size: (pairCount + 1) * 4, usage: storage }); + const leafKeys = device.createBuffer({ size: pairCount * 4, usage: storage }); + const pairGroup = bindGroup(pairKeys, pairTris, flags, leafKeys); + const sortPlan = this.sorter.plan(pairKeys, pairTris, pairCount); + const flagScan = this.scanner.plan(flags, pairCount + 1); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, pairGroup); + this.dispatch(pass, 'emit_pairs', triangleCount); + sortPlan.encode(pass); + pass.setBindGroup(0, pairGroup); + this.dispatch(pass, 'mark_unique', pairCount + 1); + flagScan.encode(pass); + pass.setBindGroup(0, pairGroup); + this.dispatch(pass, 'compact_unique', pairCount); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [leafCount] = await readBackTotals(device, [{ buffer: flags, index: pairCount }]); + + return { pointsIndex, triangles: trianglesBuf, pairKeys, pairTris, pairCount, leafKeys, leafCount, leafMin, leafMax }; + } + + private dispatch(pass: GPUComputePassEncoder, entryPoint: string, threads: number): void { + pass.setPipeline(this.pipelines[entryPoint]); + dispatch2D(pass, Math.ceil(threads / WG_SIZE)); + } +} + +/** Leaf coordinate bounds over all dilated vertices, f32 exact to the GPU math. */ +export function leafBounds( + points: Float32Array, + invVoxelSize: number, + halfWidth: number +): { leafMin: [number, number, number]; leafMax: [number, number, number] } { + const min = [Infinity, Infinity, Infinity]; + const max = [-Infinity, -Infinity, -Infinity]; + for (let i = 0; i < points.length; i += 3) { + for (let axis = 0; axis < 3; axis++) { + const p = Math.fround(points[i + axis] * invVoxelSize); + if (p < min[axis]) min[axis] = p; + if (p > max[axis]) max[axis] = p; + } + } + const lo: number[] = []; + const hi: number[] = []; + for (let axis = 0; axis < 3; axis++) { + const loLeaf = Math.ceil(Math.fround(min[axis] - halfWidth)) >> 3; + const hiLeaf = Math.floor(Math.fround(max[axis] + halfWidth)) >> 3; + if (hiLeaf - loLeaf >= 1024) { + throw new Error(`grid exceeds 1024 leaves on axis ${axis}: ${loLeaf}..${hiLeaf}`); + } + lo.push(loLeaf); + hi.push(hiLeaf); + } + return { leafMin: lo as [number, number, number], leafMax: hi as [number, number, number] }; +} diff --git a/ts/gpu/opgrid.ts b/ts/gpu/opgrid.ts new file mode 100644 index 0000000..0426bd3 --- /dev/null +++ b/ts/gpu/opgrid.ts @@ -0,0 +1,315 @@ +// The op layer grid: what the GPU ops read and write between edits. It is +// the picovdb leaf level, so files load and emit without conversion. +// +// leafKeys sorted leaf keys +// leaves leaf records in the file layout: inside and band masks, per +// word value prefixes, and the absolute value index +// data +hw, -hw, then the band values only +// +// Surface bits and surface bases stay zero until emission. Every leaf +// holds at least one band voxel. +// +// Kernels read voxels through the shared reader (readerWgsl) and write +// grids through the shared writer. A mark pass folds each candidate leaf's +// values into a record and a band count. The host scans the counts. +// opgrid_compact drops band-empty leaves and assigns value bases. An apply +// pass writes the band values into their slots. + +import { PICOVDB_LEAF_SIZE } from '../picovdb.ts'; +import type { Scanner } from './scan.ts'; +import { checkBindingSize, dispatch2D, readBackTotals } from './device.ts'; + +export const LEAF_U32 = PICOVDB_LEAF_SIZE / 4; + +export interface OpGrid { + /** Sorted leaf keys, 10 bits per axis relative to leafMin. */ + leafKeys: GPUBuffer; + /** Leaf records in the picovdb leaf layout, in key order. */ + leaves: GPUBuffer; + /** f32 bits: +half width, -half width, then the band values by leaf and bit order. */ + data: GPUBuffer; + leafCount: number; + activeVoxels: number; + leafMin: [number, number, number]; + /** Maximum leaf coordinate inclusive. */ + leafMax: [number, number, number]; +} + +export function emptyOpGrid(device: GPUDevice, leafMin: [number, number, number] = [0, 0, 0], leafMax: [number, number, number] = [0, 0, 0]): OpGrid { + const placeholder = () => device.createBuffer({ size: 4, usage: GPUBufferUsage.STORAGE }); + return { leafKeys: placeholder(), leaves: placeholder(), data: placeholder(), leafCount: 0, activeVoxels: 0, leafMin, leafMax }; +} + +/** Prelude of every kernel: indexing, key packing, the leaf layout, and the writer. Kernels must declare params.half_width. */ +export const preludeWgsl = /* wgsl */ ` +const DISPATCH_STRIDE: u32 = 65535u; +const NOT_FOUND: u32 = 0xffffffffu; +const LEAF_U32: u32 = ${LEAF_U32}u; + +fn globalIndex(wid: vec3, lid: vec3) -> u32 { + return (((wid.y * DISPATCH_STRIDE) + wid.x) * 256u) + lid.x; +} + +fn unpack(key: u32) -> vec3 { + return vec3(i32(key >> 20u), i32((key >> 10u) & 0x3ffu), i32(key & 0x3ffu)); +} + +fn pack(c: vec3) -> u32 { + return (u32(c.x) << 20u) | (u32(c.y) << 10u) | u32(c.z); +} + +fn voxelOffset(ijk: vec3) -> u32 { + return (u32(ijk.x & 7) << 6u) | (u32(ijk.y & 7) << 3u) | u32(ijk.z & 7); +} + +fn voxelLocal(n: u32) -> vec3 { + return vec3(i32(n >> 6u), i32((n >> 3u) & 7u), i32(n & 7u)); +} + +// Writer bindings. w_counts and w_kept have one extra entry for the scan +// totals. +@group(0) @binding(40) var w_cand_keys: array; +@group(0) @binding(41) var w_cand_leaves: array; +@group(0) @binding(42) var w_counts: array; // band counts, then their exclusive scan +@group(0) @binding(43) var w_kept: array; // kept flags, then their exclusive scan +@group(0) @binding(44) var w_keys: array; +@group(0) @binding(45) var w_leaves: array; +@group(0) @binding(46) var w_data: array; + +// Writes the extra entries. Returns true when i is past the candidates. +fn markSentinel(i: u32, cand: u32) -> bool { + if (i == cand) { + w_counts[i] = 0u; + w_kept[i] = 0u; + } + return i >= cand; +} + +// Folds the values of candidate leaf i into its record, one word per 32 +// voxels. Call in voxel order. +struct LeafAcc { + band: u32, + inside: u32, + local: u32, + count: u32, +} + +fn accPush(acc: ptr, i: u32, n: u32, v: f32) { + let bit = 1u << (n & 31u); + if (abs(v) < params.half_width) { + (*acc).band = (*acc).band | bit; + } + if (v < 0.0) { + (*acc).inside = (*acc).inside | bit; + } + if ((n & 31u) == 31u) { + let e = (i * LEAF_U32) + 4u + ((n >> 5u) * 3u); + w_cand_leaves[e] = (*acc).inside & ~(*acc).band; + w_cand_leaves[e + 1u] = (*acc).band; + w_cand_leaves[e + 2u] = (*acc).local; + let c = countOneBits((*acc).band); + (*acc).local = (*acc).local + c; + (*acc).count = (*acc).count + c; + (*acc).band = 0u; + (*acc).inside = 0u; + } +} + +fn accFinish(acc: ptr, i: u32) { + for (var k = 0u; k < 4u; k = k + 1u) { + w_cand_leaves[(i * LEAF_U32) + k] = 0u; + } + w_counts[i] = (*acc).count; + w_kept[i] = select(0u, 1u, (*acc).count > 0u); +} + +// Band mask word w of output leaf j. +fn bandWord(j: u32, w: u32) -> u32 { + return w_leaves[(j * LEAF_U32) + 5u + (w * 3u)]; +} + +// Copies kept candidates to the output with their value bases. +@compute @workgroup_size(256) +fn opgrid_compact(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + let cand = arrayLength(&w_counts) - 1u; + if (i >= cand || w_kept[i] == w_kept[i + 1u]) { + return; + } + let j = w_kept[i]; + w_keys[j] = w_cand_keys[i]; + for (var k = 0u; k < LEAF_U32; k = k + 1u) { + w_leaves[(j * LEAF_U32) + k] = w_cand_leaves[(i * LEAF_U32) + k]; + } + w_leaves[(j * LEAF_U32) + 2u] = 2u + w_counts[i]; +} +`; + +/** + * Reader of a grid bound as {p}_keys, {p}_leaves, {p}_data with count + * leaves. {p}_valueAt gives the value at any relative voxel: the stored + * band value, or the implicit background of the inside bit. Leafless + * space takes the sign of the nearest leaf in its column, and a column + * with no leaves is outside. This matches the CPU converter. + */ +export function readerWgsl(p: string, count: string): string { + return /* wgsl */ ` +fn ${p}_find(key: u32) -> u32 { + var lo = 0u; + var hi = ${count}; + while (lo < hi) { + let mid = (lo + hi) >> 1u; + if (${p}_keys[mid] < key) { + lo = mid + 1u; + } else { + hi = mid; + } + } + if (lo < ${count} && ${p}_keys[lo] == key) { + return lo; + } + return NOT_FOUND; +} + +fn ${p}_leafValue(i: u32, n: u32) -> f32 { + let e = (i * LEAF_U32) + 4u + ((n >> 5u) * 3u); + let value = ${p}_leaves[e + 1u]; + let bit = 1u << (n & 31u); + if ((value & bit) != 0u) { + let d = ${p}_leaves[(i * LEAF_U32) + 2u] + (${p}_leaves[e + 2u] & 0xffffu) + countOneBits(value & (bit - 1u)); + return bitcast(${p}_data[d]); + } + return select(params.half_width, -params.half_width, (${p}_leaves[e] & bit) != 0u); +} + +fn ${p}_valueAt(ijk: vec3) -> f32 { + let leaf = ijk >> vec3(3u); + if (leaf.x < 0 || leaf.x > 1023 || leaf.y < 0 || leaf.y > 1023) { + return params.half_width; + } + let n = voxelOffset(ijk); + if (leaf.z >= 0 && leaf.z <= 1023) { + let i = ${p}_find(pack(leaf)); + if (i != NOT_FOUND) { + return ${p}_leafValue(i, n); + } + } + let col = (u32(leaf.x) << 20u) | (u32(leaf.y) << 10u); + var lo = 0u; + var hi = ${count}; + while (lo < hi) { + let mid = (lo + hi) >> 1u; + if (${p}_keys[mid] < col) { + lo = mid + 1u; + } else { + hi = mid; + } + } + if (lo >= ${count} || (${p}_keys[lo] & 0xfffffc00u) != col) { + return params.half_width; + } + var best = lo; + var best_d = abs(i32(${p}_keys[lo] & 0x3ffu) - leaf.z); + var i = lo + 1u; + while (i < ${count} && (${p}_keys[i] & 0xfffffc00u) == col) { + let d = abs(i32(${p}_keys[i] & 0x3ffu) - leaf.z); + if (d >= best_d) { + break; + } + best = i; + best_d = d; + i = i + 1u; + } + let zloc = select(0u, 7u, i32(${p}_keys[best] & 0x3ffu) < leaf.z); + return select(params.half_width, -params.half_width, ${p}_leafValue(best, (n & 0x1f8u) | zloc) < 0.0); +} +`; +} + +/** Host side of the writer. Construct one per output grid, run the kernel's mark pass with markBindings, then call finish. */ +export class GridWriter { + readonly device: GPUDevice; + readonly cand: number; + readonly candLeaves: GPUBuffer; + readonly counts: GPUBuffer; + readonly kept: GPUBuffer; + + constructor(device: GPUDevice, cand: number) { + this.device = device; + this.cand = cand; + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + this.candLeaves = device.createBuffer({ size: Math.max(cand, 1) * LEAF_U32 * 4, usage: storage }); + this.counts = device.createBuffer({ size: (cand + 1) * 4, usage: storage }); + this.kept = device.createBuffer({ size: (cand + 1) * 4, usage: storage }); + } + + get markBindings(): Record { + return { 41: this.candLeaves, 42: this.counts, 43: this.kept }; + } + + /** + * Scans the mark results, allocates the grid, runs the kernel's + * opgrid_compact pipeline, then runs apply with the output bindings. + * candKeys must be the candidate keys the mark pass indexed. + */ + async finish( + scanner: Scanner, + compact: GPUComputePipeline, + candKeys: GPUBuffer, + halfWidth: number, + bounds: { leafMin: [number, number, number]; leafMax: [number, number, number] }, + apply: (pass: GPUComputePassEncoder, bindings: Record, leafCount: number) => void + ): Promise { + const device = this.device; + const n = this.cand + 1; + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + scanner.plan(this.kept, n).encode(pass); + scanner.plan(this.counts, n).encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [leafCount, activeVoxels] = await readBackTotals(device, [ + { buffer: this.kept, index: this.cand }, + { buffer: this.counts, index: this.cand }, + ]); + const release = () => { + this.candLeaves.destroy(); + this.counts.destroy(); + this.kept.destroy(); + }; + if (leafCount === 0) { + release(); + return emptyOpGrid(device, bounds.leafMin, bounds.leafMax); + } + checkBindingSize(device, leafCount * LEAF_U32 * 4, 'op grid leaves'); + checkBindingSize(device, (2 + activeVoxels) * 4, 'op grid values'); + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const leafKeys = device.createBuffer({ size: leafCount * 4, usage: storage }); + const leaves = device.createBuffer({ size: leafCount * LEAF_U32 * 4, usage: storage }); + const data = device.createBuffer({ size: (2 + activeVoxels) * 4, usage: storage }); + device.queue.writeBuffer(data, 0, new Float32Array([halfWidth, -halfWidth])); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setPipeline(compact); + pass.setBindGroup( + 0, + device.createBindGroup({ + layout: compact.getBindGroupLayout(0), + entries: Object.entries({ 40: candKeys, ...this.markBindings, 44: leafKeys, 45: leaves }).map(([binding, buffer]) => ({ + binding: Number(binding), + resource: { buffer }, + })), + }) + ); + dispatch2D(pass, Math.ceil(this.cand / 256)); + apply(pass, { 44: leafKeys, 45: leaves, 46: data }, leafCount); + pass.end(); + device.queue.submit([encoder.finish()]); + } + release(); + return { leafKeys, leaves, data, leafCount, activeVoxels, leafMin: bounds.leafMin, leafMax: bounds.leafMax }; + } +} diff --git a/ts/gpu/pipeline.test.ts b/ts/gpu/pipeline.test.ts new file mode 100644 index 0000000..dcc0551 --- /dev/null +++ b/ts/gpu/pipeline.test.ts @@ -0,0 +1,156 @@ +// Converts two meshes on the GPU in a shared key space, merges them, and +// emits the tree, comparing against the CPU converter run on the combined +// mesh. The copies sit far enough apart that bands and leaf sets stay +// disjoint, so the CPU oracle applies. + +import { hasWebGPU, requestDevice } from './device.ts'; +import { assertU32ArrayEqual, compareTreeToCpu, unpackGrid } from './test_util.ts'; +import { parseBinarySTL, refCsgMerge } from './reference.ts'; +import { Binner, leafBounds } from './mesh_to_grid.ts'; +import { Rasterizer } from './rasterize.ts'; +import { Signer } from './sign.ts'; +import { Emitter } from './emit.ts'; +import { Merger } from './merge.ts'; +import { initSTL, importSTL } from '../stl.ts'; + +const gpu = await hasWebGPU(); + +let stl: Uint8Array | null = null; +let wasm: Uint8Array | null = null; +try { + stl = Deno.readFileSync(new URL('../../data/bases/base_32mm.stl', import.meta.url)); + wasm = Deno.readFileSync(new URL('../../zig-out/wasm/picovdb.wasm', import.meta.url)); +} catch { + // skip +} + +function writeBinarySTL(points: Float32Array, triangles: Uint32Array): Uint8Array { + const triCount = triangles.length / 3; + const bytes = new Uint8Array(84 + triCount * 50); + const view = new DataView(bytes.buffer); + view.setUint32(80, triCount, true); + for (let t = 0; t < triCount; t++) { + const base = 84 + t * 50 + 12; // normal left zero + for (let v = 0; v < 3; v++) { + for (let c = 0; c < 3; c++) { + view.setFloat32(base + (v * 3 + c) * 4, points[triangles[t * 3 + v] * 3 + c], true); + } + } + } + return bytes; +} + +Deno.test({ name: 'full pipeline: convert x2, merge, re-emit matches CPU', ignore: !gpu || !stl || !wasm }, async () => { + const device = await requestDevice(); + const meshA = parseBinarySTL(stl!); + const voxelSize = 0.25; + const halfWidth = 3; + + // Mesh B is the same base translated 50 world units in x, far beyond + // the band, so leaf sets stay disjoint. + const pointsB = new Float32Array(meshA.points); + for (let i = 0; i < pointsB.length; i += 3) pointsB[i] = Math.fround(pointsB[i] + 50); + + // The CPU oracle converts the combined mesh. + const combinedPoints = new Float32Array(meshA.points.length * 2); + combinedPoints.set(meshA.points, 0); + combinedPoints.set(pointsB, meshA.points.length); + const combinedTris = new Uint32Array(meshA.triangles.length * 2); + combinedTris.set(meshA.triangles, 0); + for (let i = 0; i < meshA.triangles.length; i++) { + combinedTris[meshA.triangles.length + i] = meshA.triangles[i] + meshA.points.length / 3; + } + initSTL({ wasmBinary: wasm! }); + const cpu = (await importSTL(writeBinarySTL(combinedPoints, combinedTris), { voxelsPerUnit: 4 })).file; + + // The GPU converts both meshes in the combined key space, merges, and emits. + const bounds = leafBounds(combinedPoints, Math.fround(1 / voxelSize), halfWidth); + const opts = { voxelSize, halfWidth, bounds }; + const binner = new Binner(device); + const rasterizer = new Rasterizer(device); + const signer = new Signer(device); + const emitter = new Emitter(device); + + const convert = async (points: Float32Array, triangles: Uint32Array) => { + const bin = await binner.bin(points, triangles, opts); + const values = rasterizer.rasterize(bin, opts); + const sign = await signer.sign(bin); + return emitter.classifyOnly(bin, values, sign, opts); + }; + const gridA = await convert(meshA.points, meshA.triangles); + const gridB = await convert(pointsB, meshA.triangles); + + const merged = await new Merger(device).mergeCsg(gridA, gridB, { halfWidth }); + const tree = await emitter.reEmit(merged, { halfWidth }); + + // Compare against the CPU tree. + const maxAbs = await compareTreeToCpu(device, tree, cpu); + console.log( + ` pipeline: ${tree.leafCount} leaves / ${tree.lowerCount} lowers / ${tree.upperCount} uppers, ` + + `${tree.activeVoxels} active, max |Δv| ${maxAbs.toExponential(2)}` + ); +}); + +Deno.test({ name: 'overlapping solids: CSG merge deactivates swallowed band', ignore: !gpu || !stl }, async () => { + const device = await requestDevice(); + const meshA = parseBinarySTL(stl!); + const voxelSize = 0.25; + const halfWidth = 3; + + // Mesh B overlaps A, shifted 2.5 world units in x. + const pointsB = new Float32Array(meshA.points); + for (let i = 0; i < pointsB.length; i += 3) pointsB[i] = Math.fround(pointsB[i] + 2.5); + const combinedPoints = new Float32Array(meshA.points.length * 2); + combinedPoints.set(meshA.points, 0); + combinedPoints.set(pointsB, meshA.points.length); + + const bounds = leafBounds(combinedPoints, Math.fround(1 / voxelSize), halfWidth); + const opts = { voxelSize, halfWidth, bounds }; + const binner = new Binner(device); + const rasterizer = new Rasterizer(device); + const signer = new Signer(device); + const emitter = new Emitter(device); + + const convert = async (points: Float32Array, triangles: Uint32Array) => { + const bin = await binner.bin(points, triangles, opts); + const values = rasterizer.rasterize(bin, opts); + const sign = await signer.sign(bin); + return emitter.classifyOnly(bin, values, sign, opts); + }; + const gridA = await convert(meshA.points, meshA.triangles); + const gridB = await convert(pointsB, meshA.triangles); + const merged = await new Merger(device).mergeCsg(gridA, gridB, { halfWidth }); + + // The GPU CSG merge must match the JS reference built from the two + // source grids. + const a = await unpackGrid(device, gridA, halfWidth); + const b = await unpackGrid(device, gridB, halfWidth); + const ref = refCsgMerge(a.keys, a.values, b.keys, b.values, halfWidth); + const got = await unpackGrid(device, merged, halfWidth); + if (merged.leafCount !== ref.keys.length) throw new Error(`count ${merged.leafCount} != ${ref.keys.length}`); + assertU32ArrayEqual(got.keys, ref.keys, 'overlap keys'); + assertU32ArrayEqual(got.masks, ref.masks, 'overlap masks'); + assertU32ArrayEqual(new Uint32Array(got.values.buffer), new Uint32Array(ref.values.buffer), 'overlap values'); + + // The swallowed band must deactivate. The union band is smaller than the + // two bands combined and at least one solid's worth. + const popcount = (v: number) => { + v = v - ((v >>> 1) & 0x55555555); + v = (v & 0x33333333) + ((v >>> 2) & 0x33333333); + return (((v + (v >>> 4)) & 0x0f0f0f0f) * 0x01010101) >>> 24; + }; + const sumBits = (words: Uint32Array) => words.reduce((s, w) => s + popcount(w), 0); + const activeA = sumBits(a.masks); + const activeB = sumBits(b.masks); + const activeUnion = sumBits(ref.masks); + if (activeUnion >= activeA + activeB) throw new Error(`no deactivation: ${activeUnion} >= ${activeA} + ${activeB}`); + if (activeUnion < activeA) throw new Error(`union band implausibly small: ${activeUnion} < ${activeA}`); + + // The edited grid emits into a tree. + const tree = await emitter.reEmit(merged, { halfWidth }); + if (tree.activeVoxels !== activeUnion) throw new Error(`tree active ${tree.activeVoxels} != ${activeUnion}`); + console.log( + ` overlap: A=${activeA} B=${activeB} union=${activeUnion} active; ` + + `tree ${tree.leafCount} leaves / ${tree.lowerCount} lowers / ${tree.upperCount} uppers, ${tree.surfaceVoxels} surface` + ); +}); diff --git a/ts/gpu/prune.ts b/ts/gpu/prune.ts new file mode 100644 index 0000000..dfb225e --- /dev/null +++ b/ts/gpu/prune.ts @@ -0,0 +1,71 @@ +// Host side of wgsl/prune.wgsl. ANDs leaf masks with a retain set and +// drops leaves left empty. + +import pruneWgsl from 'picovdb/wgsl/prune.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { dispatch2D, readBackU32 } from './device.ts'; + +const WG_SIZE = 256; + +export interface PruneResult { + leafKeys: GPUBuffer; + masks: GPUBuffer; + leafCount: number; +} + +export class Pruner { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly mark: GPUComputePipeline; + readonly compact: GPUComputePipeline; + + constructor(device: GPUDevice, scanner = new Scanner(device)) { + this.device = device; + this.scanner = scanner; + const module = device.createShaderModule({ code: pruneWgsl }); + this.mark = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint: 'mark' } }); + this.compact = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint: 'compact' } }); + } + + async prune(leafKeys: GPUBuffer, masks: GPUBuffer, retain: GPUBuffer, leafCount: number): Promise { + const device = this.device; + const params = device.createBuffer({ size: 16, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([leafCount])); + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const flags = device.createBuffer({ size: (leafCount + 1) * 4, usage: storage }); + + const group = (pipeline: GPUComputePipeline, buffers: Record) => + device.createBindGroup({ + layout: pipeline.getBindGroupLayout(0), + entries: [ + { binding: 0, resource: { buffer: params } }, + ...Object.entries(buffers).map(([binding, buffer]) => ({ binding: Number(binding), resource: { buffer } })), + ], + }); + + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setPipeline(this.mark); + pass.setBindGroup(0, group(this.mark, { 2: masks, 3: retain, 4: flags })); + dispatch2D(pass, Math.ceil((leafCount + 1) / WG_SIZE)); + this.scanner.plan(flags, leafCount + 1).encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const outCount = (await readBackU32(device, flags, leafCount + 1))[leafCount]; + + const outKeys = device.createBuffer({ size: Math.max(outCount, 1) * 4, usage: storage }); + const outMasks = device.createBuffer({ size: Math.max(outCount, 1) * 16 * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setPipeline(this.compact); + pass.setBindGroup(0, group(this.compact, { 1: leafKeys, 2: masks, 3: retain, 4: flags, 5: outKeys, 6: outMasks })); + dispatch2D(pass, Math.ceil(leafCount / WG_SIZE)); + pass.end(); + device.queue.submit([encoder.finish()]); + } + return { leafKeys: outKeys, masks: outMasks, leafCount: outCount }; + } +} diff --git a/ts/gpu/radix_sort.test.ts b/ts/gpu/radix_sort.test.ts new file mode 100644 index 0000000..9750935 --- /dev/null +++ b/ts/gpu/radix_sort.test.ts @@ -0,0 +1,58 @@ +import { hasWebGPU, requestDevice, createU32Buffer, readBackU32 } from './device.ts'; +import { mulberry32, assertU32ArrayEqual } from './test_util.ts'; +import { Sorter } from './radix_sort.ts'; + +const gpu = await hasWebGPU(); + +async function sortOnGpu(sorter: Sorter, keys: Uint32Array): Promise<{ keys: Uint32Array; vals: Uint32Array }> { + const device = sorter.device; + const n = keys.length; + const vals = new Uint32Array(n); + for (let i = 0; i < n; i++) vals[i] = i; + + const keyBuf = createU32Buffer(device, keys); + const valBuf = createU32Buffer(device, vals); + const plan = sorter.plan(keyBuf, valBuf, n); + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + plan.encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + + const out = { keys: await readBackU32(device, keyBuf, n), vals: await readBackU32(device, valBuf, n) }; + keyBuf.destroy(); + valBuf.destroy(); + return out; +} + +Deno.test({ name: 'radix sort matches stable JS sort', ignore: !gpu }, async () => { + const device = await requestDevice(); + const sorter = new Sorter(device); + const rand = mulberry32(2); + + const cases: Array<{ n: number; keyBits: number }> = [ + { n: 1, keyBits: 32 }, + { n: 100, keyBits: 32 }, + { n: 1024, keyBits: 32 }, + { n: 1030, keyBits: 32 }, + { n: 65536, keyBits: 32 }, + { n: 1 << 20, keyBits: 32 }, + // Few distinct keys exercise stability with long equal runs. + { n: 100000, keyBits: 4 }, + ]; + for (const { n, keyBits } of cases) { + const keys = new Uint32Array(n); + const mask = keyBits === 32 ? 0xffffffff : (1 << keyBits) - 1; + for (let i = 0; i < n; i++) keys[i] = (Math.floor(rand() * 4294967296) & mask) >>> 0; + + // Array.prototype.sort is stable, so sorting indices by key is the + // reference for both outputs. + const order = [...Array(n).keys()].sort((a, b) => keys[a] - keys[b]); + const expectedKeys = new Uint32Array(order.map((i) => keys[i])); + const expectedVals = new Uint32Array(order); + + const got = await sortOnGpu(sorter, keys); + assertU32ArrayEqual(got.keys, expectedKeys, `sorted keys n=${n} bits=${keyBits}`); + assertU32ArrayEqual(got.vals, expectedVals, `payload order n=${n} bits=${keyBits}`); + } +}); diff --git a/ts/gpu/radix_sort.ts b/ts/gpu/radix_sort.ts new file mode 100644 index 0000000..00cf081 --- /dev/null +++ b/ts/gpu/radix_sort.ts @@ -0,0 +1,95 @@ +// Host side of wgsl/radix_sort.wgsl. Plans a stable sort of u32 keys with +// u32 payloads. The result lands back in the caller's buffers. + +import sortWgsl from 'picovdb/wgsl/radix_sort.wgsl' with { type: 'text' }; +import { Scanner, type ScanPlan } from './scan.ts'; + +const TILE = 1024; +const RADIX = 16; +const PASSES = 8; + +export class Sorter { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly layout: GPUBindGroupLayout; + readonly histogram: GPUComputePipeline; + readonly scatter: GPUComputePipeline; + + constructor(device: GPUDevice, scanner: Scanner = new Scanner(device)) { + this.device = device; + this.scanner = scanner; + this.layout = device.createBindGroupLayout({ + entries: [ + { binding: 0, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'uniform' } }, + { binding: 1, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'read-only-storage' } }, + { binding: 2, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'read-only-storage' } }, + { binding: 3, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'storage' } }, + { binding: 4, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'storage' } }, + { binding: 5, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'storage' } }, + ], + }); + const module = device.createShaderModule({ code: sortWgsl }); + const layout = device.createPipelineLayout({ bindGroupLayouts: [this.layout] }); + this.histogram = device.createComputePipeline({ layout, compute: { module, entryPoint: 'histogram' } }); + this.scatter = device.createComputePipeline({ layout, compute: { module, entryPoint: 'scatter' } }); + } + + /** Plans an in place sort of the first n key and value pairs. */ + plan(keys: GPUBuffer, vals: GPUBuffer, n: number): SortPlan { + return new SortPlan(this, keys, vals, n); + } +} + +export class SortPlan { + private readonly sorter: Sorter; + private readonly numTiles: number; + private readonly bindGroups: GPUBindGroup[] = []; // one per pass + private readonly histScan: ScanPlan; + + constructor(sorter: Sorter, keys: GPUBuffer, vals: GPUBuffer, n: number) { + this.sorter = sorter; + const { device } = sorter; + this.numTiles = Math.max(1, Math.ceil(n / TILE)); + if (this.numTiles > 65535) throw new Error(`sort of ${n} elements exceeds one dispatch dimension`); + + const scratch = { size: Math.max(n, 1) * 4, usage: GPUBufferUsage.STORAGE }; + const keysB = device.createBuffer(scratch); + const valsB = device.createBuffer(scratch); + const hist = device.createBuffer({ + size: RADIX * this.numTiles * 4, + usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC, + }); + this.histScan = sorter.scanner.plan(hist, RADIX * this.numTiles); + + for (let pass = 0; pass < PASSES; pass++) { + const params = device.createBuffer({ size: 16, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([n, pass * 4, this.numTiles])); + const forward = pass % 2 === 0; + this.bindGroups.push( + device.createBindGroup({ + layout: sorter.layout, + entries: [ + { binding: 0, resource: { buffer: params } }, + { binding: 1, resource: { buffer: forward ? keys : keysB } }, + { binding: 2, resource: { buffer: forward ? vals : valsB } }, + { binding: 3, resource: { buffer: forward ? keysB : keys } }, + { binding: 4, resource: { buffer: forward ? valsB : vals } }, + { binding: 5, resource: { buffer: hist } }, + ], + }) + ); + } + } + + encode(pass: GPUComputePassEncoder): void { + for (const bindGroup of this.bindGroups) { + pass.setPipeline(this.sorter.histogram); + pass.setBindGroup(0, bindGroup); + pass.dispatchWorkgroups(this.numTiles); + this.histScan.encode(pass); + pass.setPipeline(this.sorter.scatter); + pass.setBindGroup(0, bindGroup); + pass.dispatchWorkgroups(this.numTiles); + } + } +} diff --git a/ts/gpu/rasterize.test.ts b/ts/gpu/rasterize.test.ts new file mode 100644 index 0000000..8fe2a55 --- /dev/null +++ b/ts/gpu/rasterize.test.ts @@ -0,0 +1,88 @@ +import { hasWebGPU, requestDevice, readBackU32 } from './device.ts'; +import { mulberry32 } from './test_util.ts'; +import { refRasterize, parseBinarySTL } from './reference.ts'; +import { Binner } from './mesh_to_grid.ts'; +import { Rasterizer } from './rasterize.ts'; + +const gpu = await hasWebGPU(); + +let stl: Uint8Array | null = null; +try { + stl = Deno.readFileSync(new URL('../../data/bases/base_32mm.stl', import.meta.url)); +} catch { + // Sample data not present, so the STL test skips. +} + +async function checkDistances( + binner: Binner, + rasterizer: Rasterizer, + points: Float32Array, + triangles: Uint32Array, + voxelSize: number, + halfWidth: number, + label: string +): Promise<{ band: number; total: number }> { + const bin = await binner.bin(points, triangles, { voxelSize, halfWidth }); + const leafValues = rasterizer.rasterize(bin, { halfWidth }); + const got = await readBackU32(binner.device, leafValues, bin.leafCount * 512); + const ref = refRasterize(points, triangles, voxelSize, halfWidth, bin.leafMin); + const expected = new Uint32Array(ref.values.buffer); + + if (got.length !== expected.length) throw new Error(`${label}: slab size mismatch`); + // Band membership must match exactly. GPU compilers fuse multiply adds, + // which can flip closest feature branches in the distance function. The + // branches are continuous so the distance error stays tiny. + const INF = 0x7f800000; + const MAX_ABS_D = 1e-3; // voxel units + const gotF = new Float32Array(got.buffer); + let band = 0; + let maxAbs = 0; + let membershipFlips = 0; + let firstBad = ''; + for (let i = 0; i < got.length; i++) { + const e = expected[i]; + if (e !== INF) band++; + if ((e === INF) !== (got[i] === INF)) { + membershipFlips++; + if (!firstBad) firstBad = `[${i}] membership: got 0x${got[i].toString(16)}, expected 0x${e.toString(16)}`; + continue; + } + if (e === INF) continue; + const d = Math.abs(Math.sqrt(gotF[i]) - Math.sqrt(ref.values[i])); + if (d > maxAbs) maxAbs = d; + if (d > MAX_ABS_D && !firstBad) { + firstBad = `[${i}] |Δd|=${d}: got ${Math.sqrt(gotF[i])}, expected ${Math.sqrt(ref.values[i])}`; + } + } + if (membershipFlips > 0 || maxAbs > MAX_ABS_D) { + throw new Error(`${label}: ${membershipFlips} membership flips, max |Δd| ${maxAbs}; first ${firstBad}`); + } + return { band, total: got.length }; +} + +Deno.test({ name: 'distance rasterization matches reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const binner = new Binner(device); + const rasterizer = new Rasterizer(device); + const rand = mulberry32(4); + + const triCount = 100; + const points = new Float32Array(triCount * 9); + for (let i = 0; i < points.length; i++) points[i] = (rand() - 0.5) * 40; + const triangles = new Uint32Array([...Array(triCount * 3).keys()]); + const soup = await checkDistances(binner, rasterizer, points, triangles, 0.5, 3, 'soup'); + if (soup.band === 0) throw new Error('no band voxels in soup'); + + // A collinear triangle exercises the segment fallback. + const line = new Float32Array([0, 0, 0, 4, 0, 0, 8, 0, 0]); + await checkDistances(binner, rasterizer, line, new Uint32Array([0, 1, 2]), 1, 2, 'collinear'); +}); + +Deno.test({ name: 'STL distance rasterization matches reference', ignore: !gpu || !stl }, async () => { + const device = await requestDevice(); + const binner = new Binner(device); + const rasterizer = new Rasterizer(device); + const { points, triangles } = parseBinarySTL(stl!); + const { band, total } = await checkDistances(binner, rasterizer, points, triangles, 0.25, 3, 'stl'); + if (band < 1000) throw new Error(`implausibly few band voxels: ${band}/${total}`); +}); diff --git a/ts/gpu/rasterize.ts b/ts/gpu/rasterize.ts new file mode 100644 index 0000000..dca3ab1 --- /dev/null +++ b/ts/gpu/rasterize.ts @@ -0,0 +1,78 @@ +// Host side of wgsl/rasterize.wgsl. Fills each binned leaf's slab with +// the minimum squared distance to any incident triangle in voxel units, +// with infinity where no triangle is within the band. + +import rasterWgsl from 'picovdb/wgsl/rasterize.wgsl' with { type: 'text' }; +import { dispatch2D } from './device.ts'; +import type { BinResult } from './mesh_to_grid.ts'; + +export interface RasterizeOptions { + /** Narrow band half width in voxels. Must match the binning pass. */ + halfWidth: number; +} + +export class Rasterizer { + readonly device: GPUDevice; + readonly layout: GPUBindGroupLayout; + readonly rasterizePipeline: GPUComputePipeline; + + constructor(device: GPUDevice) { + this.device = device; + const entry = (binding: number, type: GPUBufferBindingType): GPUBindGroupLayoutEntry => ({ + binding, + visibility: GPUShaderStage.COMPUTE, + buffer: { type }, + }); + this.layout = device.createBindGroupLayout({ + entries: [ + entry(0, 'uniform'), + entry(1, 'read-only-storage'), + entry(2, 'read-only-storage'), + entry(3, 'read-only-storage'), + entry(4, 'read-only-storage'), + entry(5, 'read-only-storage'), + entry(6, 'storage'), + ], + }); + const module = device.createShaderModule({ code: rasterWgsl }); + const layout = device.createPipelineLayout({ bindGroupLayouts: [this.layout] }); + this.rasterizePipeline = device.createComputePipeline({ layout, compute: { module, entryPoint: 'rasterize' } }); + } + + /** Returns one slab of 512 squared distances per leaf as f32 bits. */ + rasterize(bin: BinResult, opts: RasterizeOptions): GPUBuffer { + const device = this.device; + const valueBytes = bin.leafCount * 512 * 4; + if (valueBytes > device.limits.maxStorageBufferBindingSize) { + throw new Error(`${bin.leafCount} leaf slabs (${valueBytes} bytes) exceed the storage binding limit`); + } + + const params = device.createBuffer({ size: 32, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([bin.pairCount, bin.leafCount])); + device.queue.writeBuffer(params, 8, new Float32Array([opts.halfWidth])); + device.queue.writeBuffer(params, 16, new Int32Array(bin.leafMin)); + + const leafValues = device.createBuffer({ size: valueBytes, usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC }); + const bindGroup = device.createBindGroup({ + layout: this.layout, + entries: [ + { binding: 0, resource: { buffer: params } }, + { binding: 1, resource: { buffer: bin.pointsIndex } }, + { binding: 2, resource: { buffer: bin.triangles } }, + { binding: 3, resource: { buffer: bin.pairKeys } }, + { binding: 4, resource: { buffer: bin.pairTris } }, + { binding: 5, resource: { buffer: bin.leafKeys } }, + { binding: 6, resource: { buffer: leafValues } }, + ], + }); + + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, bindGroup); + pass.setPipeline(this.rasterizePipeline); + dispatch2D(pass, bin.leafCount); + pass.end(); + device.queue.submit([encoder.finish()]); + return leafValues; + } +} diff --git a/ts/gpu/reference.ts b/ts/gpu/reference.ts new file mode 100644 index 0000000..9357c39 --- /dev/null +++ b/ts/gpu/reference.ts @@ -0,0 +1,350 @@ +// f32 exact reference implementations of the GPU stages, used by the +// tests. Every operation rounds through Math.fround, which matches the +// correctly rounded f32 result for f32 operands evaluated in f64, so +// these mirror the WGSL bit for bit. + +const f = Math.fround; + +type V3 = [number, number, number]; + +function sub(a: V3, b: V3): V3 { + return [f(a[0] - b[0]), f(a[1] - b[1]), f(a[2] - b[2])]; +} + +function add(a: V3, b: V3): V3 { + return [f(a[0] + b[0]), f(a[1] + b[1]), f(a[2] + b[2])]; +} + +function scale(a: V3, s: number): V3 { + return [f(a[0] * s), f(a[1] * s), f(a[2] * s)]; +} + +function dot3(a: V3, b: V3): number { + return f(f(f(a[0] * b[0]) + f(a[1] * b[1])) + f(a[2] * b[2])); +} + +function dsq(a: V3): number { + return dot3(a, a); +} + +function distSqPointSegment(p: V3, a: V3, b: V3): number { + const ab = sub(b, a); + const denom = dsq(ab); + if (denom <= 0) return dsq(sub(p, a)); + const t = Math.min(Math.max(f(dot3(sub(p, a), ab) / denom), 0), 1); + return dsq(sub(p, add(a, scale(ab, t)))); +} + +/** Mirrors distSqPointTriangle in src/mesh_to_grid.zig and wgsl/rasterize.wgsl. */ +export function distSqPointTriangle(p: V3, a: V3, b: V3, c: V3): number { + const ab = sub(b, a); + const ac = sub(c, a); + const ap = sub(p, a); + const d1 = dot3(ab, ap); + const d2 = dot3(ac, ap); + if (d1 <= 0 && d2 <= 0) return dsq(ap); // vertex a + + const bp = sub(p, b); + const d3 = dot3(ab, bp); + const d4 = dot3(ac, bp); + if (d3 >= 0 && d4 <= d3) return dsq(bp); // vertex b + + const vc = f(f(d1 * d4) - f(d3 * d2)); + if (vc <= 0 && d1 >= 0 && d3 <= 0) { + const denom = f(d1 - d3); + if (denom > 0) return dsq(sub(ap, scale(ab, f(d1 / denom)))); // edge ab + } + + const cp = sub(p, c); + const d5 = dot3(ab, cp); + const d6 = dot3(ac, cp); + if (d6 >= 0 && d5 <= d6) return dsq(cp); // vertex c + + const vb = f(f(d5 * d2) - f(d1 * d6)); + if (vb <= 0 && d2 >= 0 && d6 <= 0) { + const denom = f(d2 - d6); + if (denom > 0) return dsq(sub(ap, scale(ac, f(d2 / denom)))); // edge ac + } + + const va = f(f(d3 * d6) - f(d5 * d4)); + if (va <= 0 && f(d4 - d3) >= 0 && f(d5 - d6) >= 0) { + const denom = f(f(d4 - d3) + f(d5 - d6)); + if (denom > 0) return dsq(sub(bp, scale(sub(c, b), f(f(d4 - d3) / denom)))); // edge bc + } + + const denom = f(f(va + vb) + vc); + if (denom <= 0) { + // Degenerate triangles fall back to edge distances. + return Math.min(distSqPointSegment(p, a, b), distSqPointSegment(p, a, c), distSqPointSegment(p, b, c)); + } + const inv = f(1 / denom); + return dsq(sub(ap, add(scale(ab, f(vb * inv)), scale(ac, f(vc * inv))))); // face +} + +export interface RefBin { + pairKeys: Uint32Array; + pairTris: Uint32Array; + leafKeys: Uint32Array; +} + +/** + * Reference of the binning contract in rasterizeTriangle from + * src/mesh_to_grid.zig, with f32 arithmetic throughout. + */ +export function refBin( + points: Float32Array, + triangles: Uint32Array, + voxelSize: number, + halfWidth: number, + leafMin: [number, number, number] +): RefBin { + const pts = transform(points, voxelSize); + const pairs: Array<[number, number]> = []; + for (let t = 0; t < triangles.length / 3; t++) { + const { lo, hi } = leafRange(pts, triangles, t, halfWidth); + for (let x = lo[0]; x <= hi[0]; x++) { + for (let y = lo[1]; y <= hi[1]; y++) { + for (let z = lo[2]; z <= hi[2]; z++) { + const key = (((x - leafMin[0]) << 20) | ((y - leafMin[1]) << 10) | (z - leafMin[2])) >>> 0; + pairs.push([key, t]); + } + } + } + } + pairs.sort((a, b) => a[0] - b[0] || 0); // stable sort preserves triangle order per key + const pairKeys = new Uint32Array(pairs.map((p) => p[0])); + const pairTris = new Uint32Array(pairs.map((p) => p[1])); + const leafKeys = new Uint32Array([...new Set(pairKeys)]); + return { pairKeys, pairTris, leafKeys }; +} + +/** + * Reference of the distance stage. Returns one slab of minimum squared + * distances per leaf, with infinity where untouched. + */ +export function refRasterize( + points: Float32Array, + triangles: Uint32Array, + voxelSize: number, + halfWidth: number, + leafMin: [number, number, number] +): { bin: RefBin; values: Float32Array } { + const bin = refBin(points, triangles, voxelSize, halfWidth, leafMin); + const pts = transform(points, voxelSize); + const slot = new Map(); + bin.leafKeys.forEach((key, i) => slot.set(key, i)); + const values = new Float32Array(bin.leafKeys.length * 512).fill(Infinity); + const hw2 = f(halfWidth * halfWidth); + + for (let i = 0; i < bin.pairKeys.length; i++) { + const t = bin.pairTris[i]; + const a = vert(pts, triangles[t * 3]); + const b = vert(pts, triangles[t * 3 + 1]); + const c = vert(pts, triangles[t * 3 + 2]); + const { loV, hiV } = voxelRange(pts, triangles, t, halfWidth); + const key = bin.pairKeys[i]; + const origin = [ + (((key >>> 20) & 0x3ff) + leafMin[0]) << 3, + (((key >>> 10) & 0x3ff) + leafMin[1]) << 3, + ((key & 0x3ff) + leafMin[2]) << 3, + ]; + const base = slot.get(key)! * 512; + for (let x = Math.max(loV[0], origin[0]); x <= Math.min(hiV[0], origin[0] + 7); x++) { + for (let y = Math.max(loV[1], origin[1]); y <= Math.min(hiV[1], origin[1] + 7); y++) { + for (let z = Math.max(loV[2], origin[2]); z <= Math.min(hiV[2], origin[2] + 7); z++) { + const d2 = distSqPointTriangle([x, y, z], a, b, c); + if (d2 <= hw2) { + const n = base + ((x & 7) << 6) + ((y & 7) << 3) + (z & 7); + if (d2 < values[n]) values[n] = d2; + } + } + } + } + } + return { bin, values }; +} + +/** + * Reference of the sign stage, mirroring ColumnGrid from + * src/mesh_to_grid.zig in f64, the same precision as the CPU. Returns one + * inside mask per leaf in voxel bit order. + */ +export function refSign( + points: Float32Array, + triangles: Uint32Array, + voxelSize: number, + leafKeys: Uint32Array, + leafMin: [number, number, number] +): Uint32Array { + const pts = transform(points, voxelSize); + const edge = (px: number, py: number, qx: number, qy: number, sx: number, sy: number): number => + (qx - px) * (sy - py) - (qy - py) * (sx - px); + const accept = (w: number, ex: number, ey: number): boolean => { + if (w > 0) return true; + if (w < 0) return false; + return ey < 0 || (ey === 0 && ex > 0); + }; + + const cols = new Map(); + for (let t = 0; t < triangles.length / 3; t++) { + const [ax, ay, az] = vert(pts, triangles[t * 3]); + const [bx, by, bz] = vert(pts, triangles[t * 3 + 1]); + const [cx, cy, cz] = vert(pts, triangles[t * 3 + 2]); + const signedArea = edge(ax, ay, bx, by, cx, cy); + if (signedArea === 0) continue; // vertical triangle + const flip = signedArea < 0 ? -1 : 1; + const area = flip * signedArea; + const x0 = Math.ceil(Math.min(ax, bx, cx)); + const x1 = Math.floor(Math.max(ax, bx, cx)); + const y0 = Math.ceil(Math.min(ay, by, cy)); + const y1 = Math.floor(Math.max(ay, by, cy)); + for (let x = x0; x <= x1; x++) { + for (let y = y0; y <= y1; y++) { + const w0 = flip * edge(bx, by, cx, cy, x, y); + const w1 = flip * edge(cx, cy, ax, ay, x, y); + const w2 = flip * edge(ax, ay, bx, by, x, y); + const inside = + accept(w0, flip * (cx - bx), flip * (cy - by)) && + accept(w1, flip * (ax - cx), flip * (ay - cy)) && + accept(w2, flip * (bx - ax), flip * (by - ay)); + if (!inside) continue; + const z = (w0 * az + w1 * bz + w2 * cz) / area; + const key = `${x},${y}`; + let list = cols.get(key); + if (!list) cols.set(key, (list = [])); + list.push(z); + } + } + } + for (const list of cols.values()) list.sort((a, b) => a - b); + + const masks = new Uint32Array(leafKeys.length * 16); + leafKeys.forEach((key, li) => { + const ox = (((key >>> 20) & 0x3ff) + leafMin[0]) << 3; + const oy = (((key >>> 10) & 0x3ff) + leafMin[1]) << 3; + const oz = ((key & 0x3ff) + leafMin[2]) << 3; + for (let c = 0; c < 64; c++) { + const list = cols.get(`${ox + (c >> 3)},${oy + (c & 7)}`) ?? []; + let ptr = 0; + let count = 0; + let bits = 0; + for (let z = 0; z < 8; z++) { + while (ptr < list.length && list[ptr] < oz + z) { + ptr++; + count++; + } + bits |= (count & 1) << z; + } + const n0 = c * 8; + masks[li * 16 + (n0 >> 5)] |= (bits << (n0 & 31)) >>> 0; + } + }); + return masks; +} + +function transform(points: Float32Array, voxelSize: number): Float32Array { + const inv = f(1 / voxelSize); + const pts = new Float32Array(points.length); + for (let i = 0; i < points.length; i++) pts[i] = f(points[i] * inv); + return pts; +} + +function vert(pts: Float32Array, i: number): V3 { + return [pts[i * 3], pts[i * 3 + 1], pts[i * 3 + 2]]; +} + +function voxelRange(pts: Float32Array, triangles: Uint32Array, t: number, halfWidth: number) { + const loV = [0, 0, 0]; + const hiV = [0, 0, 0]; + for (let axis = 0; axis < 3; axis++) { + const a = pts[triangles[t * 3] * 3 + axis]; + const b = pts[triangles[t * 3 + 1] * 3 + axis]; + const c = pts[triangles[t * 3 + 2] * 3 + axis]; + loV[axis] = Math.ceil(f(Math.min(a, b, c) - halfWidth)); + hiV[axis] = Math.floor(f(Math.max(a, b, c) + halfWidth)); + } + return { loV, hiV }; +} + +function leafRange(pts: Float32Array, triangles: Uint32Array, t: number, halfWidth: number) { + const { loV, hiV } = voxelRange(pts, triangles, t, halfWidth); + return { lo: loV.map((v) => v >> 3), hi: hiV.map((v) => v >> 3) }; +} + +/** + * Reference of the csg passes in wgsl/merge.wgsl. Unions the leaf tables + * and takes per voxel minima, where a grid without the leaf contributes + * its implicit background. The band holds voxels with |v| below + * halfWidth; leaves without band voxels are dropped. + */ +export function refCsgMerge( + aKeys: Uint32Array, + aValues: Float32Array, + bKeys: Uint32Array, + bValues: Float32Array, + halfWidth: number, + op: 'union' | 'intersect' | 'subtract' = 'union' +): { keys: Uint32Array; masks: Uint32Array; values: Float32Array } { + const implicit = (keys: Uint32Array, values: Float32Array, leaf: [number, number, number], n: number): number => { + const colBase = ((leaf[0] << 20) | (leaf[1] << 10)) >>> 0; + let lo = 0; + let hi = keys.length; + while (lo < hi) { + const mid = (lo + hi) >> 1; + if (keys[mid] < colBase) lo = mid + 1; + else hi = mid; + } + if (lo >= keys.length || (keys[lo] & 0xfffffc00) >>> 0 !== colBase) return halfWidth; + let best = lo; + let bestD = Math.abs((keys[lo] & 0x3ff) - leaf[2]); + for (let i = lo + 1; i < keys.length && ((keys[i] & 0xfffffc00) >>> 0) === colBase; i++) { + const d = Math.abs((keys[i] & 0x3ff) - leaf[2]); + if (d >= bestD) break; + best = i; + bestD = d; + } + const zloc = (keys[best] & 0x3ff) < leaf[2] ? 7 : 0; + const facing = (n & 0x1f8) | zloc; + return values[best * 512 + facing] < 0 ? -halfWidth : halfWidth; + }; + + const union = new Uint32Array([...new Set([...aKeys, ...bKeys])].sort((x, y) => x - y)); + const aIdx = new Map(); + aKeys.forEach((k, i) => aIdx.set(k, i)); + const bIdx = new Map(); + bKeys.forEach((k, i) => bIdx.set(k, i)); + const masks = new Uint32Array(union.length * 16); + const values = new Float32Array(union.length * 512); + union.forEach((key, i) => { + const leaf: [number, number, number] = [(key >>> 20) & 0x3ff, (key >>> 10) & 0x3ff, key & 0x3ff]; + const ai = aIdx.get(key); + const bi = bIdx.get(key); + for (let n = 0; n < 512; n++) { + const va = ai !== undefined ? aValues[ai * 512 + n] : implicit(aKeys, aValues, leaf, n); + const vb = bi !== undefined ? bValues[bi * 512 + n] : implicit(bKeys, bValues, leaf, n); + const v = op === 'union' ? Math.min(va, vb) : op === 'intersect' ? Math.max(va, vb) : Math.max(va, -vb); + values[i * 512 + n] = v; + if (Math.abs(v) < halfWidth) masks[i * 16 + (n >> 5)] |= 1 << (n & 31); + } + }); + const kept = [...union.keys()].filter((i) => masks.subarray(i * 16, i * 16 + 16).some((m) => m !== 0)); + return { + keys: Uint32Array.from(kept, (i) => union[i]), + masks: Uint32Array.from(kept.flatMap((i) => [...masks.subarray(i * 16, i * 16 + 16)])), + values: Float32Array.from(kept.flatMap((i) => [...values.subarray(i * 512, i * 512 + 512)])), + }; +} + +/** Binary STL parser producing a triangle soup for test inputs. */ +export function parseBinarySTL(bytes: Uint8Array): { points: Float32Array; triangles: Uint32Array } { + const view = new DataView(bytes.buffer, bytes.byteOffset, bytes.byteLength); + const triCount = view.getUint32(80, true); + const points = new Float32Array(triCount * 9); + const triangles = new Uint32Array(triCount * 3); + for (let t = 0; t < triCount; t++) { + const base = 84 + t * 50 + 12; // skip normal + for (let i = 0; i < 9; i++) points[t * 9 + i] = view.getFloat32(base + i * 4, true); + for (let v = 0; v < 3; v++) triangles[t * 3 + v] = t * 3 + v; + } + return { points, triangles }; +} diff --git a/ts/gpu/remap.test.ts b/ts/gpu/remap.test.ts new file mode 100644 index 0000000..4cfadeb --- /dev/null +++ b/ts/gpu/remap.test.ts @@ -0,0 +1,44 @@ +import { hasWebGPU, requestDevice, readBackU32 } from './device.ts'; +import { checkAnalytic, emptyGrid, assertU32ArrayEqual } from './test_util.ts'; +import { Stamper, sphere as sphereShape } from './stamp.ts'; +import { Remapper } from './remap.ts'; + +const gpu = await hasWebGPU(); + +Deno.test({ name: 'offset, translate, and rebase match the analytic sphere', ignore: !gpu }, async () => { + const device = await requestDevice(); + const stamper = new Stamper(device); + const remapper = new Remapper(device); + const halfWidth = 3; + const center: [number, number, number] = [100.3, 97.2, 88.9]; + const sphere = (c: number[], r: number) => (p: [number, number, number]) => Math.hypot(p[0] - c[0], p[1] - c[1], p[2] - c[2]) - r; + const base = await stamper.stamp(emptyGrid(device), { shape: sphereShape(center, 20), mode: 'add', halfWidth }); + + // Offsets redistance from the offset surface, exact to marching cubes + // precision over the whole band: about a hundredth of a voxel per step + // at this radius, with no seam. Offsets past halfWidth - 1 run in + // steps, so their bias adds up. + const grown = await remapper.offset(base, 2, halfWidth); + const g = await checkAnalytic(device, grown, sphere(center, 22), halfWidth, 'grow', Infinity, 0.03); + const shrunk = await remapper.offset(base, -2, halfWidth); + const s = await checkAnalytic(device, shrunk, sphere(center, 18), halfWidth, 'shrink', Infinity, 0.03); + if (!(g.band > s.band)) throw new Error(`grown band ${g.band} should exceed shrunk band ${s.band}`); + const far = await remapper.offset(base, 5, halfWidth); + await checkAnalytic(device, far, sphere(center, 25), halfWidth, 'grow past the band', Infinity, 0.05); + + // Translation is exact. + const shift: [number, number, number] = [3, -5, 8]; + const moved = await remapper.translate(base, shift, halfWidth); + await checkAnalytic(device, moved, sphere(center.map((c, a) => c + shift[a]), 20), halfWidth, 'translate'); + const movedLeaf = await remapper.translate(base, [8, -16, 0], halfWidth); + if (movedLeaf.leafKeys !== base.leafKeys || movedLeaf.leafMin[1] !== base.leafMin[1] - 2) throw new Error('leaf multiple translate should only move the origin'); + await checkAnalytic(device, movedLeaf, sphere([center[0] + 8, center[1] - 16, center[2]], 20), halfWidth, 'translate leaves'); + + // Rebase moves the key origin and nothing else. + const rebased = remapper.rebase(base, [-3, 1, -7]); + await checkAnalytic(device, rebased, sphere(center, 20), halfWidth, 'rebase'); + const keys = await readBackU32(device, rebased.leafKeys, rebased.leafCount); + const expect = (await readBackU32(device, base.leafKeys, base.leafCount)).map((k) => (k + (3 << 20) + (-1 << 10) + 7) >>> 0); + assertU32ArrayEqual(keys, expect, 'rebased keys'); + console.log(` offset: base ${base.leafCount} leaves, grown ${grown.leafCount} (${g.band} band), shrunk ${shrunk.leafCount} (${s.band} band), moved ${moved.leafCount}`); +}); diff --git a/ts/gpu/remap.ts b/ts/gpu/remap.ts new file mode 100644 index 0000000..a2cb6b2 --- /dev/null +++ b/ts/gpu/remap.ts @@ -0,0 +1,242 @@ +// Host side of wgsl/remap.wgsl: SDF offset, integer translation, and key +// rebasing of op layer grids. + +import remapWgsl from 'picovdb/wgsl/remap.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { Sorter } from './radix_sort.ts'; +import { Extractor } from './extract.ts'; +import { Binner } from './mesh_to_grid.ts'; +import { Rasterizer } from './rasterize.ts'; +import { dispatch2D, readBackTotals } from './device.ts'; +import { GridWriter, preludeWgsl, readerWgsl, type OpGrid } from './opgrid.ts'; + +const WG_SIZE = 256; + +export class Remapper { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly sorter: Sorter; + private readonly pipelines: Record = {}; + private extractor?: Extractor; + private binner?: Binner; + private rasterizer?: Rasterizer; + + constructor(device: GPUDevice, scanner: Scanner = new Scanner(device), sorter: Sorter = new Sorter(device, scanner)) { + this.device = device; + this.scanner = scanner; + this.sorter = sorter; + const module = device.createShaderModule({ code: preludeWgsl + readerWgsl('old', 'params.old_count') + remapWgsl }); + for (const entryPoint of ['rebase', 'generate_candidates', 'mark_unique', 'compact_unique', 'mark', 'apply', 'opgrid_compact']) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint } }); + } + } + + /** + * Rewrites the keys relative to a new origin. The result shares the + * input's leaves and data; only leafKeys is new. Every leaf must stay + * inside 0..1023 of the new origin. + */ + rebase(grid: OpGrid, leafMin: [number, number, number]): OpGrid { + const device = this.device; + const delta = grid.leafMin.map((v, a) => v - leafMin[a]); + if (delta.every((d) => d === 0)) return grid; + const params = this.params({ oldCount: grid.leafCount, delta }); + const newKeys = device.createBuffer({ + size: Math.max(grid.leafCount, 1) * 4, + usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC, + }); + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + this.run(pass, 'rebase', grid.leafCount, params, { 1: grid.leafKeys, 5: newKeys }); + pass.end(); + device.queue.submit([encoder.finish()]); + return { ...grid, leafKeys: newKeys, leafMin }; + } + + /** + * Offsets the level set by amount voxels. Positive grows the solid. + * Each step extracts the old zero level set and redistances the whole + * new band from its triangles. So the result has no seam and does not + * depend on stored values away from the surface. Steps are at most + * halfWidth - 1, to bound the reach. The leaf table must fit the current + * key origin with leafMargin(amount, halfWidth) leaves of margin; rebase + * first if it does not. + */ + async offset(grid: OpGrid, amount: number, halfWidth: number): Promise { + const maxStep = halfWidth - 1; + if (!(maxStep > 0)) throw new Error('offset needs a half width above one voxel'); + let out = grid; + for (let remaining = amount; remaining !== 0;) { + const step = Math.sign(remaining) * Math.min(Math.abs(remaining), maxStep); + const next = await this.offsetStep(out, step, halfWidth); + if (out !== grid) { + out.leafKeys.destroy(); + out.leaves.destroy(); + out.data.destroy(); + } + out = next; + remaining -= step; + } + return out; + } + + private async offsetStep(grid: OpGrid, amount: number, halfWidth: number): Promise { + this.extractor ??= new Extractor(this.device, this.scanner); + this.binner ??= new Binner(this.device, this.scanner, this.sorter); + this.rasterizer ??= new Rasterizer(this.device); + const reach = halfWidth + Math.abs(amount); + // Distances to the old surface, in the grid's own key space. + const mesh = await this.extractor.extract(grid, halfWidth); + const bin = await this.binner.binBuffers(mesh.points, mesh.triangleCount * 3, mesh.triangles, mesh.triangleCount, { + voxelSize: 1, + halfWidth: reach, + bounds: { leafMin: [0, 0, 0], leafMax: [1023, 1023, 1023] }, + }); + const dist = this.rasterizer.rasterize(bin, { halfWidth: reach }); + mesh.points.destroy(); + mesh.triangles.destroy(); + bin.pointsIndex.destroy(); + bin.pairKeys.destroy(); + bin.pairTris.destroy(); + const out = await this.finish(grid, halfWidth, { mode: 0, amount }, bin.leafKeys, bin.leafCount, { 9: dist }); + bin.leafKeys.destroy(); + dist.destroy(); + return out; + } + + /** Leaves an offset's band can reach beyond the input's leaves. */ + static leafMargin(amount: number, halfWidth: number): number { + return Math.ceil((halfWidth + Math.abs(amount) + 1) / 8); + } + + /** Translates by whole voxels. Multiples of a leaf only move the key origin. */ + translate(grid: OpGrid, shift: [number, number, number], halfWidth: number): Promise { + if (shift.some((s) => !Number.isInteger(s))) throw new Error('translate takes whole voxels'); + if (shift.every((s) => s % 8 === 0)) { + const leafMin = grid.leafMin.map((v, a) => v + shift[a] / 8) as [number, number, number]; + const leafMax = grid.leafMax.map((v, a) => v + shift[a] / 8) as [number, number, number]; + return Promise.resolve({ ...grid, leafMin, leafMax }); + } + const nbLo = shift.map((s) => Math.floor(s / 8)); + const nbHi = shift.map((s) => Math.floor((s + 7) / 8)); + return this.remap(grid, halfWidth, { mode: 1, shift, nbLo, nbDims: nbHi.map((h, a) => h - nbLo[a] + 1) }); + } + + /** Builds the candidates from the old leaves and their neighbors, then finishes. */ + private async remap( + grid: OpGrid, + halfWidth: number, + opts: { mode: number; shift: number[]; nbLo: number[]; nbDims: number[] } + ): Promise { + const device = this.device; + const vol = opts.nbDims[0] * opts.nbDims[1] * opts.nbDims[2]; + const concatCount = grid.leafCount * (1 + vol); + const params = this.params({ oldCount: grid.leafCount, concatCount, halfWidth, ...opts }); + + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const concatKeys = device.createBuffer({ size: concatCount * 4, usage: storage }); + const sortVals = device.createBuffer({ size: concatCount * 4, usage: GPUBufferUsage.STORAGE }); + const flags = device.createBuffer({ size: (concatCount + 1) * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + encoder.copyBufferToBuffer(grid.leafKeys, 0, concatKeys, 0, grid.leafCount * 4); + const pass = encoder.beginComputePass(); + this.run(pass, 'generate_candidates', grid.leafCount * vol, params, { 1: grid.leafKeys, 3: concatKeys }); + this.sorter.plan(concatKeys, sortVals, concatCount).encode(pass); + this.run(pass, 'mark_unique', concatCount + 1, params, { 3: concatKeys, 4: flags }); + this.scanner.plan(flags, concatCount + 1).encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [candCount] = await readBackTotals(device, [{ buffer: flags, index: concatCount }]); + device.queue.writeBuffer(params, 8, new Uint32Array([candCount])); + const candKeys = device.createBuffer({ size: candCount * 4, usage: storage }); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + this.run(pass, 'compact_unique', concatCount, params, { 3: concatKeys, 4: flags, 5: candKeys }); + pass.end(); + device.queue.submit([encoder.finish()]); + } + concatKeys.destroy(); + sortVals.destroy(); + flags.destroy(); + // The distance binding exists in every entry point's auto layout. + const placeholder = device.createBuffer({ size: 4, usage: GPUBufferUsage.STORAGE }); + const out = await this.finish(grid, halfWidth, opts, candKeys, candCount, { 9: placeholder }); + candKeys.destroy(); + placeholder.destroy(); + const leafMax = grid.leafMax.map((v, a) => Math.max(v, v + opts.nbLo[a] + opts.nbDims[a] - 1)) as [number, number, number]; + return { ...out, leafMax }; + } + + /** Runs the writer over the candidates. extra holds the mode's own bindings. */ + private async finish( + grid: OpGrid, + halfWidth: number, + opts: { mode: number; amount?: number; shift?: number[] }, + candKeys: GPUBuffer, + candCount: number, + extra: Record + ): Promise { + const device = this.device; + const params = this.params({ oldCount: grid.leafCount, halfWidth, distCount: candCount, ...opts }); + device.queue.writeBuffer(params, 8, new Uint32Array([candCount])); + const run = (pass: GPUComputePassEncoder, name: string, threads: number, buffers: Record) => this.run(pass, name, threads, params, buffers); + const inputs = { 1: grid.leafKeys, 2: grid.leaves, 5: candKeys, 6: grid.data, ...extra }; + const writer = new GridWriter(device, candCount); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'mark', candCount + 1, { ...inputs, ...writer.markBindings }); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const out = await writer.finish(this.scanner, this.pipelines['opgrid_compact'], candKeys, halfWidth, grid, (pass, bindings, count) => { + run(pass, 'apply', count, { ...inputs, ...bindings }); + }); + params.destroy(); + return out; + } + + private params(p: { + oldCount: number; + concatCount?: number; + mode?: number; + shift?: number[]; + amount?: number; + nbLo?: number[]; + halfWidth?: number; + nbDims?: number[]; + distCount?: number; + delta?: number[]; + }): GPUBuffer { + const buffer = this.device.createBuffer({ size: 96, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + const q = this.device.queue; + q.writeBuffer(buffer, 0, new Uint32Array([p.oldCount, p.concatCount ?? 0, 0, p.mode ?? 0])); + q.writeBuffer(buffer, 16, new Int32Array(p.shift ?? [0, 0, 0])); + q.writeBuffer(buffer, 28, new Float32Array([p.amount ?? 0])); + q.writeBuffer(buffer, 32, new Int32Array(p.nbLo ?? [0, 0, 0])); + q.writeBuffer(buffer, 44, new Float32Array([p.halfWidth ?? 0])); + q.writeBuffer(buffer, 48, new Int32Array(p.nbDims ?? [1, 1, 1])); + q.writeBuffer(buffer, 60, new Uint32Array([p.distCount ?? 0])); + q.writeBuffer(buffer, 64, new Int32Array(p.delta ?? [0, 0, 0])); + return buffer; + } + + private run(pass: GPUComputePassEncoder, name: string, threads: number, params: GPUBuffer, buffers: Record): void { + if (threads === 0) return; + pass.setPipeline(this.pipelines[name]); + pass.setBindGroup( + 0, + this.device.createBindGroup({ + layout: this.pipelines[name].getBindGroupLayout(0), + entries: [ + { binding: 0, resource: { buffer: params } }, + ...Object.entries(buffers).map(([binding, buffer]) => ({ binding: Number(binding), resource: { buffer } })), + ], + }) + ); + dispatch2D(pass, Math.ceil(threads / WG_SIZE)); + } +} diff --git a/ts/gpu/scan.test.ts b/ts/gpu/scan.test.ts new file mode 100644 index 0000000..487d4dc --- /dev/null +++ b/ts/gpu/scan.test.ts @@ -0,0 +1,39 @@ +import { hasWebGPU, requestDevice, createU32Buffer, readBackU32 } from './device.ts'; +import { mulberry32, assertU32ArrayEqual } from './test_util.ts'; +import { Scanner } from './scan.ts'; + +const gpu = await hasWebGPU(); + +function exclusiveScanRef(input: Uint32Array): Uint32Array { + const out = new Uint32Array(input.length); + let sum = 0; + for (let i = 0; i < input.length; i++) { + out[i] = sum; + sum = (sum + input[i]) >>> 0; + } + return out; +} + +Deno.test({ name: 'exclusive scan matches JS reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const scanner = new Scanner(device); + const rand = mulberry32(1); + + for (const n of [1, 7, 256, 1024, 1025, 4096, 65536, 1 << 20]) { + const input = new Uint32Array(n); + for (let i = 0; i < n; i++) input[i] = Math.floor(rand() * 1000); + const expected = exclusiveScanRef(input); + + const buffer = createU32Buffer(device, input); + const plan = scanner.plan(buffer, n); + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + plan.encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + + const got = await readBackU32(device, buffer, n); + assertU32ArrayEqual(got, expected, `scan n=${n}`); + buffer.destroy(); + } +}); diff --git a/ts/gpu/scan.ts b/ts/gpu/scan.ts new file mode 100644 index 0000000..a25122d --- /dev/null +++ b/ts/gpu/scan.ts @@ -0,0 +1,89 @@ +// Host side of wgsl/scan.wgsl. Plans an exclusive prefix scan of u32 +// values in place over a storage buffer. + +import scanWgsl from 'picovdb/wgsl/scan.wgsl' with { type: 'text' }; + +const TILE = 1024; + +interface ScanLevel { + count: number; // tiles at this level + partials: GPUBuffer; + bindGroup: GPUBindGroup; +} + +export class Scanner { + readonly device: GPUDevice; + readonly layout: GPUBindGroupLayout; + readonly scanTile: GPUComputePipeline; + readonly addOffsets: GPUComputePipeline; + + constructor(device: GPUDevice) { + this.device = device; + this.layout = device.createBindGroupLayout({ + entries: [ + { binding: 0, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'uniform' } }, + { binding: 1, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'storage' } }, + { binding: 2, visibility: GPUShaderStage.COMPUTE, buffer: { type: 'storage' } }, + ], + }); + const module = device.createShaderModule({ code: scanWgsl }); + const layout = device.createPipelineLayout({ bindGroupLayouts: [this.layout] }); + this.scanTile = device.createComputePipeline({ layout, compute: { module, entryPoint: 'scan_tile' } }); + this.addOffsets = device.createComputePipeline({ layout, compute: { module, entryPoint: 'add_offsets' } }); + } + + /** Plans an exclusive scan of the first n u32 elements of buffer. */ + plan(buffer: GPUBuffer, n: number): ScanPlan { + return new ScanPlan(this, buffer, n); + } +} + +export class ScanPlan { + private readonly scanner: Scanner; + private readonly levels: ScanLevel[] = []; + + constructor(scanner: Scanner, buffer: GPUBuffer, n: number) { + this.scanner = scanner; + const { device } = scanner; + let data = buffer; + let size = n; + for (;;) { + const count = Math.max(1, Math.ceil(size / TILE)); + if (count > 65535) throw new Error(`scan of ${n} elements exceeds one dispatch dimension`); + const params = device.createBuffer({ size: 16, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([size])); + const partials = device.createBuffer({ + size: count * 4, + usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_SRC, + }); + this.levels.push({ + count, + partials, + bindGroup: device.createBindGroup({ + layout: scanner.layout, + entries: [ + { binding: 0, resource: { buffer: params } }, + { binding: 1, resource: { buffer: data } }, + { binding: 2, resource: { buffer: partials } }, + ], + }), + }); + if (count === 1) break; + data = partials; + size = count; + } + } + + encode(pass: GPUComputePassEncoder): void { + pass.setPipeline(this.scanner.scanTile); + for (const level of this.levels) { + pass.setBindGroup(0, level.bindGroup); + pass.dispatchWorkgroups(level.count); + } + pass.setPipeline(this.scanner.addOffsets); + for (let i = this.levels.length - 2; i >= 0; i--) { + pass.setBindGroup(0, this.levels[i].bindGroup); + pass.dispatchWorkgroups(this.levels[i].count); + } + } +} diff --git a/ts/gpu/sign.test.ts b/ts/gpu/sign.test.ts new file mode 100644 index 0000000..348e1ef --- /dev/null +++ b/ts/gpu/sign.test.ts @@ -0,0 +1,104 @@ +import { hasWebGPU, requestDevice, readBackU32 } from './device.ts'; +import { mulberry32 } from './test_util.ts'; +import { refSign, refRasterize, parseBinarySTL } from './reference.ts'; +import { Binner } from './mesh_to_grid.ts'; +import { Signer } from './sign.ts'; + +const gpu = await hasWebGPU(); + +let stl: Uint8Array | null = null; +try { + stl = Deno.readFileSync(new URL('../../data/bases/base_32mm.stl', import.meta.url)); +} catch { + // Sample data not present, so the STL test skips. +} + +async function checkSigns( + binner: Binner, + signer: Signer, + points: Float32Array, + triangles: Uint32Array, + voxelSize: number, + halfWidth: number, + label: string +): Promise<{ inside: number; total: number }> { + const bin = await binner.bin(points, triangles, { voxelSize, halfWidth }); + const sign = await signer.sign(bin); + const got = await readBackU32(binner.device, sign.inside, bin.leafCount * 16); + const leafKeys = await readBackU32(binner.device, bin.leafKeys, bin.leafCount); + const expected = refSign(points, triangles, voxelSize, leafKeys, bin.leafMin); + // The GPU computes crossings in f32 and the reference in f64. Parity may + // flip only for voxels whose column crossing sits within f32 noise of + // their center plane, and those lie on the surface. + const ref = refRasterize(points, triangles, voxelSize, halfWidth, bin.leafMin); + const ON_SURFACE = 1e-2; // voxel units + let inside = 0; + let flips = 0; + let firstBad = ''; + for (let w = 0; w < got.length; w++) { + inside += popcount(expected[w]); + let diff = (got[w] ^ expected[w]) >>> 0; + while (diff !== 0) { + const bit = 31 - Math.clz32(diff); + diff = (diff & ~(1 << bit)) >>> 0; + flips++; + const n = ((w % 16) * 32) + bit; + const d2 = ref.values[(w >> 4) * 512 + n]; + if (Math.sqrt(d2) > ON_SURFACE && !firstBad) { + firstBad = `leaf ${w >> 4} voxel ${n}: sign flip at distance ${Math.sqrt(d2)}`; + } + } + } + if (firstBad || flips > got.length * 32 * 0.001) { + throw new Error(`${label}: ${flips} sign flips; ${firstBad || 'all on-surface but too many'}`); + } + return { inside, total: got.length * 32 }; +} + +function popcount(v: number): number { + v = v - ((v >>> 1) & 0x55555555); + v = (v & 0x33333333) + ((v >>> 2) & 0x33333333); + return (((v + (v >>> 4)) & 0x0f0f0f0f) * 0x01010101) >>> 24; +} + +Deno.test({ name: 'parity signing matches f64 reference', ignore: !gpu }, async () => { + const device = await requestDevice(); + const binner = new Binner(device); + const signer = new Signer(device); + + // A closed cube with interior voxels. + // deno-fmt-ignore + const cubePts = new Float32Array([ + -6, -6, -6, 6, -6, -6, 6, 6, -6, -6, 6, -6, // corners of the low z face + -6, -6, 6, 6, -6, 6, 6, 6, 6, -6, 6, 6, // corners of the high z face + ]); + const quads = [ + [0, 1, 2, 3], // low z face + [4, 6, 5, 7], // high z face, winding does not matter for parity + [0, 4, 1, 5], + [1, 5, 2, 6], + [2, 6, 3, 7], + [3, 7, 0, 4], + ]; + const cubeTris: number[] = []; + for (const [a, b, c, d] of quads) cubeTris.push(a, b, c, b, c, d); + const cube = await checkSigns(binner, signer, cubePts, new Uint32Array(cubeTris), 1, 3, 'cube'); + if (cube.inside === 0) throw new Error('cube has no inside voxels'); + + // Random soup parity is arbitrary but deterministic and must match the + // reference. + const rand = mulberry32(5); + const triCount = 100; + const points = new Float32Array(triCount * 9); + for (let i = 0; i < points.length; i++) points[i] = (rand() - 0.5) * 40; + await checkSigns(binner, signer, points, new Uint32Array([...Array(triCount * 3).keys()]), 0.5, 3, 'soup'); +}); + +Deno.test({ name: 'STL parity signing matches f64 reference', ignore: !gpu || !stl }, async () => { + const device = await requestDevice(); + const binner = new Binner(device); + const signer = new Signer(device); + const { points, triangles } = parseBinarySTL(stl!); + const { inside, total } = await checkSigns(binner, signer, points, triangles, 0.25, 3, 'stl'); + if (inside < 1000) throw new Error(`implausibly few inside voxels: ${inside}/${total}`); +}); diff --git a/ts/gpu/sign.ts b/ts/gpu/sign.ts new file mode 100644 index 0000000..fefa619 --- /dev/null +++ b/ts/gpu/sign.ts @@ -0,0 +1,128 @@ +// Host side of wgsl/sign.wgsl. Computes inside parity for every voxel of +// the binned leaves as one mask per leaf in slab bit order. + +import signWgsl from 'picovdb/wgsl/sign.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { Sorter } from './radix_sort.ts'; +import { dispatch2D, readBackTotals } from './device.ts'; +import type { BinResult } from './mesh_to_grid.ts'; + +const WG_SIZE = 256; + +export interface SignResult { + /** One 512 bit mask per leaf. A set bit marks an inside voxel. */ + inside: GPUBuffer; + crossingCount: number; +} + +export class Signer { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly sorter: Sorter; + readonly layout: GPUBindGroupLayout; + private readonly pipelines: Record = {}; + + constructor(device: GPUDevice, scanner = new Scanner(device), sorter = new Sorter(device, scanner)) { + this.device = device; + this.scanner = scanner; + this.sorter = sorter; + const entry = (binding: number, type: GPUBufferBindingType): GPUBindGroupLayoutEntry => ({ + binding, + visibility: GPUShaderStage.COMPUTE, + buffer: { type }, + }); + this.layout = device.createBindGroupLayout({ + entries: [ + entry(0, 'uniform'), + entry(1, 'read-only-storage'), + entry(2, 'read-only-storage'), + entry(3, 'storage'), + entry(4, 'storage'), + entry(5, 'storage'), + entry(6, 'read-only-storage'), + entry(7, 'storage'), + ], + }); + const module = device.createShaderModule({ code: signWgsl }); + const layout = device.createPipelineLayout({ bindGroupLayouts: [this.layout] }); + for (const entryPoint of ['count_crossings', 'emit_crossings', 'sign_leaves']) { + this.pipelines[entryPoint] = device.createComputePipeline({ layout, compute: { module, entryPoint } }); + } + } + + async sign(bin: BinResult): Promise { + const device = this.device; + const triangleCount = bin.triangles.size / 12; // 3 u32 indices per triangle + // Column grid covering all candidate leaves. Parity per column does + // not depend on the grid bounds. + const minX = bin.leafMin[0] * 8; + const minY = bin.leafMin[1] * 8; + const nx = (bin.leafMax[0] - bin.leafMin[0] + 1) * 8; + const ny = (bin.leafMax[1] - bin.leafMin[1] + 1) * 8; + + const params = device.createBuffer({ size: 48, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([triangleCount, 0])); + device.queue.writeBuffer(params, 8, new Int32Array([minX, minY])); + device.queue.writeBuffer(params, 16, new Uint32Array([nx, ny, bin.leafCount, 0])); + device.queue.writeBuffer(params, 32, new Int32Array(bin.leafMin)); + + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const counts = device.createBuffer({ size: (triangleCount + 1) * 4, usage: storage }); + // Distinct placeholders: writable bindings may not alias one buffer. + const placeholders = [0, 1].map(() => device.createBuffer({ size: 4, usage: GPUBufferUsage.STORAGE })); + const inside = device.createBuffer({ size: bin.leafCount * 16 * 4, usage: storage }); + + const bindGroup = (crossCols: GPUBuffer, crossZ: GPUBuffer) => + device.createBindGroup({ + layout: this.layout, + entries: [ + { binding: 0, resource: { buffer: params } }, + { binding: 1, resource: { buffer: bin.pointsIndex } }, + { binding: 2, resource: { buffer: bin.triangles } }, + { binding: 3, resource: { buffer: counts } }, + { binding: 4, resource: { buffer: crossCols } }, + { binding: 5, resource: { buffer: crossZ } }, + { binding: 6, resource: { buffer: bin.leafKeys } }, + { binding: 7, resource: { buffer: inside } }, + ], + }); + + // Count crossings per triangle, scan, and read the total. + const countScan = this.scanner.plan(counts, triangleCount + 1); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, bindGroup(placeholders[0], placeholders[1])); + pass.setPipeline(this.pipelines['count_crossings']); + dispatch2D(pass, Math.ceil((triangleCount + 1) / WG_SIZE)); + countScan.encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [crossingCount] = await readBackTotals(device, [{ buffer: counts, index: triangleCount }]); + + // Emit, sort by column then height, and walk each leaf column. + device.queue.writeBuffer(params, 4, new Uint32Array([crossingCount])); + const crossCols = device.createBuffer({ size: Math.max(crossingCount, 1) * 4, usage: storage }); + const crossZ = device.createBuffer({ size: Math.max(crossingCount, 1) * 4, usage: storage }); + const group = bindGroup(crossCols, crossZ); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + pass.setBindGroup(0, group); + pass.setPipeline(this.pipelines['emit_crossings']); + dispatch2D(pass, Math.ceil(triangleCount / WG_SIZE)); + if (crossingCount > 0) { + // Two stable sorts give column then height order. + this.sorter.plan(crossZ, crossCols, crossingCount).encode(pass); + this.sorter.plan(crossCols, crossZ, crossingCount).encode(pass); + } + pass.setBindGroup(0, group); + pass.setPipeline(this.pipelines['sign_leaves']); + dispatch2D(pass, bin.leafCount); + pass.end(); + device.queue.submit([encoder.finish()]); + } + return { inside, crossingCount }; + } +} diff --git a/ts/gpu/stamp.test.ts b/ts/gpu/stamp.test.ts new file mode 100644 index 0000000..db8885b --- /dev/null +++ b/ts/gpu/stamp.test.ts @@ -0,0 +1,82 @@ +import { hasWebGPU, requestDevice } from './device.ts'; +import { checkAnalytic, emptyGrid } from './test_util.ts'; +import { Stamper, box as boxShape, capsule as capsuleShape, cylinder as cylinderShape, sphere as sphereShape } from './stamp.ts'; +import { Emitter } from './emit.ts'; + +const gpu = await hasWebGPU(); + +Deno.test({ name: 'brush stamps sculpt from an empty grid', ignore: !gpu }, async () => { + const device = await requestDevice(); + const stamper = new Stamper(device); + const emitter = new Emitter(device); + const halfWidth = 3; + const center: [number, number, number] = [100.3, 97.2, 88.9]; + + const sphere = (r: number) => (p: [number, number, number]) => + Math.hypot(p[0] - center[0], p[1] - center[1], p[2] - center[2]) - r; + + // Add a sphere to empty space. + const added = await stamper.stamp(emptyGrid(device), { shape: sphereShape(center, 20), mode: 'add', halfWidth }); + const a = await checkAnalytic(device, added, sphere(20), halfWidth, 'add'); + if (a.band === 0) throw new Error('no band voxels after add'); + + // Carve a concentric hole so a shell remains. + const carved = await stamper.stamp(added, { shape: sphereShape(center, 12), mode: 'carve', halfWidth }); + const shell = (p: [number, number, number]) => Math.max(sphere(20)(p), -sphere(12)(p)); + const c = await checkAnalytic(device, carved, shell, halfWidth, 'carve'); + if (c.band <= a.band) throw new Error(`carving should grow the band: ${c.band} <= ${a.band}`); + + // The sculpted grid emits into a tree. + const tree = await emitter.reEmit(carved, { halfWidth }); + if (tree.surfaceVoxels === 0) throw new Error('no surface voxels in sculpted tree'); + const spanOk = tree.indexBoundsMax.every((v, axis) => v - tree.indexBoundsMin[axis] > 40); + if (!spanOk) throw new Error(`implausible bounds: ${tree.indexBoundsMin} .. ${tree.indexBoundsMax}`); + console.log( + ` sculpt: add band=${a.band}, shell band=${c.band}; tree ${tree.leafCount} leaves, ` + + `${tree.activeVoxels} active, ${tree.surfaceVoxels} surface` + ); +}); + +Deno.test({ name: 'box, capsule, and cylinder stamps match their SDFs', ignore: !gpu }, async () => { + const device = await requestDevice(); + const stamper = new Stamper(device); + const halfWidth = 3; + const v = (a: number[], b: number[], s = 1) => a.map((x, i) => (x - b[i]) * s); + const len = (a: number[]) => Math.hypot(a[0], a[1], a[2]); + const dot = (a: number[], b: number[]) => a[0] * b[0] + a[1] * b[1] + a[2] * b[2]; + + const center: [number, number, number] = [60.5, 70.25, 80]; + const half: [number, number, number] = [15, 9, 6]; + const box = (p: number[]) => { + const q = v(p, center).map((x, i) => Math.abs(x) - half[i]); + return len(q.map((x) => Math.max(x, 0))) + Math.min(Math.max(q[0], q[1], q[2]), 0) - 1.5; + }; + const boxed = await stamper.stamp(emptyGrid(device), { shape: boxShape(center, half, 1.5), mode: 'add', halfWidth }); + await checkAnalytic(device, boxed, box, halfWidth, 'box'); + + const a: [number, number, number] = [40, 40, 40]; + const b: [number, number, number] = [90.7, 63.1, 55]; + const capsule = (p: number[]) => { + const pa = v(p, a); + const ba = v(b, a); + const h = Math.max(0, Math.min(1, dot(pa, ba) / dot(ba, ba))); + return len(pa.map((x, i) => x - ba[i] * h)) - 7; + }; + const capsuled = await stamper.stamp(emptyGrid(device), { shape: capsuleShape(a, b, 7), mode: 'add', halfWidth }); + await checkAnalytic(device, capsuled, capsule, halfWidth, 'capsule'); + + const cylinder = (p: number[]) => { + const pa = v(p, a); + const ba = v(b, a); + const baba = dot(ba, ba); + const paba = dot(pa, ba); + const x = len(pa.map((c, i) => c * baba - ba[i] * paba)) - 7 * baba; + const y = Math.abs(paba - baba * 0.5) - baba * 0.5; + const x2 = x * x; + const y2 = y * y * baba; + const d = Math.max(x, y) < 0 ? -Math.min(x2, y2) : (x > 0 ? x2 : 0) + (y > 0 ? y2 : 0); + return Math.sign(d) * Math.sqrt(Math.abs(d)) / baba; + }; + const cylindered = await stamper.stamp(emptyGrid(device), { shape: cylinderShape(a, b, 7), mode: 'add', halfWidth }); + await checkAnalytic(device, cylindered, cylinder, halfWidth, 'cylinder'); +}); diff --git a/ts/gpu/stamp.ts b/ts/gpu/stamp.ts new file mode 100644 index 0000000..0c6a720 --- /dev/null +++ b/ts/gpu/stamp.ts @@ -0,0 +1,234 @@ +// Host side of wgsl/stamp.wgsl: stamps shapes into op layer grids, +// adding or carving material. A shape is a WGSL distance function, from +// the built-in library or the caller's. + +import stampWgsl from 'picovdb/wgsl/stamp.wgsl' with { type: 'text' }; +import { Scanner } from './scan.ts'; +import { Sorter } from './radix_sort.ts'; +import { dispatch2D, readBackTotals } from './device.ts'; +import { GridWriter, preludeWgsl, readerWgsl, type OpGrid } from './opgrid.ts'; + +const WG_SIZE = 256; + +export type Vec3 = [number, number, number]; + +/** + * A shape: a WGSL signed distance function by name. The function has the + * form `fn name(p: vec3f) -> f32`, takes absolute voxel coordinates, and + * returns the signed distance in voxels. Its arguments arrive in the uniform + * `args`, an `array`. + */ +export interface Shape { + fn: string; + /** Up to 32 numbers, four per slot: args[0..3] is args[0] in WGSL, and so on. */ + args?: number[]; + /** Voxel bounds of the zero level set. Needed to add; a carve clips to the grid. */ + bounds?: { min: Vec3; max: Vec3 }; +} + +/** The built-in shapes. Their helpers below fill args to match. */ +export const shapesWgsl = /* wgsl */ ` +fn picovdb_sphere(p: vec3) -> f32 { + return length(p - args[0].xyz) - args[0].w; +} + +fn picovdb_box(p: vec3) -> f32 { + let q = abs(p - args[0].xyz) - args[1].xyz; + return length(max(q, vec3(0.0))) + min(max(q.x, max(q.y, q.z)), 0.0) - args[1].w; +} + +fn picovdb_capsule(p: vec3) -> f32 { + let ba = args[1].xyz - args[0].xyz; + let pa = p - args[0].xyz; + let h = clamp(dot(pa, ba) / dot(ba, ba), 0.0, 1.0); + return length(pa - (ba * h)) - args[0].w; +} + +fn picovdb_cylinder(p: vec3) -> f32 { + let ba = args[1].xyz - args[0].xyz; + let pa = p - args[0].xyz; + let baba = dot(ba, ba); + let paba = dot(pa, ba); + let x = length((pa * baba) - (ba * paba)) - (args[0].w * baba); + let y = abs(paba - (baba * 0.5)) - (baba * 0.5); + let x2 = x * x; + let y2 = y * y * baba; + var d: f32; + if (max(x, y) < 0.0) { + d = -min(x2, y2); + } else { + d = select(0.0, x2, x > 0.0) + select(0.0, y2, y > 0.0); + } + return sign(d) * sqrt(abs(d)) / baba; +} +`; + +function aabb(points: Vec3[], radius: number): { min: Vec3; max: Vec3 } { + const min = [Infinity, Infinity, Infinity] as Vec3; + const max = [-Infinity, -Infinity, -Infinity] as Vec3; + for (const p of points) { + for (let a = 0; a < 3; a++) { + min[a] = Math.min(min[a], p[a] - radius); + max[a] = Math.max(max[a], p[a] + radius); + } + } + return { min, max }; +} + +export function sphere(center: Vec3, radius: number): Shape { + return { fn: 'picovdb_sphere', args: [...center, radius], bounds: aabb([center], radius) }; +} + +/** Half extents per axis, edges rounded by radius. */ +export function box(center: Vec3, half: Vec3, radius = 0): Shape { + const corner = center.map((c, a) => c + half[a]) as Vec3; + const opposite = center.map((c, a) => c - half[a]) as Vec3; + return { fn: 'picovdb_box', args: [...center, 0, ...half, radius], bounds: aabb([corner, opposite], radius) }; +} + +export function capsule(a: Vec3, b: Vec3, radius: number): Shape { + return { fn: 'picovdb_capsule', args: [...a, radius, ...b, 0], bounds: aabb([a, b], radius) }; +} + +export function cylinder(a: Vec3, b: Vec3, radius: number): Shape { + return { fn: 'picovdb_cylinder', args: [...a, radius, ...b, 0], bounds: aabb([a, b], radius) }; +} + +export interface StampOptions { + shape: Shape; + mode: 'add' | 'carve'; + halfWidth: number; +} + +export class Stamper { + readonly device: GPUDevice; + readonly scanner: Scanner; + readonly sorter: Sorter; + /** The shape library: the built-ins plus the caller's functions. */ + readonly shapes: string; + private readonly pipelines = new Map>(); + + constructor(device: GPUDevice, scanner: Scanner = new Scanner(device), sorter: Sorter = new Sorter(device, scanner), shapes = '') { + this.device = device; + this.scanner = scanner; + this.sorter = sorter; + this.shapes = shapesWgsl + shapes; + } + + /** The pipelines for one shape function, compiled on first use. */ + private async pipelinesFor(fn: string): Promise> { + const cached = this.pipelines.get(fn); + if (cached) return cached; + if (!/^[A-Za-z_][A-Za-z0-9_]*$/.test(fn)) throw new Error(`shape function name ${JSON.stringify(fn)} is not an identifier`); + const code = preludeWgsl + readerWgsl('old', 'params.old_count') + this.shapes + `fn sdf(p: vec3) -> f32 { return ${fn}(p); }\n` + stampWgsl; + const module = this.device.createShaderModule({ code }); + const info = await module.getCompilationInfo(); + const errors = info.messages.filter((m) => m.type === 'error'); + if (errors.length > 0) { + throw new Error(`shapes do not compile for ${fn}:\n` + errors.map((m) => ` line ${m.lineNum}: ${m.message}`).join('\n')); + } + const pipelines: Record = {}; + for (const entryPoint of ['generate_candidates', 'mark_unique', 'compact_unique', 'mark', 'apply', 'opgrid_compact']) { + pipelines[entryPoint] = this.device.createComputePipeline({ layout: 'auto', compute: { module, entryPoint } }); + } + this.pipelines.set(fn, pipelines); + return pipelines; + } + + async stamp(grid: OpGrid, opts: StampOptions): Promise { + const device = this.device; + const hw = opts.halfWidth; + const shape = opts.shape; + const pipelines = await this.pipelinesFor(shape.fn); + // The box of leaves the shape's band can touch. A carve only changes + // existing material, so its box clips to the grid's leaves. An add + // that reaches the key range throws, so nothing truncates silently. + let lo = [0, 0, 0]; + let hi = grid.leafMax.map((v, a) => v - grid.leafMin[a]); + if (shape.bounds) { + lo = shape.bounds.min.map((c, a) => Math.floor((c - hw) / 8) - grid.leafMin[a]); + hi = shape.bounds.max.map((c, a) => Math.floor((c + hw) / 8) - grid.leafMin[a]); + } + if (opts.mode === 'carve') { + lo = lo.map((v) => Math.max(v, 0)); + hi = hi.map((v, a) => Math.min(v, grid.leafMax[a] - grid.leafMin[a])); + } else if (!shape.bounds) { + throw new Error(`shape ${shape.fn} needs bounds to add material`); + } else if (lo.some((v) => v < 0) || hi.some((v) => v > 1023)) { + throw new Error('stamp reaches the leaf key space boundary'); + } + const dims = lo.map((v, a) => Math.max(hi[a] - v + 1, 0)); + const boxVol = dims[0] * dims[1] * dims[2]; + const concatCount = grid.leafCount + boxVol; + + const params = device.createBuffer({ size: 64, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(params, 0, new Uint32Array([grid.leafCount, concatCount, 0, opts.mode === 'carve' ? 1 : 0])); + device.queue.writeBuffer(params, 16, new Float32Array([grid.leafMin[0] * 8, grid.leafMin[1] * 8, grid.leafMin[2] * 8, hw])); + device.queue.writeBuffer(params, 32, new Int32Array(lo)); + device.queue.writeBuffer(params, 48, new Int32Array(dims)); + const u = device.createBuffer({ size: 128, usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST }); + if ((shape.args?.length ?? 0) > 32) throw new Error(`shape ${shape.fn} has more than 32 arguments`); + const packed = new Float32Array(32); + packed.set(shape.args ?? []); + device.queue.writeBuffer(u, 0, packed); + + const storage = GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC; + const concatKeys = device.createBuffer({ size: concatCount * 4, usage: storage }); + const sortVals = device.createBuffer({ size: concatCount * 4, usage: GPUBufferUsage.STORAGE }); + const flags = device.createBuffer({ size: (concatCount + 1) * 4, usage: storage }); + + const run = (pass: GPUComputePassEncoder, name: string, threads: number, buffers: Record) => { + pass.setPipeline(pipelines[name]); + pass.setBindGroup( + 0, + device.createBindGroup({ + layout: pipelines[name].getBindGroupLayout(0), + entries: [ + { binding: 0, resource: { buffer: params } }, + ...Object.entries(buffers).map(([binding, buffer]) => ({ binding: Number(binding), resource: { buffer } })), + ], + }) + ); + dispatch2D(pass, Math.ceil(threads / WG_SIZE)); + }; + + { + const encoder = device.createCommandEncoder(); + if (grid.leafCount > 0) { + encoder.copyBufferToBuffer(grid.leafKeys, 0, concatKeys, 0, grid.leafCount * 4); + } + const pass = encoder.beginComputePass(); + run(pass, 'generate_candidates', boxVol, { 3: concatKeys, 7: u }); + this.sorter.plan(concatKeys, sortVals, concatCount).encode(pass); + run(pass, 'mark_unique', concatCount + 1, { 3: concatKeys, 4: flags }); + this.scanner.plan(flags, concatCount + 1).encode(pass); + pass.end(); + device.queue.submit([encoder.finish()]); + } + const [newCount] = await readBackTotals(device, [{ buffer: flags, index: concatCount }]); + + device.queue.writeBuffer(params, 8, new Uint32Array([newCount])); + const newKeys = device.createBuffer({ size: newCount * 4, usage: storage }); + const old = { 1: grid.leafKeys, 2: grid.leaves, 6: grid.data }; + const writer = new GridWriter(device, newCount); + { + const encoder = device.createCommandEncoder(); + const pass = encoder.beginComputePass(); + run(pass, 'compact_unique', concatCount, { 3: concatKeys, 4: flags, 5: newKeys }); + run(pass, 'mark', newCount + 1, { ...old, 5: newKeys, 7: u, ...writer.markBindings }); + pass.end(); + device.queue.submit([encoder.finish()]); + } + concatKeys.destroy(); + sortVals.destroy(); + flags.destroy(); + const leafMax = grid.leafMax.map((v, a) => (opts.mode === 'add' ? Math.max(v, hi[a] + grid.leafMin[a]) : v)) as Vec3; + const out = await writer.finish(this.scanner, pipelines['opgrid_compact'], newKeys, hw, { leafMin: grid.leafMin, leafMax }, (pass, bindings, count) => { + run(pass, 'apply', count, { ...old, 7: u, ...bindings }); + }); + newKeys.destroy(); + params.destroy(); + u.destroy(); + return out; + } +} diff --git a/ts/gpu/test_util.ts b/ts/gpu/test_util.ts new file mode 100644 index 0000000..480bfba --- /dev/null +++ b/ts/gpu/test_util.ts @@ -0,0 +1,209 @@ +// Shared helpers for the GPU tests. + +import assert from 'node:assert/strict'; +import { readBackU32 } from './device.ts'; +import { LOWER_U32, UPPER_U32, type EmitResult } from './emit.ts'; +import { LEAF_U32, emptyOpGrid, type OpGrid } from './opgrid.ts'; +import { createU32Buffer } from './device.ts'; +import type { PicoVDBFile } from '../picovdb.ts'; + +/** Fast equality for large typed arrays. assert.deepEqual takes minutes at this size. */ +export function assertU32ArrayEqual(got: Uint32Array, expected: Uint32Array, label: string): void { + assert.equal(got.length, expected.length, `${label}: length`); + for (let i = 0; i < got.length; i++) { + if (got[i] !== expected[i]) { + assert.fail(`${label}: first mismatch at [${i}]: got ${got[i]}, expected ${expected[i]}`); + } + } +} + +/** Deterministic PRNG so failures reproduce. */ +export function mulberry32(seed: number): () => number { + let a = seed >>> 0; + return () => { + a = (a + 0x6d2b79f5) >>> 0; + let t = a; + t = Math.imul(t ^ (t >>> 15), t | 1); + t ^= t + Math.imul(t ^ (t >>> 7), t | 61); + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + }; +} + +/** + * Compares an emitted tree against a CPU converted file. Node buffers must + * match byte for byte and values must match in sign and within tolerance. + * Returns the largest value difference. + */ +export async function compareTreeToCpu(device: GPUDevice, tree: EmitResult, cpu: PicoVDBFile): Promise { + const h = cpu.header; + if (tree.leafCount !== h.leafCount || tree.lowerCount !== h.lowerCount || tree.upperCount !== h.upperCount) { + throw new Error( + `counts: gpu ${tree.leafCount}/${tree.lowerCount}/${tree.upperCount} != cpu ${h.leafCount}/${h.lowerCount}/${h.upperCount}` + ); + } + const grid = cpu.getGrid(0); + if (tree.dataElemCount !== grid.dataElemCount) { + throw new Error(`dataElemCount: gpu ${tree.dataElemCount} != cpu ${grid.dataElemCount}`); + } + for (let a = 0; a < 3; a++) { + if (tree.indexBoundsMin[a] !== grid.indexBoundsMin[a] || tree.indexBoundsMax[a] !== grid.indexBoundsMax[a]) { + throw new Error(`index bounds mismatch on axis ${a}`); + } + } + const cpuU32 = (bytes: Uint8Array) => new Uint32Array(bytes.buffer, bytes.byteOffset, bytes.byteLength / 4); + assertU32ArrayEqual( + await readBackU32(device, tree.roots, tree.upperCount * 2), + cpuU32(cpu.rootsBuffer).slice(0, tree.upperCount * 2), + 'roots' + ); + assertU32ArrayEqual(await readBackU32(device, tree.uppers, tree.upperCount * UPPER_U32), cpuU32(cpu.uppersBuffer), 'uppers'); + assertU32ArrayEqual(await readBackU32(device, tree.lowers, tree.lowerCount * LOWER_U32), cpuU32(cpu.lowersBuffer), 'lowers'); + assertU32ArrayEqual(await readBackU32(device, tree.leaves, tree.leafCount * LEAF_U32), cpuU32(cpu.leavesBuffer), 'leaves'); + + const gpuData = new Float32Array((await readBackU32(device, tree.data, tree.dataElemCount)).buffer); + const cpuData = new Float32Array(cpu.dataBuffer.buffer, cpu.dataBuffer.byteOffset, tree.dataElemCount); + let maxAbs = 0; + for (let i = 0; i < tree.dataElemCount; i++) { + if ((gpuData[i] < 0) !== (cpuData[i] < 0)) { + throw new Error(`value sign mismatch at ${i}: ${gpuData[i]} vs ${cpuData[i]}`); + } + maxAbs = Math.max(maxAbs, Math.abs(gpuData[i] - cpuData[i])); + } + if (maxAbs > 1e-3) throw new Error(`value divergence ${maxAbs}`); + return maxAbs; +} + +export function emptyGrid(device: GPUDevice): OpGrid { + return emptyOpGrid(device, [0, 0, 0], [1023, 1023, 1023]); +} + +export interface GridView { + keys: Uint32Array; + leaves: Uint32Array; + data: Float32Array; + /** Value of voxel n of leaf i, mirroring the WGSL reader. */ + value(i: number, n: number): number; + band(i: number, n: number): boolean; +} + +/** Reads an op layer grid back with a voxel accessor. */ +export async function readGrid(device: GPUDevice, grid: OpGrid, halfWidth: number): Promise { + const keys = await readBackU32(device, grid.leafKeys, grid.leafCount); + const leaves = await readBackU32(device, grid.leaves, grid.leafCount * LEAF_U32); + const data = new Float32Array((await readBackU32(device, grid.data, 2 + grid.activeVoxels)).buffer); + const popcount = (v: number) => { + v = v - ((v >>> 1) & 0x55555555); + v = (v & 0x33333333) + ((v >>> 2) & 0x33333333); + return (((v + (v >>> 4)) & 0x0f0f0f0f) * 0x01010101) >>> 24; + }; + const band = (i: number, n: number) => ((leaves[i * LEAF_U32 + 5 + (n >> 5) * 3] >>> (n & 31)) & 1) === 1; + const value = (i: number, n: number) => { + const e = i * LEAF_U32 + 4 + (n >> 5) * 3; + const bit = n & 31; + if (band(i, n)) { + const below = bit === 0 ? 0 : (leaves[e + 1] & ((1 << bit) - 1)) >>> 0; + return data[leaves[i * LEAF_U32 + 2] + (leaves[e + 2] & 0xffff) + popcount(below)]; + } + return ((leaves[e] >>> bit) & 1) === 1 ? -halfWidth : halfWidth; + }; + return { keys, leaves, data, value, band }; +} + +/** Dense view of a grid: a 512 value slab and 16 band mask words per leaf. */ +export async function unpackGrid(device: GPUDevice, grid: OpGrid, halfWidth: number): Promise<{ keys: Uint32Array; masks: Uint32Array; values: Float32Array }> { + const view = await readGrid(device, grid, halfWidth); + const values = new Float32Array(grid.leafCount * 512); + const masks = new Uint32Array(grid.leafCount * 16); + for (let i = 0; i < grid.leafCount; i++) { + for (let n = 0; n < 512; n++) { + values[i * 512 + n] = view.value(i, n); + if (view.band(i, n)) masks[i * 16 + (n >> 5)] |= 1 << (n & 31); + } + } + return { keys: view.keys, masks, values }; +} + +/** Uploads a dense grid of sorted keys and 512 value slabs as an op layer grid, dropping leaves without band voxels. */ +export function packGrid(device: GPUDevice, keys: Uint32Array, values: Float32Array, halfWidth: number): OpGrid { + const outKeys: number[] = []; + const leaves: number[] = []; + const data: number[] = [halfWidth, -halfWidth]; + for (let i = 0; i < keys.length; i++) { + const record = [0, 0, data.length, 0]; + const slab: number[] = []; + for (let w = 0; w < 16; w++) { + let band = 0; + let inside = 0; + for (let b = 0; b < 32; b++) { + const v = values[i * 512 + w * 32 + b]; + if (Math.abs(v) < halfWidth) { + band |= 1 << b; + slab.push(v); + } + if (v < 0) inside |= 1 << b; + } + record.push((inside & ~band) >>> 0, band >>> 0, 0); + } + if (slab.length === 0) continue; + // Recompute the per word prefixes from the band words. + let prefix = 0; + for (let w = 0; w < 16; w++) { + record[4 + w * 3 + 2] = prefix; + let m = record[4 + w * 3 + 1]; + let c = 0; + while (m) { m &= m - 1; c++; } + prefix += c; + } + outKeys.push(keys[i]); + leaves.push(...record); + data.push(...slab); + } + const f32 = new Float32Array(data); + return { + leafKeys: createU32Buffer(device, new Uint32Array(outKeys)), + leaves: createU32Buffer(device, new Uint32Array(leaves)), + data: createU32Buffer(device, new Uint32Array(f32.buffer)), + leafCount: outKeys.length, + activeVoxels: data.length - 2, + leafMin: [0, 0, 0], + leafMax: [1023, 1023, 1023], + }; +} + +// Every stored voxel must match the expected SDF within tolerance, and +// its band bit must match away from the band edge. With exactBelow, +// values only need to match where |expected| is below it. Elsewhere the +// sign must agree. +export async function checkAnalytic( + device: GPUDevice, + grid: OpGrid, + expected: (p: [number, number, number]) => number, + halfWidth: number, + label: string, + exactBelow = Infinity, + tolerance = 1e-3 +): Promise<{ band: number }> { + const view = await readGrid(device, grid, halfWidth); + const TOL = tolerance; + let band = 0; + for (let i = 0; i < grid.leafCount; i++) { + const ox = (((view.keys[i] >>> 20) & 0x3ff) + grid.leafMin[0]) * 8; + const oy = (((view.keys[i] >>> 10) & 0x3ff) + grid.leafMin[1]) * 8; + const oz = ((view.keys[i] & 0x3ff) + grid.leafMin[2]) * 8; + for (let n = 0; n < 512; n++) { + const p: [number, number, number] = [ox + (n >> 6), oy + ((n >> 3) & 7), oz + (n & 7)]; + const e = Math.max(-halfWidth, Math.min(halfWidth, expected(p))); + const v = view.value(i, n); + const exact = Math.abs(e) < exactBelow; + if (exact ? Math.abs(v - e) > TOL : (v < 0) !== (e < 0) && Math.abs(e) > TOL) { + throw new Error(`${label}: value at ${p}: got ${v}, expected ${e}`); + } + const bit = view.band(i, n) ? 1 : 0; + if (bit) band++; + if (exact && Math.abs(Math.abs(e) - halfWidth) > 2 * TOL && bit !== (Math.abs(e) < halfWidth ? 1 : 0)) { + throw new Error(`${label}: band bit at ${p}: got ${bit}, |v|=${Math.abs(v)}`); + } + } + } + return { band }; +} diff --git a/ts/model.test.ts b/ts/model.test.ts new file mode 100644 index 0000000..8049a1a --- /dev/null +++ b/ts/model.test.ts @@ -0,0 +1,168 @@ +import { hasWebGPU, requestDevice } from './gpu/device.ts'; +import { checkAnalytic, compareTreeToCpu } from './gpu/test_util.ts'; +import { Space, box as boxShape, sphere as sphereShape } from './model.ts'; +import { PICOVDB_LEAF_SIZE, PICOVDB_LOWER_SIZE, PICOVDB_MAGIC, PICOVDB_UPPER_SIZE, PicoVDBFile } from './picovdb.ts'; + +const gpu = await hasWebGPU(); + +let bunny: Uint8Array | null = null; +try { + bunny = Deno.readFileSync(new URL('../data/bunny.pvdb', import.meta.url)); +} catch { + // skip +} + +type P = [number, number, number]; +const sphere = (c: P, r: number) => (p: P) => Math.hypot(p[0] - c[0], p[1] - c[1], p[2] - c[2]) - r; +const box = (c: P, h: P) => (p: P) => { + const q = p.map((x, i) => Math.abs(x - c[i]) - h[i]); + return Math.hypot(...q.map((x) => Math.max(x, 0))) + Math.min(Math.max(q[0], q[1], q[2]), 0); +}; + +Deno.test({ name: 'booleans between distant solids rebase and match the SDFs', ignore: !gpu }, async () => { + const device = await requestDevice(); + const space = new Space(device); + const hw = space.halfWidth; + + // Two solids in unrelated key regions, far from any shared origin. + const cs: P = [100.3, 97.2, 88.9]; + const cb: P = [3000, -2000.5, 500]; + using a = await space.solid(sphereShape(cs, 20)); + using b = await space.solid(boxShape(cb, [12, 7, 9])); + using u = await a.union(b); + await checkAnalytic(device, u.grid, (p) => Math.min(sphere(cs, 20)(p), box(cb, [12, 7, 9])(p)), hw, 'union'); + if (!u.bounds || u.bounds.min[1] !== Math.floor((cb[1] - 7 - hw) / 8)) throw new Error(`union bounds ${JSON.stringify(u.bounds)}`); + + // Overlapping booleans, with solid, op, and shape operands. + const cc: P = [cs[0] + 15, cs[1], cs[2]]; + using c = await space.solid(sphereShape(cc, 20)); + using sub = await a.subtract(c); + await checkAnalytic(device, sub.grid, (p) => Math.max(sphere(cs, 20)(p), -sphere(cc, 20)(p)), hw, 'subtract'); + using inter = await a.intersect(space.solid(sphereShape(cc, 20))); + await checkAnalytic(device, inter.grid, (p) => Math.max(sphere(cs, 20)(p), sphere(cc, 20)(p)), hw, 'intersect'); + using shell = await a.subtract(sphereShape(cs, 12)); + await checkAnalytic(device, shell.grid, (p) => Math.max(sphere(cs, 20)(p), -sphere(cs, 12)(p)), hw, 'shape subtract'); + // A carve clips to the solid, so a half space far beyond the key range works. + const hs: P = [cs[0] + 4000, 0, 0]; + using half = await a.subtract(boxShape(hs, [4000, 4000, 4000])); + await checkAnalytic(device, half.grid, (p) => Math.max(sphere(cs, 20)(p), -box(hs, [4000, 4000, 4000])(p)), hw, 'half space carve'); + if (!(half.leafCount < a.leafCount)) throw new Error(`half space carve should drop leaves: ${half.leafCount} vs ${a.leafCount}`); + + // A chain frees its intermediates and resolves to the same result. + using chained = await space.solid(sphereShape(cs, 20)).subtract(c).union(sphereShape(cs, 12)).intersect(a); + await checkAnalytic(device, chained.grid, (p) => Math.max(Math.min(Math.max(sphere(cs, 20)(p), -sphere(cc, 20)(p)), sphere(cs, 12)(p)), sphere(cs, 20)(p)), hw, 'chain'); + + // Empty solids. + using e = space.empty(); + using eu = await e.union(a); + await checkAnalytic(device, eu.grid, sphere(cs, 20), hw, 'empty union'); + using ei = await a.intersect(e); + using es = await e.subtract(a); + if (ei.leafCount !== 0 || es.leafCount !== 0) throw new Error('empty intersect or subtract should be empty'); + + // toPvdb round trips the tree byte for byte. + const tree = await u.toTree(); + const file = new PicoVDBFile(await u.toPvdb()); + await compareTreeToCpu(device, tree, file); + using back = space.fromPvdb(file); + await checkAnalytic(device, back.grid, (p) => Math.min(sphere(cs, 20)(p), box(cb, [12, 7, 9])(p)), hw, 'fromPvdb(toPvdb)'); + console.log(` union: ${u.leafCount} leaves -> tree ${tree.leafCount} leaves / ${tree.upperCount} uppers, ${tree.surfaceVoxels} surface, ${file.getSize()} bytes`); +}); + +Deno.test({ name: 'bunny grows, hollows, moves, and emits', ignore: !gpu || !bunny }, async () => { + const device = await requestDevice(); + const space = new Space(device); + using bunny_ = space.fromPvdb(new PicoVDBFile(bunny!.buffer)); + const t0 = performance.now(); + using moved = await bunny_.offset(1.5).subtract(bunny_).translate([1000, 5, -3]); + const tree = await moved.toTree(); + const ms = performance.now() - t0; + const base = await bunny_.toTree(); + // The shell has an inner and an outer surface. + if (tree.surfaceVoxels < 1.5 * base.surfaceVoxels) throw new Error(`hollow shell should double the surface: ${tree.surfaceVoxels} vs ${base.surfaceVoxels}`); + if (tree.indexBoundsMin[0] < base.indexBoundsMin[0] + 990) throw new Error(`translate did not move the bounds: ${tree.indexBoundsMin}`); + console.log(` bunny: ${base.leafCount} leaves -> shell ${tree.leafCount} leaves, ${tree.surfaceVoxels} surface, ${ms.toFixed(0)} ms for offset + subtract + translate + emit`); +}); + +Deno.test({ name: 'custom shape functions stamp by name and report compile errors', ignore: !gpu }, async () => { + const device = await requestDevice(); + const space = new Space(device, { + shapes: /* wgsl */ ` + // A torus in the xz plane: args[0].xyz center, args[1].x ring radius, args[1].y tube radius. + fn torus(p: vec3) -> f32 { + let d = p - args[0].xyz; + let q = vec2(length(d.xz) - args[1].x, d.y); + return length(q) - args[1].y; + } + `, + }); + const hw = space.halfWidth; + const c: P = [100.3, 97.2, 88.9]; + const torus = (p: P) => Math.hypot(Math.hypot(p[0] - c[0], p[2] - c[2]) - 18, p[1] - c[1]) - 6; + const shape = { fn: 'torus', args: [...c, 0, 18, 6], bounds: { min: [c[0] - 24, c[1] - 6, c[2] - 24] as P, max: [c[0] + 24, c[1] + 6, c[2] + 24] as P } }; + using ring = await space.solid(shape); + await checkAnalytic(device, ring.grid, torus, hw, 'torus'); + // As an operand the same function carves. + using ball = await space.solid(sphereShape(c, 20)); + using notched = await ball.subtract(shape); + await checkAnalytic(device, notched.grid, (p) => Math.max(sphere(c, 20)(p), -torus(p)), hw, 'torus carve'); + // Adding needs bounds; carving does not. + let message = ''; + try { await space.solid({ fn: 'torus', args: shape.args }); } catch (e) { message = (e as Error).message; } + if (!message.includes('bounds')) throw new Error(`expected a bounds error, got: ${message}`); + using whole = await ball.subtract({ fn: 'torus', args: shape.args }); + await checkAnalytic(device, whole.grid, (p) => Math.max(sphere(c, 20)(p), -torus(p)), hw, 'unbounded torus carve'); + // An unknown function, and a library that does not compile, report the WGSL error. + message = ''; + try { await ball.subtract({ fn: 'missing' }); } catch (e) { message = (e as Error).message; } + if (!message.includes('missing')) throw new Error(`expected an unknown function error, got: ${message}`); + const bad = new Space(device, { shapes: 'fn broken(p: vec3) -> f32 { return p; }' }); + message = ''; + try { await bad.empty().subtract({ fn: 'broken' }); } catch (e) { message = (e as Error).message; } + if (!message.includes('broken') || !message.includes('line')) throw new Error(`expected a compile error, got: ${message}`); +}); + +/** A two grid file from two single grid files, as a multi-grid writer would lay it out. */ +function stitch(a: PicoVDBFile, b: PicoVDBFile): ArrayBuffer { + const ha = a.header; + const hb = b.header; + const upperCount = ha.upperCount + hb.upperCount; + const rootsPadded = Math.ceil(upperCount / 2) * 2; + const sections = [ + [a.rootsBuffer.subarray(0, ha.upperCount * 8), b.rootsBuffer.subarray(0, hb.upperCount * 8), new Uint8Array((rootsPadded - upperCount) * 8)], + [a.uppersBuffer, b.uppersBuffer], + [a.lowersBuffer, b.lowersBuffer], + [a.leavesBuffer, b.leavesBuffer], + [a.dataBuffer, b.dataBuffer], + ]; + const bodyBytes = sections.flat().reduce((n, s) => n + s.byteLength, 0); + const out = new Uint8Array(32 + 2 * 64 + bodyBytes); + new Uint32Array(out.buffer, 0, 8).set([PICOVDB_MAGIC[0], PICOVDB_MAGIC[1], 0, 2, upperCount, ha.lowerCount + hb.lowerCount, ha.leafCount + hb.leafCount, ha.dataCount + hb.dataCount]); + out.set(a.gridsBuffer, 32); + out.set(b.gridsBuffer, 96); + new Uint32Array(out.buffer, 96 + 4, 4).set([ha.upperCount, ha.lowerCount, ha.leafCount, ha.dataCount]); + let offset = 160; + for (const part of sections.flat()) { + out.set(part, offset); + offset += part.byteLength; + } + if (a.uppersBuffer.byteLength !== ha.upperCount * PICOVDB_UPPER_SIZE || a.lowersBuffer.byteLength !== ha.lowerCount * PICOVDB_LOWER_SIZE || a.leavesBuffer.byteLength !== ha.leafCount * PICOVDB_LEAF_SIZE) throw new Error('unexpected node buffer sizes'); + return out.buffer; +} + +Deno.test({ name: 'fromPvdb loads a chosen grid of a multi-grid file', ignore: !gpu }, async () => { + const device = await requestDevice(); + const space = new Space(device); + const hw = space.halfWidth; + const cs: P = [100.3, 97.2, 88.9]; + const cb: P = [-200, 50, 30]; + using a = await space.solid(sphereShape(cs, 20)); + using b = await space.solid(boxShape(cb, [12, 7, 9])); + const file = new PicoVDBFile(stitch(new PicoVDBFile(await a.toPvdb()), new PicoVDBFile(await b.toPvdb()))); + if (file.header.gridCount !== 2) throw new Error(`stitched file has ${file.header.gridCount} grids`); + using g0 = space.fromPvdb(file, 0); + await checkAnalytic(device, g0.grid, sphere(cs, 20), hw, 'grid 0'); + using g1 = space.fromPvdb(file, 1); + await checkAnalytic(device, g1.grid, box(cb, [12, 7, 9]), hw, 'grid 1'); + if (g0.leafCount !== a.leafCount || g1.leafCount !== b.leafCount) throw new Error(`leaf counts ${g0.leafCount}/${g1.leafCount} vs ${a.leafCount}/${b.leafCount}`); +}); diff --git a/ts/model.ts b/ts/model.ts new file mode 100644 index 0000000..5465bde --- /dev/null +++ b/ts/model.ts @@ -0,0 +1,438 @@ +// A csg.js style modelling API over the GPU grid ops. +// +// A Space is a device plus a narrow band half width. A Solid is an +// immutable GPU resident SDF grid that you own. Destroy it, or declare it +// with `using`. Operations on a Solid return an Op: a recipe that runs +// when awaited and resolves to a new Solid. A chain of ops frees its own +// intermediates, so only the solids you keep need destroying. All +// coordinates are voxels. +// +// import { Space, box, cylinder, sphere } from '@emcfarlane/picovdb/model'; +// +// const space = new Space(device, { halfWidth: 3 }); +// +// // Shapes are values. As operands they stamp straight into the solid. +// // space.solid makes a solid from one. +// using bolt = await space.solid(sphere([0, 0, 0], 20)) +// .union(cylinder([0, -30, 0], [0, 30, 0], 6)) +// .subtract(box([0, 0, 0], [30, 4, 4])); +// +// // Edit a file: grow the bunny by two voxels, hollow it, and move it. +// using bunny = space.fromPvdb(bytes); +// using shell = await bunny.offset(2).subtract(bunny).translate([0, 0, -10]); +// +// // Outputs: file bytes, or the picovdb tree left on the GPU for a renderer. +// await Deno.writeFile('shell.pvdb', new Uint8Array(await shell.toPvdb())); +// const tree = await shell.toTree(); // roots, uppers, lowers, leaves, data +// +// Solids carry their own key origin, so position is unbounded. The extent +// of one solid, or of the two operands of a boolean, is limited to 1024 +// leaves (8192 voxels) per axis. + +import { Binner } from './gpu/mesh_to_grid.ts'; +import { Rasterizer } from './gpu/rasterize.ts'; +import { Signer } from './gpu/sign.ts'; +import { Emitter, type EmitResult, LOWER_U32, UPPER_U32 } from './gpu/emit.ts'; +import { LEAF_U32, emptyOpGrid, type OpGrid } from './gpu/opgrid.ts'; +import { Merger, type CsgOp } from './gpu/merge.ts'; +import { Stamper, box, capsule, cylinder, sphere, type Shape, type Vec3 } from './gpu/stamp.ts'; +import { Remapper } from './gpu/remap.ts'; +import { Loader } from './gpu/load.ts'; +import { Scanner } from './gpu/scan.ts'; +import { Sorter } from './gpu/radix_sort.ts'; +import { + GRID_TYPE_SDF_FLOAT, + PICOVDB_FILE_HEADER_SIZE, + PICOVDB_GRID_SIZE, + PICOVDB_MAGIC, + PICOVDB_ROOT_SIZE, + PicoVDBFile, +} from './picovdb.ts'; + +export type { Shape, Vec3, OpGrid }; + +// Shapes are plain values: a WGSL distance function by name, its +// arguments, and its bounds. They stamp straight into a solid as +// operands, or become a solid with space.solid(shape). These make the +// built-in shapes; SpaceOptions.shapes adds your own functions. +export { box, capsule, cylinder, sphere }; + +/** The picovdb node buffers of a solid, GPU resident in the file layout. */ +export type PicoVDBTree = EmitResult; + +/** Inclusive leaf coordinate bounds, or null for an empty solid. */ +export type Bounds = { min: Vec3; max: Vec3 } | null; + +/** A boolean operand: a solid, a pending op, or a shape, which stamps directly. */ +export type Operand = Solid | Op | Shape; + +export interface SpaceOptions { + /** Narrow band half width in voxels. */ + halfWidth?: number; + /** + * WGSL functions of the form `fn name(p: vec3f) -> f32` to use as + * shapes by name. p is in absolute voxels, `args` is the shape's + * arguments as an `array`, and the result is the signed + * distance in voxels. + */ + shapes?: string; +} + +const KEY_RANGE = 1024; + +function unionBounds(a: Bounds, b: Bounds): Bounds { + if (!a) return b; + if (!b) return a; + return { + min: a.min.map((v, i) => Math.min(v, b.min[i])) as Vec3, + max: a.max.map((v, i) => Math.max(v, b.max[i])) as Vec3, + }; +} + +function growBounds(b: Bounds, lo: Vec3, hi: Vec3): Bounds { + if (!b) return b; + return { min: b.min.map((v, i) => v + lo[i]) as Vec3, max: b.max.map((v, i) => v + hi[i]) as Vec3 }; +} + +/** Leaf bounds a shape's band can touch. */ +function shapeLeafBounds(shape: Shape, halfWidth: number): Bounds { + if (!shape.bounds) throw new Error(`shape ${shape.fn} needs bounds to add material`); + const { min, max } = shape.bounds; + return { + min: min.map((v) => Math.floor((v - halfWidth) / 8)) as Vec3, + max: max.map((v) => Math.floor((v + halfWidth) / 8)) as Vec3, + }; +} + +/** The key origin that fits bounds, or throws when the extent is too large. */ +function originFor(bounds: Bounds): Vec3 { + if (!bounds) return [0, 0, 0]; + for (let a = 0; a < 3; a++) { + const extent = bounds.max[a] - bounds.min[a] + 1; + if (extent > KEY_RANGE) { + throw new Error(`solid extent of ${extent} leaves on axis ${a} exceeds the ${KEY_RANGE} leaf key range`); + } + } + return [...bounds.min] as Vec3; +} + +export class Space { + readonly device: GPUDevice; + readonly halfWidth: number; + readonly scanner: Scanner; + readonly sorter: Sorter; + readonly emitter: Emitter; + readonly merger: Merger; + readonly stamper: Stamper; + readonly remapper: Remapper; + readonly loader: Loader; + private binner?: Binner; + private rasterizer?: Rasterizer; + private signer?: Signer; + + constructor(device: GPUDevice, opts: SpaceOptions = {}) { + this.device = device; + this.halfWidth = opts.halfWidth ?? 3; + this.scanner = new Scanner(device); + this.sorter = new Sorter(device, this.scanner); + this.emitter = new Emitter(device, this.scanner, this.sorter); + this.merger = new Merger(device, this.scanner, this.sorter); + this.stamper = new Stamper(device, this.scanner, this.sorter, opts.shapes); + this.remapper = new Remapper(device, this.scanner, this.sorter); + this.loader = new Loader(device); + } + + /** A solid with no leaves. */ + empty(): Solid { + return new Solid(this, emptyOpGrid(this.device), null); + } + + /** A solid of one shape. */ + solid(shape: Shape): Op { + return new Op(this, async () => { + const empty = this.empty(); + try { + return await stamp(empty, shape, 'add'); + } finally { + empty.destroy(); + } + }); + } + + /** One grid of an f32 or u8 SDF picovdb file. Values rescale to this half width. */ + fromPvdb(file: PicoVDBFile | ArrayBuffer, grid = 0): Solid { + const loaded = this.loader.load(file instanceof PicoVDBFile ? file : new PicoVDBFile(file), { halfWidth: this.halfWidth, grid }); + return new Solid(this, loaded, { min: loaded.leafMin, max: loaded.leafMax }); + } + + /** A closed triangle mesh in world units, voxelized at voxelSize world units per voxel. */ + fromMesh(points: Float32Array, triangles: Uint32Array, voxelSize: number): Op { + return new Op(this, async () => { + this.binner ??= new Binner(this.device, this.scanner, this.sorter); + this.rasterizer ??= new Rasterizer(this.device); + this.signer ??= new Signer(this.device, this.scanner, this.sorter); + const halfWidth = this.halfWidth; + const bin = await this.binner.bin(points, triangles, { voxelSize, halfWidth }); + const dist2 = this.rasterizer.rasterize(bin, { halfWidth }); + const sign = await this.signer.sign(bin); + const grid = await this.emitter.classifyOnly(bin, dist2, sign, { halfWidth }); + for (const b of [bin.pointsIndex, bin.triangles, bin.pairKeys, bin.pairTris, bin.leafKeys, dist2, sign.inside]) b.destroy(); + return new Solid(this, grid, { min: grid.leafMin, max: grid.leafMax }); + }); + } +} + +export class Solid { + readonly space: Space; + readonly grid: OpGrid; + readonly bounds: Bounds; + + constructor(space: Space, grid: OpGrid, bounds: Bounds) { + this.space = space; + this.grid = grid; + this.bounds = bounds; + } + + get leafCount(): number { + return this.grid.leafCount; + } + + get activeVoxels(): number { + return this.grid.activeVoxels; + } + + union(other: Operand): Op { + return new Op(this.space, () => combine(this, other, 'union')); + } + + subtract(other: Operand): Op { + return new Op(this.space, () => combine(this, other, 'subtract')); + } + + intersect(other: Solid | Op): Op { + return new Op(this.space, () => combine(this, other, 'intersect')); + } + + /** + * Offsets the surface by amount voxels. Positive grows. The new band is + * redistanced from the old surface, exact to marching cubes precision. + * Amounts beyond half width - 1 run in several steps. + */ + offset(amount: number): Op { + return new Op(this.space, () => offset(this, amount)); + } + + /** Translates by whole voxels. */ + translate(shift: Vec3): Op { + return new Op(this.space, () => translate(this, shift)); + } + + /** The picovdb tree on the GPU, for a renderer. The caller destroys its buffers. */ + toTree(): Promise { + return this.space.emitter.reEmit(this.grid, { halfWidth: this.space.halfWidth }); + } + + /** The solid as a single grid .pvdb file. */ + async toPvdb(): Promise { + const tree = await this.toTree(); + try { + return await writePvdb(this.space.device, tree); + } finally { + for (const b of [tree.roots, tree.uppers, tree.lowers, tree.leaves, tree.data]) b.destroy(); + } + } + + /** Releases the GPU buffers. The solid is unusable afterwards. */ + destroy(): void { + this.grid.leafKeys.destroy(); + this.grid.leaves.destroy(); + this.grid.data.destroy(); + } + + [Symbol.dispose](): void { + this.destroy(); + } +} + +/** + * A pending operation. Awaiting it runs the chain and resolves to a new + * Solid that the caller owns. Intermediates and Op operands are destroyed + * along the way. Each await runs the recipe again. + */ +export class Op implements PromiseLike { + readonly space: Space; + private readonly run: () => Promise; + + constructor(space: Space, run: () => Promise) { + this.space = space; + this.run = run; + } + + union(other: Operand): Op { + return this.then_((s) => combine(s, other, 'union')); + } + + subtract(other: Operand): Op { + return this.then_((s) => combine(s, other, 'subtract')); + } + + intersect(other: Solid | Op): Op { + return this.then_((s) => combine(s, other, 'intersect')); + } + + offset(amount: number): Op { + return this.then_((s) => offset(s, amount)); + } + + translate(shift: Vec3): Op { + return this.then_((s) => translate(s, shift)); + } + + then( + onfulfilled?: ((value: Solid) => R1 | PromiseLike) | null, + onrejected?: ((reason: unknown) => R2 | PromiseLike) | null, + ): Promise { + return this.run().then(onfulfilled, onrejected); + } + + /** The next step, which owns and frees this step's result. */ + private then_(step: (s: Solid) => Promise): Op { + return new Op(this.space, async () => { + const s = await this.run(); + try { + return await step(s); + } finally { + s.destroy(); + } + }); + } +} + +// The immediate operations. Each returns a new Solid and leaves its inputs as they are. + +async function combine(self: Solid, other: Operand, op: CsgOp): Promise { + if (other instanceof Op) { + const s = await other; + try { + return await combine(self, s, op); + } finally { + s.destroy(); + } + } + if (!(other instanceof Solid)) { + if (op === 'intersect') throw new Error('intersect takes a solid'); + return stamp(self, other, op === 'union' ? 'add' : 'carve'); + } + const space = self.space; + if (other.space !== space) throw new Error('solids belong to different spaces'); + if (self.grid.leafCount === 0 || other.grid.leafCount === 0) { + // An empty solid is all outside: union and subtract keep this, + // intersect empties it. Subtracting from empty stays empty. + const keep = op === 'intersect' ? null : self.grid.leafCount === 0 && op === 'union' ? other : self; + return keep ? copy(keep) : space.empty(); + } + const bounds = unionBounds(self.bounds, other.bounds); + const a = rebased(self, bounds); + const b = rebased(other, bounds); + const out = await space.merger.mergeCsg(a.grid, b.grid, { halfWidth: space.halfWidth, op }); + a.release(); + b.release(); + return new Solid(space, { ...out, leafMax: bounds!.max }, bounds); +} + +async function stamp(self: Solid, shape: Shape, mode: 'add' | 'carve'): Promise { + const space = self.space; + // Carving cannot extend the solid, so a carve keeps its bounds. + const bounds = mode === 'carve' ? self.bounds : unionBounds(self.bounds, shapeLeafBounds(shape, space.halfWidth)); + const { grid, release } = rebased(self, bounds); + const out = await space.stamper.stamp(grid, { shape, mode, halfWidth: space.halfWidth }); + release(); + return new Solid(space, out, bounds); +} + +async function offset(self: Solid, amount: number): Promise { + const space = self.space; + if (self.grid.leafCount === 0) return space.empty(); + const m = Remapper.leafMargin(amount, space.halfWidth); + const bounds = growBounds(self.bounds, [-m, -m, -m], [m, m, m]); + const { grid, release } = rebased(self, bounds); + const out = await space.remapper.offset(grid, amount, space.halfWidth); + release(); + return new Solid(space, out, bounds); +} + +async function translate(self: Solid, shift: Vec3): Promise { + const space = self.space; + if (self.grid.leafCount === 0) return space.empty(); + if (shift.some((s) => !Number.isInteger(s))) throw new Error('translate takes whole voxels'); + // Leaf multiples move the key origin; the remainder remaps voxels. + const q = shift.map((s) => Math.floor(s / 8)) as Vec3; + const r = shift.map((s, a) => s - 8 * q[a]) as Vec3; + const shifted: OpGrid = { ...self.grid, leafMin: self.grid.leafMin.map((v, a) => v + q[a]) as Vec3, leafMax: self.grid.leafMax.map((v, a) => v + q[a]) as Vec3 }; + let bounds = growBounds(self.bounds, q, q); + if (r.every((v) => v === 0)) return new Solid(space, copyGrid(space.device, shifted), bounds); + bounds = growBounds(bounds, [0, 0, 0], r.map((v) => (v > 0 ? 1 : 0)) as Vec3); + const grid = space.remapper.rebase(shifted, originFor(bounds)); + const out = await space.remapper.translate(grid, r, space.halfWidth); + if (grid.leafKeys !== self.grid.leafKeys) grid.leafKeys.destroy(); + return new Solid(space, out, bounds); +} + +/** The grid keyed from an origin that fits bounds, carrying the solid's exact leaf bounds. release frees the temporary keys. */ +function rebased(self: Solid, bounds: Bounds): { grid: OpGrid; release: () => void } { + const origin = originFor(bounds); + const grid = { ...self.space.remapper.rebase(self.grid, origin), leafMax: bounds ? bounds.max : origin }; + return { grid, release: () => { if (grid.leafKeys !== self.grid.leafKeys) grid.leafKeys.destroy(); } }; +} + +function copy(self: Solid): Solid { + return new Solid(self.space, copyGrid(self.space.device, self.grid), self.bounds); +} + +function copyGrid(device: GPUDevice, grid: OpGrid): OpGrid { + const dup = (src: GPUBuffer, size: number) => { + const dst = device.createBuffer({ size: Math.max(size, 4), usage: GPUBufferUsage.STORAGE | GPUBufferUsage.COPY_DST | GPUBufferUsage.COPY_SRC }); + if (size > 0) { + const encoder = device.createCommandEncoder(); + encoder.copyBufferToBuffer(src, 0, dst, 0, size); + device.queue.submit([encoder.finish()]); + } + return dst; + }; + const n = grid.leafCount; + return { ...grid, leafKeys: dup(grid.leafKeys, n * 4), leaves: dup(grid.leaves, n * LEAF_U32 * 4), data: dup(grid.data, (2 + grid.activeVoxels) * 4) }; +} + +/** Reads a tree back into .pvdb bytes: header, one grid record, then the node buffers. */ +async function writePvdb(device: GPUDevice, tree: PicoVDBTree): Promise { + const rootsPadded = Math.ceil(tree.upperCount / 2) * 2; + const dataCount = Math.ceil((tree.dataElemCount * 4) / 16); + const sections: [GPUBuffer, number, number][] = [ + [tree.roots, tree.upperCount * PICOVDB_ROOT_SIZE, rootsPadded * PICOVDB_ROOT_SIZE], + [tree.uppers, tree.upperCount * UPPER_U32 * 4, tree.upperCount * UPPER_U32 * 4], + [tree.lowers, tree.lowerCount * LOWER_U32 * 4, tree.lowerCount * LOWER_U32 * 4], + [tree.leaves, tree.leafCount * LEAF_U32 * 4, tree.leafCount * LEAF_U32 * 4], + [tree.data, tree.dataElemCount * 4, dataCount * 16], + ]; + const headSize = PICOVDB_FILE_HEADER_SIZE + PICOVDB_GRID_SIZE; + const head = new ArrayBuffer(headSize); + new Uint32Array(head, 0, 8).set([PICOVDB_MAGIC[0], PICOVDB_MAGIC[1], 0, 1, tree.upperCount, tree.lowerCount, tree.leafCount, dataCount]); + new Uint32Array(head, PICOVDB_FILE_HEADER_SIZE, 8).set([0, 0, 0, 0, 0, tree.dataElemCount, GRID_TYPE_SDF_FLOAT, 0]); + new Int32Array(head, PICOVDB_FILE_HEADER_SIZE + 32, 3).set(tree.indexBoundsMin); + new Int32Array(head, PICOVDB_FILE_HEADER_SIZE + 48, 3).set(tree.indexBoundsMax); + + const size = headSize + sections.reduce((n, [, , padded]) => n + padded, 0); + const staging = device.createBuffer({ size, usage: GPUBufferUsage.MAP_READ | GPUBufferUsage.COPY_DST }); + device.queue.writeBuffer(staging, 0, head); + const encoder = device.createCommandEncoder(); + let offset = headSize; + for (const [src, bytes, padded] of sections) { + if (bytes > 0) encoder.copyBufferToBuffer(src, 0, staging, offset, bytes); + offset += padded; + } + device.queue.submit([encoder.finish()]); + await staging.mapAsync(GPUMapMode.READ); + const out = staging.getMappedRange().slice(0); + staging.destroy(); + return out; +} diff --git a/ts/picovdb.ts b/ts/picovdb.ts index c31bfc2..c1a7fb8 100644 --- a/ts/picovdb.ts +++ b/ts/picovdb.ts @@ -143,7 +143,7 @@ export class PicoVDBFile { } const baseOffset = PICOVDB_FILE_HEADER_SIZE + index * PICOVDB_GRID_SIZE; - let offset = baseOffset; + const offset = baseOffset; return { gridIndex: this.view.getUint32(offset + 0, true), @@ -158,6 +158,22 @@ export class PicoVDBFile { }; } + /** Node and value ranges of one grid. Node indices inside a grid's nodes are relative to these starts. */ + getGridRange(index: number): { upperStart: number; upperCount: number; lowerStart: number; lowerCount: number; leafStart: number; leafCount: number; dataStart: number; dataElemCount: number } { + const grid = this.getGrid(index); + const next = index + 1 < this.header.gridCount ? this.getGrid(index + 1) : null; + return { + upperStart: grid.upperStart, + upperCount: (next ? next.upperStart : this.header.upperCount) - grid.upperStart, + lowerStart: grid.lowerStart, + lowerCount: (next ? next.lowerStart : this.header.lowerCount) - grid.lowerStart, + leafStart: grid.leafStart, + leafCount: (next ? next.leafStart : this.header.leafCount) - grid.leafStart, + dataStart: grid.dataStart, + dataElemCount: grid.dataElemCount, + }; + } + getRootCountPadded(): number { return ((this.header.upperCount + 1) / 2 | 0) * 2 // Padding to even number } @@ -268,7 +284,7 @@ export class PicoVDBFile { } getVoxelCount(): number { - var count = 0 + let count = 0 for (let i = 0; i < this.header.gridCount; i++) { count += this.getGrid(i).dataElemCount - 2 // Minus background values } diff --git a/ts/shaders.ts b/ts/shaders.ts new file mode 100644 index 0000000..293f09b --- /dev/null +++ b/ts/shaders.ts @@ -0,0 +1,5 @@ +// WGSL shader sources as strings. + +import picovdbWgsl from 'picovdb/wgsl/picovdb.wgsl' with { type: 'text' }; + +export { picovdbWgsl }; diff --git a/ts/wgsl.d.ts b/ts/wgsl.d.ts deleted file mode 100644 index 4ceaac0..0000000 --- a/ts/wgsl.d.ts +++ /dev/null @@ -1,4 +0,0 @@ -declare module '*.wgsl' { - const shader: string; - export default shader; -} diff --git a/tsconfig.json b/tsconfig.json deleted file mode 100644 index 05a820b..0000000 --- a/tsconfig.json +++ /dev/null @@ -1,30 +0,0 @@ -{ - "compilerOptions": { - // Enable latest features - "lib": ["ESNext", "DOM"], - "target": "ESNext", - "module": "ESNext", - "moduleDetection": "force", - "jsx": "react-jsx", - "allowJs": true, - - // Bundler mode - "moduleResolution": "bundler", - "allowImportingTsExtensions": true, - "verbatimModuleSyntax": true, - "noEmit": true, - - // Best practices - "strict": true, - "skipLibCheck": true, - "noFallthroughCasesInSwitch": true, - - // Some stricter flags (disabled by default) - "noUnusedLocals": false, - "noUnusedParameters": false, - "noPropertyAccessFromIndexSignature": false, - - // Custom - "types": ["@webgpu/types"] - } -} diff --git a/wgsl/dilate.wgsl b/wgsl/dilate.wgsl new file mode 100644 index 0000000..514c99c --- /dev/null +++ b/wgsl/dilate.wgsl @@ -0,0 +1,258 @@ +// Dilates an active voxel mask set by one voxel across face neighbors, +// the word shift approach of NanoVDB DilateGrid on u32 words. +// +// Input is a sorted unique leaf key table with one 512 bit mask per leaf. +// Each u32 word covers one x and four y values, one byte per column of 8 +// z bits. count_spawn and emit_spawn emit each leaf plus any face neighbor +// its boundary planes spill into. The host sorts and dedupes the keys and +// dilate_masks builds each output leaf's mask from its own shifted words +// plus the six neighbors' boundary planes. +// +// A spawn falling outside the packed key range increments clipped and the +// host fails the op, so callers keep a margin of one leaf. + +struct DilateParams { + old_count: u32, + spawn_count: u32, + new_count: u32, + pad: u32, +} + +@group(0) @binding(0) var params: DilateParams; +@group(0) @binding(1) var old_keys: array; +@group(0) @binding(2) var old_masks: array; +@group(0) @binding(3) var counts: array; +@group(0) @binding(4) var spawn_keys: array; +@group(0) @binding(5) var flags: array; +@group(0) @binding(6) var new_keys: array; +@group(0) @binding(7) var new_masks: array; +@group(0) @binding(8) var clipped: atomic; + +const DISPATCH_STRIDE: u32 = 65535u; + +fn globalIndex(wid: vec3, lid: vec3) -> u32 { + return (((wid.y * DISPATCH_STRIDE) + wid.x) * 256u) + lid.x; +} + +fn unpack(key: u32) -> vec3 { + return vec3(i32(key >> 20u), i32((key >> 10u) & 0x3ffu), i32(key & 0x3ffu)); +} + +fn pack(c: vec3) -> u32 { + return (u32(c.x) << 20u) | (u32(c.y) << 10u) | u32(c.z); +} + +fn inRange(c: vec3) -> bool { + return all(c >= vec3(0)) && all(c <= vec3(1023)); +} + +// True when the leaf's boundary plane facing direction d has any active +// voxel. Directions order as negative then positive x, then y, then z. +fn spills(leaf: u32, d: u32) -> bool { + let base = leaf * 16u; + var acc = 0u; + switch (d) { + case 0u: { acc = old_masks[base] | old_masks[base + 1u]; } + case 1u: { acc = old_masks[base + 14u] | old_masks[base + 15u]; } + case 2u: { + for (var k = 0u; k < 8u; k = k + 1u) { + acc = acc | (old_masks[base + (2u * k)] & 0x000000ffu); + } + } + case 3u: { + for (var k = 0u; k < 8u; k = k + 1u) { + acc = acc | (old_masks[base + (2u * k) + 1u] & 0xff000000u); + } + } + case 4u: { + for (var w = 0u; w < 16u; w = w + 1u) { + acc = acc | (old_masks[base + w] & 0x01010101u); + } + } + default: { + for (var w = 0u; w < 16u; w = w + 1u) { + acc = acc | (old_masks[base + w] & 0x80808080u); + } + } + } + return acc != 0u; +} + +const DIRS = array, 6>( + vec3(-1, 0, 0), vec3(1, 0, 0), + vec3(0, -1, 0), vec3(0, 1, 0), + vec3(0, 0, -1), vec3(0, 0, 1), +); + +// Shared by the count and emit passes, which must agree. +fn spawn(i: u32, emit: bool, offset: u32) -> u32 { + let c = unpack(old_keys[i]); + var w = offset; + if (emit) { + spawn_keys[w] = old_keys[i]; + } + w = w + 1u; + var n = 1u; + for (var d = 0u; d < 6u; d = d + 1u) { + let nc = c + DIRS[d]; + if (!spills(i, d)) { + continue; + } + if (!inRange(nc)) { + if (!emit) { + atomicAdd(&clipped, 1u); + } + continue; + } + if (emit) { + spawn_keys[w] = pack(nc); + } + w = w + 1u; + n = n + 1u; + } + return n; +} + +// counts has one extra entry so the scanned total lands in the last slot. +@compute @workgroup_size(256) +fn count_spawn(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.old_count) { + return; + } + if (i == params.old_count) { + counts[i] = 0u; + return; + } + counts[i] = spawn(i, false, 0u); +} + +// counts now holds the scanned write offsets. +@compute @workgroup_size(256) +fn emit_spawn(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i < params.old_count) { + let unused = spawn(i, true, counts[i]); + } +} + +// flags has one extra entry so the scanned total lands in the last slot. +@compute @workgroup_size(256) +fn mark_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.spawn_count) { + return; + } + if (i == params.spawn_count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, i == 0u || spawn_keys[i] != spawn_keys[i - 1u]); +} + +// flags now holds the scanned unique positions. +@compute @workgroup_size(256) +fn compact_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.spawn_count) { + return; + } + if (i == 0u || spawn_keys[i] != spawn_keys[i - 1u]) { + new_keys[flags[i]] = spawn_keys[i]; + } +} + +// Loads the leaf's 16 words, or zeros when the leaf is absent. +fn loadMask(key: u32, out: ptr>) { + var lo = 0u; + var hi = params.old_count; + while (lo < hi) { + let mid = (lo + hi) >> 1u; + if (old_keys[mid] < key) { + lo = mid + 1u; + } else { + hi = mid; + } + } + let found = lo < params.old_count && old_keys[lo] == key; + for (var w = 0u; w < 16u; w = w + 1u) { + (*out)[w] = select(0u, old_masks[(lo * 16u) + w], found); + } +} + +@compute @workgroup_size(256) +fn dilate_masks( + @builtin(workgroup_id) wid: vec3, + @builtin(local_invocation_id) lid: vec3, +) { + let i = globalIndex(wid, lid); + if (i >= params.new_count) { + return; + } + let c = unpack(new_keys[i]); + var s: array; + loadMask(new_keys[i], &s); + + var out: array; + // Within the leaf, identity plus one voxel shifts in y and z. + for (var k = 0u; k < 8u; k = k + 1u) { + let lo = s[2u * k]; + let hi = s[(2u * k) + 1u]; + out[2u * k] = lo + | (lo << 8u) | ((lo >> 8u) | ((hi & 0xffu) << 24u)) + | ((lo << 1u) & 0xfefefefeu) | ((lo >> 1u) & 0x7f7f7f7fu); + out[(2u * k) + 1u] = hi + | ((hi << 8u) | (lo >> 24u)) | (hi >> 8u) + | ((hi << 1u) & 0xfefefefeu) | ((hi >> 1u) & 0x7f7f7f7fu); + } + // x shifts move whole word pairs. + for (var w = 0u; w < 16u; w = w + 1u) { + if (w >= 2u) { + out[w] = out[w] | s[w - 2u]; + } + if (w < 14u) { + out[w] = out[w] | s[w + 2u]; + } + } + + // Boundary planes from the six face neighbors. + var nb: array; + if (c.x > 0) { + loadMask(pack(c + vec3(-1, 0, 0)), &nb); + out[0] = out[0] | nb[14]; + out[1] = out[1] | nb[15]; + } + if (c.x < 1023) { + loadMask(pack(c + vec3(1, 0, 0)), &nb); + out[14] = out[14] | nb[0]; + out[15] = out[15] | nb[1]; + } + if (c.y > 0) { + loadMask(pack(c + vec3(0, -1, 0)), &nb); + for (var k = 0u; k < 8u; k = k + 1u) { + out[2u * k] = out[2u * k] | (nb[(2u * k) + 1u] >> 24u); + } + } + if (c.y < 1023) { + loadMask(pack(c + vec3(0, 1, 0)), &nb); + for (var k = 0u; k < 8u; k = k + 1u) { + out[(2u * k) + 1u] = out[(2u * k) + 1u] | ((nb[2u * k] & 0xffu) << 24u); + } + } + if (c.z > 0) { + loadMask(pack(c + vec3(0, 0, -1)), &nb); + for (var w = 0u; w < 16u; w = w + 1u) { + out[w] = out[w] | ((nb[w] & 0x80808080u) >> 7u); + } + } + if (c.z < 1023) { + loadMask(pack(c + vec3(0, 0, 1)), &nb); + for (var w = 0u; w < 16u; w = w + 1u) { + out[w] = out[w] | ((nb[w] & 0x01010101u) << 7u); + } + } + + for (var w = 0u; w < 16u; w = w + 1u) { + new_masks[(i * 16u) + w] = out[w]; + } +} diff --git a/wgsl/emit.wgsl b/wgsl/emit.wgsl new file mode 100644 index 0000000..94a91f8 --- /dev/null +++ b/wgsl/emit.wgsl @@ -0,0 +1,466 @@ +// Builds the picovdb tree of an op layer grid: the leaf, lower, upper, +// and root node buffers and the value array, in the file layout. Leaves +// come out in the CPU converter's order, so the output matches it byte +// for byte. +// +// The grid binds as cand_keys, cand_leaves, cand_data. classify_mark and +// classify_apply are the mesh converter's entry: they build an op layer +// grid from per leaf squared distance slabs and inside masks. +// +// Each entry point uses a few of the bindings. Pipelines use auto layouts, +// which keeps storage buffer counts under the WebGPU limit. + +struct EmitParams { + cand_count: u32, + lower_count: u32, + upper_count: u32, + half_width: f32, + leaf_min: vec3, + pad0: u32, + lower_min: vec3, + pad1: u32, + upper_min: vec3, + pad2: u32, +} + +@group(0) @binding(0) var params: EmitParams; +@group(0) @binding(1) var cand_keys: array; +@group(0) @binding(2) var cand_leaves: array; +@group(0) @binding(3) var cand_data: array; +@group(0) @binding(4) var dist2: array; // classify: squared distances as f32 bits, 512 per leaf +@group(0) @binding(5) var band_counts: array; +@group(0) @binding(6) var bounds: array>; // min xyz, max xyz +@group(0) @binding(7) var flags: array; +@group(0) @binding(8) var final_keys: array; +@group(0) @binding(9) var final_cand: array; +@group(0) @binding(10) var value_counts: array; +@group(0) @binding(11) var inside_masks: array; // classify: 16 words per leaf +@group(0) @binding(13) var surf_masks: array; +@group(0) @binding(14) var surf_counts: array; +@group(0) @binding(15) var leaves_out: array; +@group(0) @binding(16) var data_out: array; // f32 bits +@group(0) @binding(17) var lower_keys: array; +@group(0) @binding(18) var lower_first: array; +@group(0) @binding(19) var lowers_out: array; +@group(0) @binding(20) var upper_keys: array; +@group(0) @binding(21) var upper_first: array; +@group(0) @binding(22) var uppers_out: array; +@group(0) @binding(23) var roots_out: array; +@group(0) @binding(26) var hier: array; +@group(0) @binding(27) var idx: array; +@group(0) @binding(28) var flat_lower: array; + +const LOWER_U32: u32 = 388u; // 1552 bytes +const UPPER_U32: u32 = 3076u; // 12304 bytes + +// Signed value of voxel n of binned leaf c: its distance inside the band, +// the background elsewhere. +fn classifyValue(c: u32, n: u32) -> f32 { + let hw = params.half_width; + let inside = ((inside_masks[(c * 16u) + (n >> 5u)] >> (n & 31u)) & 1u) == 1u; + let d2 = bitcast(dist2[(c * 512u) + n]); + var v = select(hw, -hw, inside); + if (d2 <= hw * hw) { + v = select(sqrt(d2), -sqrt(d2), inside); + } + return v; +} + +@compute @workgroup_size(256) +fn classify_mark(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (markSentinel(i, params.cand_count)) { + return; + } + var acc: LeafAcc; + for (var n = 0u; n < 512u; n = n + 1u) { + accPush(&acc, i, n, classifyValue(i, n)); + } + accFinish(&acc, i); +} + +@compute @workgroup_size(256) +fn classify_apply(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j >= arrayLength(&w_keys)) { + return; + } + let c = cand_find(w_keys[j]); + let base = w_leaves[(j * LEAF_U32) + 2u]; + var k = 0u; + for (var w = 0u; w < 16u; w = w + 1u) { + let band = bandWord(j, w); + for (var b = 0u; b < 32u; b = b + 1u) { + if (((band >> b) & 1u) == 0u) { + continue; + } + w_data[base + k] = bitcast(classifyValue(c, (w * 32u) + b)); + k = k + 1u; + } + } +} + +// Band count and voxel bounds per leaf. band_counts has an extra entry +// for the scan total. +@compute @workgroup_size(256) +fn leaf_stats(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.cand_count) { + return; + } + if (i == params.cand_count) { + band_counts[i] = 0u; + return; + } + let origin = (unpack(cand_keys[i]) + params.leaf_min) << vec3(3u); + var count = 0u; + var leaf_min = vec3(2147483647); + var leaf_max = vec3(-2147483648); + for (var w = 0u; w < 16u; w = w + 1u) { + let band = cand_leaves[(i * LEAF_U32) + 5u + (w * 3u)]; + count = count + countOneBits(band); + for (var b = 0u; b < 32u; b = b + 1u) { + if (((band >> b) & 1u) == 1u) { + let ijk = origin + voxelLocal((w * 32u) + b); + leaf_min = min(leaf_min, ijk); + leaf_max = max(leaf_max, ijk); + } + } + } + band_counts[i] = count; + if (count > 0u) { + for (var a = 0u; a < 3u; a = a + 1u) { + atomicMin(&bounds[a], leaf_min[a]); + atomicMax(&bounds[a + 3u], leaf_max[a]); + } + } +} + +// The CPU emits leaves depth first: uppers in coordinate order, then +// child slots per level. Flat key order interleaves parents. So the leaves +// sort by a hierarchical key: upper coordinate, lower slot, leaf slot. It +// spans two u32 words, sorted in two stable passes. +fn hierLoOf(key: u32) -> u32 { + let a = unpack(key) + params.leaf_min; + let low = (a >> vec3(4u)) & vec3(31); + let lf = a & vec3(15); + return (u32(low.x) << 22u) | (u32(low.y) << 17u) | (u32(low.z) << 12u) + | (u32(lf.x) << 8u) | (u32(lf.y) << 4u) | u32(lf.z); +} + +fn hierHiOf(key: u32) -> u32 { + let a = unpack(key) + params.leaf_min; + let up = (a >> vec3(9u)) - params.upper_min; + return (u32(up.x) << 20u) | (u32(up.y) << 10u) | u32(up.z); +} + +@compute @workgroup_size(256) +fn hier_lo(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j < params.cand_count) { + hier[j] = hierLoOf(cand_keys[j]); + idx[j] = j; + } +} + +@compute @workgroup_size(256) +fn hier_hi(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j < params.cand_count) { + hier[j] = hierHiOf(cand_keys[idx[j]]); + } +} + +@compute @workgroup_size(256) +fn reorder_final(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j < params.cand_count) { + final_keys[j] = cand_keys[idx[j]]; + final_cand[j] = idx[j]; + } +} + +@compute @workgroup_size(256) +fn leaf_value_counts(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j > params.cand_count) { + return; + } + if (j == params.cand_count) { + value_counts[j] = 0u; + return; + } + value_counts[j] = band_counts[final_cand[j]]; +} + +fn lowerKeyOf(leaf_key: u32) -> u32 { + let abs = unpack(leaf_key) + params.leaf_min; + return pack((abs >> vec3(4u)) - params.lower_min); +} + +fn upperKeyOf(lower_key: u32) -> u32 { + let abs = unpack(lower_key) + params.lower_min; + return pack((abs >> vec3(5u)) - params.upper_min); +} + +@compute @workgroup_size(256) +fn mark_lower(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.cand_count) { + return; + } + if (i == params.cand_count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, i == 0u || lowerKeyOf(final_keys[i]) != lowerKeyOf(final_keys[i - 1u])); +} + +@compute @workgroup_size(256) +fn compact_lower(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.cand_count) { + return; + } + if (i == 0u || lowerKeyOf(final_keys[i]) != lowerKeyOf(final_keys[i - 1u])) { + lower_keys[flags[i]] = lowerKeyOf(final_keys[i]); + lower_first[flags[i]] = i; + } +} + +@compute @workgroup_size(256) +fn mark_upper(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.lower_count) { + return; + } + if (i == params.lower_count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, i == 0u || upperKeyOf(lower_keys[i]) != upperKeyOf(lower_keys[i - 1u])); +} + +@compute @workgroup_size(256) +fn compact_upper(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.lower_count) { + return; + } + if (i == 0u || upperKeyOf(lower_keys[i]) != upperKeyOf(lower_keys[i - 1u])) { + upper_keys[flags[i]] = upperKeyOf(lower_keys[i]); + upper_first[flags[i]] = i; + } +} + +// Sign of the empty space at a voxel, from the grid's implicit background. +fn leafInside(ijk: vec3) -> bool { + return cand_valueAt(ijk - (params.leaf_min << vec3(3u))) < 0.0; +} + +const NEIGHBORS = array, 7>( + vec3(1, 0, 0), vec3(0, 1, 0), vec3(0, 0, 1), + vec3(1, 1, 0), vec3(1, 0, 1), vec3(0, 1, 1), + vec3(1, 1, 1), +); + +// A voxel is surface when its sign differs from any positive neighbor. +// Mirrors Builder.emitLeaf. +@compute @workgroup_size(256) +fn surface(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j > params.cand_count) { + return; + } + if (j == params.cand_count) { + surf_counts[j] = 0u; + return; + } + let c = final_cand[j]; + let origin = unpack(cand_keys[c]) << vec3(3u); + var total = 0u; + for (var w = 0u; w < 16u; w = w + 1u) { + let band = cand_leaves[(c * LEAF_U32) + 5u + (w * 3u)]; + var surf = 0u; + for (var b = 0u; b < 32u; b = b + 1u) { + if (((band >> b) & 1u) == 0u) { + continue; + } + let n = (w * 32u) + b; + let v = cand_leafValue(c, n); + let local = voxelLocal(n); + for (var k = 0u; k < 7u; k = k + 1u) { + let nl = local + NEIGHBORS[k]; + var nv: f32; + if (all(nl < vec3(8))) { + nv = cand_leafValue(c, voxelOffset(nl)); + } else { + nv = cand_valueAt(origin + nl); + } + let strict = (v < 0.0) != (nv < 0.0); + let nonstrict = (v <= 0.0) != (nv <= 0.0); + if (strict || nonstrict) { + surf = surf | (1u << b); + break; + } + } + } + surf_masks[(j * 16u) + w] = surf; + total = total + countOneBits(surf); + } + surf_counts[j] = total; +} + +// Writes leaf nodes: the op layer records plus surface bits and final +// value bases. surf_counts and value_counts hold exclusive scans. +@compute @workgroup_size(256) +fn write_leaves(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j >= params.cand_count) { + return; + } + let c = final_cand[j]; + let base = j * LEAF_U32; + leaves_out[base] = surf_counts[j]; // running surface count + leaves_out[base + 1u] = 0u; + leaves_out[base + 2u] = 2u + value_counts[j]; + leaves_out[base + 3u] = 0u; + var local_state = 0u; + var local_value = 0u; + for (var w = 0u; w < 16u; w = w + 1u) { + let e = (c * LEAF_U32) + 4u + (w * 3u); + let band = cand_leaves[e + 1u]; + let surf = surf_masks[(j * 16u) + w]; + let state = (cand_leaves[e] & ~band) | (surf & band); + leaves_out[base + 4u + (w * 3u)] = state; + leaves_out[base + 5u + (w * 3u)] = band; + leaves_out[base + 6u + (w * 3u)] = (local_state << 16u) | local_value; + local_value = local_value + countOneBits(band); + local_state = local_state + countOneBits(band & state); + } +} + +// Copies each leaf's values to their final slot. The first two slots hold +// the implicit background values. +@compute @workgroup_size(256) +fn write_data(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j >= params.cand_count) { + return; + } + if (j == 0u) { + data_out[0] = bitcast(params.half_width); + data_out[1] = bitcast(-params.half_width); + } + let c = final_cand[j]; + let src = cand_leaves[(c * LEAF_U32) + 2u]; + let dst = 2u + value_counts[j]; + let count = band_counts[c]; + for (var k = 0u; k < count; k = k + 1u) { + data_out[dst + k] = cand_data[src + k]; + } +} + +fn hasFinalLeaf(key: u32) -> bool { + return cand_find(key) != NOT_FOUND; +} + +// flat_lower is the flat sorted copy of the lower keys. +fn hasLower(key: u32) -> bool { + var lo = 0u; + var hi = params.lower_count; + while (lo < hi) { + let mid = (lo + hi) >> 1u; + if (flat_lower[mid] < key) { + lo = mid + 1u; + } else { + hi = mid; + } + } + return lo < params.lower_count && flat_lower[lo] == key; +} + +// Writes lower nodes in the CPU's child slot order. +@compute @workgroup_size(256) +fn write_lowers(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let l = globalIndex(wid, lid); + if (l >= params.lower_count) { + return; + } + let lower_abs = unpack(lower_keys[l]) + params.lower_min; + let base = l * LOWER_U32; + lowers_out[base] = lower_first[l]; + lowers_out[base + 1u] = 0u; + lowers_out[base + 2u] = 2u + value_counts[lower_first[l]]; + lowers_out[base + 3u] = 0u; + var local_state = 0u; + for (var w = 0u; w < 128u; w = w + 1u) { + var state = 0u; + var value = 0u; + for (var b = 0u; b < 32u; b = b + 1u) { + let n = (w * 32u) + b; + let local = vec3(i32((n >> 8u) & 15u), i32((n >> 4u) & 15u), i32(n & 15u)); + let leaf_abs = (lower_abs << vec3(4u)) + local; + if (hasFinalLeaf(pack(leaf_abs - params.leaf_min))) { + state = state | (1u << b); + value = value | (1u << b); + } else if (leafInside((leaf_abs << vec3(3u)) + vec3(4))) { + state = state | (1u << b); + } + } + lowers_out[base + 4u + (w * 3u)] = state; + lowers_out[base + 5u + (w * 3u)] = value; + lowers_out[base + 6u + (w * 3u)] = local_state << 16u; + local_state = local_state + countOneBits(value & state); + } +} + +// Writes upper nodes in the CPU's child slot order. +@compute @workgroup_size(256) +fn write_uppers(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let u = globalIndex(wid, lid); + if (u >= params.upper_count) { + return; + } + let upper_abs = unpack(upper_keys[u]) + params.upper_min; + let base = u * UPPER_U32; + uppers_out[base] = upper_first[u]; + uppers_out[base + 1u] = 0u; + uppers_out[base + 2u] = lowers_out[(upper_first[u] * LOWER_U32) + 2u]; + uppers_out[base + 3u] = 0u; + var local_state = 0u; + for (var w = 0u; w < 1024u; w = w + 1u) { + var state = 0u; + var value = 0u; + for (var b = 0u; b < 32u; b = b + 1u) { + let n = (w * 32u) + b; + let local = vec3(i32((n >> 10u) & 31u), i32((n >> 5u) & 31u), i32(n & 31u)); + let lower_abs = (upper_abs << vec3(5u)) + local; + if (hasLower(pack(lower_abs - params.lower_min))) { + state = state | (1u << b); + value = value | (1u << b); + } else if (leafInside((lower_abs << vec3(7u)) + vec3(64))) { + state = state | (1u << b); + } + } + uppers_out[base + 4u + (w * 3u)] = state; + uppers_out[base + 5u + (w * 3u)] = value; + uppers_out[base + 6u + (w * 3u)] = local_state << 16u; + local_state = local_state + countOneBits(value & state); + } +} + +// Root keys, mirroring picovdb coordToKey. +@compute @workgroup_size(256) +fn write_roots(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let u = globalIndex(wid, lid); + if (u >= params.upper_count) { + return; + } + let origin = (unpack(upper_keys[u]) + params.upper_min) << vec3(12u); + let iu = bitcast(origin.x) >> 12u; + let ju = bitcast(origin.y) >> 12u; + let ku = bitcast(origin.z) >> 12u; + roots_out[u * 2u] = ku | (ju << 21u); + roots_out[(u * 2u) + 1u] = (iu << 10u) | (ju >> 11u); +} diff --git a/wgsl/extract.wgsl b/wgsl/extract.wgsl new file mode 100644 index 0000000..02ac134 --- /dev/null +++ b/wgsl/extract.wgsl @@ -0,0 +1,155 @@ +// Extracts a level set of an op layer grid as triangles, with marching +// cubes. Used to redistance after an offset, and to export meshes. +// +// The grid binds as old_keys, old_leaves, old_data. params.iso is the +// level and must lie inside the band. Run count, scan counts on the host, +// then run emit. The output is three points per triangle in the grid's +// relative voxel coordinates, plus triangle indices. + +struct ExtractParams { + leaf_count: u32, + half_width: f32, + iso: f32, + pad1: u32, +} + +@group(0) @binding(0) var params: ExtractParams; +@group(0) @binding(1) var old_keys: array; +@group(0) @binding(2) var old_leaves: array; +@group(0) @binding(3) var tri_table: array; // 256 x 16 edge triples, then 256 counts +@group(0) @binding(4) var counts: array; // per leaf, one extra for the scan total +@group(0) @binding(5) var out_points: array; +@group(0) @binding(6) var out_tris: array; +@group(0) @binding(7) var old_data: array; + +const CELLS: u32 = 729u; // 9^3 cell bases in the block + +// One workgroup per leaf. The block caches the leaf's voxels and a one +// voxel margin around them. +var block: array; // voxel origin - 1 first +var wg_count: atomic; + +fn hasLeaf(leaf: vec3) -> bool { + if (any(leaf < vec3(0)) || any(leaf > vec3(1023))) { + return false; + } + return old_find(pack(leaf)) != NOT_FOUND; +} + +// Fills the block for leaf l and returns the leaf's voxel origin. +fn loadBlock(l: u32, lid: u32) -> vec3 { + let leaf = unpack(old_keys[l]); + let origin = leaf << vec3(3u); + for (var b = lid; b < 1000u; b = b + 256u) { + let c = vec3(i32(b / 100u), i32((b / 10u) % 10u), i32(b % 10u)); + if (all(c >= vec3(1)) && all(c <= vec3(8))) { + block[b] = old_leafValue(l, voxelOffset(c - vec3(1))) - params.iso; + } else { + block[b] = old_valueAt(origin - vec3(1) + c) - params.iso; + } + } + return origin; +} + +const CORNER = array, 8>( + vec3(0, 0, 0), vec3(1, 0, 0), vec3(1, 1, 0), vec3(0, 1, 0), + vec3(0, 0, 1), vec3(1, 0, 1), vec3(1, 1, 1), vec3(0, 1, 1), +); +const EDGE_A = array(0u, 1u, 2u, 3u, 4u, 5u, 6u, 7u, 0u, 1u, 2u, 3u); +const EDGE_B = array(1u, 2u, 3u, 0u, 5u, 6u, 7u, 4u, 4u, 5u, 6u, 7u); + +fn blockAt(c: vec3) -> f32 { + return block[(u32(c.x) * 100u) + (u32(c.y) * 10u) + u32(c.z)]; +} + +// Cube index of the cell with block base b, or -1 when another leaf owns +// the cell. A cell belongs to the leaf that holds its base voxel, so each +// cell is emitted once. When that leaf is missing, this leaf takes it. +fn cellIndex(leaf: vec3, b: vec3) -> i32 { + if (any(b == vec3(0))) { + let owner = leaf - vec3(select(0, 1, b.x == 0), select(0, 1, b.y == 0), select(0, 1, b.z == 0)); + if (hasLeaf(owner)) { + return -1; + } + } + var index = 0u; + for (var i = 0u; i < 8u; i = i + 1u) { + if (blockAt(b + CORNER[i]) < 0.0) { + index = index | (1u << i); + } + } + return i32(index); +} + +@compute @workgroup_size(256) +fn count(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let l = (wid.y * DISPATCH_STRIDE) + wid.x; + if (l >= params.leaf_count) { + return; + } + if (lid.x == 0u) { + atomicStore(&wg_count, 0u); + } + let origin = loadBlock(l, lid.x); + workgroupBarrier(); + let leaf = origin >> vec3(3u); + var n = 0u; + for (var c = lid.x; c < CELLS; c = c + 256u) { + let b = vec3(i32(c / 81u), i32((c / 9u) % 9u), i32(c % 9u)); + let index = cellIndex(leaf, b); + if (index > 0 && index < 255) { + n = n + u32(tri_table[4096 + index]); + } + } + atomicAdd(&wg_count, n); + workgroupBarrier(); + if (lid.x == 0u) { + counts[l] = atomicLoad(&wg_count); + } +} + +fn writeVertex(t: u32, v: u32, p: vec3) { + let o = ((t * 3u) + v) * 3u; + out_points[o] = p.x; + out_points[o + 1u] = p.y; + out_points[o + 2u] = p.z; + out_tris[(t * 3u) + v] = (t * 3u) + v; +} + +// counts holds the scanned triangle offsets. +@compute @workgroup_size(256) +fn emit(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let l = (wid.y * DISPATCH_STRIDE) + wid.x; + if (l >= params.leaf_count) { + return; + } + if (lid.x == 0u) { + atomicStore(&wg_count, 0u); + } + let origin = loadBlock(l, lid.x); + workgroupBarrier(); + let leaf = origin >> vec3(3u); + let base = counts[l]; + for (var c = lid.x; c < CELLS; c = c + 256u) { + let b = vec3(i32(c / 81u), i32((c / 9u) % 9u), i32(c % 9u)); + let index = cellIndex(leaf, b); + if (index <= 0 || index >= 255) { + continue; + } + let ntri = u32(tri_table[4096 + index]); + var t = base + atomicAdd(&wg_count, ntri); + let cell = origin - vec3(1) + b; // base voxel, relative coords + for (var k = 0u; k < ntri; k = k + 1u) { + for (var v = 0u; v < 3u; v = v + 1u) { + let e = u32(tri_table[(u32(index) * 16u) + (k * 3u) + v]); + let ca = CORNER[EDGE_A[e]]; + let cb = CORNER[EDGE_B[e]]; + let va = blockAt(b + ca); + let vb = blockAt(b + cb); + let s = clamp(va / (va - vb), 0.0, 1.0); + writeVertex(t, v, vec3(cell + ca) + (vec3(cb - ca) * s)); + } + t = t + 1u; + } + } +} diff --git a/wgsl/load.wgsl b/wgsl/load.wgsl new file mode 100644 index 0000000..ed54668 --- /dev/null +++ b/wgsl/load.wgsl @@ -0,0 +1,54 @@ +// Loads a picovdb tree into the op layer. The host orders the leaves by +// key. gather_leaves copies the leaf records in that order. convert_data +// rescales the values to the op layer's half width, and turns u8 values +// into f32. Records keep their value indices, so the value array stays in +// file order. + +struct LoadParams { + leaf_count: u32, + data_count: u32, // value slots including the two implicit entries + scale: f32, // multiplies stored values, maps the file's background to the half width + grid_type: u32, // 1 f32, 2 u8 (unorm bytes mapped to [-3, 3] as the renderer reads them) +} + +@group(0) @binding(0) var params: LoadParams; +@group(0) @binding(1) var order: array; // output slot -> tree leaf index +@group(0) @binding(2) var leaves: array; // tree leaf nodes +@group(0) @binding(3) var data: array; // f32 bits, or packed u8 +@group(0) @binding(4) var out_leaves: array; +@group(0) @binding(5) var out_data: array; + +const DISPATCH_STRIDE: u32 = 65535u; +const LEAF_U32: u32 = 52u; // 208 bytes + +fn globalIndex(wid: vec3, lid: vec3) -> u32 { + return (((wid.y * DISPATCH_STRIDE) + wid.x) * 256u) + lid.x; +} + +@compute @workgroup_size(256) +fn gather_leaves(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j >= params.leaf_count) { + return; + } + let src = order[j] * LEAF_U32; + let dst = j * LEAF_U32; + for (var k = 0u; k < LEAF_U32; k = k + 1u) { + out_leaves[dst + k] = leaves[src + k]; + } +} + +@compute @workgroup_size(256) +fn convert_data(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let d = globalIndex(wid, lid); + if (d >= params.data_count) { + return; + } + var v: f32; + if (params.grid_type == 2u) { + v = fma(unpack4x8unorm(data[d >> 2u])[d & 3u], 6.0, -3.0); + } else { + v = bitcast(data[d]); + } + out_data[d] = bitcast(v * params.scale); +} diff --git a/wgsl/merge.wgsl b/wgsl/merge.wgsl new file mode 100644 index 0000000..79d87e1 --- /dev/null +++ b/wgsl/merge.wgsl @@ -0,0 +1,144 @@ +// Merges two grids over the union of their leaf tables. +// +// merge_masks ORs 16 word masks, for topology. csg_mark and csg_apply +// combine two op layer grids, bound as a_* and b_*: per voxel min for a +// union, max for an intersection, max(a, -b) for a subtraction. A grid +// without the leaf contributes its implicit background, so band voxels +// inside the other solid deactivate. + +struct MergeParams { + a_count: u32, + b_count: u32, + concat_count: u32, + out_count: u32, + half_width: f32, + op: u32, // 0 union, 1 intersect, 2 subtract + pad1: u32, + pad2: u32, +} + +@group(0) @binding(0) var params: MergeParams; +@group(0) @binding(1) var concat_keys: array; +@group(0) @binding(2) var flags: array; +@group(0) @binding(3) var out_keys: array; +@group(0) @binding(4) var a_keys: array; +@group(0) @binding(5) var a_masks: array; +@group(0) @binding(6) var b_keys: array; +@group(0) @binding(7) var b_masks: array; +@group(0) @binding(8) var out_masks: array; +@group(0) @binding(10) var a_leaves: array; +@group(0) @binding(11) var a_data: array; +@group(0) @binding(12) var b_leaves: array; +@group(0) @binding(13) var b_data: array; + +// flags has one extra entry for the scan total. +@compute @workgroup_size(256) +fn mark_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.concat_count) { + return; + } + if (i == params.concat_count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, i == 0u || concat_keys[i] != concat_keys[i - 1u]); +} + +@compute @workgroup_size(256) +fn compact_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.concat_count) { + return; + } + if (i == 0u || concat_keys[i] != concat_keys[i - 1u]) { + out_keys[flags[i]] = concat_keys[i]; + } +} + +// Topology merge: ORs the 16 mask words per leaf. +@compute @workgroup_size(256) +fn merge_masks(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.out_count) { + return; + } + let a = a_find(out_keys[i]); + let b = b_find(out_keys[i]); + for (var w = 0u; w < 16u; w = w + 1u) { + var m = 0u; + if (a != NOT_FOUND) { + m = m | a_masks[(a * 16u) + w]; + } + if (b != NOT_FOUND) { + m = m | b_masks[(b * 16u) + w]; + } + out_masks[(i * 16u) + w] = m; + } +} + +// Boolean of voxel n of leaf `leaf`. a and b are the leaf's indices in +// the two grids, or NOT_FOUND. +fn csgValue(leaf: vec3, a: u32, b: u32, n: u32) -> f32 { + let p = (leaf << vec3(3u)) + voxelLocal(n); + var va: f32; + if (a != NOT_FOUND) { + va = a_leafValue(a, n); + } else { + va = a_valueAt(p); + } + var vb: f32; + if (b != NOT_FOUND) { + vb = b_leafValue(b, n); + } else { + vb = b_valueAt(p); + } + if (params.op == 0u) { + return min(va, vb); + } + if (params.op == 1u) { + return max(va, vb); + } + return max(va, -vb); +} + +@compute @workgroup_size(256) +fn csg_mark(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (markSentinel(i, params.out_count)) { + return; + } + let key = out_keys[i]; + let leaf = unpack(key); + let a = a_find(key); + let b = b_find(key); + var acc: LeafAcc; + for (var n = 0u; n < 512u; n = n + 1u) { + accPush(&acc, i, n, csgValue(leaf, a, b, n)); + } + accFinish(&acc, i); +} + +@compute @workgroup_size(256) +fn csg_apply(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j >= arrayLength(&w_keys)) { + return; + } + let key = w_keys[j]; + let leaf = unpack(key); + let a = a_find(key); + let b = b_find(key); + let base = w_leaves[(j * LEAF_U32) + 2u]; + var k = 0u; + for (var w = 0u; w < 16u; w = w + 1u) { + let band = bandWord(j, w); + for (var b_ = 0u; b_ < 32u; b_ = b_ + 1u) { + if (((band >> b_) & 1u) == 0u) { + continue; + } + w_data[base + k] = bitcast(csgValue(leaf, a, b, (w * 32u) + b_)); + k = k + 1u; + } + } +} diff --git a/wgsl/mesh_to_grid.wgsl b/wgsl/mesh_to_grid.wgsl new file mode 100644 index 0000000..cb411b3 --- /dev/null +++ b/wgsl/mesh_to_grid.wgsl @@ -0,0 +1,128 @@ +// Bins triangles into the 8^3 leaf blocks their dilated bounding boxes +// touch. +// +// The bounds match rasterizeTriangle in src/mesh_to_grid.zig. A triangle +// covers the voxels inside its index space bounds dilated by the half +// width and touches the leaves containing them. count_pairs counts leaves +// per triangle, the host scans the counts into write offsets, emit_pairs +// writes (leaf key, triangle) pairs, radix sort orders them by key, and +// mark_unique plus compact_unique build the deduplicated leaf table. +// +// A leaf key packs the leaf coordinate relative to leaf_min with 10 bits +// per axis, so grids may span up to 1024 leaves per axis. The host +// validates the range. + +struct BinParams { + point_count: u32, + triangle_count: u32, + inv_voxel_size: f32, + half_width: f32, + leaf_min: vec3, + pair_count: u32, +} + +@group(0) @binding(0) var params: BinParams; +@group(0) @binding(1) var points_world: array; +@group(0) @binding(2) var points_index: array; +@group(0) @binding(3) var triangles: array; +@group(0) @binding(4) var counts: array; +@group(0) @binding(5) var pair_keys: array; +@group(0) @binding(6) var pair_tris: array; +@group(0) @binding(7) var flags: array; +@group(0) @binding(8) var unique_keys: array; + +const DISPATCH_STRIDE: u32 = 65535u; + +fn globalIndex(wid: vec3, lid: vec3) -> u32 { + return (((wid.y * DISPATCH_STRIDE) + wid.x) * 256u) + lid.x; +} + +@compute @workgroup_size(256) +fn transform_points(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i < (params.point_count * 3u)) { + points_index[i] = points_world[i] * params.inv_voxel_size; + } +} + +fn loadPoint(i: u32) -> vec3 { + return vec3(points_index[i * 3u], points_index[(i * 3u) + 1u], points_index[(i * 3u) + 2u]); +} + +struct LeafRange { + lo: vec3, + hi: vec3, +} + +fn leafRange(t: u32) -> LeafRange { + let a = loadPoint(triangles[t * 3u]); + let b = loadPoint(triangles[(t * 3u) + 1u]); + let c = loadPoint(triangles[(t * 3u) + 2u]); + let hw = vec3(params.half_width); + let lo = vec3(ceil(min(a, min(b, c)) - hw)); + let hi = vec3(floor(max(a, max(b, c)) + hw)); + return LeafRange(lo >> vec3(3u), hi >> vec3(3u)); +} + +// counts has one extra entry so the scanned total lands in the last slot. +@compute @workgroup_size(256) +fn count_pairs(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let t = globalIndex(wid, lid); + if (t > params.triangle_count) { + return; + } + if (t == params.triangle_count) { + counts[t] = 0u; + return; + } + let r = leafRange(t); + let span = max((r.hi - r.lo) + vec3(1), vec3(0)); + counts[t] = u32(span.x * span.y * span.z); +} + +// counts now holds the scanned write offsets. +@compute @workgroup_size(256) +fn emit_pairs(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let t = globalIndex(wid, lid); + if (t >= params.triangle_count) { + return; + } + let r = leafRange(t); + var w = counts[t]; + for (var x = r.lo.x; x <= r.hi.x; x = x + 1) { + for (var y = r.lo.y; y <= r.hi.y; y = y + 1) { + for (var z = r.lo.z; z <= r.hi.z; z = z + 1) { + let rel = vec3(vec3(x, y, z) - params.leaf_min); + pair_keys[w] = (rel.x << 20u) | (rel.y << 10u) | rel.z; + pair_tris[w] = t; + w = w + 1u; + } + } + } +} + +// flags has one extra entry so the scanned total lands in the last slot. +@compute @workgroup_size(256) +fn mark_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.pair_count) { + return; + } + if (i == params.pair_count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, i == 0u || pair_keys[i] != pair_keys[i - 1u]); +} + +// flags now holds the scanned unique positions. +@compute @workgroup_size(256) +fn compact_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.pair_count) { + return; + } + if (i == 0u || pair_keys[i] != pair_keys[i - 1u]) { + unique_keys[flags[i]] = pair_keys[i]; + } +} diff --git a/wgsl/prune.wgsl b/wgsl/prune.wgsl new file mode 100644 index 0000000..a7eb611 --- /dev/null +++ b/wgsl/prune.wgsl @@ -0,0 +1,57 @@ +// ANDs each leaf's mask with a retain mask and drops leaves left empty, +// compacting the leaf table. The host scans flags between mark and +// compact. + +struct PruneParams { + count: u32, +} + +@group(0) @binding(0) var params: PruneParams; +@group(0) @binding(1) var keys: array; +@group(0) @binding(2) var masks: array; +@group(0) @binding(3) var retain: array; +@group(0) @binding(4) var flags: array; +@group(0) @binding(5) var out_keys: array; +@group(0) @binding(6) var out_masks: array; + +const DISPATCH_STRIDE: u32 = 65535u; + +fn globalIndex(wid: vec3, lid: vec3) -> u32 { + return (((wid.y * DISPATCH_STRIDE) + wid.x) * 256u) + lid.x; +} + +fn survives(i: u32) -> bool { + var any = 0u; + for (var w = 0u; w < 16u; w = w + 1u) { + any = any | (masks[(i * 16u) + w] & retain[(i * 16u) + w]); + } + return any != 0u; +} + +// flags has one extra entry so the scanned total lands in the last slot. +@compute @workgroup_size(256) +fn mark(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.count) { + return; + } + if (i == params.count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, survives(i)); +} + +// flags now holds the scanned output positions. +@compute @workgroup_size(256) +fn compact(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.count || !survives(i)) { + return; + } + let o = flags[i]; + out_keys[o] = keys[i]; + for (var w = 0u; w < 16u; w = w + 1u) { + out_masks[(o * 16u) + w] = masks[(i * 16u) + w] & retain[(i * 16u) + w]; + } +} diff --git a/wgsl/radix_sort.wgsl b/wgsl/radix_sort.wgsl new file mode 100644 index 0000000..2e3972f --- /dev/null +++ b/wgsl/radix_sort.wgsl @@ -0,0 +1,115 @@ +// Stable LSD radix sort of u32 keys with u32 payloads. Eight passes of 4 +// bit digits ping pong between two key and payload buffer pairs. +// +// Each pass runs histogram, then the scan.wgsl exclusive scan over hist +// (digit major layout, so scanned counts become global scatter bases), +// then scatter, which ranks each tile's items stably in workgroup memory +// and writes them out. The host side lives in ts/gpu/radix_sort.ts. + +const WG_SIZE: u32 = 256u; +const ITEMS: u32 = 4u; +const TILE: u32 = 1024u; +const RADIX: u32 = 16u; +const NO_ITEM: u32 = 0xffffffffu; + +struct SortParams { + n: u32, + shift: u32, + num_tiles: u32, +} + +@group(0) @binding(0) var params: SortParams; +@group(0) @binding(1) var keys_in: array; +@group(0) @binding(2) var vals_in: array; +@group(0) @binding(3) var keys_out: array; +@group(0) @binding(4) var vals_out: array; +@group(0) @binding(5) var hist: array; + +var counts: array, RADIX>; + +@compute @workgroup_size(256) +fn histogram( + @builtin(local_invocation_id) lid: vec3, + @builtin(workgroup_id) wid: vec3, +) { + if (lid.x < RADIX) { + atomicStore(&counts[lid.x], 0u); + } + workgroupBarrier(); + let base = (wid.x * TILE) + (lid.x * ITEMS); + for (var i = 0u; i < ITEMS; i = i + 1u) { + if ((base + i) < params.n) { + let d = (keys_in[base + i] >> params.shift) & (RADIX - 1u); + atomicAdd(&counts[d], 1u); + } + } + workgroupBarrier(); + if (lid.x < RADIX) { + hist[(lid.x * params.num_tiles) + wid.x] = atomicLoad(&counts[lid.x]); + } +} + +var scan_buf: array; + +@compute @workgroup_size(256) +fn scatter( + @builtin(local_invocation_id) lid: vec3, + @builtin(workgroup_id) wid: vec3, +) { + let tid = lid.x; + let base = (wid.x * TILE) + (tid * ITEMS); + + // Thread t owns ITEMS consecutive elements, so thread order then item + // order is the stable order within a tile. + var k: array; + var v: array; + var d: array; + var cnt: array; + for (var b = 0u; b < RADIX; b = b + 1u) { + cnt[b] = 0u; + } + for (var i = 0u; i < ITEMS; i = i + 1u) { + d[i] = NO_ITEM; + if ((base + i) < params.n) { + k[i] = keys_in[base + i]; + v[i] = vals_in[base + i]; + let digit = (k[i] >> params.shift) & (RADIX - 1u); + d[i] = digit; + cnt[digit] = cnt[digit] + 1u; + } + } + + // For each digit an exclusive scan across the workgroup gives each + // thread the count of matching items in earlier threads. + var thread_base: array; + for (var b = 0u; b < RADIX; b = b + 1u) { + scan_buf[tid] = cnt[b]; + workgroupBarrier(); + for (var offset = 1u; offset < WG_SIZE; offset = offset << 1u) { + var s = scan_buf[tid]; + if (tid >= offset) { + s = s + scan_buf[tid - offset]; + } + workgroupBarrier(); + scan_buf[tid] = s; + workgroupBarrier(); + } + thread_base[b] = scan_buf[tid] - cnt[b]; + workgroupBarrier(); + } + + // hist now holds globally scanned bases. cnt is reused to count this + // thread's items already placed per digit. + for (var b = 0u; b < RADIX; b = b + 1u) { + cnt[b] = 0u; + } + for (var i = 0u; i < ITEMS; i = i + 1u) { + if (d[i] != NO_ITEM) { + let digit = d[i]; + let dst = hist[(digit * params.num_tiles) + wid.x] + thread_base[digit] + cnt[digit]; + keys_out[dst] = k[i]; + vals_out[dst] = v[i]; + cnt[digit] = cnt[digit] + 1u; + } + } +} diff --git a/wgsl/rasterize.wgsl b/wgsl/rasterize.wgsl new file mode 100644 index 0000000..b616b80 --- /dev/null +++ b/wgsl/rasterize.wgsl @@ -0,0 +1,179 @@ +// Computes the narrow band squared distances for binned leaves. +// +// One workgroup per leaf walks its contiguous run of the key sorted pair +// list, each thread covers two of the 512 voxels, and minima accumulate in +// a workgroup memory slab written back with plain stores. The workgroup +// owns its leaf, so there are no global atomics and no slot searches. +// +// Distances are nonnegative, so the u32 bit pattern of an f32 orders the +// same as the float value and atomicMin on the bitcast is an exact float +// min. +// +// The distance function mirrors distSqPointTriangle in src/mesh_to_grid.zig +// with an explicit summation order. A voxel updates only when it lies in +// the triangle's dilated bounds and its squared distance is within the +// squared half width, the same gate the CPU uses. + +struct RasterParams { + pair_count: u32, + leaf_count: u32, + half_width: f32, + pad: u32, + leaf_min: vec3, + pad2: u32, +} + +@group(0) @binding(0) var params: RasterParams; +@group(0) @binding(1) var points_index: array; +@group(0) @binding(2) var triangles: array; +@group(0) @binding(3) var pair_keys: array; +@group(0) @binding(4) var pair_tris: array; +@group(0) @binding(5) var leaf_keys: array; +@group(0) @binding(6) var leaf_values: array; + +const INF_BITS: u32 = 0x7f800000u; +const DISPATCH_STRIDE: u32 = 65535u; + +fn dot3(a: vec3, b: vec3) -> f32 { + return ((a.x * b.x) + (a.y * b.y)) + (a.z * b.z); +} + +fn dsq(a: vec3) -> f32 { + return dot3(a, a); +} + +fn distSqPointSegment(p: vec3, a: vec3, b: vec3) -> f32 { + let ab = b - a; + let denom = dsq(ab); + if (denom <= 0.0) { + return dsq(p - a); + } + let t = clamp(dot3(p - a, ab) / denom, 0.0, 1.0); + return dsq(p - (a + (ab * t))); +} + +fn distSqPointTriangle(p: vec3, a: vec3, b: vec3, c: vec3) -> f32 { + let ab = b - a; + let ac = c - a; + let ap = p - a; + let d1 = dot3(ab, ap); + let d2 = dot3(ac, ap); + if (d1 <= 0.0 && d2 <= 0.0) { + return dsq(ap); // vertex a + } + + let bp = p - b; + let d3 = dot3(ab, bp); + let d4 = dot3(ac, bp); + if (d3 >= 0.0 && d4 <= d3) { + return dsq(bp); // vertex b + } + + let vc = (d1 * d4) - (d3 * d2); + if (vc <= 0.0 && d1 >= 0.0 && d3 <= 0.0) { + let denom = d1 - d3; + if (denom > 0.0) { + return dsq(ap - (ab * (d1 / denom))); // edge ab + } + } + + let cp = p - c; + let d5 = dot3(ab, cp); + let d6 = dot3(ac, cp); + if (d6 >= 0.0 && d5 <= d6) { + return dsq(cp); // vertex c + } + + let vb = (d5 * d2) - (d1 * d6); + if (vb <= 0.0 && d2 >= 0.0 && d6 <= 0.0) { + let denom = d2 - d6; + if (denom > 0.0) { + return dsq(ap - (ac * (d2 / denom))); // edge ac + } + } + + let va = (d3 * d6) - (d5 * d4); + if (va <= 0.0 && (d4 - d3) >= 0.0 && (d5 - d6) >= 0.0) { + let denom = (d4 - d3) + (d5 - d6); + if (denom > 0.0) { + return dsq(bp - ((c - b) * ((d4 - d3) / denom))); // edge bc + } + } + + let denom = (va + vb) + vc; + if (denom <= 0.0) { + // Degenerate triangles fall back to edge distances. + let e0 = distSqPointSegment(p, a, b); + let e1 = distSqPointSegment(p, a, c); + let e2 = distSqPointSegment(p, b, c); + return min(e0, min(e1, e2)); + } + let inv = 1.0 / denom; + let v = vb * inv; + let w = vc * inv; + return dsq(ap - ((ab * v) + (ac * w))); // face +} + +fn loadPoint(i: u32) -> vec3 { + return vec3(points_index[i * 3u], points_index[(i * 3u) + 1u], points_index[(i * 3u) + 2u]); +} + +var slab: array, 512>; + +@compute @workgroup_size(256) +fn rasterize( + @builtin(workgroup_id) wid: vec3, + @builtin(local_invocation_id) lid: vec3, +) { + let i = (wid.y * DISPATCH_STRIDE) + wid.x; + if (i >= params.leaf_count) { + return; + } + let key = leaf_keys[i]; + let leaf = vec3(i32(key >> 20u), i32((key >> 10u) & 0x3ffu), i32(key & 0x3ffu)) + params.leaf_min; + let origin = leaf << vec3(3u); + + atomicStore(&slab[lid.x], INF_BITS); + atomicStore(&slab[lid.x + 256u], INF_BITS); + workgroupBarrier(); + + // Find this leaf's run in the key sorted pair list. + var lo = 0u; + var hi = params.pair_count; + while (lo < hi) { + let mid = (lo + hi) >> 1u; + if (pair_keys[mid] < key) { + lo = mid + 1u; + } else { + hi = mid; + } + } + + let hw = params.half_width; + let hw2 = hw * hw; + for (var p = lo; p < params.pair_count && pair_keys[p] == key; p = p + 1u) { + let t = pair_tris[p]; + let a = loadPoint(triangles[t * 3u]); + let b = loadPoint(triangles[(t * 3u) + 1u]); + let c = loadPoint(triangles[(t * 3u) + 2u]); + let lo_v = vec3(ceil(min(a, min(b, c)) - vec3(hw))); + let hi_v = vec3(floor(max(a, max(b, c)) + vec3(hw))); + + for (var n = lid.x; n < 512u; n = n + 256u) { + // Voxel order matches picovdb leafCoordToOffset. + let local = vec3(i32(n >> 6u), i32((n >> 3u) & 7u), i32(n & 7u)); + let ijk = origin + local; + if (any(ijk < lo_v) || any(ijk > hi_v)) { + continue; + } + let d2 = distSqPointTriangle(vec3(ijk), a, b, c); + if (d2 <= hw2) { + atomicMin(&slab[n], bitcast(d2)); + } + } + } + + workgroupBarrier(); + leaf_values[(i * 512u) + lid.x] = atomicLoad(&slab[lid.x]); + leaf_values[(i * 512u) + lid.x + 256u] = atomicLoad(&slab[lid.x + 256u]); +} diff --git a/wgsl/remap.wgsl b/wgsl/remap.wgsl new file mode 100644 index 0000000..337611f --- /dev/null +++ b/wgsl/remap.wgsl @@ -0,0 +1,180 @@ +// Remaps a grid into a new grid by a per voxel rule. Two modes: an SDF +// offset, and a translation by whole voxels. Only candidate leaves that +// end up with band voxels are kept. +// +// The old grid binds as old_keys, old_leaves, old_data. The host supplies +// the candidate leaves. For a translation they are the old leaves and +// their neighbors, deduped here. For an offset they are the leaves of the +// distance table. +// +// An offset uses only the old zero level set. The host extracts it +// (extract.wgsl) and rasterizes exact distances to its triangles out to +// half_width + |amount| (mesh_to_grid.wgsl, rasterize.wgsl). The new value +// is that distance, signed by the old grid, minus the amount. Stored +// values away from the surface are not used: files and earlier ops may +// carry them with less accuracy. The error is the marching cubes chord. +// The host splits large offsets into steps to bound the reach. + +struct RemapParams { + old_count: u32, + concat_count: u32, + new_count: u32, + mode: u32, // 0 offset, 1 translate + shift: vec3, // translate, voxels + amount: f32, // offset, voxels + nb_lo: vec3, // neighborhood box, leaf units + half_width: f32, + nb_dims: vec3, + dist_count: u32, // offset, leaves in the distance table + delta: vec3, // rebase, leaf units + pad1: u32, +} + +@group(0) @binding(0) var params: RemapParams; +@group(0) @binding(1) var old_keys: array; +@group(0) @binding(2) var old_leaves: array; +@group(0) @binding(3) var concat_keys: array; +@group(0) @binding(4) var flags: array; +@group(0) @binding(5) var new_keys: array; +@group(0) @binding(6) var old_data: array; +// Offset only. new_keys is the sorted leaf table of the distances, +// dist_count long. dist_values holds their squared distances as f32 bits, +// INF_BITS beyond the reach. +@group(0) @binding(9) var dist_values: array; + +const INF_BITS: u32 = 0x7f800000u; + +// Shifts every key by the same delta, so the order holds. +@compute @workgroup_size(256) +fn rebase(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.old_count) { + return; + } + new_keys[i] = pack(unpack(old_keys[i]) + params.delta); +} + +// One candidate per old leaf and neighborhood cell, after the old keys. +// Cells outside the key range repeat the leaf's own key. +@compute @workgroup_size(256) +fn generate_candidates(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + let vol = u32(params.nb_dims.x * params.nb_dims.y * params.nb_dims.z); + if (i >= params.old_count * vol) { + return; + } + let leaf = unpack(old_keys[i / vol]); + let j = i % vol; + let dy = u32(params.nb_dims.y); + let dz = u32(params.nb_dims.z); + let c = leaf + params.nb_lo + vec3(i32(j / (dy * dz)), i32((j / dz) % dy), i32(j % dz)); + var key = old_keys[i / vol]; + if (all(c >= vec3(0)) && all(c <= vec3(1023))) { + key = pack(c); + } + concat_keys[params.old_count + i] = key; +} + +// flags has one extra entry for the scan total. +@compute @workgroup_size(256) +fn mark_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.concat_count) { + return; + } + if (i == params.concat_count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, i == 0u || concat_keys[i] != concat_keys[i - 1u]); +} + +// flags now holds the scanned unique positions. +@compute @workgroup_size(256) +fn compact_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.concat_count) { + return; + } + if (i == 0u || concat_keys[i] != concat_keys[i - 1u]) { + new_keys[flags[i]] = concat_keys[i]; + } +} + +// Squared distance to the old surface at a relative voxel coordinate, as +// f32 bits. INF_BITS beyond the reach. +fn distSqBitsAt(ijk: vec3) -> u32 { + let leaf = ijk >> vec3(3u); + if (any(leaf < vec3(0)) || any(leaf > vec3(1023))) { + return INF_BITS; + } + let key = pack(leaf); + var lo = 0u; + var hi = params.dist_count; + while (lo < hi) { + let mid = (lo + hi) >> 1u; + if (new_keys[mid] < key) { + lo = mid + 1u; + } else { + hi = mid; + } + } + if (lo < params.dist_count && new_keys[lo] == key) { + return dist_values[(lo * 512u) + voxelOffset(ijk)]; + } + return INF_BITS; +} + +// New value at a relative voxel coordinate. +fn newValueAt(p: vec3) -> f32 { + let hw = params.half_width; + var v: f32; + if (params.mode == 0u) { + let s = select(1.0, -1.0, old_valueAt(p) < 0.0); + let d2 = distSqBitsAt(p); + if (d2 == INF_BITS) { + v = s * hw; + } else { + v = (s * sqrt(bitcast(d2))) - params.amount; + } + } else { + v = old_valueAt(p - params.shift); + } + return clamp(v, -hw, hw); +} + +// The writer passes. See the writer in opgrid.ts. +@compute @workgroup_size(256) +fn mark(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (markSentinel(i, params.new_count)) { + return; + } + let origin = unpack(new_keys[i]) << vec3(3u); + var acc: LeafAcc; + for (var n = 0u; n < 512u; n = n + 1u) { + accPush(&acc, i, n, newValueAt(origin + voxelLocal(n))); + } + accFinish(&acc, i); +} + +@compute @workgroup_size(256) +fn apply(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j >= arrayLength(&w_keys)) { + return; + } + let origin = unpack(w_keys[j]) << vec3(3u); + let base = w_leaves[(j * LEAF_U32) + 2u]; + var k = 0u; + for (var w = 0u; w < 16u; w = w + 1u) { + let band = bandWord(j, w); + for (var b = 0u; b < 32u; b = b + 1u) { + if (((band >> b) & 1u) == 0u) { + continue; + } + w_data[base + k] = bitcast(newValueAt(origin + voxelLocal((w * 32u) + b))); + k = k + 1u; + } + } +} diff --git a/wgsl/scan.wgsl b/wgsl/scan.wgsl new file mode 100644 index 0000000..6c33c75 --- /dev/null +++ b/wgsl/scan.wgsl @@ -0,0 +1,78 @@ +// Exclusive prefix scan of u32 values with wrapping addition. +// +// scan_tile writes each 1024 element tile's exclusive scan in place and the +// tile total to partials. The host scans partials recursively with the same +// kernels and add_offsets folds the scanned totals back in. The dispatch +// plan lives in ts/gpu/scan.ts. + +const WG_SIZE: u32 = 256u; +const ITEMS: u32 = 4u; +const TILE: u32 = 1024u; + +struct ScanParams { + n: u32, +} + +@group(0) @binding(0) var params: ScanParams; +@group(0) @binding(1) var data: array; +@group(0) @binding(2) var partials: array; + +var thread_sums: array; + +@compute @workgroup_size(256) +fn scan_tile( + @builtin(local_invocation_id) lid: vec3, + @builtin(workgroup_id) wid: vec3, +) { + let tid = lid.x; + let base = (wid.x * TILE) + (tid * ITEMS); + + var v: array; + var sum = 0u; + for (var i = 0u; i < ITEMS; i = i + 1u) { + var x = 0u; + if ((base + i) < params.n) { + x = data[base + i]; + } + v[i] = x; + sum = sum + x; + } + + thread_sums[tid] = sum; + workgroupBarrier(); + // Inclusive scan over the 256 thread sums. + for (var offset = 1u; offset < WG_SIZE; offset = offset << 1u) { + var s = thread_sums[tid]; + if (tid >= offset) { + s = s + thread_sums[tid - offset]; + } + workgroupBarrier(); + thread_sums[tid] = s; + workgroupBarrier(); + } + + var running = thread_sums[tid] - sum; // exclusive offset of this thread + for (var i = 0u; i < ITEMS; i = i + 1u) { + if ((base + i) < params.n) { + data[base + i] = running; + } + running = running + v[i]; + } + if (tid == (WG_SIZE - 1u)) { + partials[wid.x] = thread_sums[tid]; + } +} + +@compute @workgroup_size(256) +fn add_offsets( + @builtin(local_invocation_id) lid: vec3, + @builtin(workgroup_id) wid: vec3, +) { + let offset = partials[wid.x]; + let base = (wid.x * TILE) + (lid.x * ITEMS); + for (var i = 0u; i < ITEMS; i = i + 1u) { + if ((base + i) < params.n) { + data[base + i] = data[base + i] + offset; + } + } +} diff --git a/wgsl/sign.wgsl b/wgsl/sign.wgsl new file mode 100644 index 0000000..78483f8 --- /dev/null +++ b/wgsl/sign.wgsl @@ -0,0 +1,203 @@ +// Computes inside or outside parity per voxel from vertical ray +// crossings, mirroring ColumnGrid in src/mesh_to_grid.zig. +// +// Each triangle records a surface crossing height for every lattice column +// its XY projection covers. The tie break matches the CPU, so an edge +// shared by two triangles counts exactly once and vertical triangles +// contribute nothing. The host scans the counts, sorts the crossings by +// column then height with two stable radix passes, and sign_leaves walks +// each candidate leaf column counting crossings below each voxel center. +// An odd count means inside. +// +// The CPU computes crossings in f64. WGSL has no f64, so this runs in f32. +// All triangles evaluate identical expressions, which keeps the parity +// consistent. Only voxels whose column crossing sits within f32 noise of +// their center plane can differ from the CPU, and those lie on the +// surface. + +struct SignParams { + triangle_count: u32, + crossing_count: u32, + min_x: i32, + min_y: i32, + nx: u32, + ny: u32, + leaf_count: u32, + pad: u32, + leaf_min: vec3, + pad2: u32, +} + +@group(0) @binding(0) var params: SignParams; +@group(0) @binding(1) var points_index: array; +@group(0) @binding(2) var triangles: array; +@group(0) @binding(3) var counts: array; +@group(0) @binding(4) var cross_cols: array; +@group(0) @binding(5) var cross_z: array; +@group(0) @binding(6) var leaf_keys: array; +@group(0) @binding(7) var inside: array>; + +const DISPATCH_STRIDE: u32 = 65535u; + +fn globalIndex(wid: vec3, lid: vec3) -> u32 { + return (((wid.y * DISPATCH_STRIDE) + wid.x) * 256u) + lid.x; +} + +fn loadPoint(i: u32) -> vec3 { + return vec3(points_index[i * 3u], points_index[(i * 3u) + 1u], points_index[(i * 3u) + 2u]); +} + +fn edgeFn(px: f32, py: f32, qx: f32, qy: f32, sx: f32, sy: f32) -> f32 { + return ((qx - px) * (sy - py)) - ((qy - py) * (sx - px)); +} + +fn accept(w: f32, ex: f32, ey: f32) -> bool { + if (w > 0.0) { + return true; + } + if (w < 0.0) { + return false; + } + return ey < 0.0 || (ey == 0.0 && ex > 0.0); +} + +// Monotone f32 to u32 transform so crossing order survives the u32 sort. +// Negative zero maps to the positive zero key, matching the CPU float +// compare where the two are equal. +fn sortableFromF32(v: f32) -> u32 { + let b = bitcast(v); + if (b == 0x80000000u) { + return 0x80000000u; + } + if ((b >> 31u) == 1u) { + return ~b; + } + return b | 0x80000000u; +} + +// Shared by the count and emit passes, which must agree. +fn binTriangle(t: u32, emit: bool, offset: u32) -> u32 { + let a = loadPoint(triangles[t * 3u]); + let b = loadPoint(triangles[(t * 3u) + 1u]); + let c = loadPoint(triangles[(t * 3u) + 2u]); + + let signed_area = edgeFn(a.x, a.y, b.x, b.y, c.x, c.y); + if (signed_area == 0.0) { + return 0u; // vertical triangle + } + var flip = 1.0; + if (signed_area < 0.0) { + flip = -1.0; + } + let area = flip * signed_area; + + let x0 = i32(ceil(min(a.x, min(b.x, c.x)))); + let x1 = i32(floor(max(a.x, max(b.x, c.x)))); + let y0 = i32(ceil(min(a.y, min(b.y, c.y)))); + let y1 = i32(floor(max(a.y, max(b.y, c.y)))); + + var w = offset; + var n = 0u; + for (var x = x0; x <= x1; x = x + 1) { + for (var y = y0; y <= y1; y = y + 1) { + let px = f32(x); + let py = f32(y); + let w0 = flip * edgeFn(b.x, b.y, c.x, c.y, px, py); + let w1 = flip * edgeFn(c.x, c.y, a.x, a.y, px, py); + let w2 = flip * edgeFn(a.x, a.y, b.x, b.y, px, py); + let ins = accept(w0, flip * (c.x - b.x), flip * (c.y - b.y)) + && accept(w1, flip * (a.x - c.x), flip * (a.y - c.y)) + && accept(w2, flip * (b.x - a.x), flip * (b.y - a.y)); + if (!ins) { + continue; + } + if (x < params.min_x || y < params.min_y) { + continue; + } + let ix = u32(x - params.min_x); + let iy = u32(y - params.min_y); + if (ix >= params.nx || iy >= params.ny) { + continue; + } + if (emit) { + let z = (((w0 * a.z) + (w1 * b.z)) + (w2 * c.z)) / area; + cross_cols[w] = (ix * params.ny) + iy; + cross_z[w] = sortableFromF32(z); + w = w + 1u; + } + n = n + 1u; + } + } + return n; +} + +// counts has one extra entry so the scanned total lands in the last slot. +@compute @workgroup_size(256) +fn count_crossings(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let t = globalIndex(wid, lid); + if (t > params.triangle_count) { + return; + } + if (t == params.triangle_count) { + counts[t] = 0u; + return; + } + counts[t] = binTriangle(t, false, 0u); +} + +// counts now holds the scanned write offsets. +@compute @workgroup_size(256) +fn emit_crossings(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let t = globalIndex(wid, lid); + if (t < params.triangle_count) { + let unused = binTriangle(t, true, counts[t]); + } +} + +// One workgroup per candidate leaf and one thread per column. Crossings +// arrive sorted by column then height. Each thread ORs the parity bits of +// its column's 8 voxels into the zero initialized mask. +@compute @workgroup_size(64) +fn sign_leaves( + @builtin(workgroup_id) wid: vec3, + @builtin(local_invocation_id) lid: vec3, +) { + let leaf_i = (wid.y * DISPATCH_STRIDE) + wid.x; + if (leaf_i >= params.leaf_count) { + return; + } + let key = leaf_keys[leaf_i]; + let leaf = vec3(i32(key >> 20u), i32((key >> 10u) & 0x3ffu), i32(key & 0x3ffu)) + params.leaf_min; + let origin = leaf << vec3(3u); + + let x = origin.x + i32(lid.x >> 3u); + let y = origin.y + i32(lid.x & 7u); + let col = (u32(x - params.min_x) * params.ny) + u32(y - params.min_y); + + // Lower bound of this column's crossing run. + var lo = 0u; + var hi = params.crossing_count; + while (lo < hi) { + let mid = (lo + hi) >> 1u; + if (cross_cols[mid] < col) { + lo = mid + 1u; + } else { + hi = mid; + } + } + + var i = lo; + var count = 0u; + var bits = 0u; + for (var z = 0u; z < 8u; z = z + 1u) { + let voxel_z = sortableFromF32(f32(origin.z + i32(z))); + while (i < params.crossing_count && cross_cols[i] == col && cross_z[i] < voxel_z) { + i = i + 1u; + count = count + 1u; + } + bits = bits | ((count & 1u) << z); + } + // This column's 8 voxels form one byte of the leaf mask. + let n0 = lid.x << 3u; + atomicOr(&inside[(leaf_i * 16u) + (n0 >> 5u)], bits << (n0 & 31u)); +} diff --git a/wgsl/stamp.wgsl b/wgsl/stamp.wgsl new file mode 100644 index 0000000..26d6162 --- /dev/null +++ b/wgsl/stamp.wgsl @@ -0,0 +1,149 @@ +// Stamps a shape into a grid. The shape is a WGSL signed distance +// function, sdf(p), in absolute voxels; the host generates it from the +// shape library and the shape's name. Add takes the min with the shape's +// distance. Carve takes the max with its negation. Stamping an empty grid +// makes the shape. +// +// The old grid binds as old_keys, old_leaves, old_data. args holds the +// shape's arguments. The candidates are the old leaves plus the leaves +// the shape's shell crosses, deduped here. Voxels read the old leaf where +// one exists and the implicit background elsewhere, so a carve through a +// leafless interior forms a correct band. + +struct StampParams { + old_count: u32, + concat_count: u32, + new_count: u32, + mode: u32, // 0 adds material and 1 carves + origin: vec3, // voxel origin of the key space, absolute voxels + half_width: f32, + box_lo: vec3, // candidate box, relative leaf coords + pad0: u32, + box_dims: vec3, + pad1: u32, +} + +@group(0) @binding(0) var params: StampParams; +@group(0) @binding(1) var old_keys: array; +@group(0) @binding(2) var old_leaves: array; +@group(0) @binding(3) var concat_keys: array; +@group(0) @binding(4) var flags: array; +@group(0) @binding(5) var new_keys: array; +@group(0) @binding(6) var old_data: array; +@group(0) @binding(7) var args: array, 8>; // the shape's arguments + +// Candidate leaves go after the old keys in concat_keys. Only leaves the +// shell crosses can gain band voxels. A leaf fully inside or outside the +// shape keeps its old band, and old leaves are candidates already. The +// rest repeat the box corner key, which the dedupe drops. +@compute @workgroup_size(256) +fn generate_candidates(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + let vol = u32(params.box_dims.x * params.box_dims.y * params.box_dims.z); + if (i >= vol) { + return; + } + let dy = u32(params.box_dims.y); + let dz = u32(params.box_dims.z); + let c = params.box_lo + vec3(i32(i / (dy * dz)), i32((i / dz) % dy), i32(i % dz)); + let center = vec3(c << vec3(3u)) + vec3(3.5); + var key = pack(params.box_lo); + if (abs(shapeDistance(center)) < params.half_width + 6.1) { // 3.5 * sqrt(3) to any voxel of the leaf + key = pack(c); + } + concat_keys[params.old_count + i] = key; +} + +// flags has one extra entry for the scan total. +@compute @workgroup_size(256) +fn mark_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i > params.concat_count) { + return; + } + if (i == params.concat_count) { + flags[i] = 0u; + return; + } + flags[i] = select(0u, 1u, i == 0u || concat_keys[i] != concat_keys[i - 1u]); +} + +// flags now holds the scanned unique positions. +@compute @workgroup_size(256) +fn compact_unique(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (i >= params.concat_count) { + return; + } + if (i == 0u || concat_keys[i] != concat_keys[i - 1u]) { + new_keys[flags[i]] = concat_keys[i]; + } +} + +// Shape distance at a relative voxel position. +fn shapeDistance(p: vec3) -> f32 { + return sdf(p + params.origin); +} + +// New value of voxel n of leaf `leaf`. old is the leaf's index in the old +// grid, or NOT_FOUND. +fn newValue(leaf: vec3, old: u32, n: u32) -> f32 { + let p = (leaf << vec3(3u)) + voxelLocal(n); + var v: f32; + if (old != NOT_FOUND) { + v = old_leafValue(old, n); + } else { + v = old_valueAt(p); + } + // Leaves outside the candidate box cannot change. + let box_hi = params.box_lo + params.box_dims - vec3(1); + if (any(leaf < params.box_lo) || any(leaf > box_hi)) { + return v; + } + let d = shapeDistance(vec3(p)); + if (params.mode == 0u) { + v = min(v, d); + } else { + v = max(v, -d); + } + return clamp(v, -params.half_width, params.half_width); +} + +@compute @workgroup_size(256) +fn mark(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let i = globalIndex(wid, lid); + if (markSentinel(i, params.new_count)) { + return; + } + let key = new_keys[i]; + let leaf = unpack(key); + let old = old_find(key); + var acc: LeafAcc; + for (var n = 0u; n < 512u; n = n + 1u) { + accPush(&acc, i, n, newValue(leaf, old, n)); + } + accFinish(&acc, i); +} + +@compute @workgroup_size(256) +fn apply(@builtin(workgroup_id) wid: vec3, @builtin(local_invocation_id) lid: vec3) { + let j = globalIndex(wid, lid); + if (j >= arrayLength(&w_keys)) { + return; + } + let key = w_keys[j]; + let leaf = unpack(key); + let old = old_find(key); + let base = w_leaves[(j * LEAF_U32) + 2u]; + var k = 0u; + for (var w = 0u; w < 16u; w = w + 1u) { + let band = bandWord(j, w); + for (var b = 0u; b < 32u; b = b + 1u) { + if (((band >> b) & 1u) == 0u) { + continue; + } + w_data[base + k] = bitcast(newValue(leaf, old, (w * 32u) + b)); + k = k + 1u; + } + } +}