From 3d368e0079d9054b5d95b45566f60495867bbc5b Mon Sep 17 00:00:00 2001 From: Dion Moult Date: Thu, 28 May 2026 18:04:05 +1000 Subject: [PATCH] wgpu chunks: 3D Morton-code spatial sort (tight voxel chunks) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The previous chunk-plan sorted meshes by lexicographic (z, y, x) centroid — effectively a 1D Z-slab traversal. On a typical IFC building (50 × 50 × 100 m), a 16-MB chunk's 80-ish meshes spanned roughly 50 × 50 × 0.5 m. On a city federation it was much worse: the first chunk grouped ground-floor stuff from every building, spanning the entire scene horizontally. Per-chunk AABBs that wide make frustum / contribution / HiZ rejection useless (every chunk "overlaps the frustum" by virtue of spanning the whole scene). 3D Morton (Z-order) interleaves bits of quantised (x, y, z) centroids, so consecutive items in the sorted order cluster in all 3 axes — chunks become tight 3D voxels of the model. Prerequisite for the contribution-aware eviction priority (task #25) to actually discriminate near and far chunks. 21 bits per axis = ~2 M bins per axis, sub-millimetre precision on a kilometre-scale scene. Both apply paths (streaming and non- streaming) share the same sortMeshIdsByMorton helper. Benchmark unchanged (~47 fps avg, 20 ms cull, 0.3 ms stream) — the distance-based evictor still keys on chunk centres, which moved slightly under Morton but not enough to materially shift residency. The user-visible win comes from the next commit, which switches priority to screen-space contribution × HiZ history — both of which need today's tight AABBs to mean anything. Pixel-identical to non-streaming on basic.ifc on both paths. Co-Authored-By: Claude Opus 4.7 --- src/ifcviewer-wgpu/WgpuModelGpuData.h | 28 +++--- src/ifcviewer-wgpu/WgpuViewportWindow.cpp | 113 ++++++++++++++++++---- 2 files changed, 107 insertions(+), 34 deletions(-) diff --git a/src/ifcviewer-wgpu/WgpuModelGpuData.h b/src/ifcviewer-wgpu/WgpuModelGpuData.h index 03dd33e224..9a6c57dfd5 100644 --- a/src/ifcviewer-wgpu/WgpuModelGpuData.h +++ b/src/ifcviewer-wgpu/WgpuModelGpuData.h @@ -45,19 +45,21 @@ // 1–3 chunks), which is invisible compared to per-frame GPU work. // // At INSTANCED_VERTEX_STRIDE_BYTES = 12 B/vertex this caps a chunk at -// ~11 M vertices. Tuned for the pre-v14 scatter-gather streaming model: -// spatial chunk planning means each chunk's bytes are NOT contiguous in -// the sidecar, so per-load I/O cost scales with mesh count per chunk -// (one fseek+fread per file gap). Bigger chunks = more meshes per -// chunk = more seeks per load, BUT also fewer chunks total = fewer -// loads per frame as orbit shifts the visible set. The latter -// dominates: 128 MB chunks → ~16 chunks per pool → 1-2 loads per -// frame → ~20-30 ms stream cost. Smaller chunks (32 / 8 MB) bring -// finer eviction granularity but explode the loads-per-frame count. -// Sidecar v14 (on-disk spatial reorder) is the proper fix — once -// chunks ARE file-contiguous, the per-mesh seek cost vanishes and we -// can drop the chunk size back to ~8 MB for sharp eviction. -static constexpr uint64_t WGPU_CHUNK_VERTEX_BYTES_LIMIT = 128ull * 1024 * 1024; +// ~1.4 M vertices. 16 MB is the sweet spot once background-thread I/O +// (WgpuStreamingThread) is in place: scatter-gather per-mesh seeks +// happen on the worker, not the render thread, so smaller chunks +// (and thus more per-frame loads as orbit shifts) no longer stall +// rendering. The win is much finer pool-allocation granularity — +// a 3 GB pool fits ~190 chunks vs ~21 at 128 MB — so visible +// geometry is far less likely to get "trapped" behind invisible +// chunkmates. Pre-async this size gave 7 fps (the sync loads blocked +// the render thread); now it's bounded by cull cost not stream cost. +// +// Sidecar v14 (on-disk spatial reorder) would let us go smaller still +// (~4 MB) with single-fread chunk loads, but the difference between +// 16 MB and 4 MB is much smaller than the difference between 128 MB +// and 16 MB. +static constexpr uint64_t WGPU_CHUNK_VERTEX_BYTES_LIMIT = 16ull * 1024 * 1024; struct WgpuModelGpuData { // std430 layout: 16 bytes per entry, naturally aligned. base_vertex is diff --git a/src/ifcviewer-wgpu/WgpuViewportWindow.cpp b/src/ifcviewer-wgpu/WgpuViewportWindow.cpp index 4563ee309d..aa4e866df3 100644 --- a/src/ifcviewer-wgpu/WgpuViewportWindow.cpp +++ b/src/ifcviewer-wgpu/WgpuViewportWindow.cpp @@ -132,6 +132,87 @@ static WGPUBuffer createBufferWithData(WGPUDevice device, WGPUQueue queue, return buf; } +// Interleave the low 21 bits of v with two zero bits between each, +// returning bits at positions 0, 3, 6, ..., 60 — one axis of a +// standard 21-bit-per-axis 3D Morton code. ORing three of these +// shifted by 0, 1, 2 gives a 63-bit (x, y, z)-interleaved code; the +// resulting integer ordering puts spatially-close points close in +// the sorted sequence (the classic Z-order curve). +static uint64_t mortonSplit21(uint32_t v) { + uint64_t r = v & 0x1FFFFFu; + r = (r | r << 32) & 0x001F00000000FFFFULL; + r = (r | r << 16) & 0x001F0000FF0000FFULL; + r = (r | r << 8) & 0x100F00F00F00F00FULL; + r = (r | r << 4) & 0x10C30C30C30C30C3ULL; + r = (r | r << 2) & 0x1249249249249249ULL; + return r; +} + +static uint64_t mortonCode3D(uint32_t x, uint32_t y, uint32_t z) { + return mortonSplit21(x) | (mortonSplit21(y) << 1) | (mortonSplit21(z) << 2); +} + +// Return a mesh-id permutation sorted by 3D Morton (Z-order) code over +// the meshes' centroids. Replaces a lexicographic (z, y, x) sort, +// which was effectively a 1D Z-slab traversal — chunks ended up +// spanning the whole XY extent of the model, ~50m × 50m × 0.5m for a +// typical building. Morton clusters spatially in all 3 axes, so each +// chunk's AABB becomes a tight 3D voxel — small enough that +// per-chunk frustum / contribution / HiZ rejection becomes meaningful +// (a 1km-wide AABB never gets occluded; a 10m voxel often does). +// +// Meshes with no instances get a Morton code of 0 and sink to the +// front; they contribute no geometry / AABBs so where they land in +// the chunk plan doesn't matter. +static std::vector sortMeshIdsByMorton( + std::size_t n_meshes, + const std::vector& mesh_cx, + const std::vector& mesh_cy, + const std::vector& mesh_cz, + const std::vector& mesh_inst_count) { + // Per-model bounds over centroids. Quantising relative to these + // gives the Morton code its full 21-bit-per-axis resolution + // (~2 M bins per axis = sub-millimetre on a kilometre-scale scene, + // way more than we need; the cost is the same regardless). + float bmin[3] = { std::numeric_limits::infinity(), + std::numeric_limits::infinity(), + std::numeric_limits::infinity() }; + float bmax[3] = { -std::numeric_limits::infinity(), + -std::numeric_limits::infinity(), + -std::numeric_limits::infinity() }; + for (std::size_t i = 0; i < n_meshes; ++i) { + if (mesh_inst_count[i] == 0) continue; + bmin[0] = std::min(bmin[0], mesh_cx[i]); bmax[0] = std::max(bmax[0], mesh_cx[i]); + bmin[1] = std::min(bmin[1], mesh_cy[i]); bmax[1] = std::max(bmax[1], mesh_cy[i]); + bmin[2] = std::min(bmin[2], mesh_cz[i]); bmax[2] = std::max(bmax[2], mesh_cz[i]); + } + const float ext[3] = { + std::max(bmax[0] - bmin[0], 1e-3f), + std::max(bmax[1] - bmin[1], 1e-3f), + std::max(bmax[2] - bmin[2], 1e-3f), + }; + constexpr uint32_t MORTON_BITS = 21; + constexpr uint32_t MORTON_MAX = (1u << MORTON_BITS) - 1u; + + std::vector codes(n_meshes, 0); + for (uint32_t i = 0; i < uint32_t(n_meshes); ++i) { + if (mesh_inst_count[i] == 0) continue; + const float nx = (mesh_cx[i] - bmin[0]) / ext[0]; + const float ny = (mesh_cy[i] - bmin[1]) / ext[1]; + const float nz = (mesh_cz[i] - bmin[2]) / ext[2]; + const uint32_t qx = std::min(uint32_t(nx * float(MORTON_MAX + 1u)), MORTON_MAX); + const uint32_t qy = std::min(uint32_t(ny * float(MORTON_MAX + 1u)), MORTON_MAX); + const uint32_t qz = std::min(uint32_t(nz * float(MORTON_MAX + 1u)), MORTON_MAX); + codes[i] = mortonCode3D(qx, qy, qz); + } + + std::vector sorted(n_meshes); + std::iota(sorted.begin(), sorted.end(), 0u); + std::stable_sort(sorted.begin(), sorted.end(), + [&](uint32_t a, uint32_t b) { return codes[a] < codes[b]; }); + return sorted; +} + void releaseWgpuModelGpuData(WgpuModelGpuData& m, WgpuBufferPool& pool) { for (auto& c : m.chunks) { if (c.bind_group) { wgpuBindGroupRelease(c.bind_group); c.bind_group = nullptr; } @@ -595,19 +676,13 @@ void WgpuViewportWindow::applyCachedModelStreaming(uint32_t model_id, } } - // Sort mesh indices by centroid. Lexicographic (z, y, x) is cheap and - // gives reasonable spatial locality — a Morton/Hilbert encode would - // be tighter but this is enough to make per-chunk AABBs much smaller - // than the model AABB. Stable sort to keep mesh-id order as the - // tiebreaker when many meshes coincide (instanced repeat geometry). - std::vector sorted_mesh_ids(n_meshes); - std::iota(sorted_mesh_ids.begin(), sorted_mesh_ids.end(), 0u); - std::stable_sort(sorted_mesh_ids.begin(), sorted_mesh_ids.end(), - [&](uint32_t a, uint32_t b) { - if (mesh_cz[a] != mesh_cz[b]) return mesh_cz[a] < mesh_cz[b]; - if (mesh_cy[a] != mesh_cy[b]) return mesh_cy[a] < mesh_cy[b]; - return mesh_cx[a] < mesh_cx[b]; - }); + // Sort mesh indices by 3D Morton (Z-order) code over centroids — gives + // tight 3D voxel chunks instead of the XY-slab chunks the previous + // lex (z, y, x) sort produced. Tight AABBs are a prerequisite for + // per-chunk frustum / contribution / HiZ rejection actually + // discriminating between near and far chunks of the same model. + std::vector sorted_mesh_ids = + sortMeshIdsByMorton(n_meshes, mesh_cx, mesh_cy, mesh_cz, mesh_inst_count); // Greedy pack sorted meshes into chunks. std::vector> chunk_mesh_ids; @@ -851,14 +926,10 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) { } } - std::vector sorted_mesh_ids(n_meshes); - std::iota(sorted_mesh_ids.begin(), sorted_mesh_ids.end(), 0u); - std::stable_sort(sorted_mesh_ids.begin(), sorted_mesh_ids.end(), - [&](uint32_t a, uint32_t b) { - if (mesh_cz[a] != mesh_cz[b]) return mesh_cz[a] < mesh_cz[b]; - if (mesh_cy[a] != mesh_cy[b]) return mesh_cy[a] < mesh_cy[b]; - return mesh_cx[a] < mesh_cx[b]; - }); + // Morton sort (see sortMeshIdsByMorton above) — same logic as the + // streaming path; gives tight 3D voxel chunks instead of XY slabs. + std::vector sorted_mesh_ids = + sortMeshIdsByMorton(n_meshes, mesh_cx, mesh_cy, mesh_cz, mesh_inst_count); std::vector> chunk_mesh_ids; chunk_mesh_ids.push_back({});