From 10f8fc88e81c9e9b3155a7581e2ef96adde02b5d Mon Sep 17 00:00:00 2001 From: Dion Moult Date: Fri, 29 May 2026 09:02:27 +1000 Subject: [PATCH] wgpu chunks: pack LOD1 indices alongside LOD0; cull picks per-instance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sidecar LOD1 is per-mesh, index-only — meshoptimizer-baked decimated index slices that share the LOD0 VBO. Before this change the wgpu chunk path force-disabled LOD1 (effective_lod1 = false; tagged in a comment as "until per-chunk LOD1 storage lands"), so it had to walk every visible instance at full LOD0 even when the per-instance LOD selection said the projected radius was below the LOD1 threshold. Per-chunk layout: append LOD1 indices for the chunk's meshes after all LOD0 indices in the same pool slice. A new mesh_chunk_local_lod1_first_u32 array records each mesh's LOD1 starting offset (in u32s) within the chunk's index slice; LOD0 offsets stay where they were. The cull's emit then sets VisibleDrawGpu.ebo_first_u32 to whichever side matches the per- instance use_lod1 decision the prior #8 commit already computed. Vertex pulling is oblivious to the LOD split — it just reads the indices the cull pointed it at. c.index_count is repurposed as the LOD0+LOD1 total so the pool allocation, eviction's pool-fit check, and the VRAM accounting all scale automatically. c.lod1_index_count exposes the LOD1 portion for stats. makeChunkRequest appends LOD1 byte ranges to req.i_ranges in the same per-mesh order; the streaming worker concatenates ranges in order, so the assembled idx blob lands LOD0-first / LOD1-second which matches the chunk-local packing. On a 10-model regen with meshoptimizer-baked sidecars, the bench [frame] heartbeat shows lod1 firing on ~80-90% of LOD1-eligible instances and saving 10-30M tris per frame versus the prior LOD0-only ceiling. Same camera/scene on basic.ifc is still pixel-identical (no mesh in basic.ifc is large enough to bake a LOD1, so the cull just follows the LOD0 path it always did). Temporary debug counters (lod1_dbg_count_ et al.) print "lod1 X/Y (saved Z tris, N no-lod1)" in both interactive and benchmark [frame] heartbeats — kept on while LOD1 correctness gets confirmed across more scenes, will come out once trust is built. The non-streaming applyCachedModel path runs the same LOD1 plumbing but no longer fits the full federation scene in pool (extra index bytes push past the 2 GB single-buffer cap); that mode was already streaming-only on that scene before. Co-Authored-By: Claude Opus 4.7 --- src/ifcviewer-wgpu/WgpuModelGpuData.h | 9 ++ src/ifcviewer-wgpu/WgpuViewportWindow.cpp | 117 ++++++++++++++++------ src/ifcviewer-wgpu/WgpuViewportWindow.h | 9 ++ 3 files changed, 106 insertions(+), 29 deletions(-) diff --git a/src/ifcviewer-wgpu/WgpuModelGpuData.h b/src/ifcviewer-wgpu/WgpuModelGpuData.h index 0c7ba15e53..7b8c74b2c2 100644 --- a/src/ifcviewer-wgpu/WgpuModelGpuData.h +++ b/src/ifcviewer-wgpu/WgpuModelGpuData.h @@ -139,6 +139,11 @@ struct WgpuModelGpuData { // is recovered by walking mesh_ids and the model's MeshInfo[]. uint64_t vertex_byte_size = 0; uint64_t index_count = 0; + // Of `index_count`, how many are LOD1 indices. LOD0 indices occupy + // chunk-local u32 offsets [0, index_count - lod1_index_count); LOD1 + // indices occupy [index_count - lod1_index_count, index_count). 0 + // when no mesh in this chunk had a baked LOD1 slice. + uint32_t lod1_index_count = 0; // World-space AABB covering every instance whose mesh lives in // this chunk. With spatial chunk planning this AABB is tight @@ -217,6 +222,10 @@ struct WgpuModelGpuData { std::vector mesh_chunk_idx; std::vector mesh_chunk_local_base_vertex; std::vector mesh_chunk_local_ebo_first_u32; + // Where in the chunk's index slice this mesh's LOD1 indices start + // (in u32 units). Only meaningful when m.meshes[mi].lod1_index_count > 0; + // entries for meshes without LOD1 are 0 and unused. + std::vector mesh_chunk_local_lod1_first_u32; // Model-shared buffers. Mesh + instance storage are small (<10 MB on // any real scene we've seen); the chunked index buffer lives in Chunk diff --git a/src/ifcviewer-wgpu/WgpuViewportWindow.cpp b/src/ifcviewer-wgpu/WgpuViewportWindow.cpp index 70b6c1de97..70b66d3636 100644 --- a/src/ifcviewer-wgpu/WgpuViewportWindow.cpp +++ b/src/ifcviewer-wgpu/WgpuViewportWindow.cpp @@ -232,6 +232,7 @@ void releaseWgpuModelGpuData(WgpuModelGpuData& m, WgpuBufferPool& pool) { m.mesh_chunk_idx.clear(); m.mesh_chunk_local_base_vertex.clear(); m.mesh_chunk_local_ebo_first_u32.clear(); + m.mesh_chunk_local_lod1_first_u32.clear(); if (m.mesh_storage) { wgpuBufferRelease(m.mesh_storage); m.mesh_storage = nullptr; } if (m.instance_storage) { wgpuBufferRelease(m.instance_storage); m.instance_storage = nullptr; } m.vertex_bytes = 0; @@ -654,6 +655,7 @@ void WgpuViewportWindow::applyCachedModelStreaming(uint32_t model_id, m.mesh_chunk_idx.assign(n_meshes, 0); m.mesh_chunk_local_base_vertex.assign(n_meshes, 0); m.mesh_chunk_local_ebo_first_u32.assign(n_meshes, 0); + m.mesh_chunk_local_lod1_first_u32.assign(n_meshes, 0); // Per-mesh centroid = mean of its instances' world AABB centres. // Meshes with no instances stay at (0,0,0) — they're dead weight but @@ -688,14 +690,8 @@ void WgpuViewportWindow::applyCachedModelStreaming(uint32_t model_id, std::vector> chunk_mesh_ids; chunk_mesh_ids.push_back({}); uint64_t current_chunk_bytes = 0; - bool warned_lod1 = false; for (uint32_t mi : sorted_mesh_ids) { const MeshInfo& mesh = metadata.meta.meshes[mi]; - if (!warned_lod1 && mesh.lod1_index_count > 0) { - qWarning() << "[wgpu stream] LOD1 indices present but per-chunk " - "buffers only carry LOD0; LOD1 will be ignored."; - warned_lod1 = true; - } const uint64_t mesh_bytes = uint64_t(mesh.vertex_count) * INSTANCED_VERTEX_STRIDE_BYTES; if (current_chunk_bytes > 0 && current_chunk_bytes + mesh_bytes > WGPU_CHUNK_VERTEX_BYTES_LIMIT) { @@ -729,7 +725,11 @@ void WgpuViewportWindow::applyCachedModelStreaming(uint32_t model_id, // Walk this chunk's meshes in chunk-local layout order, computing // each mesh's chunk-local base_vertex / ebo_first_u32 and the - // chunk's aggregate vertex/index totals. + // chunk's aggregate vertex/index totals. LOD1 indices (if any + // mesh has them baked) get a second pass and pack AFTER all the + // LOD0 indices in the chunk's index slice — so a single slice + // carries both LODs and cull picks per-instance by chunk-local + // u32 offset. uint32_t chunk_local_v = 0; uint32_t chunk_local_i = 0; for (uint32_t mi : c.mesh_ids) { @@ -740,9 +740,17 @@ void WgpuViewportWindow::applyCachedModelStreaming(uint32_t model_id, chunk_local_v += mesh.vertex_count; chunk_local_i += mesh.index_count; } + uint32_t chunk_local_lod1 = 0; + for (uint32_t mi : c.mesh_ids) { + const MeshInfo& mesh = metadata.meta.meshes[mi]; + if (mesh.lod1_index_count == 0) continue; + m.mesh_chunk_local_lod1_first_u32[mi] = chunk_local_i + chunk_local_lod1; + chunk_local_lod1 += mesh.lod1_index_count; + } c.vertex_count = chunk_local_v; c.vertex_byte_size = uint64_t(chunk_local_v) * INSTANCED_VERTEX_STRIDE_BYTES; - c.index_count = chunk_local_i; + c.index_count = chunk_local_i + chunk_local_lod1; + c.lod1_index_count = chunk_local_lod1; // Small per-chunk buffers, allocated upfront so cull can write into // them. visible_draws_buffer cap = chunk's instance count (worst- @@ -907,6 +915,7 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) { m.mesh_chunk_idx.assign(n_meshes, 0); m.mesh_chunk_local_base_vertex.assign(n_meshes, 0); m.mesh_chunk_local_ebo_first_u32.assign(n_meshes, 0); + m.mesh_chunk_local_lod1_first_u32.assign(n_meshes, 0); std::vector mesh_cx(n_meshes, 0.0f), mesh_cy(n_meshes, 0.0f), @@ -934,14 +943,8 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) { std::vector> chunk_mesh_ids; chunk_mesh_ids.push_back({}); uint64_t current_chunk_bytes = 0; - bool warned_lod1 = false; for (uint32_t mi : sorted_mesh_ids) { const MeshInfo& mesh = data.meshes[mi]; - if (!warned_lod1 && mesh.lod1_index_count > 0) { - qWarning() << "[wgpu] LOD1 indices present but per-chunk buffers " - "only carry LOD0; LOD1 will be ignored this load."; - warned_lod1 = true; - } const uint64_t mesh_vertex_bytes = uint64_t(mesh.vertex_count) * INSTANCED_VERTEX_STRIDE_BYTES; if (mesh_vertex_bytes > WGPU_CHUNK_VERTEX_BYTES_LIMIT) { qWarning().noquote().nospace() @@ -976,6 +979,9 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) { // Walk meshes in chunk-local layout order, computing each mesh's // chunk-local offsets and the chunk's aggregate vertex/index totals. + // LOD1 indices (if any mesh has them baked) pack AFTER all the + // LOD0 indices in the chunk's index slice — single slice carries + // both LODs, cull picks per-instance by chunk-local u32 offset. uint32_t chunk_local_v = 0; uint32_t chunk_local_i = 0; for (uint32_t mi : c.mesh_ids) { @@ -986,9 +992,17 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) { chunk_local_v += mesh.vertex_count; chunk_local_i += mesh.index_count; } + uint32_t chunk_local_lod1 = 0; + for (uint32_t mi : c.mesh_ids) { + const MeshInfo& mesh = data.meshes[mi]; + if (mesh.lod1_index_count == 0) continue; + m.mesh_chunk_local_lod1_first_u32[mi] = chunk_local_i + chunk_local_lod1; + chunk_local_lod1 += mesh.lod1_index_count; + } c.vertex_count = chunk_local_v; c.vertex_byte_size = uint64_t(chunk_local_v) * INSTANCED_VERTEX_STRIDE_BYTES; - c.index_count = chunk_local_i; + c.index_count = chunk_local_i + chunk_local_lod1; + c.lod1_index_count = chunk_local_lod1; c.vertex_slice = pool_.alloc(c.vertex_byte_size, 256); if (!c.vertex_slice.valid()) { @@ -1038,6 +1052,19 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) { i_off += i_bytes; } } + // LOD1 indices second, packed after all LOD0 indices in the slice. + // mesh_chunk_local_lod1_first_u32[mi] already encodes this layout — + // we just have to copy in the same order it was assigned. + for (uint32_t mi : c.mesh_ids) { + const MeshInfo& mesh = data.meshes[mi]; + if (mesh.lod1_index_count == 0) continue; + const size_t l1_bytes = size_t(mesh.lod1_index_count) * sizeof(uint32_t); + wgpuQueueWriteBuffer(queue_, c.index_slice.buffer, + c.index_slice.offset + i_off, + data.indices.data() + (mesh.lod1_ebo_byte_offset / sizeof(uint32_t)), + l1_bytes); + i_off += l1_bytes; + } m.vram_bytes_vbo += c.vertex_byte_size; m.vram_bytes_ebo += c.index_count * sizeof(uint32_t); } @@ -2658,32 +2685,39 @@ uint32_t WgpuViewportWindow::cullModelCpuCompute(WgpuModelGpuData& m, && mesh.lod1_index_count > 0 && projected_px < lod1_threshold_px; - // LOD1 is incompatible with the current per-chunk index layout - // (LOD1 indices are appended at the end of sd.indices, not - // contiguous with their chunk's range). Force LOD0 until per- - // chunk LOD1 storage lands. - const bool effective_lod1 = false; - (void)use_lod1; - // Emit one VisibleDraw entry into the chunk that owns this mesh's // vertex range. base_vertex AND ebo_first_u32 are both CHUNK-LOCAL // — the chunk's bind group points at its own vertex_storage and - // index_buffer slices so the shader indexes them directly. + // index_buffer slices so the shader indexes them directly. When + // use_lod1, ebo_first_u32 routes into the LOD1 section of the + // chunk's index slice (which is packed after the LOD0 section at + // chunk-build time); the shader is oblivious to the LOD split. const uint32_t chunk_idx = m.mesh_chunk_idx[inst.mesh_id]; WgpuModelGpuData::Chunk& c = m.chunks[chunk_idx]; WgpuModelGpuData::VisibleDrawGpu d; d.mesh_id = inst.mesh_id; d.instance_idx = i; - d.ebo_first_u32 = m.mesh_chunk_local_ebo_first_u32[inst.mesh_id]; + d.ebo_first_u32 = use_lod1 + ? m.mesh_chunk_local_lod1_first_u32[inst.mesh_id] + : m.mesh_chunk_local_ebo_first_u32[inst.mesh_id]; d.base_vertex = m.mesh_chunk_local_base_vertex[inst.mesh_id]; c.visible_draws_scratch.push_back(d); - const uint32_t entry_vert_count = effective_lod1 - ? mesh.lod1_index_count - : mesh.index_count; + const uint32_t entry_vert_count = use_lod1 ? mesh.lod1_index_count + : mesh.index_count; running_vertex_count[chunk_idx] += entry_vert_count; c.prefix_sums_scratch.push_back(running_vertex_count[chunk_idx]); + if (use_lod1) { + ++lod1_dbg_count_; + lod1_dbg_tris_saved_ += (mesh.index_count > mesh.lod1_index_count + ? (mesh.index_count - mesh.lod1_index_count) / 3 + : 0); + } else if (mesh.lod1_index_count > 0) { + ++lod0_dbg_eligible_count_; + } else { + ++lod0_dbg_no_lod1_count_; + } }; // Chunk-driven walk: frustum-test each chunk's AABB once, and skip @@ -3135,7 +3169,14 @@ void WgpuViewportWindow::render() { << " chunks " << chunks_resident << "/" << chunks_frustum_vis << "/" << chunks_total << " (missing " << chunks_missing << ")" << " vram " << QString::number(double(total_vbo + total_ebo + total_ssbo) * mb, 'f', 1) << "MB" - << " models " << models_gpu_.size(); + << " models " << models_gpu_.size() + << " lod1 " << lod1_dbg_count_ << "/" << (lod1_dbg_count_ + lod0_dbg_eligible_count_) + << " (saved " << lod1_dbg_tris_saved_ << " tris, " + << lod0_dbg_no_lod1_count_ << " no-lod1)"; + lod1_dbg_count_ = 0; + lod0_dbg_eligible_count_ = 0; + lod0_dbg_no_lod1_count_ = 0; + lod1_dbg_tris_saved_ = 0; // Every ~120 frames, when something's missing, dump the // top-8 models by missing-chunk count + top/bottom chunks @@ -3405,7 +3446,14 @@ void WgpuViewportWindow::render() { << " (vbo " << QString::number(double(total_vbo) * mb, 'f', 1) << " + ebo " << QString::number(double(total_ebo) * mb, 'f', 1) << " + ssbo " << QString::number(double(total_ssbo) * mb, 'f', 1) << ")" - << " models " << models_gpu_.size(); + << " models " << models_gpu_.size() + << " lod1 " << lod1_dbg_count_ << "/" << (lod1_dbg_count_ + lod0_dbg_eligible_count_) + << " (saved " << lod1_dbg_tris_saved_ << " tris, " + << lod0_dbg_no_lod1_count_ << " no-lod1)"; + lod1_dbg_count_ = 0; + lod0_dbg_eligible_count_ = 0; + lod0_dbg_no_lod1_count_ = 0; + lod1_dbg_tris_saved_ = 0; } camera_yaw_deg_ = bench_yaw_start_ + bench_yaw_speed_ * float(bench_count_ + 1); @@ -3732,6 +3780,17 @@ static WgpuStreamingThread::Request makeChunkRequest( uint64_t(mesh.index_count)); } } + // LOD1 indices second pass — matches the chunk-local packing order + // (all LOD0 first, then LOD1) so the worker's concatenated index + // result lands at the offsets recorded in + // m.mesh_chunk_local_lod1_first_u32. + for (uint32_t mi : c.mesh_ids) { + const MeshInfo& mesh = m.meshes[mi]; + if (mesh.lod1_index_count == 0) continue; + req.i_ranges.emplace_back( + uint64_t(mesh.lod1_ebo_byte_offset / sizeof(uint32_t)), + uint64_t(mesh.lod1_index_count)); + } return req; } diff --git a/src/ifcviewer-wgpu/WgpuViewportWindow.h b/src/ifcviewer-wgpu/WgpuViewportWindow.h index d92df6c174..85fb7f0b8b 100644 --- a/src/ifcviewer-wgpu/WgpuViewportWindow.h +++ b/src/ifcviewer-wgpu/WgpuViewportWindow.h @@ -530,6 +530,15 @@ private: // Tick count for the interactive (non-bench) [frame] heartbeat log. // Increments every render() and prints stats every N frames. int interactive_frame_count_ = 0; + + // Per-frame LOD selection counts, mutated from cullModelCpuCompute + // and reset after the [frame] heartbeat prints them. Keeps an eye + // on whether LOD1 is actually firing on real scenes — early-days + // diagnostic while we trust the new code path. + mutable uint32_t lod1_dbg_count_ = 0; + mutable uint32_t lod0_dbg_eligible_count_ = 0; + mutable uint32_t lod0_dbg_no_lod1_count_ = 0; + mutable uint64_t lod1_dbg_tris_saved_ = 0; }; #endif // WGPUVIEWPORTWINDOW_H