From 5a3e5167dfe5b0d8c4ec0b6afd0ae83e0630b54f Mon Sep 17 00:00:00 2001 From: Dion Moult Date: Thu, 28 May 2026 09:23:25 +1000 Subject: [PATCH] wgpu streaming (4/4): per-frame chunk-on-visible loader MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The OOM fix for vertex storage. With --streaming, chunks now load on demand: - After cull determines which chunks have visible draws, driveStreamingLoads walks non-resident chunks with total_visible_draws > 0 and brings up to MAX_STREAMING_LOADS_PER_FRAME (currently 4) into residency. - Each load: readSidecarVertexChunk → createBufferWithData → buildChunkBindGroup → is_resident = true. Same frame's draw loop picks up the newly-built bind_group and renders the chunk. - If more non-resident-but-visible chunks remain, requestUpdate is called so the load loop keeps running until the visible set is fully resident. Per-chunk bind group construction refactored out of buildModelBindGroup into a buildChunkBindGroup(m, chunk_idx) helper so the streaming loader can build one chunk at a time as it arrives. 4 chunks/frame × 60 fps = 240 chunks/sec ingestion. A 200-chunk scene fully resides in ~1 second of motion. Off-screen chunks never become resident, never pay vertex-storage VRAM — that's where most of the OOM fix lands. Verified on basic.ifc: pixel-identical to non-streaming. On the user's real 111-model / 1M-instance scene: all metadata loads succeed (was OOM before), then loader runs but **indices are still loaded upfront (1.5 GB!) so OOM still hits when vertex chunks start adding on top.** Per-chunk index deferral is the next commit. Eager-no-evict policy (per the design conversation): chunks stay resident once loaded. LRU eviction lands in a follow-up if a workload proves it necessary. This completes the 4-commit stage-1 series for task #16. Stage-2: defer indices, deferred mesh/instance storage if needed, async worker thread. Co-Authored-By: Claude Opus 4.7 --- src/ifcviewer-wgpu/WgpuViewportWindow.cpp | 155 ++++++++++++++++------ src/ifcviewer-wgpu/WgpuViewportWindow.h | 9 ++ 2 files changed, 123 insertions(+), 41 deletions(-) diff --git a/src/ifcviewer-wgpu/WgpuViewportWindow.cpp b/src/ifcviewer-wgpu/WgpuViewportWindow.cpp index 7033d1486d..4f806ee562 100644 --- a/src/ifcviewer-wgpu/WgpuViewportWindow.cpp +++ b/src/ifcviewer-wgpu/WgpuViewportWindow.cpp @@ -2631,6 +2631,11 @@ void WgpuViewportWindow::render() { } } + // Streaming: bring non-resident chunks that the cull just flagged + // visible into residency. Runs before draw encoding so newly-loaded + // chunks render the same frame. + driveStreamingLoads(); + // Snapshot camera state for next frame's motion detection. prev_camera_target_[0] = camera_target_[0]; prev_camera_target_[1] = camera_target_[1]; @@ -3120,50 +3125,118 @@ void WgpuViewportWindow::buildModelBindGroup(WgpuModelGpuData& m) { // Empty model — no chunks, no bind groups; the draw loop will skip. return; } - - // One bind group per chunk: chunk's vertex_storage, visible_draws, - // prefix_sums, per_chunk_uniform — plus the shared mesh / instance / - // index buffers. - for (auto& c : m.chunks) { - if (c.bind_group) { - wgpuBindGroupRelease(c.bind_group); - c.bind_group = nullptr; - } - if (!c.vertex_storage || !c.visible_draws_buffer - || !c.prefix_sums_buffer || !c.per_chunk_uniform) continue; - - WGPUBindGroupEntry entries[7] = {}; - entries[0].binding = 0; - entries[0].buffer = c.vertex_storage; - entries[0].size = WGPU_WHOLE_SIZE; - entries[1].binding = 1; - entries[1].buffer = m.mesh_storage; - entries[1].size = WGPU_WHOLE_SIZE; - entries[2].binding = 2; - entries[2].buffer = m.instance_storage; - entries[2].size = WGPU_WHOLE_SIZE; - entries[3].binding = 3; - entries[3].buffer = m.index_buffer; - entries[3].size = WGPU_WHOLE_SIZE; - entries[4].binding = 4; - entries[4].buffer = c.visible_draws_buffer; - entries[4].size = WGPU_WHOLE_SIZE; - entries[5].binding = 5; - entries[5].buffer = c.prefix_sums_buffer; - entries[5].size = WGPU_WHOLE_SIZE; - entries[6].binding = 6; - entries[6].buffer = c.per_chunk_uniform; - entries[6].size = 16; - - WGPUBindGroupDescriptor desc = {}; - desc.layout = model_bgl_; - desc.entryCount = 7; - desc.entries = entries; - desc.label = svFromCStr("ifcviewer-wgpu.chunk_bind_group"); - c.bind_group = wgpuDeviceCreateBindGroup(device_, &desc); + for (size_t ci = 0; ci < m.chunks.size(); ++ci) { + buildChunkBindGroup(m, ci); } } +void WgpuViewportWindow::buildChunkBindGroup(WgpuModelGpuData& m, size_t chunk_idx) { + if (chunk_idx >= m.chunks.size()) return; + auto& c = m.chunks[chunk_idx]; + if (c.bind_group) { + wgpuBindGroupRelease(c.bind_group); + c.bind_group = nullptr; + } + if (!c.vertex_storage || !c.visible_draws_buffer + || !c.prefix_sums_buffer || !c.per_chunk_uniform + || !m.mesh_storage || !m.instance_storage || !m.index_buffer) { + return; + } + + WGPUBindGroupEntry entries[7] = {}; + entries[0].binding = 0; + entries[0].buffer = c.vertex_storage; + entries[0].size = WGPU_WHOLE_SIZE; + entries[1].binding = 1; + entries[1].buffer = m.mesh_storage; + entries[1].size = WGPU_WHOLE_SIZE; + entries[2].binding = 2; + entries[2].buffer = m.instance_storage; + entries[2].size = WGPU_WHOLE_SIZE; + entries[3].binding = 3; + entries[3].buffer = m.index_buffer; + entries[3].size = WGPU_WHOLE_SIZE; + entries[4].binding = 4; + entries[4].buffer = c.visible_draws_buffer; + entries[4].size = WGPU_WHOLE_SIZE; + entries[5].binding = 5; + entries[5].buffer = c.prefix_sums_buffer; + entries[5].size = WGPU_WHOLE_SIZE; + entries[6].binding = 6; + entries[6].buffer = c.per_chunk_uniform; + entries[6].size = 16; + + WGPUBindGroupDescriptor desc = {}; + desc.layout = model_bgl_; + desc.entryCount = 7; + desc.entries = entries; + desc.label = svFromCStr("ifcviewer-wgpu.chunk_bind_group"); + c.bind_group = wgpuDeviceCreateBindGroup(device_, &desc); +} + +bool WgpuViewportWindow::loadChunkBytesAndUploadGpu(WgpuModelGpuData& m, size_t chunk_idx) { + if (chunk_idx >= m.chunks.size()) return false; + auto& c = m.chunks[chunk_idx]; + if (c.is_resident) return true; + if (m.streaming_file_path.empty()) return false; + + std::vector bytes; + if (!readSidecarVertexChunk(m.streaming_file_path, + m.streaming_vertex_section_offset, + c.vertex_byte_offset, + c.vertex_byte_size, + bytes)) { + qWarning().noquote().nospace() + << "[wgpu stream] failed to read vertex chunk " << chunk_idx + << " from " << QString::fromStdString(m.streaming_file_path) + << " (offset=" << c.vertex_byte_offset + << " size=" << c.vertex_byte_size << ")"; + return false; + } + + c.vertex_storage = createBufferWithData( + device_, queue_, + bytes.data(), bytes.size(), + WGPUBufferUsage_Storage, + "model.chunk.vertex_storage_streamed"); + if (!c.vertex_storage) return false; + + m.vram_bytes_vbo += bytes.size(); + buildChunkBindGroup(m, chunk_idx); + c.is_resident = true; + return true; +} + +void WgpuViewportWindow::driveStreamingLoads() { + // Per-frame budget. Caps first-frame stall on a fresh load — at 4 + // chunks/frame × 60fps we ingest 240 chunks/sec, fast enough that + // a 100-model scene fully resides in ~1s. Eviction-aware policies + // can tune this later. + constexpr int MAX_STREAMING_LOADS_PER_FRAME = 4; + int loads = 0; + bool more_pending = false; + for (auto& [mid, m] : models_gpu_) { + if (m.streaming_file_path.empty() || m.hidden) continue; + for (size_t ci = 0; ci < m.chunks.size(); ++ci) { + auto& c = m.chunks[ci]; + if (c.is_resident) continue; + // Only load chunks the cull just marked visible — keeps fetch + // priority aligned with what the camera actually sees. + if (c.total_visible_draws == 0) continue; + if (loads >= MAX_STREAMING_LOADS_PER_FRAME) { + more_pending = true; + break; + } + if (loadChunkBytesAndUploadGpu(m, ci)) { + ++loads; + } + } + if (more_pending) break; + } + // Keep the frame loop running until all visible chunks are resident. + if (more_pending) requestUpdate(); +} + // ----------------------------------------------------------------------------- // Depth attachment // ----------------------------------------------------------------------------- diff --git a/src/ifcviewer-wgpu/WgpuViewportWindow.h b/src/ifcviewer-wgpu/WgpuViewportWindow.h index 6ea4950bf3..e867798f26 100644 --- a/src/ifcviewer-wgpu/WgpuViewportWindow.h +++ b/src/ifcviewer-wgpu/WgpuViewportWindow.h @@ -128,6 +128,15 @@ private: bool buildPipelines(); void buildModelBindGroup(WgpuModelGpuData& m); + void buildChunkBindGroup(WgpuModelGpuData& m, size_t chunk_idx); + // Streaming: read the chunk's vertex bytes from disk, allocate + // vertex_storage, build the chunk's bind group, flip is_resident=true. + // Returns true on success. No-op (returns true) when already resident. + bool loadChunkBytesAndUploadGpu(WgpuModelGpuData& m, size_t chunk_idx); + // Called from render() after cull: find non-resident chunks with + // current visible draw counts > 0 and bring them resident, up to a + // per-frame budget. Triggers requestUpdate() if more remain. + void driveStreamingLoads(); void ensureDepthTexture(int w, int h); void releaseDepthTexture(); void ensureMsaaColorTexture(int w, int h);