2026-05-27 12:48:03 +10:00
|
|
|
|
/********************************************************************************
|
|
|
|
|
|
* *
|
|
|
|
|
|
* This file is part of IfcOpenShell. *
|
|
|
|
|
|
* *
|
|
|
|
|
|
* IfcOpenShell is free software: you can redistribute it and/or modify *
|
|
|
|
|
|
* it under the terms of the Lesser GNU General Public License as published by *
|
|
|
|
|
|
* the Free Software Foundation, either version 3.0 of the License, or *
|
|
|
|
|
|
* (at your option) any later version. *
|
|
|
|
|
|
* *
|
|
|
|
|
|
* IfcOpenShell is distributed in the hope that it will be useful, *
|
|
|
|
|
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of *
|
|
|
|
|
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the *
|
|
|
|
|
|
* Lesser GNU General Public License for more details. *
|
|
|
|
|
|
* *
|
|
|
|
|
|
* You should have received a copy of the Lesser GNU General Public License *
|
|
|
|
|
|
* along with this program. If not, see <http://www.gnu.org/licenses/>. *
|
|
|
|
|
|
* *
|
|
|
|
|
|
********************************************************************************/
|
|
|
|
|
|
|
|
|
|
|
|
#ifndef WGPUMODELGPUDATA_H
|
|
|
|
|
|
#define WGPUMODELGPUDATA_H
|
|
|
|
|
|
|
|
|
|
|
|
#include <webgpu/webgpu.h>
|
|
|
|
|
|
|
|
|
|
|
|
#include <cstddef>
|
|
|
|
|
|
#include <cstdint>
|
2026-05-28 09:10:53 +10:00
|
|
|
|
#include <limits>
|
|
|
|
|
|
#include <string>
|
2026-05-27 12:48:03 +10:00
|
|
|
|
#include <vector>
|
|
|
|
|
|
|
|
|
|
|
|
#include "InstancedGeometry.h"
|
2026-05-28 14:08:16 +10:00
|
|
|
|
#include "WgpuBufferPool.h"
|
2026-05-27 12:48:03 +10:00
|
|
|
|
|
|
|
|
|
|
// Per-model wgpu state. Mirrors the GL backend's ModelGpuData but with
|
|
|
|
|
|
// wgpu handles. Stage 2 only allocates and uploads the four core buffers;
|
|
|
|
|
|
// bind groups, pipelines, BVH and cull scratch land in later stages.
|
|
|
|
|
|
//
|
|
|
|
|
|
// All vertex/index/mesh/instance bytes are uploaded once at load time via
|
|
|
|
|
|
// wgpuQueueWriteBuffer. The vertex storage buffer is read by the vertex
|
|
|
|
|
|
// shader (vertex pulling), not used as a classic vertex buffer — there is
|
|
|
|
|
|
// no input-assembler vertex layout to match.
|
2026-05-27 21:05:38 +10:00
|
|
|
|
// Web (WebGPU) mandates `maxStorageBufferBindingSize` ≥ 128 MB; some browsers
|
|
|
|
|
|
// grant more, but we plan for the floor. Applied identically on desktop —
|
|
|
|
|
|
// the cost is a few extra draws per frame (1 per chunk; typical models =
|
|
|
|
|
|
// 1–3 chunks), which is invisible compared to per-frame GPU work.
|
|
|
|
|
|
//
|
|
|
|
|
|
// At INSTANCED_VERTEX_STRIDE_BYTES = 12 B/vertex this caps a chunk at
|
2026-05-28 14:45:15 +10:00
|
|
|
|
// ~11 M vertices. Tuned for the pre-v14 scatter-gather streaming model:
|
|
|
|
|
|
// spatial chunk planning means each chunk's bytes are NOT contiguous in
|
|
|
|
|
|
// the sidecar, so per-load I/O cost scales with mesh count per chunk
|
|
|
|
|
|
// (one fseek+fread per file gap). Bigger chunks = more meshes per
|
|
|
|
|
|
// chunk = more seeks per load, BUT also fewer chunks total = fewer
|
|
|
|
|
|
// loads per frame as orbit shifts the visible set. The latter
|
|
|
|
|
|
// dominates: 128 MB chunks → ~16 chunks per pool → 1-2 loads per
|
|
|
|
|
|
// frame → ~20-30 ms stream cost. Smaller chunks (32 / 8 MB) bring
|
|
|
|
|
|
// finer eviction granularity but explode the loads-per-frame count.
|
|
|
|
|
|
// Sidecar v14 (on-disk spatial reorder) is the proper fix — once
|
|
|
|
|
|
// chunks ARE file-contiguous, the per-mesh seek cost vanishes and we
|
|
|
|
|
|
// can drop the chunk size back to ~8 MB for sharp eviction.
|
2026-05-27 21:05:38 +10:00
|
|
|
|
static constexpr uint64_t WGPU_CHUNK_VERTEX_BYTES_LIMIT = 128ull * 1024 * 1024;
|
|
|
|
|
|
|
2026-05-27 12:48:03 +10:00
|
|
|
|
struct WgpuModelGpuData {
|
2026-05-27 21:05:38 +10:00
|
|
|
|
// std430 layout: 16 bytes per entry, naturally aligned. base_vertex is
|
|
|
|
|
|
// CHUNK-LOCAL — the bound vertex_storage on that chunk's bind group
|
|
|
|
|
|
// gives the right slice when the shader indexes vertices[].
|
2026-05-27 19:46:13 +10:00
|
|
|
|
struct alignas(16) VisibleDrawGpu {
|
|
|
|
|
|
uint32_t mesh_id; // -> meshes[] for quantisation basis
|
|
|
|
|
|
uint32_t instance_idx; // -> instances[] for transform + ids
|
2026-05-27 21:05:38 +10:00
|
|
|
|
uint32_t ebo_first_u32; // start of this entry's slice in indices[] (global)
|
|
|
|
|
|
uint32_t base_vertex; // chunk-local start of this mesh's slice in vertex_storage
|
2026-05-27 19:46:13 +10:00
|
|
|
|
};
|
|
|
|
|
|
static_assert(sizeof(VisibleDrawGpu) == 16, "VisibleDrawGpu must be 16 bytes");
|
|
|
|
|
|
|
2026-05-28 12:21:52 +10:00
|
|
|
|
// Per-chunk state. Each chunk references a vertex range and an
|
|
|
|
|
|
// index range inside WgpuViewportWindow::pool_, plus a small set of
|
|
|
|
|
|
// per-frame buffers (visible_draws, prefix_sums, uniform) and a bind
|
|
|
|
|
|
// group that binds the pool ranges alongside the model-shared
|
|
|
|
|
|
// mesh/instance storage. Rendering issues one drawcall per non-empty
|
|
|
|
|
|
// chunk.
|
2026-05-28 09:10:53 +10:00
|
|
|
|
//
|
|
|
|
|
|
// Streaming (task #16): a chunk may be marked is_resident=false; its
|
2026-05-28 12:21:52 +10:00
|
|
|
|
// pool ranges (pool_*_size == 0) and bind_group are then unclaimed
|
|
|
|
|
|
// until the streaming loader brings it in. Other per-chunk buffers
|
|
|
|
|
|
// (visible_draws etc.) stay allocated regardless because cull still
|
|
|
|
|
|
// needs them. Non-streaming path always sets is_resident=true and
|
|
|
|
|
|
// populates pool ranges at applyCachedModel time.
|
2026-05-27 21:05:38 +10:00
|
|
|
|
struct Chunk {
|
2026-05-28 14:08:16 +10:00
|
|
|
|
// Pool-allocated vertex + index bytes. Both slices land in the
|
|
|
|
|
|
// shared WgpuViewportWindow::pool_; the slice tells us which
|
|
|
|
|
|
// sub-buffer they live in (the pool may span multiple sub-buffers
|
|
|
|
|
|
// when scenes exceed wgpu's single-buffer cap). When non-resident,
|
|
|
|
|
|
// both .size are 0.
|
|
|
|
|
|
WgpuBufferPool::Slice vertex_slice;
|
|
|
|
|
|
WgpuBufferPool::Slice index_slice;
|
2026-05-28 12:21:52 +10:00
|
|
|
|
|
2026-05-27 21:05:38 +10:00
|
|
|
|
WGPUBuffer visible_draws_buffer = nullptr;
|
|
|
|
|
|
WGPUBuffer prefix_sums_buffer = nullptr;
|
|
|
|
|
|
WGPUBuffer per_chunk_uniform = nullptr;
|
|
|
|
|
|
WGPUBindGroup bind_group = nullptr;
|
|
|
|
|
|
|
|
|
|
|
|
uint32_t vertex_count = 0; // chunk capacity (vertices)
|
|
|
|
|
|
size_t visible_draws_capacity = 0;
|
|
|
|
|
|
size_t prefix_sums_capacity = 0;
|
|
|
|
|
|
|
|
|
|
|
|
// Per-frame, populated by cullModelCpuCompute and consumed by render().
|
2026-05-28 14:08:16 +10:00
|
|
|
|
// total_visible_* are post-frustum + contribution + HiZ — used to size
|
|
|
|
|
|
// the actual draw call. frustum_visible_count is bumped immediately
|
|
|
|
|
|
// after the frustum check (before contribution / HiZ), and is what
|
|
|
|
|
|
// driveStreamingLoads keys on for residency decisions. Streaming
|
|
|
|
|
|
// must NOT use the HiZ-post counters: HiZ visibility flips
|
|
|
|
|
|
// frame-to-frame as occluders shift, which would otherwise thrash
|
|
|
|
|
|
// the loader (evict-then-reload every frame even with the camera
|
|
|
|
|
|
// stationary, killing FPS and producing visible flicker).
|
2026-05-27 21:05:38 +10:00
|
|
|
|
uint32_t total_visible_vertices = 0;
|
|
|
|
|
|
uint32_t total_visible_draws = 0;
|
2026-05-28 14:08:16 +10:00
|
|
|
|
uint32_t frustum_visible_count = 0;
|
2026-05-27 21:05:38 +10:00
|
|
|
|
|
|
|
|
|
|
std::vector<VisibleDrawGpu> visible_draws_scratch;
|
|
|
|
|
|
std::vector<uint32_t> prefix_sums_scratch;
|
2026-05-28 09:10:53 +10:00
|
|
|
|
|
|
|
|
|
|
// Residency. Streaming sets is_resident=false at applyCachedModel
|
|
|
|
|
|
// and flips true once the chunk's vertex bytes are uploaded.
|
|
|
|
|
|
// Render and pick skip chunks where !is_resident.
|
|
|
|
|
|
bool is_resident = true;
|
|
|
|
|
|
|
2026-05-28 14:45:15 +10:00
|
|
|
|
// Aggregate vertex / index sizes across all meshes in this chunk
|
|
|
|
|
|
// (sum of mesh.vertex_count * stride / mesh.index_count for each
|
|
|
|
|
|
// mesh in mesh_ids). Used to size the pool allocation and to
|
|
|
|
|
|
// compute the cull's per-chunk free-room check. Per-mesh layout
|
|
|
|
|
|
// is recovered by walking mesh_ids and the model's MeshInfo[].
|
2026-05-28 09:10:53 +10:00
|
|
|
|
uint64_t vertex_byte_size = 0;
|
2026-05-28 11:15:11 +10:00
|
|
|
|
uint64_t index_count = 0;
|
|
|
|
|
|
|
2026-05-28 09:10:53 +10:00
|
|
|
|
// World-space AABB covering every instance whose mesh lives in
|
2026-05-28 14:45:15 +10:00
|
|
|
|
// this chunk. With spatial chunk planning this AABB is tight
|
|
|
|
|
|
// (chunks group meshes by world centroid, not mesh-id), so the
|
|
|
|
|
|
// distance-based evictor can meaningfully tell chunks apart.
|
|
|
|
|
|
// Used by cull to reject whole chunks against the frustum before
|
|
|
|
|
|
// iterating instances — and by the streaming loader to
|
|
|
|
|
|
// prioritise which non-resident chunks to fetch first.
|
2026-05-28 09:10:53 +10:00
|
|
|
|
float aabb_min[3] = { std::numeric_limits<float>::infinity(),
|
|
|
|
|
|
std::numeric_limits<float>::infinity(),
|
|
|
|
|
|
std::numeric_limits<float>::infinity() };
|
|
|
|
|
|
float aabb_max[3] = { -std::numeric_limits<float>::infinity(),
|
|
|
|
|
|
-std::numeric_limits<float>::infinity(),
|
|
|
|
|
|
-std::numeric_limits<float>::infinity() };
|
2026-05-28 11:15:11 +10:00
|
|
|
|
|
2026-05-28 14:45:15 +10:00
|
|
|
|
// Mesh IDs assigned to this chunk, in chunk-local layout order.
|
|
|
|
|
|
// Spatial chunk planning sorts meshes by world centroid first,
|
|
|
|
|
|
// so this list is not in mesh-id order in general — each mesh's
|
|
|
|
|
|
// bytes live at scattered offsets in the sidecar file. The
|
|
|
|
|
|
// loader walks this list to scatter-gather the chunk's vertex
|
|
|
|
|
|
// + index bytes; mesh_chunk_local_base_vertex /
|
|
|
|
|
|
// mesh_chunk_local_ebo_first_u32 are computed in this same
|
|
|
|
|
|
// order at planning time so the cull's VisibleDrawGpu entries
|
|
|
|
|
|
// point at the correct chunk-local offsets.
|
|
|
|
|
|
std::vector<uint32_t> mesh_ids;
|
|
|
|
|
|
|
2026-05-28 15:20:30 +10:00
|
|
|
|
// Instance indices belonging to this chunk (i.e. whose mesh lives
|
|
|
|
|
|
// in this chunk). Built at chunk-planning time. Lets cull iterate
|
|
|
|
|
|
// chunks as the outer loop, frustum-test the chunk AABB once,
|
|
|
|
|
|
// and skip every instance inside in one shot when the chunk is
|
|
|
|
|
|
// off-screen — far cheaper than the per-instance frustum check
|
|
|
|
|
|
// on flat-scan culls of 1M+ instance scenes.
|
|
|
|
|
|
std::vector<uint32_t> instance_ids;
|
|
|
|
|
|
|
2026-05-28 11:15:11 +10:00
|
|
|
|
// LRU marker for streaming eviction. Updated to the window's
|
|
|
|
|
|
// streaming_frame_idx_ every frame the chunk is rendered (i.e.
|
|
|
|
|
|
// total_visible_draws > 0). The evictor picks the smallest value
|
|
|
|
|
|
// among non-visible resident chunks when it needs to free VRAM.
|
|
|
|
|
|
uint64_t last_visible_frame_idx = 0;
|
2026-05-27 21:05:38 +10:00
|
|
|
|
};
|
|
|
|
|
|
std::vector<Chunk> chunks;
|
|
|
|
|
|
|
2026-05-28 09:10:53 +10:00
|
|
|
|
// Streaming source. Non-empty path means this model was loaded via the
|
|
|
|
|
|
// streaming path: chunks may be non-resident and need byte-range reads
|
|
|
|
|
|
// from this file. Empty path = legacy non-streaming load.
|
|
|
|
|
|
std::string streaming_file_path;
|
|
|
|
|
|
uint64_t streaming_vertex_section_offset = 0;
|
2026-05-28 11:15:11 +10:00
|
|
|
|
uint64_t streaming_index_section_offset = 0;
|
2026-05-28 09:10:53 +10:00
|
|
|
|
|
2026-05-27 21:05:38 +10:00
|
|
|
|
// For each mesh in meshes[], the chunk it lives in plus the chunk-local
|
2026-05-28 11:15:11 +10:00
|
|
|
|
// offsets into that chunk's vertex_storage and index_buffer. Populated
|
|
|
|
|
|
// at applyCachedModel time; consumed by cullModelCpuCompute when it
|
|
|
|
|
|
// populates VisibleDrawGpu entries.
|
2026-05-27 21:05:38 +10:00
|
|
|
|
std::vector<uint32_t> mesh_chunk_idx;
|
|
|
|
|
|
std::vector<uint32_t> mesh_chunk_local_base_vertex;
|
2026-05-28 11:15:11 +10:00
|
|
|
|
std::vector<uint32_t> mesh_chunk_local_ebo_first_u32;
|
2026-05-27 21:05:38 +10:00
|
|
|
|
|
2026-05-28 11:15:11 +10:00
|
|
|
|
// Model-shared buffers. Mesh + instance storage are small (<10 MB on
|
|
|
|
|
|
// any real scene we've seen); the chunked index buffer lives in Chunk
|
|
|
|
|
|
// alongside vertex_storage so streaming can defer both together.
|
2026-05-27 21:05:38 +10:00
|
|
|
|
WGPUBuffer mesh_storage = nullptr; // MeshGpu[]: aabb_min/max
|
|
|
|
|
|
WGPUBuffer instance_storage = nullptr; // InstanceGpu[]: transform + ids
|
|
|
|
|
|
|
2026-05-28 09:10:53 +10:00
|
|
|
|
// Cumulative VRAM accounting (bytes), populated at applyCachedModel
|
|
|
|
|
|
// time. Sum of vertex_storage across chunks + index_buffer + mesh_storage
|
|
|
|
|
|
// + instance_storage + per-chunk visible_draws + prefix_sums + uniforms.
|
|
|
|
|
|
// Used by the per-frame stats log to attribute total VRAM.
|
|
|
|
|
|
uint64_t vram_bytes_vbo = 0; // vertex storage total
|
|
|
|
|
|
uint64_t vram_bytes_ebo = 0; // index buffer
|
|
|
|
|
|
uint64_t vram_bytes_ssbo = 0; // mesh + instance + per-chunk small buffers
|
|
|
|
|
|
|
2026-05-27 21:05:38 +10:00
|
|
|
|
// Size mirrors for stats / range checks. vertex_bytes is the sum across
|
|
|
|
|
|
// all chunks; index_count / mesh_count / instance_count are unchanged.
|
2026-05-27 12:48:03 +10:00
|
|
|
|
size_t vertex_bytes = 0;
|
|
|
|
|
|
uint32_t index_count = 0;
|
|
|
|
|
|
uint32_t mesh_count = 0;
|
|
|
|
|
|
uint32_t instance_count = 0;
|
|
|
|
|
|
|
|
|
|
|
|
// CPU side, kept for cull / picking / federation recompose.
|
|
|
|
|
|
std::vector<MeshInfo> meshes;
|
|
|
|
|
|
std::vector<InstanceCpu> instances;
|
|
|
|
|
|
|
2026-05-28 15:20:30 +10:00
|
|
|
|
// Spatial chunk-cull replaced the per-model BVH walk — chunks are
|
|
|
|
|
|
// already a one-level spatial partition of the instances, so a
|
|
|
|
|
|
// single frustum test per chunk gives the same wholesale-reject
|
|
|
|
|
|
// win without the BVH's per-node traversal overhead. The BVH field
|
|
|
|
|
|
// is gone; cull iterates m.chunks instead.
|
2026-05-27 22:04:44 +10:00
|
|
|
|
|
2026-05-27 12:48:03 +10:00
|
|
|
|
bool hidden = false;
|
|
|
|
|
|
};
|
|
|
|
|
|
|
2026-05-28 12:21:52 +10:00
|
|
|
|
// Release every wgpu handle in `m` (including per-chunk and per-model pool
|
|
|
|
|
|
// ranges via `pool.free()`) and clear its size mirrors. Safe to call
|
2026-05-27 12:48:03 +10:00
|
|
|
|
// repeatedly; idempotent on already-released entries.
|
2026-05-28 12:21:52 +10:00
|
|
|
|
void releaseWgpuModelGpuData(WgpuModelGpuData& m, WgpuBufferPool& pool);
|
2026-05-27 12:48:03 +10:00
|
|
|
|
|
|
|
|
|
|
#endif // WGPUMODELGPUDATA_H
|