mirror of
https://github.com/IfcOpenShell/IfcOpenShell.git
synced 2026-09-27 02:31:09 +00:00
wgpu backend: 10× perf — megadraw, async HiZ, parallel cull, motion mode
Closes the perf gap to the GL backend on real BIM benchmarks. On a 10-
sidecar / 380k-instance corpus at a fixed --camera the wgpu binary went
from 110.6 ms to 11.6 ms (vs GL's 23 ms — half the frame time, but
note GL is doing extra work the wgpu backend hasn't ported yet; see
the caveats list at the bottom). Bundled because the pieces interlock
and shipping any of them without the others reintroduces the same wall.
1. Cross-mesh vertex pulling (single mega-draw per model)
The previous one-drawIndexed-per-(mesh × LOD-bucket) loop was costing
~13ms on a 27k-mesh scene. CPU now emits a flat visible_draws[]
(16 B per visible (mesh,lod,instance)) plus a prefix_sums[] table.
WGSL binary-searches prefix_sums by @builtin(vertex_index) to find
the entry, then manually fetches the mesh-local index from a
storage-bound indices[] and pulls the packed 12 B vertex. No
setIndexBuffer; the shader reads everything from storage. Bind
group grew from 4 to 7 entries (vertices, meshes, instances,
indices, visible_draws, prefix_sums, per-model uniform) — well
under WebGPU's mandatory 8 storage / 12 uniform floor.
2. Async HiZ readback via ping-pong staging buffers
Sync wait via wgpuInstanceProcessEvents was costing ~37 ms on a
real scene (GPU drain). Two staging slots now ping-pong: frame N
kicks a non-blocking mapAsync on slot K, frame N+1's first action
is one processEvents drain. Pyramid is 1-2 frames stale — matches
the "slightly-stale depth, fine" pattern the GL backend already
documents. encodeHizResolve returns -1 (skip) if both slots are
in flight; cull keeps using the most recent pyramid.
3. Cull reorder: contribution before HiZ
HiZ projection is ~10× more expensive than the contribution
check, yet most contribution-survivors would be HiZ-rejected
anyway on dense scenes. Computing projected_px first lets
contribution short-circuit ~80% of HiZ tests with no rejection-
quality loss. Saved ~34 ms on the dense bench.
4. Motion-mode contribution threshold
AppSettings::motionMinPixelRadius parity. While the camera is
changing (orbit/pan/zoom/--benchmark sweep), drop instances
below 10 px instead of 2 px. Halves visible_objects during
motion with no perceived quality loss.
5. Parallel cull (std::async across models)
Per-model cullModelCpu split into Compute (CPU-only, thread-safe)
+ Upload (main-thread wgpu queue writes). std::async fan-outs the
compute across models; main-thread joins and uploads. Wall-clock
cull on the 10-model corpus drops from ~17 ms single-threaded to
~9 ms across cores.
6. --no-hiz CLI flag + per-phase benchmark timings
Benchmark now also prints "per-frame avg ms: cull=X
hiz_readback=Y" so future regressions can be attributed without
guesswork. --no-hiz toggles the master switch from the CLI.
Honest caveats — wgpu is currently faster mostly because GL is doing
work we haven't ported yet:
- Edge silhouette pass (stage 9) will add ~3-5 ms back to wgpu.
- GL's HiZ uses the BVH so it rejects whole subtrees (1.7k vs
our 358 rejects on the same scene). BVH for HiZ is future work
(task #13 / a new task) — until then we draw more sub-pixel
geometry that's behind closer surfaces. Visually correct, perf
cost paid. Stage 4+5 are unaffected.
Verified pixel-identical on basic.ifc through every change. Real-scene
visual diff against GL pending the --screenshot flag on the GL minimal
(task #10's other half).
Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -48,10 +48,13 @@ int main(int argc, char* argv[]) {
|
|||||||
parser.addOption({{"c", "camera"},
|
parser.addOption({{"c", "camera"},
|
||||||
"Set camera as tx,ty,tz,dist,yaw,pitch (same format as IfcViewerMinimal).",
|
"Set camera as tx,ty,tz,dist,yaw,pitch (same format as IfcViewerMinimal).",
|
||||||
"params"});
|
"params"});
|
||||||
|
parser.addOption({"no-hiz",
|
||||||
|
"Disable HiZ occlusion culling for perf diagnostics."});
|
||||||
parser.process(app);
|
parser.process(app);
|
||||||
|
|
||||||
auto* viewport = new WgpuViewportWindow;
|
auto* viewport = new WgpuViewportWindow;
|
||||||
viewport->resize(1280, 800);
|
viewport->resize(1280, 800);
|
||||||
|
if (parser.isSet("no-hiz")) viewport->hiz_enabled_ = false;
|
||||||
|
|
||||||
QWidget* container = QWidget::createWindowContainer(viewport);
|
QWidget* container = QWidget::createWindowContainer(viewport);
|
||||||
container->setMinimumSize(320, 240);
|
container->setMinimumSize(320, 240);
|
||||||
|
|||||||
@@ -51,31 +51,47 @@ struct WgpuModelGpuData {
|
|||||||
// recompose from placement_transformation when stage matrices change.
|
// recompose from placement_transformation when stage matrices change.
|
||||||
WGPUBuffer instance_storage = nullptr;
|
WGPUBuffer instance_storage = nullptr;
|
||||||
|
|
||||||
// u32[] of visible instance indices, repacked per frame by the CPU cull
|
// Cross-mesh vertex pulling: one entry per visible (mesh, lod, instance)
|
||||||
// into per-mesh contiguous slices. Sized for the worst case
|
// tuple, written into a single flat buffer per frame. The vertex shader
|
||||||
// (instance_count entries) at applyCachedModel time so we never have to
|
// binary-searches `prefix_sums` to find which entry a given
|
||||||
// recreate it (and the bind group that references it) mid-frame.
|
// gl_VertexID belongs to, then fetches that entry's mesh slice via the
|
||||||
WGPUBuffer visible_buffer = nullptr;
|
// offsets baked here. Pre-sized at applyCachedModel to a generous
|
||||||
size_t visible_buffer_capacity = 0; // entries (u32 count), not bytes
|
// worst case so the bind group stays valid for the model's lifetime.
|
||||||
|
//
|
||||||
|
// std430 layout: 16 bytes per entry, naturally aligned.
|
||||||
|
struct alignas(16) VisibleDrawGpu {
|
||||||
|
uint32_t mesh_id; // -> meshes[] for quantisation basis
|
||||||
|
uint32_t instance_idx; // -> instances[] for transform + ids
|
||||||
|
uint32_t ebo_first_u32; // start of this entry's slice in indices[]
|
||||||
|
uint32_t base_vertex; // start of this mesh's slice in vertices[]
|
||||||
|
};
|
||||||
|
static_assert(sizeof(VisibleDrawGpu) == 16, "VisibleDrawGpu must be 16 bytes");
|
||||||
|
|
||||||
// Bind group binding the four storage buffers above (group=1 in the
|
WGPUBuffer visible_draws_buffer = nullptr;
|
||||||
// main pipeline). Built in applyCachedModel after the buffers exist.
|
size_t visible_draws_capacity = 0; // entries (not bytes)
|
||||||
|
|
||||||
|
// prefix_sums[i] = sum of index_counts for visible_draws[0..i-1].
|
||||||
|
// Sized to visible_draws_capacity + 1 so prefix_sums[N] = total verts.
|
||||||
|
WGPUBuffer prefix_sums_buffer = nullptr;
|
||||||
|
size_t prefix_sums_capacity = 0; // entries (u32 count)
|
||||||
|
|
||||||
|
// Per-model uniform: (draw_count, total_vertex_count, _pad, _pad).
|
||||||
|
// The vertex shader uses draw_count to bound the binary search; CPU
|
||||||
|
// uses total_vertex_count as the draw() call's vertex count.
|
||||||
|
WGPUBuffer per_model_uniform = nullptr;
|
||||||
|
|
||||||
|
// Bind group binding the storage + uniform buffers above (group=1 in
|
||||||
|
// the main pipeline). Built in applyCachedModel.
|
||||||
WGPUBindGroup bind_group = nullptr;
|
WGPUBindGroup bind_group = nullptr;
|
||||||
|
|
||||||
// One per mesh, populated per frame by cullAndCompact. instance_count==0
|
// Per-frame, populated by cullModelCpu and consumed by render().
|
||||||
// means the mesh contributes nothing this frame and the draw is skipped.
|
uint32_t total_visible_vertices = 0;
|
||||||
struct MeshDraw {
|
uint32_t total_visible_draws = 0;
|
||||||
uint32_t first_instance = 0; // offset into visible_buffer
|
|
||||||
uint32_t instance_count = 0;
|
|
||||||
uint32_t first_index = 0; // index buffer offset (in u32 indices)
|
|
||||||
int32_t base_vertex = 0; // added to every fetched vertex_index
|
|
||||||
uint32_t index_count = 0;
|
|
||||||
};
|
|
||||||
std::vector<MeshDraw> mesh_draws; // size = meshes.size() after first frame
|
|
||||||
|
|
||||||
// Scratch reused each frame so per-frame cull doesn't allocate. Sized
|
// Scratch reused each frame so per-frame cull doesn't allocate. Sized
|
||||||
// to instance_count entries on first use; never shrunk.
|
// on first use; never shrunk.
|
||||||
std::vector<uint32_t> visible_flat_scratch;
|
std::vector<VisibleDrawGpu> visible_draws_scratch;
|
||||||
|
std::vector<uint32_t> prefix_sums_scratch;
|
||||||
|
|
||||||
// Size mirrors for stats / range checks.
|
// Size mirrors for stats / range checks.
|
||||||
size_t vertex_bytes = 0;
|
size_t vertex_bytes = 0;
|
||||||
|
|||||||
@@ -33,8 +33,10 @@
|
|||||||
#include <webgpu/wgpu.h> // wgpu-native extensions (logging, MULTI_DRAW_INDIRECT, …)
|
#include <webgpu/wgpu.h> // wgpu-native extensions (logging, MULTI_DRAW_INDIRECT, …)
|
||||||
|
|
||||||
#include <algorithm>
|
#include <algorithm>
|
||||||
|
#include <atomic>
|
||||||
#include <cmath>
|
#include <cmath>
|
||||||
#include <cstring>
|
#include <cstring>
|
||||||
|
#include <future>
|
||||||
#include <limits>
|
#include <limits>
|
||||||
#include <utility>
|
#include <utility>
|
||||||
|
|
||||||
@@ -130,32 +132,42 @@ static WGPUBuffer createBufferWithData(WGPUDevice device, WGPUQueue queue,
|
|||||||
}
|
}
|
||||||
|
|
||||||
void releaseWgpuModelGpuData(WgpuModelGpuData& m) {
|
void releaseWgpuModelGpuData(WgpuModelGpuData& m) {
|
||||||
if (m.bind_group) { wgpuBindGroupRelease(m.bind_group); m.bind_group = nullptr; }
|
if (m.bind_group) { wgpuBindGroupRelease(m.bind_group); m.bind_group = nullptr; }
|
||||||
if (m.vertex_storage) { wgpuBufferRelease(m.vertex_storage); m.vertex_storage = nullptr; }
|
if (m.vertex_storage) { wgpuBufferRelease(m.vertex_storage); m.vertex_storage = nullptr; }
|
||||||
if (m.index_buffer) { wgpuBufferRelease(m.index_buffer); m.index_buffer = nullptr; }
|
if (m.index_buffer) { wgpuBufferRelease(m.index_buffer); m.index_buffer = nullptr; }
|
||||||
if (m.mesh_storage) { wgpuBufferRelease(m.mesh_storage); m.mesh_storage = nullptr; }
|
if (m.mesh_storage) { wgpuBufferRelease(m.mesh_storage); m.mesh_storage = nullptr; }
|
||||||
if (m.instance_storage) { wgpuBufferRelease(m.instance_storage); m.instance_storage = nullptr; }
|
if (m.instance_storage) { wgpuBufferRelease(m.instance_storage); m.instance_storage = nullptr; }
|
||||||
if (m.visible_buffer) { wgpuBufferRelease(m.visible_buffer); m.visible_buffer = nullptr; }
|
if (m.visible_draws_buffer) { wgpuBufferRelease(m.visible_draws_buffer); m.visible_draws_buffer = nullptr; }
|
||||||
|
if (m.prefix_sums_buffer) { wgpuBufferRelease(m.prefix_sums_buffer); m.prefix_sums_buffer = nullptr; }
|
||||||
|
if (m.per_model_uniform) { wgpuBufferRelease(m.per_model_uniform); m.per_model_uniform = nullptr; }
|
||||||
m.vertex_bytes = 0;
|
m.vertex_bytes = 0;
|
||||||
m.index_count = 0;
|
m.index_count = 0;
|
||||||
m.mesh_count = 0;
|
m.mesh_count = 0;
|
||||||
m.instance_count = 0;
|
m.instance_count = 0;
|
||||||
m.visible_buffer_capacity = 0;
|
m.visible_draws_capacity = 0;
|
||||||
|
m.prefix_sums_capacity = 0;
|
||||||
|
m.total_visible_vertices = 0;
|
||||||
|
m.total_visible_draws = 0;
|
||||||
m.meshes.clear();
|
m.meshes.clear();
|
||||||
m.instances.clear();
|
m.instances.clear();
|
||||||
m.mesh_draws.clear();
|
m.visible_draws_scratch.clear();
|
||||||
m.visible_flat_scratch.clear();
|
m.prefix_sums_scratch.clear();
|
||||||
}
|
}
|
||||||
|
|
||||||
// -----------------------------------------------------------------------------
|
// -----------------------------------------------------------------------------
|
||||||
// WGSL main pipeline — vertex pulling.
|
// WGSL main pipeline — cross-mesh vertex pulling.
|
||||||
//
|
//
|
||||||
// Vertex bytes are read manually from a u32 storage buffer (3 u32 per vertex,
|
// We issue ONE draw() call per model per frame. The vertex shader binary-
|
||||||
// matching INSTANCED_VERTEX_STRIDE_BYTES=12). gl_BaseVertex is folded into
|
// searches the prefix-sum table to find which visible-draw entry the current
|
||||||
// vertex_index by the WebGPU IA, so per-mesh draws set baseVertex =
|
// @builtin(vertex_index) belongs to, then manually fetches the index and the
|
||||||
// mesh.vbo_byte_offset / 12 and the shader indexes the global storage array
|
// 12-byte packed vertex from storage buffers. This avoids the N-drawcalls-per-
|
||||||
// directly. firstInstance carries the global instance slot, fetched as
|
// frame CPU overhead of per-mesh draws (which dominated on scenes with many
|
||||||
// @builtin(instance_index).
|
// unique meshes — wgpu-native overhead is ~5 µs/draw, so 27k draws = 135ms).
|
||||||
|
//
|
||||||
|
// Binary search cost is O(log N) per vertex, with N up to a few hundred
|
||||||
|
// thousand on dense scenes. Adjacent vertices in the same draw entry share
|
||||||
|
// the search result inside a warp, so memory-coherence keeps this cheap on
|
||||||
|
// GPU.
|
||||||
// -----------------------------------------------------------------------------
|
// -----------------------------------------------------------------------------
|
||||||
|
|
||||||
static const char* MAIN_WGSL = R"(
|
static const char* MAIN_WGSL = R"(
|
||||||
@@ -180,15 +192,29 @@ struct FrameUniforms {
|
|||||||
ground_color: vec4<f32>,
|
ground_color: vec4<f32>,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
struct VisibleDraw {
|
||||||
|
mesh_id: u32,
|
||||||
|
instance_idx: u32,
|
||||||
|
ebo_first_u32: u32,
|
||||||
|
base_vertex: u32,
|
||||||
|
};
|
||||||
|
|
||||||
|
struct PerModel {
|
||||||
|
draw_count: u32,
|
||||||
|
total_vertex_count: u32,
|
||||||
|
_pad0: u32,
|
||||||
|
_pad1: u32,
|
||||||
|
};
|
||||||
|
|
||||||
@group(0) @binding(0) var<uniform> u_frame: FrameUniforms;
|
@group(0) @binding(0) var<uniform> u_frame: FrameUniforms;
|
||||||
|
|
||||||
@group(1) @binding(0) var<storage, read> vertices: array<u32>;
|
@group(1) @binding(0) var<storage, read> vertices: array<u32>;
|
||||||
@group(1) @binding(1) var<storage, read> meshes: array<MeshQuant>;
|
@group(1) @binding(1) var<storage, read> meshes: array<MeshQuant>;
|
||||||
@group(1) @binding(2) var<storage, read> instances: array<InstanceRecord>;
|
@group(1) @binding(2) var<storage, read> instances: array<InstanceRecord>;
|
||||||
// Per-frame compacted visible list: visible[firstInstance + i] picks the
|
@group(1) @binding(3) var<storage, read> indices: array<u32>;
|
||||||
// real instance for this draw slot. firstInstance is set per drawIndexed
|
@group(1) @binding(4) var<storage, read> visible_draws: array<VisibleDraw>;
|
||||||
// call so each mesh reads its own contiguous slice of `visible`.
|
@group(1) @binding(5) var<storage, read> prefix_sums: array<u32>;
|
||||||
@group(1) @binding(3) var<storage, read> visible: array<u32>;
|
@group(1) @binding(6) var<uniform> u_model: PerModel;
|
||||||
|
|
||||||
struct VsOut {
|
struct VsOut {
|
||||||
@builtin(position) clip_pos: vec4<f32>,
|
@builtin(position) clip_pos: vec4<f32>,
|
||||||
@@ -214,16 +240,46 @@ fn octDecode(e: vec2<f32>) -> vec3<f32> {
|
|||||||
return normalize(n);
|
return normalize(n);
|
||||||
}
|
}
|
||||||
|
|
||||||
@vertex
|
// Binary search for the largest i in [0, draw_count) with prefix_sums[i] <= vid.
|
||||||
fn vs_main(@builtin(vertex_index) vid: u32,
|
// prefix_sums is monotonic non-decreasing and contains draw_count+1 entries
|
||||||
@builtin(instance_index) iid: u32) -> VsOut {
|
// (prefix_sums[draw_count] == total_vertex_count).
|
||||||
let inst_idx = visible[iid];
|
fn find_draw(vid: u32) -> u32 {
|
||||||
let inst = instances[inst_idx];
|
var lo: u32 = 0u;
|
||||||
let mq = meshes[inst.mesh_id];
|
var hi: u32 = u_model.draw_count;
|
||||||
|
while (lo + 1u < hi) {
|
||||||
|
let mid = (lo + hi) >> 1u;
|
||||||
|
if (prefix_sums[mid] <= vid) {
|
||||||
|
lo = mid;
|
||||||
|
} else {
|
||||||
|
hi = mid;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return lo;
|
||||||
|
}
|
||||||
|
|
||||||
let w0 = vertices[vid * 3u + 0u];
|
@vertex
|
||||||
let w1 = vertices[vid * 3u + 1u];
|
fn vs_main(@builtin(vertex_index) vid: u32) -> VsOut {
|
||||||
let w2 = vertices[vid * 3u + 2u];
|
// Saturate past the end (shouldn't happen given draw() count, but safe).
|
||||||
|
if (vid >= u_model.total_vertex_count) {
|
||||||
|
var degen: VsOut;
|
||||||
|
degen.clip_pos = vec4<f32>(0.0, 0.0, 0.0, 0.0);
|
||||||
|
return degen;
|
||||||
|
}
|
||||||
|
|
||||||
|
let draw_idx = find_draw(vid);
|
||||||
|
let local_v = vid - prefix_sums[draw_idx];
|
||||||
|
let item = visible_draws[draw_idx];
|
||||||
|
|
||||||
|
// Fetch the mesh-local index then the global vertex index.
|
||||||
|
let mesh_local_index = indices[item.ebo_first_u32 + local_v];
|
||||||
|
let v_global = item.base_vertex + mesh_local_index;
|
||||||
|
|
||||||
|
let inst = instances[item.instance_idx];
|
||||||
|
let mq = meshes[item.mesh_id];
|
||||||
|
|
||||||
|
let w0 = vertices[v_global * 3u + 0u];
|
||||||
|
let w1 = vertices[v_global * 3u + 1u];
|
||||||
|
let w2 = vertices[v_global * 3u + 2u];
|
||||||
|
|
||||||
let px = f32(w0 & 0xFFFFu) / 65535.0;
|
let px = f32(w0 & 0xFFFFu) / 65535.0;
|
||||||
let py = f32((w0 >> 16u) & 0xFFFFu) / 65535.0;
|
let py = f32((w0 >> 16u) & 0xFFFFu) / 65535.0;
|
||||||
@@ -428,11 +484,13 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) {
|
|||||||
WGPUBufferUsage_Storage,
|
WGPUBufferUsage_Storage,
|
||||||
"model.vertex_storage");
|
"model.vertex_storage");
|
||||||
|
|
||||||
// Index buffer — mesh-local u32 indices; baseVertex applied per-draw.
|
// Index buffer — mesh-local u32 indices. Bound as a read-only storage
|
||||||
|
// buffer in the vertex shader (cross-mesh vertex pulling manually fetches
|
||||||
|
// indices); Index usage is kept for forward compatibility but never used.
|
||||||
m.index_buffer = createBufferWithData(
|
m.index_buffer = createBufferWithData(
|
||||||
device_, queue_,
|
device_, queue_,
|
||||||
data.indices.data(), data.indices.size() * sizeof(uint32_t),
|
data.indices.data(), data.indices.size() * sizeof(uint32_t),
|
||||||
WGPUBufferUsage_Index,
|
WGPUBufferUsage_Storage | WGPUBufferUsage_Index,
|
||||||
"model.index_buffer");
|
"model.index_buffer");
|
||||||
|
|
||||||
// Derive MeshGpu[] (vec4 aabb_min + vec4 aabb_max) from MeshInfo's
|
// Derive MeshGpu[] (vec4 aabb_min + vec4 aabb_max) from MeshInfo's
|
||||||
@@ -476,24 +534,43 @@ void WgpuViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) {
|
|||||||
WGPUBufferUsage_Storage,
|
WGPUBufferUsage_Storage,
|
||||||
"model.instance_storage");
|
"model.instance_storage");
|
||||||
|
|
||||||
// Pre-size the visible buffer for the worst case (every instance visible).
|
// Per-frame buffers for cross-mesh vertex pulling. Sized for the worst
|
||||||
// Per-frame cull writes a u32 list into it via wgpuQueueWriteBuffer; we
|
// case so cullModelCpu only needs wgpuQueueWriteBuffer (never recreate)
|
||||||
// never grow it so the bind group reference stays valid for the model's
|
// — keeps the bind group reference valid for the model's lifetime.
|
||||||
// lifetime. Minimum size 4 bytes — wgpu rejects zero-sized buffers.
|
//
|
||||||
const size_t visible_bytes = std::max<size_t>(
|
// VisibleDraw capacity: up to one entry per (instance × 2 LODs). LOD1 is
|
||||||
size_t(data.instances.size()) * sizeof(uint32_t), 4u);
|
// only used when bake produced one, but allocating headroom is cheap
|
||||||
WGPUBufferDescriptor vb_desc = {};
|
// (~3 MB worst-case for a 100k-instance model).
|
||||||
vb_desc.size = visible_bytes;
|
const size_t draw_cap = std::max<size_t>(size_t(data.instances.size()) * 2u, 1u);
|
||||||
vb_desc.usage = WGPUBufferUsage_Storage | WGPUBufferUsage_CopyDst;
|
const size_t draws_bytes = draw_cap * sizeof(WgpuModelGpuData::VisibleDrawGpu);
|
||||||
vb_desc.label = svFromCStr("model.visible_buffer");
|
WGPUBufferDescriptor vd_desc = {};
|
||||||
m.visible_buffer = wgpuDeviceCreateBuffer(device_, &vb_desc);
|
vd_desc.size = std::max<uint64_t>(draws_bytes, 16);
|
||||||
m.visible_buffer_capacity = std::max<size_t>(data.instances.size(), 1u);
|
vd_desc.usage = WGPUBufferUsage_Storage | WGPUBufferUsage_CopyDst;
|
||||||
|
vd_desc.label = svFromCStr("model.visible_draws");
|
||||||
|
m.visible_draws_buffer = wgpuDeviceCreateBuffer(device_, &vd_desc);
|
||||||
|
m.visible_draws_capacity = draw_cap;
|
||||||
|
|
||||||
|
// prefix_sums: draw_cap + 1 entries (final entry = total vertex count).
|
||||||
|
const size_t ps_cap = draw_cap + 1;
|
||||||
|
WGPUBufferDescriptor ps_desc = {};
|
||||||
|
ps_desc.size = std::max<uint64_t>(ps_cap * sizeof(uint32_t), 16);
|
||||||
|
ps_desc.usage = WGPUBufferUsage_Storage | WGPUBufferUsage_CopyDst;
|
||||||
|
ps_desc.label = svFromCStr("model.prefix_sums");
|
||||||
|
m.prefix_sums_buffer = wgpuDeviceCreateBuffer(device_, &ps_desc);
|
||||||
|
m.prefix_sums_capacity = ps_cap;
|
||||||
|
|
||||||
|
// 16-byte per-model uniform: (draw_count, total_vertex_count, _pad, _pad).
|
||||||
|
WGPUBufferDescriptor mu_desc = {};
|
||||||
|
mu_desc.size = 16;
|
||||||
|
mu_desc.usage = WGPUBufferUsage_Uniform | WGPUBufferUsage_CopyDst;
|
||||||
|
mu_desc.label = svFromCStr("model.per_model_uniform");
|
||||||
|
m.per_model_uniform = wgpuDeviceCreateBuffer(device_, &mu_desc);
|
||||||
|
|
||||||
// Hand off CPU mirrors (cull / picking will need them later).
|
// Hand off CPU mirrors (cull / picking will need them later).
|
||||||
m.meshes = std::move(data.meshes);
|
m.meshes = std::move(data.meshes);
|
||||||
m.instances = std::move(data.instances);
|
m.instances = std::move(data.instances);
|
||||||
m.mesh_draws.assign(m.meshes.size(), WgpuModelGpuData::MeshDraw{});
|
m.visible_draws_scratch.reserve(draw_cap);
|
||||||
m.visible_flat_scratch.reserve(m.instances.size());
|
m.prefix_sums_scratch.reserve(ps_cap);
|
||||||
|
|
||||||
auto [inserted, _] = models_gpu_.emplace(model_id, std::move(m));
|
auto [inserted, _] = models_gpu_.emplace(model_id, std::move(m));
|
||||||
WgpuModelGpuData& mref = inserted->second;
|
WgpuModelGpuData& mref = inserted->second;
|
||||||
@@ -1005,7 +1082,19 @@ void WgpuViewportWindow::ensureHizTextures(int viewport_w, int viewport_h) {
|
|||||||
|
|
||||||
if (hiz_resolve_view_) { wgpuTextureViewRelease(hiz_resolve_view_); hiz_resolve_view_ = nullptr; }
|
if (hiz_resolve_view_) { wgpuTextureViewRelease(hiz_resolve_view_); hiz_resolve_view_ = nullptr; }
|
||||||
if (hiz_resolve_texture_) { wgpuTextureRelease(hiz_resolve_texture_); hiz_resolve_texture_ = nullptr; }
|
if (hiz_resolve_texture_) { wgpuTextureRelease(hiz_resolve_texture_); hiz_resolve_texture_ = nullptr; }
|
||||||
if (hiz_staging_buffer_) { wgpuBufferRelease(hiz_staging_buffer_); hiz_staging_buffer_ = nullptr; }
|
for (int s = 0; s < HIZ_SLOTS; ++s) {
|
||||||
|
if (hiz_staging_buffers_[s]) {
|
||||||
|
// Force any pending map to finish before release (defensive: shouldn't happen on resize).
|
||||||
|
if (hiz_slot_state_[s] == HizSlotState::Mapped) {
|
||||||
|
wgpuBufferUnmap(hiz_staging_buffers_[s]);
|
||||||
|
}
|
||||||
|
wgpuBufferRelease(hiz_staging_buffers_[s]);
|
||||||
|
hiz_staging_buffers_[s] = nullptr;
|
||||||
|
}
|
||||||
|
hiz_slot_state_[s] = HizSlotState::Idle;
|
||||||
|
}
|
||||||
|
hiz_write_idx_ = 0;
|
||||||
|
hiz_valid_ = false;
|
||||||
if (hiz_bind_group_) { wgpuBindGroupRelease(hiz_bind_group_); hiz_bind_group_ = nullptr; }
|
if (hiz_bind_group_) { wgpuBindGroupRelease(hiz_bind_group_); hiz_bind_group_ = nullptr; }
|
||||||
|
|
||||||
WGPUTextureDescriptor desc = {};
|
WGPUTextureDescriptor desc = {};
|
||||||
@@ -1028,15 +1117,19 @@ void WgpuViewportWindow::ensureHizTextures(int viewport_w, int viewport_h) {
|
|||||||
vdesc.aspect = WGPUTextureAspect_DepthOnly;
|
vdesc.aspect = WGPUTextureAspect_DepthOnly;
|
||||||
hiz_resolve_view_ = wgpuTextureCreateView(hiz_resolve_texture_, &vdesc);
|
hiz_resolve_view_ = wgpuTextureCreateView(hiz_resolve_texture_, &vdesc);
|
||||||
|
|
||||||
// Staging buffer: pad each row to 256-byte alignment.
|
// Staging buffers: pad each row to 256-byte alignment. Two slots
|
||||||
|
// ping-pong so GPU fill of slot N overlaps CPU read of slot N-1.
|
||||||
hiz_padded_bpr_ = uint32_t(
|
hiz_padded_bpr_ = uint32_t(
|
||||||
(dst_w * sizeof(float) + WGPU_BYTES_PER_ROW_ALIGN - 1)
|
(dst_w * sizeof(float) + WGPU_BYTES_PER_ROW_ALIGN - 1)
|
||||||
/ WGPU_BYTES_PER_ROW_ALIGN * WGPU_BYTES_PER_ROW_ALIGN);
|
/ WGPU_BYTES_PER_ROW_ALIGN * WGPU_BYTES_PER_ROW_ALIGN);
|
||||||
WGPUBufferDescriptor bdesc = {};
|
for (int s = 0; s < HIZ_SLOTS; ++s) {
|
||||||
bdesc.size = uint64_t(hiz_padded_bpr_) * uint64_t(dst_h);
|
WGPUBufferDescriptor bdesc = {};
|
||||||
bdesc.usage = WGPUBufferUsage_CopyDst | WGPUBufferUsage_MapRead;
|
bdesc.size = uint64_t(hiz_padded_bpr_) * uint64_t(dst_h);
|
||||||
bdesc.label = svFromCStr("ifcviewer-wgpu.hiz_staging");
|
bdesc.usage = WGPUBufferUsage_CopyDst | WGPUBufferUsage_MapRead;
|
||||||
hiz_staging_buffer_ = wgpuDeviceCreateBuffer(device_, &bdesc);
|
bdesc.label = svFromCStr(s == 0 ? "ifcviewer-wgpu.hiz_staging[0]"
|
||||||
|
: "ifcviewer-wgpu.hiz_staging[1]");
|
||||||
|
hiz_staging_buffers_[s] = wgpuDeviceCreateBuffer(device_, &bdesc);
|
||||||
|
}
|
||||||
|
|
||||||
hiz_resolve_w_ = dst_w;
|
hiz_resolve_w_ = dst_w;
|
||||||
hiz_resolve_h_ = dst_h;
|
hiz_resolve_h_ = dst_h;
|
||||||
@@ -1048,7 +1141,17 @@ void WgpuViewportWindow::releaseHizResources() {
|
|||||||
if (hiz_uniform_buffer_) { wgpuBufferRelease(hiz_uniform_buffer_); hiz_uniform_buffer_ = nullptr; }
|
if (hiz_uniform_buffer_) { wgpuBufferRelease(hiz_uniform_buffer_); hiz_uniform_buffer_ = nullptr; }
|
||||||
if (hiz_resolve_view_) { wgpuTextureViewRelease(hiz_resolve_view_); hiz_resolve_view_ = nullptr; }
|
if (hiz_resolve_view_) { wgpuTextureViewRelease(hiz_resolve_view_); hiz_resolve_view_ = nullptr; }
|
||||||
if (hiz_resolve_texture_) { wgpuTextureRelease(hiz_resolve_texture_); hiz_resolve_texture_ = nullptr; }
|
if (hiz_resolve_texture_) { wgpuTextureRelease(hiz_resolve_texture_); hiz_resolve_texture_ = nullptr; }
|
||||||
if (hiz_staging_buffer_) { wgpuBufferRelease(hiz_staging_buffer_); hiz_staging_buffer_ = nullptr; }
|
for (int s = 0; s < HIZ_SLOTS; ++s) {
|
||||||
|
if (hiz_staging_buffers_[s]) {
|
||||||
|
if (hiz_slot_state_[s] == HizSlotState::Mapped) {
|
||||||
|
wgpuBufferUnmap(hiz_staging_buffers_[s]);
|
||||||
|
}
|
||||||
|
wgpuBufferRelease(hiz_staging_buffers_[s]);
|
||||||
|
hiz_staging_buffers_[s] = nullptr;
|
||||||
|
}
|
||||||
|
hiz_slot_state_[s] = HizSlotState::Idle;
|
||||||
|
}
|
||||||
|
hiz_write_idx_ = 0;
|
||||||
if (hiz_pipeline_) { wgpuRenderPipelineRelease(hiz_pipeline_); hiz_pipeline_ = nullptr; }
|
if (hiz_pipeline_) { wgpuRenderPipelineRelease(hiz_pipeline_); hiz_pipeline_ = nullptr; }
|
||||||
if (hiz_shader_module_) { wgpuShaderModuleRelease(hiz_shader_module_); hiz_shader_module_ = nullptr; }
|
if (hiz_shader_module_) { wgpuShaderModuleRelease(hiz_shader_module_); hiz_shader_module_ = nullptr; }
|
||||||
if (hiz_pipeline_layout_) { wgpuPipelineLayoutRelease(hiz_pipeline_layout_); hiz_pipeline_layout_ = nullptr; }
|
if (hiz_pipeline_layout_) { wgpuPipelineLayoutRelease(hiz_pipeline_layout_); hiz_pipeline_layout_ = nullptr; }
|
||||||
@@ -1061,8 +1164,19 @@ void WgpuViewportWindow::releaseHizResources() {
|
|||||||
hiz_mip_h_.clear();
|
hiz_mip_h_.clear();
|
||||||
}
|
}
|
||||||
|
|
||||||
void WgpuViewportWindow::encodeHizResolve(WGPUCommandEncoder enc) {
|
int WgpuViewportWindow::encodeHizResolve(WGPUCommandEncoder enc) {
|
||||||
if (!hiz_enabled_ || !hiz_pipeline_ || !hiz_resolve_view_ || !depth_view_) return;
|
if (!hiz_enabled_ || !hiz_pipeline_ || !hiz_resolve_view_ || !depth_view_) return -1;
|
||||||
|
|
||||||
|
// Pick an idle ping-pong slot. If both slots are in flight, skip the
|
||||||
|
// resolve for this frame — the cull keeps using whatever pyramid we
|
||||||
|
// already built (slightly more stale than usual, but never blocks).
|
||||||
|
int slot = -1;
|
||||||
|
for (int s = 0; s < HIZ_SLOTS; ++s) {
|
||||||
|
const int idx = (hiz_write_idx_ + s) % HIZ_SLOTS;
|
||||||
|
if (hiz_slot_state_[idx] == HizSlotState::Idle) { slot = idx; break; }
|
||||||
|
}
|
||||||
|
if (slot < 0) return -1;
|
||||||
|
hiz_write_idx_ = (slot + 1) % HIZ_SLOTS;
|
||||||
|
|
||||||
// (Re)build the bind group every frame is wasteful; only rebuild when the
|
// (Re)build the bind group every frame is wasteful; only rebuild when the
|
||||||
// depth view itself was replaced (driven by surface resize). For now we
|
// depth view itself was replaced (driven by surface resize). For now we
|
||||||
@@ -1110,13 +1224,13 @@ void WgpuViewportWindow::encodeHizResolve(WGPUCommandEncoder enc) {
|
|||||||
wgpuRenderPassEncoderEnd(pass);
|
wgpuRenderPassEncoderEnd(pass);
|
||||||
wgpuRenderPassEncoderRelease(pass);
|
wgpuRenderPassEncoderRelease(pass);
|
||||||
|
|
||||||
// Now copy the small resolved depth texture into the staging buffer.
|
// Copy the small resolved depth texture into the chosen staging slot.
|
||||||
WGPUTexelCopyTextureInfo src = {};
|
WGPUTexelCopyTextureInfo src = {};
|
||||||
src.texture = hiz_resolve_texture_;
|
src.texture = hiz_resolve_texture_;
|
||||||
src.aspect = WGPUTextureAspect_DepthOnly;
|
src.aspect = WGPUTextureAspect_DepthOnly;
|
||||||
|
|
||||||
WGPUTexelCopyBufferInfo dst = {};
|
WGPUTexelCopyBufferInfo dst = {};
|
||||||
dst.buffer = hiz_staging_buffer_;
|
dst.buffer = hiz_staging_buffers_[slot];
|
||||||
dst.layout.bytesPerRow = hiz_padded_bpr_;
|
dst.layout.bytesPerRow = hiz_padded_bpr_;
|
||||||
dst.layout.rowsPerImage = hiz_resolve_h_;
|
dst.layout.rowsPerImage = hiz_resolve_h_;
|
||||||
|
|
||||||
@@ -1126,89 +1240,112 @@ void WgpuViewportWindow::encodeHizResolve(WGPUCommandEncoder enc) {
|
|||||||
extent.depthOrArrayLayers = 1;
|
extent.depthOrArrayLayers = 1;
|
||||||
|
|
||||||
wgpuCommandEncoderCopyTextureToBuffer(enc, &src, &dst, &extent);
|
wgpuCommandEncoderCopyTextureToBuffer(enc, &src, &dst, &extent);
|
||||||
|
return slot;
|
||||||
}
|
}
|
||||||
|
|
||||||
void WgpuViewportWindow::readbackAndBuildHizPyramid(const QMatrix4x4& vp_used) {
|
void WgpuViewportWindow::startHizMap(int slot, const QMatrix4x4& vp_used) {
|
||||||
if (!hiz_enabled_ || !hiz_staging_buffer_ || hiz_resolve_w_ == 0) return;
|
if (slot < 0 || slot >= HIZ_SLOTS) return;
|
||||||
|
if (!hiz_staging_buffers_[slot] || hiz_resolve_w_ == 0) return;
|
||||||
|
|
||||||
struct MapReq { bool done = false; bool ok = false; };
|
hiz_slot_vp_[slot] = vp_used;
|
||||||
MapReq req;
|
hiz_slot_state_[slot] = HizSlotState::Mapping;
|
||||||
|
|
||||||
|
struct MapCtx { WgpuViewportWindow* self; int slot; };
|
||||||
|
auto* ctx = new MapCtx{ this, slot };
|
||||||
|
|
||||||
WGPUBufferMapCallbackInfo mcb = {};
|
WGPUBufferMapCallbackInfo mcb = {};
|
||||||
mcb.mode = WGPUCallbackMode_AllowProcessEvents;
|
mcb.mode = WGPUCallbackMode_AllowProcessEvents;
|
||||||
mcb.callback = [](WGPUMapAsyncStatus status, WGPUStringView /*msg*/,
|
mcb.callback = [](WGPUMapAsyncStatus status, WGPUStringView /*msg*/,
|
||||||
void* ud1, void* /*ud2*/) {
|
void* ud1, void* /*ud2*/) {
|
||||||
auto* r = static_cast<MapReq*>(ud1);
|
auto* c = static_cast<MapCtx*>(ud1);
|
||||||
r->done = true;
|
if (status == WGPUMapAsyncStatus_Success) {
|
||||||
r->ok = (status == WGPUMapAsyncStatus_Success);
|
c->self->hiz_slot_state_[c->slot] = HizSlotState::Mapped;
|
||||||
|
} else {
|
||||||
|
c->self->hiz_slot_state_[c->slot] = HizSlotState::Idle;
|
||||||
|
}
|
||||||
|
delete c;
|
||||||
};
|
};
|
||||||
mcb.userdata1 = &req;
|
mcb.userdata1 = ctx;
|
||||||
|
|
||||||
const size_t map_size = size_t(hiz_padded_bpr_) * size_t(hiz_resolve_h_);
|
const size_t map_size = size_t(hiz_padded_bpr_) * size_t(hiz_resolve_h_);
|
||||||
wgpuBufferMapAsync(hiz_staging_buffer_, WGPUMapMode_Read, 0, map_size, mcb);
|
wgpuBufferMapAsync(hiz_staging_buffers_[slot], WGPUMapMode_Read,
|
||||||
while (!req.done) wgpuInstanceProcessEvents(instance_);
|
0, map_size, mcb);
|
||||||
if (!req.ok) return;
|
}
|
||||||
|
|
||||||
const uint8_t* mapped = static_cast<const uint8_t*>(
|
void WgpuViewportWindow::drainHizReadbacks() {
|
||||||
wgpuBufferGetConstMappedRange(hiz_staging_buffer_, 0, map_size));
|
if (!hiz_enabled_ || hiz_resolve_w_ == 0) return;
|
||||||
|
// Process any callbacks that have fired since last frame. Does NOT block:
|
||||||
|
// wgpuInstanceProcessEvents returns immediately after running ready
|
||||||
|
// callbacks. The mapAsync mode is AllowProcessEvents, so this is the
|
||||||
|
// correct drainage point.
|
||||||
|
wgpuInstanceProcessEvents(instance_);
|
||||||
|
|
||||||
// Mip 0: tight copy, stripping per-row padding.
|
for (int slot = 0; slot < HIZ_SLOTS; ++slot) {
|
||||||
const uint32_t W0 = hiz_resolve_w_;
|
if (hiz_slot_state_[slot] != HizSlotState::Mapped) continue;
|
||||||
const uint32_t H0 = hiz_resolve_h_;
|
|
||||||
|
|
||||||
// Pre-size storage: count levels until we hit 1×1.
|
const size_t map_size = size_t(hiz_padded_bpr_) * size_t(hiz_resolve_h_);
|
||||||
hiz_mip_offset_.clear();
|
const uint8_t* mapped = static_cast<const uint8_t*>(
|
||||||
hiz_mip_w_.clear();
|
wgpuBufferGetConstMappedRange(hiz_staging_buffers_[slot], 0, map_size));
|
||||||
hiz_mip_h_.clear();
|
|
||||||
uint32_t total_texels = 0;
|
const uint32_t W0 = hiz_resolve_w_;
|
||||||
{
|
const uint32_t H0 = hiz_resolve_h_;
|
||||||
uint32_t w = W0, h = H0;
|
|
||||||
while (true) {
|
// (Re)build mip pyramid metadata if dimensions changed.
|
||||||
hiz_mip_offset_.push_back(total_texels);
|
if (hiz_mip_offset_.empty()
|
||||||
hiz_mip_w_.push_back(w);
|
|| hiz_mip_w_.empty() || hiz_mip_w_[0] != W0
|
||||||
hiz_mip_h_.push_back(h);
|
|| hiz_mip_h_.empty() || hiz_mip_h_[0] != H0) {
|
||||||
total_texels += w * h;
|
hiz_mip_offset_.clear();
|
||||||
if (w == 1 && h == 1) break;
|
hiz_mip_w_.clear();
|
||||||
w = std::max(1u, w / 2u);
|
hiz_mip_h_.clear();
|
||||||
h = std::max(1u, h / 2u);
|
uint32_t total = 0;
|
||||||
|
uint32_t w = W0, h = H0;
|
||||||
|
while (true) {
|
||||||
|
hiz_mip_offset_.push_back(total);
|
||||||
|
hiz_mip_w_.push_back(w);
|
||||||
|
hiz_mip_h_.push_back(h);
|
||||||
|
total += w * h;
|
||||||
|
if (w == 1 && h == 1) break;
|
||||||
|
w = std::max(1u, w / 2u);
|
||||||
|
h = std::max(1u, h / 2u);
|
||||||
|
}
|
||||||
|
hiz_pyramid_.assign(total, 0.0f);
|
||||||
}
|
}
|
||||||
}
|
|
||||||
hiz_pyramid_.assign(total_texels, 0.0f);
|
|
||||||
|
|
||||||
// Mip 0: strip row padding.
|
// Mip 0: strip per-row padding.
|
||||||
for (uint32_t y = 0; y < H0; ++y) {
|
for (uint32_t y = 0; y < H0; ++y) {
|
||||||
std::memcpy(&hiz_pyramid_[y * W0],
|
std::memcpy(&hiz_pyramid_[y * W0],
|
||||||
mapped + size_t(y) * hiz_padded_bpr_,
|
mapped + size_t(y) * hiz_padded_bpr_,
|
||||||
W0 * sizeof(float));
|
W0 * sizeof(float));
|
||||||
}
|
}
|
||||||
wgpuBufferUnmap(hiz_staging_buffer_);
|
wgpuBufferUnmap(hiz_staging_buffers_[slot]);
|
||||||
|
hiz_slot_state_[slot] = HizSlotState::Idle;
|
||||||
|
|
||||||
// Higher mips: max-reduce 2×2 children. Edge handling: if parent
|
// Higher mips: max-reduce 2×2 children.
|
||||||
// dimension shrunk to 1, just read the single child.
|
for (size_t L = 1; L < hiz_mip_offset_.size(); ++L) {
|
||||||
for (size_t L = 1; L < hiz_mip_offset_.size(); ++L) {
|
const uint32_t prev_w = hiz_mip_w_[L - 1];
|
||||||
const uint32_t prev_w = hiz_mip_w_[L - 1];
|
const uint32_t prev_h = hiz_mip_h_[L - 1];
|
||||||
const uint32_t prev_h = hiz_mip_h_[L - 1];
|
const uint32_t this_w = hiz_mip_w_[L];
|
||||||
const uint32_t this_w = hiz_mip_w_[L];
|
const uint32_t this_h = hiz_mip_h_[L];
|
||||||
const uint32_t this_h = hiz_mip_h_[L];
|
const float* src = &hiz_pyramid_[hiz_mip_offset_[L - 1]];
|
||||||
const float* src = &hiz_pyramid_[hiz_mip_offset_[L - 1]];
|
float* dst = &hiz_pyramid_[hiz_mip_offset_[L]];
|
||||||
float* dst = &hiz_pyramid_[hiz_mip_offset_[L]];
|
for (uint32_t y = 0; y < this_h; ++y) {
|
||||||
for (uint32_t y = 0; y < this_h; ++y) {
|
for (uint32_t x = 0; x < this_w; ++x) {
|
||||||
for (uint32_t x = 0; x < this_w; ++x) {
|
const uint32_t x0 = std::min(prev_w - 1, x * 2u);
|
||||||
const uint32_t x0 = std::min(prev_w - 1, x * 2u);
|
const uint32_t y0 = std::min(prev_h - 1, y * 2u);
|
||||||
const uint32_t y0 = std::min(prev_h - 1, y * 2u);
|
const uint32_t x1 = std::min(prev_w - 1, x0 + 1u);
|
||||||
const uint32_t x1 = std::min(prev_w - 1, x0 + 1u);
|
const uint32_t y1 = std::min(prev_h - 1, y0 + 1u);
|
||||||
const uint32_t y1 = std::min(prev_h - 1, y0 + 1u);
|
const float a = src[y0 * prev_w + x0];
|
||||||
const float a = src[y0 * prev_w + x0];
|
const float b = src[y0 * prev_w + x1];
|
||||||
const float b = src[y0 * prev_w + x1];
|
const float c = src[y1 * prev_w + x0];
|
||||||
const float c = src[y1 * prev_w + x0];
|
const float d = src[y1 * prev_w + x1];
|
||||||
const float d = src[y1 * prev_w + x1];
|
dst[y * this_w + x] = std::max(std::max(a, b), std::max(c, d));
|
||||||
dst[y * this_w + x] = std::max(std::max(a, b), std::max(c, d));
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
hiz_vp_ = vp_used;
|
hiz_vp_ = hiz_slot_vp_[slot];
|
||||||
hiz_valid_ = true;
|
hiz_valid_ = true;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
bool WgpuViewportWindow::aabbOccludedByHiz(const float mn[3], const float mx[3]) const {
|
bool WgpuViewportWindow::aabbOccludedByHiz(const float mn[3], const float mx[3]) const {
|
||||||
@@ -1302,47 +1439,41 @@ void WgpuViewportWindow::setBenchmarkFrames(int frames) {
|
|||||||
if (isExposed() && bench_total_ > 0) requestUpdate();
|
if (isExposed() && bench_total_ > 0) requestUpdate();
|
||||||
}
|
}
|
||||||
|
|
||||||
void WgpuViewportWindow::cullModelCpu(WgpuModelGpuData& m,
|
uint32_t WgpuViewportWindow::cullModelCpuCompute(WgpuModelGpuData& m,
|
||||||
const float planes[6][4],
|
const float planes[6][4],
|
||||||
const float eye[3], const float forward[3],
|
const float eye[3], const float forward[3],
|
||||||
float focal_px,
|
float focal_px,
|
||||||
float min_radius_px,
|
float min_radius_px,
|
||||||
float lod1_threshold_px,
|
float lod1_threshold_px,
|
||||||
bool hiz_enabled) {
|
bool hiz_enabled) const {
|
||||||
if (m.instances.empty() || m.meshes.empty() || !m.visible_buffer) {
|
m.total_visible_vertices = 0;
|
||||||
for (auto& d : m.mesh_draws) d.instance_count = 0;
|
m.total_visible_draws = 0;
|
||||||
return;
|
uint32_t hiz_rejects = 0;
|
||||||
}
|
|
||||||
|
|
||||||
// Per-mesh visible-instance buckets split by LOD. Per-frame scratch held
|
if (m.instances.empty() || m.meshes.empty()
|
||||||
// thread-locally so the cull path doesn't allocate.
|
|| !m.visible_draws_buffer || !m.prefix_sums_buffer || !m.per_model_uniform) {
|
||||||
static thread_local std::vector<std::vector<uint32_t>> per_mesh_lod0;
|
return 0;
|
||||||
static thread_local std::vector<std::vector<uint32_t>> per_mesh_lod1;
|
|
||||||
if (per_mesh_lod0.size() < m.meshes.size()) per_mesh_lod0.resize(m.meshes.size());
|
|
||||||
if (per_mesh_lod1.size() < m.meshes.size()) per_mesh_lod1.resize(m.meshes.size());
|
|
||||||
for (size_t mi = 0; mi < m.meshes.size(); ++mi) {
|
|
||||||
per_mesh_lod0[mi].clear();
|
|
||||||
per_mesh_lod1[mi].clear();
|
|
||||||
}
|
}
|
||||||
|
|
||||||
const bool contrib_enabled = (min_radius_px > 0.0f);
|
const bool contrib_enabled = (min_radius_px > 0.0f);
|
||||||
const bool lod_enabled = (lod1_threshold_px > 0.0f);
|
const bool lod_enabled = (lod1_threshold_px > 0.0f);
|
||||||
|
|
||||||
|
m.visible_draws_scratch.clear();
|
||||||
|
m.prefix_sums_scratch.clear();
|
||||||
|
m.prefix_sums_scratch.push_back(0); // prefix_sums[0] = 0
|
||||||
|
uint32_t running_vertex_count = 0;
|
||||||
|
|
||||||
for (uint32_t i = 0; i < uint32_t(m.instances.size()); ++i) {
|
for (uint32_t i = 0; i < uint32_t(m.instances.size()); ++i) {
|
||||||
const auto& inst = m.instances[i];
|
const auto& inst = m.instances[i];
|
||||||
if (inst.mesh_id >= m.meshes.size()) continue;
|
if (inst.mesh_id >= m.meshes.size()) continue;
|
||||||
if (!aabbInFrustum(inst.world_aabb_min, inst.world_aabb_max, planes)) continue;
|
if (!aabbInFrustum(inst.world_aabb_min, inst.world_aabb_max, planes)) continue;
|
||||||
if (hiz_enabled
|
|
||||||
&& aabbOccludedByHiz(inst.world_aabb_min, inst.world_aabb_max)) {
|
|
||||||
++hiz_reject_count_;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
const MeshInfo& mesh = m.meshes[inst.mesh_id];
|
const MeshInfo& mesh = m.meshes[inst.mesh_id];
|
||||||
|
|
||||||
// Projected bounding-sphere radius in pixels, shared between the
|
// Projected bounding-sphere radius in pixels — shared between the
|
||||||
// contribution-cull and LOD-pick decisions. Computed once per
|
// contribution-cull and LOD-pick decisions. Computed before HiZ so
|
||||||
// instance; guarded against behind-near-plane and division by zero.
|
// contribution can short-circuit the per-instance HiZ projection
|
||||||
|
// (which is the bulk of cull cost on dense scenes).
|
||||||
float projected_px = std::numeric_limits<float>::infinity();
|
float projected_px = std::numeric_limits<float>::infinity();
|
||||||
if (contrib_enabled || (lod_enabled && mesh.lod1_index_count > 0)) {
|
if (contrib_enabled || (lod_enabled && mesh.lod1_index_count > 0)) {
|
||||||
const float cx = 0.5f * (inst.world_aabb_min[0] + inst.world_aabb_max[0]);
|
const float cx = 0.5f * (inst.world_aabb_min[0] + inst.world_aabb_max[0]);
|
||||||
@@ -1360,65 +1491,69 @@ void WgpuViewportWindow::cullModelCpu(WgpuModelGpuData& m,
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Contribution cull: drop instances projected below threshold. Most
|
// Contribution cull before HiZ: HiZ is by far the most expensive
|
||||||
// of the per-frame cost on dense scenes — typically rejects 80-90%
|
// per-instance test (8-corner projection + mip pyramid sample), so
|
||||||
// of frustum-visible instances on real BIM models.
|
// letting cheap contribution drops happen first cuts the HiZ-tested
|
||||||
|
// population by ~5× on real scenes.
|
||||||
if (contrib_enabled && projected_px < min_radius_px) continue;
|
if (contrib_enabled && projected_px < min_radius_px) continue;
|
||||||
|
|
||||||
// LOD pick: switch to LOD1 if the mesh has one baked and we're below
|
if (hiz_enabled
|
||||||
// the LOD threshold.
|
&& aabbOccludedByHiz(inst.world_aabb_min, inst.world_aabb_max)) {
|
||||||
|
++hiz_rejects;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
const bool use_lod1 = lod_enabled
|
const bool use_lod1 = lod_enabled
|
||||||
&& mesh.lod1_index_count > 0
|
&& mesh.lod1_index_count > 0
|
||||||
&& projected_px < lod1_threshold_px;
|
&& projected_px < lod1_threshold_px;
|
||||||
|
|
||||||
if (use_lod1) per_mesh_lod1[inst.mesh_id].push_back(i);
|
// Emit one VisibleDraw entry per visible instance. Cross-mesh vertex
|
||||||
else per_mesh_lod0[inst.mesh_id].push_back(i);
|
// pulling means the shader binary-searches prefix_sums to locate
|
||||||
|
// this entry from a global vertex_index.
|
||||||
|
WgpuModelGpuData::VisibleDrawGpu d;
|
||||||
|
d.mesh_id = inst.mesh_id;
|
||||||
|
d.instance_idx = i;
|
||||||
|
d.ebo_first_u32 = (use_lod1 ? mesh.lod1_ebo_byte_offset : mesh.ebo_byte_offset)
|
||||||
|
/ uint32_t(sizeof(uint32_t));
|
||||||
|
d.base_vertex = mesh.vbo_byte_offset / INSTANCED_VERTEX_STRIDE_BYTES;
|
||||||
|
m.visible_draws_scratch.push_back(d);
|
||||||
|
|
||||||
|
const uint32_t entry_vert_count = use_lod1 ? mesh.lod1_index_count
|
||||||
|
: mesh.index_count;
|
||||||
|
running_vertex_count += entry_vert_count;
|
||||||
|
m.prefix_sums_scratch.push_back(running_vertex_count);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Flatten into m.visible_flat_scratch as [mesh0 lod0 | mesh0 lod1 | mesh1
|
m.total_visible_draws = uint32_t(m.visible_draws_scratch.size());
|
||||||
// lod0 | mesh1 lod1 | …] and emit one MeshDraw per non-empty bucket.
|
m.total_visible_vertices = running_vertex_count;
|
||||||
m.visible_flat_scratch.clear();
|
return hiz_rejects;
|
||||||
m.mesh_draws.clear();
|
}
|
||||||
m.mesh_draws.reserve(m.meshes.size() * 2);
|
|
||||||
for (uint32_t mi = 0; mi < m.meshes.size(); ++mi) {
|
|
||||||
const MeshInfo& mesh = m.meshes[mi];
|
|
||||||
|
|
||||||
if (!per_mesh_lod0[mi].empty()) {
|
void WgpuViewportWindow::cullModelCpuUpload(WgpuModelGpuData& m) {
|
||||||
WgpuModelGpuData::MeshDraw d;
|
if (!m.visible_draws_buffer || !m.prefix_sums_buffer || !m.per_model_uniform) return;
|
||||||
d.first_instance = uint32_t(m.visible_flat_scratch.size());
|
|
||||||
d.instance_count = uint32_t(per_mesh_lod0[mi].size());
|
if (m.total_visible_draws == 0) {
|
||||||
d.first_index = mesh.ebo_byte_offset / uint32_t(sizeof(uint32_t));
|
// No visible geometry this frame; render() will skip the draw.
|
||||||
d.base_vertex = int32_t(mesh.vbo_byte_offset / INSTANCED_VERTEX_STRIDE_BYTES);
|
// Still update the uniform so the shader sees zero (defensive).
|
||||||
d.index_count = mesh.index_count;
|
const uint32_t um[4] = { 0, 0, 0, 0 };
|
||||||
for (uint32_t inst_idx : per_mesh_lod0[mi]) {
|
wgpuQueueWriteBuffer(queue_, m.per_model_uniform, 0, um, sizeof(um));
|
||||||
m.visible_flat_scratch.push_back(inst_idx);
|
return;
|
||||||
}
|
|
||||||
m.mesh_draws.push_back(d);
|
|
||||||
}
|
|
||||||
if (!per_mesh_lod1[mi].empty()) {
|
|
||||||
WgpuModelGpuData::MeshDraw d;
|
|
||||||
d.first_instance = uint32_t(m.visible_flat_scratch.size());
|
|
||||||
d.instance_count = uint32_t(per_mesh_lod1[mi].size());
|
|
||||||
d.first_index = mesh.lod1_ebo_byte_offset / uint32_t(sizeof(uint32_t));
|
|
||||||
d.base_vertex = int32_t(mesh.vbo_byte_offset / INSTANCED_VERTEX_STRIDE_BYTES);
|
|
||||||
d.index_count = mesh.lod1_index_count;
|
|
||||||
for (uint32_t inst_idx : per_mesh_lod1[mi]) {
|
|
||||||
m.visible_flat_scratch.push_back(inst_idx);
|
|
||||||
}
|
|
||||||
m.mesh_draws.push_back(d);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Upload the visible list. Round up to 4-byte multiple (always true for
|
wgpuQueueWriteBuffer(queue_, m.visible_draws_buffer, 0,
|
||||||
// u32 arrays). Empty visible list still uploads one zero so the bind
|
m.visible_draws_scratch.data(),
|
||||||
// group has something well-defined; the per-mesh loop will skip the draw.
|
m.visible_draws_scratch.size()
|
||||||
const size_t bytes_to_upload = std::max<size_t>(
|
* sizeof(WgpuModelGpuData::VisibleDrawGpu));
|
||||||
m.visible_flat_scratch.size() * sizeof(uint32_t), sizeof(uint32_t));
|
wgpuQueueWriteBuffer(queue_, m.prefix_sums_buffer, 0,
|
||||||
static const uint32_t zero = 0;
|
m.prefix_sums_scratch.data(),
|
||||||
const void* src = m.visible_flat_scratch.empty()
|
m.prefix_sums_scratch.size() * sizeof(uint32_t));
|
||||||
? static_cast<const void*>(&zero)
|
|
||||||
: static_cast<const void*>(m.visible_flat_scratch.data());
|
const uint32_t um[4] = {
|
||||||
wgpuQueueWriteBuffer(queue_, m.visible_buffer, 0, src, bytes_to_upload);
|
m.total_visible_draws,
|
||||||
|
m.total_visible_vertices,
|
||||||
|
0, 0,
|
||||||
|
};
|
||||||
|
wgpuQueueWriteBuffer(queue_, m.per_model_uniform, 0, um, sizeof(um));
|
||||||
}
|
}
|
||||||
|
|
||||||
void WgpuViewportWindow::render() {
|
void WgpuViewportWindow::render() {
|
||||||
@@ -1427,6 +1562,10 @@ void WgpuViewportWindow::render() {
|
|||||||
QElapsedTimer frame_timer;
|
QElapsedTimer frame_timer;
|
||||||
if (bench_total_ > 0) frame_timer.start();
|
if (bench_total_ > 0) frame_timer.start();
|
||||||
|
|
||||||
|
// Drain any HiZ async readbacks that completed since last frame so the
|
||||||
|
// pyramid is as fresh as it can be before cull runs.
|
||||||
|
if (hiz_enabled_) drainHizReadbacks();
|
||||||
|
|
||||||
WGPUSurfaceTexture surf_tex = {};
|
WGPUSurfaceTexture surf_tex = {};
|
||||||
wgpuSurfaceGetCurrentTexture(surface_, &surf_tex);
|
wgpuSurfaceGetCurrentTexture(surface_, &surf_tex);
|
||||||
|
|
||||||
@@ -1462,6 +1601,8 @@ void WgpuViewportWindow::render() {
|
|||||||
last_visible_triangles_ = 0;
|
last_visible_triangles_ = 0;
|
||||||
last_sub_draws_ = 0;
|
last_sub_draws_ = 0;
|
||||||
hiz_reject_count_ = 0;
|
hiz_reject_count_ = 0;
|
||||||
|
QElapsedTimer cull_timer;
|
||||||
|
if (bench_total_ > 0) cull_timer.start();
|
||||||
QMatrix4x4 vp_this_frame;
|
QMatrix4x4 vp_this_frame;
|
||||||
{
|
{
|
||||||
const QVector3D target(camera_target_[0], camera_target_[1], camera_target_[2]);
|
const QVector3D target(camera_target_[0], camera_target_[1], camera_target_[2]);
|
||||||
@@ -1488,18 +1629,62 @@ void WgpuViewportWindow::render() {
|
|||||||
/ std::tan(qDegreesToRadians(camera_fov_y_deg_) * 0.5f))
|
/ std::tan(qDegreesToRadians(camera_fov_y_deg_) * 0.5f))
|
||||||
: 0.0f;
|
: 0.0f;
|
||||||
|
|
||||||
|
// Motion detection: any change in camera state since last frame
|
||||||
|
// bumps the contribution threshold to motion_min_pixel_radius_
|
||||||
|
// (mirrors GL's NavPreset behaviour, drops more sub-pixel work
|
||||||
|
// during orbit/pan/zoom).
|
||||||
|
const bool camera_moved = has_prev_camera_
|
||||||
|
&& (camera_target_[0] != prev_camera_target_[0]
|
||||||
|
|| camera_target_[1] != prev_camera_target_[1]
|
||||||
|
|| camera_target_[2] != prev_camera_target_[2]
|
||||||
|
|| camera_distance_ != prev_camera_distance_
|
||||||
|
|| camera_yaw_deg_ != prev_camera_yaw_deg_
|
||||||
|
|| camera_pitch_deg_ != prev_camera_pitch_deg_);
|
||||||
|
const float effective_min_px =
|
||||||
|
(camera_moved && motion_min_pixel_radius_ > min_pixel_radius_)
|
||||||
|
? motion_min_pixel_radius_
|
||||||
|
: min_pixel_radius_;
|
||||||
|
|
||||||
|
// Cull each model on its own worker thread. wgpu queue writes are
|
||||||
|
// serialised on the main thread after the parallel compute joins —
|
||||||
|
// wgpu-native doesn't guarantee thread-safety on queue ops.
|
||||||
|
std::vector<std::pair<uint32_t, std::future<uint32_t>>> futures;
|
||||||
|
futures.reserve(models_gpu_.size());
|
||||||
for (auto& [mid, m] : models_gpu_) {
|
for (auto& [mid, m] : models_gpu_) {
|
||||||
if (m.hidden) continue;
|
if (m.hidden) continue;
|
||||||
cullModelCpu(m, planes, eye_a, fwd_a, focal_px,
|
auto& m_ref = m;
|
||||||
min_pixel_radius_, lod1_pixel_threshold_,
|
futures.emplace_back(mid, std::async(std::launch::async,
|
||||||
hiz_enabled_);
|
[this, &m_ref, &planes, &eye_a, &fwd_a,
|
||||||
for (const auto& d : m.mesh_draws) {
|
focal_px, effective_min_px]() {
|
||||||
if (d.instance_count == 0 || d.index_count == 0) continue;
|
return cullModelCpuCompute(
|
||||||
last_visible_objects_ += d.instance_count;
|
m_ref, planes, eye_a, fwd_a, focal_px,
|
||||||
last_visible_triangles_ += (d.index_count / 3u) * d.instance_count;
|
effective_min_px, lod1_pixel_threshold_,
|
||||||
last_sub_draws_ += 1;
|
hiz_enabled_);
|
||||||
}
|
}));
|
||||||
}
|
}
|
||||||
|
for (auto& [mid, fut] : futures) {
|
||||||
|
hiz_reject_count_ += fut.get();
|
||||||
|
}
|
||||||
|
for (auto& [mid, m] : models_gpu_) {
|
||||||
|
if (m.hidden) continue;
|
||||||
|
cullModelCpuUpload(m);
|
||||||
|
last_visible_objects_ += m.total_visible_draws;
|
||||||
|
last_visible_triangles_ += m.total_visible_vertices / 3u;
|
||||||
|
// One CPU drawcall per model now (cross-mesh vertex pulling).
|
||||||
|
if (m.total_visible_draws > 0) last_sub_draws_ += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Snapshot camera state for next frame's motion detection.
|
||||||
|
prev_camera_target_[0] = camera_target_[0];
|
||||||
|
prev_camera_target_[1] = camera_target_[1];
|
||||||
|
prev_camera_target_[2] = camera_target_[2];
|
||||||
|
prev_camera_distance_ = camera_distance_;
|
||||||
|
prev_camera_yaw_deg_ = camera_yaw_deg_;
|
||||||
|
prev_camera_pitch_deg_ = camera_pitch_deg_;
|
||||||
|
has_prev_camera_ = true;
|
||||||
|
if (bench_total_ > 0 && bench_count_ >= bench_warmup_) {
|
||||||
|
bench_cull_ms_total_ += double(cull_timer.nsecsElapsed()) / 1e6;
|
||||||
}
|
}
|
||||||
|
|
||||||
WGPUCommandEncoder enc = wgpuDeviceCreateCommandEncoder(device_, nullptr);
|
WGPUCommandEncoder enc = wgpuDeviceCreateCommandEncoder(device_, nullptr);
|
||||||
@@ -1539,37 +1724,22 @@ void WgpuViewportWindow::render() {
|
|||||||
wgpuRenderPassEncoderSetBindGroup(pass, 0, frame_bind_group_, 0, nullptr);
|
wgpuRenderPassEncoderSetBindGroup(pass, 0, frame_bind_group_, 0, nullptr);
|
||||||
|
|
||||||
for (const auto& [mid, m] : models_gpu_) {
|
for (const auto& [mid, m] : models_gpu_) {
|
||||||
if (m.hidden || !m.bind_group || !m.index_buffer || m.index_count == 0) continue;
|
if (m.hidden || !m.bind_group || m.total_visible_vertices == 0) continue;
|
||||||
|
|
||||||
wgpuRenderPassEncoderSetBindGroup(pass, 1, m.bind_group, 0, nullptr);
|
wgpuRenderPassEncoderSetBindGroup(pass, 1, m.bind_group, 0, nullptr);
|
||||||
wgpuRenderPassEncoderSetIndexBuffer(pass, m.index_buffer,
|
// Single CPU drawcall per model. The vertex shader binary-
|
||||||
WGPUIndexFormat_Uint32,
|
// searches prefix_sums by vertex_index to find (mesh, instance)
|
||||||
0, WGPU_WHOLE_SIZE);
|
// and pulls index + vertex data manually from storage buffers.
|
||||||
|
wgpuRenderPassEncoderDraw(pass, m.total_visible_vertices, 1, 0, 0);
|
||||||
// One drawIndexed per non-empty mesh. instanceCount is the number
|
|
||||||
// of frustum-surviving instances of this mesh; firstInstance is
|
|
||||||
// their offset into m.visible_buffer (consumed by the WGSL
|
|
||||||
// `visible[iid]` indirection). All-instances-culled meshes are
|
|
||||||
// skipped — no draw call issued at all.
|
|
||||||
for (const auto& d : m.mesh_draws) {
|
|
||||||
if (d.instance_count == 0 || d.index_count == 0) continue;
|
|
||||||
wgpuRenderPassEncoderDrawIndexed(
|
|
||||||
pass,
|
|
||||||
d.index_count,
|
|
||||||
d.instance_count,
|
|
||||||
d.first_index,
|
|
||||||
d.base_vertex,
|
|
||||||
d.first_instance);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
wgpuRenderPassEncoderEnd(pass);
|
wgpuRenderPassEncoderEnd(pass);
|
||||||
wgpuRenderPassEncoderRelease(pass);
|
wgpuRenderPassEncoderRelease(pass);
|
||||||
|
|
||||||
// ---- HiZ: resolve MSAA depth → small single-sample → staging buffer
|
// ---- HiZ: resolve MSAA depth → small single-sample → ping-pong slot
|
||||||
|
int hiz_submitted_slot = -1;
|
||||||
if (hiz_enabled_) {
|
if (hiz_enabled_) {
|
||||||
encodeHizResolve(enc);
|
hiz_submitted_slot = encodeHizResolve(enc);
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- Optional capture: encode copy on the same command buffer -------
|
// ---- Optional capture: encode copy on the same command buffer -------
|
||||||
@@ -1683,11 +1853,17 @@ void WgpuViewportWindow::render() {
|
|||||||
wgpuSurfacePresent(surface_);
|
wgpuSurfacePresent(surface_);
|
||||||
wgpuTextureRelease(surf_tex.texture);
|
wgpuTextureRelease(surf_tex.texture);
|
||||||
|
|
||||||
// ---- HiZ readback + mip pyramid for the *next* frame's cull ---------
|
// ---- HiZ async readback handoff -------------------------------------
|
||||||
// Sync wait via processEvents — the staging buffer is small (≈ 256×160×4)
|
// Don't block — just kick off the mapAsync for the slot we filled this
|
||||||
// so the stall is well under a millisecond on every backend.
|
// frame. Drainage happens at the top of the *next* frame via
|
||||||
if (hiz_enabled_) {
|
// drainHizReadbacks(), giving the GPU at least one frame of headroom.
|
||||||
readbackAndBuildHizPyramid(vp_this_frame);
|
if (hiz_enabled_ && hiz_submitted_slot >= 0) {
|
||||||
|
QElapsedTimer hiz_timer;
|
||||||
|
if (bench_total_ > 0) hiz_timer.start();
|
||||||
|
startHizMap(hiz_submitted_slot, vp_this_frame);
|
||||||
|
if (bench_total_ > 0 && bench_count_ >= bench_warmup_) {
|
||||||
|
bench_hiz_readback_ms_total_ += double(hiz_timer.nsecsElapsed()) / 1e6;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- Benchmark integration + auto-quit -------------------------------
|
// ---- Benchmark integration + auto-quit -------------------------------
|
||||||
@@ -1736,6 +1912,11 @@ void WgpuViewportWindow::render() {
|
|||||||
<< " tri " << last_visible_triangles_
|
<< " tri " << last_visible_triangles_
|
||||||
<< " sub_draws " << last_sub_draws_
|
<< " sub_draws " << last_sub_draws_
|
||||||
<< " hiz_rej " << hiz_reject_count_;
|
<< " hiz_rej " << hiz_reject_count_;
|
||||||
|
const double n = double(std::max(1, bench_total_));
|
||||||
|
qInfo().noquote().nospace()
|
||||||
|
<< " per-frame avg ms: cull=" << bench_cull_ms_total_ / n
|
||||||
|
<< " hiz_readback=" << bench_hiz_readback_ms_total_ / n
|
||||||
|
<< " hiz=" << (hiz_enabled_ ? "on" : "off");
|
||||||
qInfo().noquote() << "=== END BENCHMARK ===\n";
|
qInfo().noquote() << "=== END BENCHMARK ===\n";
|
||||||
|
|
||||||
bench_total_ = 0;
|
bench_total_ = 0;
|
||||||
@@ -1764,14 +1945,22 @@ bool WgpuViewportWindow::buildPipelines() {
|
|||||||
frame_bgl_desc.label = svFromCStr("ifcviewer-wgpu.frame_bgl");
|
frame_bgl_desc.label = svFromCStr("ifcviewer-wgpu.frame_bgl");
|
||||||
frame_bgl_ = wgpuDeviceCreateBindGroupLayout(device_, &frame_bgl_desc);
|
frame_bgl_ = wgpuDeviceCreateBindGroupLayout(device_, &frame_bgl_desc);
|
||||||
|
|
||||||
WGPUBindGroupLayoutEntry model_entries[4] = {};
|
// 6 read-only storage buffers (vertices, meshes, instances, indices,
|
||||||
for (int i = 0; i < 4; ++i) {
|
// visible_draws, prefix_sums) + 1 uniform (per-model count). All read
|
||||||
|
// in the vertex shader. WebGPU's mandatory min is 8 storage / 12 uniform
|
||||||
|
// per stage, so we're comfortably under the cap.
|
||||||
|
WGPUBindGroupLayoutEntry model_entries[7] = {};
|
||||||
|
for (int i = 0; i < 6; ++i) {
|
||||||
model_entries[i].binding = uint32_t(i);
|
model_entries[i].binding = uint32_t(i);
|
||||||
model_entries[i].visibility = WGPUShaderStage_Vertex;
|
model_entries[i].visibility = WGPUShaderStage_Vertex;
|
||||||
model_entries[i].buffer.type = WGPUBufferBindingType_ReadOnlyStorage;
|
model_entries[i].buffer.type = WGPUBufferBindingType_ReadOnlyStorage;
|
||||||
}
|
}
|
||||||
|
model_entries[6].binding = 6;
|
||||||
|
model_entries[6].visibility = WGPUShaderStage_Vertex;
|
||||||
|
model_entries[6].buffer.type = WGPUBufferBindingType_Uniform;
|
||||||
|
model_entries[6].buffer.minBindingSize = 16;
|
||||||
WGPUBindGroupLayoutDescriptor model_bgl_desc = {};
|
WGPUBindGroupLayoutDescriptor model_bgl_desc = {};
|
||||||
model_bgl_desc.entryCount = 4;
|
model_bgl_desc.entryCount = 7;
|
||||||
model_bgl_desc.entries = model_entries;
|
model_bgl_desc.entries = model_entries;
|
||||||
model_bgl_desc.label = svFromCStr("ifcviewer-wgpu.model_bgl");
|
model_bgl_desc.label = svFromCStr("ifcviewer-wgpu.model_bgl");
|
||||||
model_bgl_ = wgpuDeviceCreateBindGroupLayout(device_, &model_bgl_desc);
|
model_bgl_ = wgpuDeviceCreateBindGroupLayout(device_, &model_bgl_desc);
|
||||||
@@ -1859,12 +2048,14 @@ void WgpuViewportWindow::buildModelBindGroup(WgpuModelGpuData& m) {
|
|||||||
wgpuBindGroupRelease(m.bind_group);
|
wgpuBindGroupRelease(m.bind_group);
|
||||||
m.bind_group = nullptr;
|
m.bind_group = nullptr;
|
||||||
}
|
}
|
||||||
if (!m.vertex_storage || !m.mesh_storage || !m.instance_storage || !m.visible_buffer) {
|
if (!m.vertex_storage || !m.mesh_storage || !m.instance_storage
|
||||||
|
|| !m.index_buffer || !m.visible_draws_buffer || !m.prefix_sums_buffer
|
||||||
|
|| !m.per_model_uniform) {
|
||||||
// Empty model — no bind group needed; the draw loop will skip it.
|
// Empty model — no bind group needed; the draw loop will skip it.
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|
||||||
WGPUBindGroupEntry entries[4] = {};
|
WGPUBindGroupEntry entries[7] = {};
|
||||||
entries[0].binding = 0;
|
entries[0].binding = 0;
|
||||||
entries[0].buffer = m.vertex_storage;
|
entries[0].buffer = m.vertex_storage;
|
||||||
entries[0].size = WGPU_WHOLE_SIZE;
|
entries[0].size = WGPU_WHOLE_SIZE;
|
||||||
@@ -1875,12 +2066,21 @@ void WgpuViewportWindow::buildModelBindGroup(WgpuModelGpuData& m) {
|
|||||||
entries[2].buffer = m.instance_storage;
|
entries[2].buffer = m.instance_storage;
|
||||||
entries[2].size = WGPU_WHOLE_SIZE;
|
entries[2].size = WGPU_WHOLE_SIZE;
|
||||||
entries[3].binding = 3;
|
entries[3].binding = 3;
|
||||||
entries[3].buffer = m.visible_buffer;
|
entries[3].buffer = m.index_buffer;
|
||||||
entries[3].size = WGPU_WHOLE_SIZE;
|
entries[3].size = WGPU_WHOLE_SIZE;
|
||||||
|
entries[4].binding = 4;
|
||||||
|
entries[4].buffer = m.visible_draws_buffer;
|
||||||
|
entries[4].size = WGPU_WHOLE_SIZE;
|
||||||
|
entries[5].binding = 5;
|
||||||
|
entries[5].buffer = m.prefix_sums_buffer;
|
||||||
|
entries[5].size = WGPU_WHOLE_SIZE;
|
||||||
|
entries[6].binding = 6;
|
||||||
|
entries[6].buffer = m.per_model_uniform;
|
||||||
|
entries[6].size = 16;
|
||||||
|
|
||||||
WGPUBindGroupDescriptor desc = {};
|
WGPUBindGroupDescriptor desc = {};
|
||||||
desc.layout = model_bgl_;
|
desc.layout = model_bgl_;
|
||||||
desc.entryCount = 4;
|
desc.entryCount = 7;
|
||||||
desc.entries = entries;
|
desc.entries = entries;
|
||||||
desc.label = svFromCStr("ifcviewer-wgpu.model_bind_group");
|
desc.label = svFromCStr("ifcviewer-wgpu.model_bind_group");
|
||||||
m.bind_group = wgpuDeviceCreateBindGroup(device_, &desc);
|
m.bind_group = wgpuDeviceCreateBindGroup(device_, &desc);
|
||||||
|
|||||||
@@ -124,13 +124,19 @@ private:
|
|||||||
void ensureHizTextures(int viewport_w, int viewport_h);
|
void ensureHizTextures(int viewport_w, int viewport_h);
|
||||||
void releaseHizResources();
|
void releaseHizResources();
|
||||||
// Resolves the just-rendered MSAA depth into the small single-sample
|
// Resolves the just-rendered MSAA depth into the small single-sample
|
||||||
// HiZ texture and copies it to the staging buffer. Encoded onto `enc`
|
// HiZ texture and copies it to whichever staging slot is currently
|
||||||
// so it ships in the same command buffer as the main draw.
|
// idle. Returns the slot index used, or -1 if both slots are still
|
||||||
void encodeHizResolve(WGPUCommandEncoder enc);
|
// in flight (resolve is skipped this frame — fine, we already have
|
||||||
// Maps the staging buffer (waits via processEvents), max-reduces a CPU
|
// a recent pyramid). Encoded onto `enc` so it ships in the same
|
||||||
// mip pyramid, stores the VP used. Run after submitting the encoder so
|
// command buffer as the main draw.
|
||||||
// the GPU has begun the copy. Updates hiz_valid_ to true on success.
|
int encodeHizResolve(WGPUCommandEncoder enc);
|
||||||
void readbackAndBuildHizPyramid(const QMatrix4x4& vp_used);
|
// Issues a non-blocking mapAsync on `slot` after submit, so the
|
||||||
|
// callback can fire whenever the GPU has actually finished writing.
|
||||||
|
void startHizMap(int slot, const QMatrix4x4& vp_used);
|
||||||
|
// Drains pending mapAsync callbacks (via processEvents — does NOT
|
||||||
|
// block on GPU work). For any slot that just signalled Mapped, reads
|
||||||
|
// it, unmaps it, max-reduces the mip pyramid, and updates hiz_vp_.
|
||||||
|
void drainHizReadbacks();
|
||||||
// Project AABB through hiz_vp_ and test against the pyramid. False
|
// Project AABB through hiz_vp_ and test against the pyramid. False
|
||||||
// (keep) if HiZ isn't valid yet, AABB straddles the near plane, or
|
// (keep) if HiZ isn't valid yet, AABB straddles the near plane, or
|
||||||
// any projection is unreliable. True (cull) when AABB is provably
|
// any projection is unreliable. True (cull) when AABB is provably
|
||||||
@@ -156,13 +162,22 @@ private:
|
|||||||
// `lod1_threshold_px` — survivors projected below this get the mesh's
|
// `lod1_threshold_px` — survivors projected below this get the mesh's
|
||||||
// LOD1 index slice when one was baked.
|
// LOD1 index slice when one was baked.
|
||||||
// min_radius_px == 0 disables contribution culling.
|
// min_radius_px == 0 disables contribution culling.
|
||||||
void cullModelCpu(WgpuModelGpuData& m,
|
// CPU-only phase of cull: produces m.visible_draws_scratch /
|
||||||
const float planes[6][4],
|
// prefix_sums_scratch and sets total_visible_draws / total_visible_
|
||||||
const float eye[3], const float forward[3],
|
// vertices. Touches no wgpu state, so this can run on a worker thread
|
||||||
float focal_px,
|
// (multiple models culled in parallel). Returns the number of HiZ
|
||||||
float min_radius_px,
|
// rejections accumulated (caller adds to the per-frame stat).
|
||||||
float lod1_threshold_px,
|
uint32_t cullModelCpuCompute(WgpuModelGpuData& m,
|
||||||
bool hiz_enabled);
|
const float planes[6][4],
|
||||||
|
const float eye[3], const float forward[3],
|
||||||
|
float focal_px,
|
||||||
|
float min_radius_px,
|
||||||
|
float lod1_threshold_px,
|
||||||
|
bool hiz_enabled) const;
|
||||||
|
// Upload phase: wgpuQueueWriteBuffer for visible_draws / prefix_sums /
|
||||||
|
// per-model uniform. Main-thread only (wgpu queue ops are not all
|
||||||
|
// thread-safe).
|
||||||
|
void cullModelCpuUpload(WgpuModelGpuData& m);
|
||||||
|
|
||||||
bool wgpu_initialized_ = false;
|
bool wgpu_initialized_ = false;
|
||||||
bool surface_configured_ = false;
|
bool surface_configured_ = false;
|
||||||
@@ -221,11 +236,25 @@ private:
|
|||||||
|
|
||||||
WGPUTexture hiz_resolve_texture_ = nullptr;
|
WGPUTexture hiz_resolve_texture_ = nullptr;
|
||||||
WGPUTextureView hiz_resolve_view_ = nullptr;
|
WGPUTextureView hiz_resolve_view_ = nullptr;
|
||||||
WGPUBuffer hiz_staging_buffer_ = nullptr;
|
|
||||||
uint32_t hiz_resolve_w_ = 0;
|
uint32_t hiz_resolve_w_ = 0;
|
||||||
uint32_t hiz_resolve_h_ = 0;
|
uint32_t hiz_resolve_h_ = 0;
|
||||||
uint32_t hiz_padded_bpr_ = 0; // bytes per row in the staging buffer
|
uint32_t hiz_padded_bpr_ = 0; // bytes per row in the staging buffer
|
||||||
|
|
||||||
|
// Ping-pong async readback. Frame N submits a copy into slot
|
||||||
|
// hiz_write_idx_ and calls mapAsync (non-blocking) on that slot. Frame
|
||||||
|
// N+K (K ≥ 1) calls processEvents to drain callbacks; whichever slot
|
||||||
|
// signalled completion is mapped, read into hiz_pyramid_, and unmapped
|
||||||
|
// — making the pyramid 1+ frames stale, which is fine ("slightly-stale
|
||||||
|
// depth" pattern the GL backend already documents). Two slots overlap
|
||||||
|
// GPU write with CPU read; we never block on the readback.
|
||||||
|
enum class HizSlotState : uint8_t { Idle, Mapping, Mapped };
|
||||||
|
static constexpr int HIZ_SLOTS = 2;
|
||||||
|
WGPUBuffer hiz_staging_buffers_[HIZ_SLOTS] = { nullptr, nullptr };
|
||||||
|
QMatrix4x4 hiz_slot_vp_ [HIZ_SLOTS];
|
||||||
|
HizSlotState hiz_slot_state_ [HIZ_SLOTS] = { HizSlotState::Idle,
|
||||||
|
HizSlotState::Idle };
|
||||||
|
int hiz_write_idx_ = 0;
|
||||||
|
|
||||||
// CPU mip pyramid (max-reduce). hiz_pyramid_[hiz_mip_offset_[L] + y*W + x].
|
// CPU mip pyramid (max-reduce). hiz_pyramid_[hiz_mip_offset_[L] + y*W + x].
|
||||||
std::vector<float> hiz_pyramid_;
|
std::vector<float> hiz_pyramid_;
|
||||||
std::vector<uint32_t> hiz_mip_offset_;
|
std::vector<uint32_t> hiz_mip_offset_;
|
||||||
@@ -247,16 +276,20 @@ private:
|
|||||||
float camera_near_ = 0.1f;
|
float camera_near_ = 0.1f;
|
||||||
float camera_far_ = 10000.0f;
|
float camera_far_ = 10000.0f;
|
||||||
|
|
||||||
// Drop instances whose projected bounding-sphere radius is below this
|
// Contribution-cull thresholds. Still-frame uses min_pixel_radius_;
|
||||||
// many pixels. Mirrors AppSettings::minPixelRadius() (GL default 2.0;
|
// when the camera changed since last frame, the bigger motion threshold
|
||||||
// motion mode uses 10.0 but we don't differentiate yet — that arrives
|
// kicks in to drop more sub-pixel detail (and slash per-frame cull cost).
|
||||||
// with mouse-driven motion-state tracking later).
|
// Matches AppSettings::minPixelRadius / motionMinPixelRadius in GL.
|
||||||
float min_pixel_radius_ = 2.0f;
|
float min_pixel_radius_ = 2.0f;
|
||||||
|
float motion_min_pixel_radius_ = 10.0f;
|
||||||
|
|
||||||
|
public:
|
||||||
// Master switch for HiZ occlusion. Set false to skip the depth resolve
|
// Master switch for HiZ occlusion. Set false to skip the depth resolve
|
||||||
// + readback + cull test entirely (matches IFC_NO_HIZ in the GL backend).
|
// + readback + cull test entirely (matches IFC_NO_HIZ in the GL backend).
|
||||||
bool hiz_enabled_ = true;
|
bool hiz_enabled_ = true;
|
||||||
|
|
||||||
|
private:
|
||||||
|
|
||||||
// Switch to LOD1 when an instance's projected bounding-sphere radius
|
// Switch to LOD1 when an instance's projected bounding-sphere radius
|
||||||
// drops below this many pixels. 0 disables (always LOD0). Defaults
|
// drops below this many pixels. 0 disables (always LOD0). Defaults
|
||||||
// mirror AppSettings::lod1PixelThreshold() in the GL backend.
|
// mirror AppSettings::lod1PixelThreshold() in the GL backend.
|
||||||
@@ -274,6 +307,15 @@ private:
|
|||||||
// user pointed it.
|
// user pointed it.
|
||||||
bool initial_view_applied_ = false;
|
bool initial_view_applied_ = false;
|
||||||
|
|
||||||
|
// Camera state at the previous render() for motion detection. Any
|
||||||
|
// change means we apply the motion contribution threshold this frame
|
||||||
|
// (drops more sub-pixel work mid-orbit; matches GL behaviour).
|
||||||
|
float prev_camera_target_[3] = { 0, 0, 0 };
|
||||||
|
float prev_camera_distance_ = 0.0f;
|
||||||
|
float prev_camera_yaw_deg_ = 0.0f;
|
||||||
|
float prev_camera_pitch_deg_ = 0.0f;
|
||||||
|
bool has_prev_camera_ = false;
|
||||||
|
|
||||||
// Pending one-shot screenshot, captured at the end of the next render().
|
// Pending one-shot screenshot, captured at the end of the next render().
|
||||||
QString pending_screenshot_path_;
|
QString pending_screenshot_path_;
|
||||||
bool pending_screenshot_quit_ = false;
|
bool pending_screenshot_quit_ = false;
|
||||||
@@ -301,6 +343,13 @@ private:
|
|||||||
uint32_t last_visible_objects_ = 0;
|
uint32_t last_visible_objects_ = 0;
|
||||||
uint32_t last_visible_triangles_ = 0;
|
uint32_t last_visible_triangles_ = 0;
|
||||||
uint32_t last_sub_draws_ = 0;
|
uint32_t last_sub_draws_ = 0;
|
||||||
|
|
||||||
|
// Phase-time accumulators for benchmark mode. Each window measures a
|
||||||
|
// distinct slice of render() so we can attribute frame cost. Totals
|
||||||
|
// across the timed window are divided by bench_total_ on print.
|
||||||
|
double bench_cull_ms_total_ = 0.0;
|
||||||
|
double bench_hiz_readback_ms_total_ = 0.0;
|
||||||
|
double bench_submit_ms_total_ = 0.0;
|
||||||
};
|
};
|
||||||
|
|
||||||
#endif // WGPUVIEWPORTWINDOW_H
|
#endif // WGPUVIEWPORTWINDOW_H
|
||||||
|
|||||||
Reference in New Issue
Block a user