From c78e16eafbf13d3d5f88c53005a5d92b69c32657 Mon Sep 17 00:00:00 2001 From: Dion Moult Date: Mon, 13 Apr 2026 18:31:43 +1000 Subject: [PATCH] Phase 3B: per-instance LOD via meshoptimizer simplifySloppy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Decimate each unique mesh once at sidecar-build time and swap to the reduced index slice per-instance per-frame when projected sphere radius drops below IFC_LOD1_PX (default 30). Same VBO, same SSBO, just a different firstIndex/count in the indirect command. Extends MeshInfo (48→56 B) with lod1_ebo_byte_offset + lod1_index_count and bumps the sidecar to v5. buildLods() runs inside onStreamingFinished, appends decimated indices to sd.indices, applyLodExtension pushes the EBO suffix to the live GPU state, and the sidecar is written with LOD1 baked in. simplifySloppy (voxel clustering) is used instead of the default edge-collapse meshopt_simplify because BIM brep output is per-triangle- unwelded and non-manifold after welding — simplify returned the input unchanged for every mesh tested. Sloppy ignores topology. Knobs (IFC_LOD_SLOPPY, IFC_LOD_ERROR, IFC_LOD_RATIO, IFC_LOD_MIN_SAVINGS, IFC_LOD_LOCK_BORDER, IFC_LOD_DEBUG) are available for A/B tuning. Result on the 128M-tri 10-model test scene (GTX 1650, 2px contribution cull): 20.2 → 43.2 fps, 40M → 14M visible triangles, no change in object count. LOD build adds 100–600 ms per model on first open, cached thereafter. README Phase 3B section is now a full writeup of pipeline, selection, decimator-choice rationale, env vars, and measured numbers. Co-Authored-By: Claude Opus 4.6 --- src/ifcviewer/CMakeLists.txt | 2 + src/ifcviewer/InstancedGeometry.h | 18 ++- src/ifcviewer/LodBuilder.cpp | 203 ++++++++++++++++++++++++++++++ src/ifcviewer/LodBuilder.h | 56 +++++++++ src/ifcviewer/MainWindow.cpp | 14 +++ src/ifcviewer/README.md | 135 +++++++++++++++++--- src/ifcviewer/SidecarCache.cpp | 6 +- src/ifcviewer/SidecarCache.h | 5 +- src/ifcviewer/ViewportWindow.cpp | 117 ++++++++++++++--- src/ifcviewer/ViewportWindow.h | 16 ++- 10 files changed, 532 insertions(+), 40 deletions(-) create mode 100644 src/ifcviewer/LodBuilder.cpp create mode 100644 src/ifcviewer/LodBuilder.h diff --git a/src/ifcviewer/CMakeLists.txt b/src/ifcviewer/CMakeLists.txt index 9f1c4dac50..70642acabf 100644 --- a/src/ifcviewer/CMakeLists.txt +++ b/src/ifcviewer/CMakeLists.txt @@ -26,6 +26,7 @@ set(QT_VERSION 6 CACHE STRING "Qt version") find_package(Qt${QT_VERSION} COMPONENTS Core Gui Widgets OpenGL REQUIRED PATHS ${QT_DIR}) find_package(OpenGL REQUIRED) +find_package(meshoptimizer REQUIRED) file(GLOB IFCVIEWER_CPP_FILES ${CMAKE_CURRENT_SOURCE_DIR}/*.cpp) file(GLOB IFCVIEWER_H_FILES ${CMAKE_CURRENT_SOURCE_DIR}/*.h) @@ -51,6 +52,7 @@ target_link_libraries(IfcViewer PRIVATE Qt${QT_VERSION}::Widgets Qt${QT_VERSION}::OpenGL OpenGL::GL + meshoptimizer::meshoptimizer ) if(UNIX AND NOT APPLE) diff --git a/src/ifcviewer/InstancedGeometry.h b/src/ifcviewer/InstancedGeometry.h index 1c027976ef..ef79751806 100644 --- a/src/ifcviewer/InstancedGeometry.h +++ b/src/ifcviewer/InstancedGeometry.h @@ -33,18 +33,28 @@ static constexpr int INSTANCED_VERTEX_STRIDE_BYTES = 28; static constexpr int INSTANCED_VERTEX_STRIDE_FLOATS = 7; // Per-mesh metadata on the CPU side. Meshes own a slice of the model's -// VBO and EBO (both local-coords/mesh-local indices). +// VBO (shared across LODs) and one or more slices of the EBO, one per LOD. +// +// LOD0 is the original, full-resolution tessellation — the fields +// `ebo_byte_offset` / `index_count` describe it. +// +// LOD1 is an optional decimated copy of the same triangles referencing the +// same vertex buffer. Built at sidecar time via meshoptimizer for meshes +// whose triangle count crosses a threshold. `lod1_index_count == 0` +// means no LOD1 was built; the renderer must use LOD0 at every distance. struct MeshInfo { uint32_t vbo_byte_offset = 0; // where this mesh's vertices start uint32_t vertex_count = 0; - uint32_t ebo_byte_offset = 0; // where this mesh's indices start - uint32_t index_count = 0; + uint32_t ebo_byte_offset = 0; // LOD0 indices + uint32_t index_count = 0; // LOD0 index count float local_aabb_min[3]{}; float local_aabb_max[3]{}; uint32_t first_instance = 0; // index into per-model instances array uint32_t instance_count = 0; + uint32_t lod1_ebo_byte_offset = 0; + uint32_t lod1_index_count = 0; // 0 = no LOD1 available }; -static_assert(sizeof(MeshInfo) == 48, "MeshInfo must be 48 bytes"); +static_assert(sizeof(MeshInfo) == 56, "MeshInfo must be 56 bytes"); // Per-instance record uploaded to an SSBO and read by the vertex shader. // Layout deliberately matches std430 expectations: diff --git a/src/ifcviewer/LodBuilder.cpp b/src/ifcviewer/LodBuilder.cpp new file mode 100644 index 0000000000..88b8c9f046 --- /dev/null +++ b/src/ifcviewer/LodBuilder.cpp @@ -0,0 +1,203 @@ +/******************************************************************************** + * * + * This file is part of IfcOpenShell. * + * * + * IfcOpenShell is free software: you can redistribute it and/or modify * + * it under the terms of the Lesser GNU General Public License as published by * + * the Free Software Foundation, either version 3.0 of the License, or * + * (at your option) any later version. * + * * + * IfcOpenShell is distributed in the hope that it will be useful, * + * but WITHOUT ANY WARRANTY; without even the implied warranty of * + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * + * Lesser GNU General Public License for more details. * + * * + * You should have received a copy of the Lesser GNU General Public License * + * along with this program. If not, see . * + * * + ********************************************************************************/ + +#include "LodBuilder.h" + +#include + +#include +#include +#include +#include +#include + +void buildLods(SidecarData& sd, + int min_triangles, + float target_ratio, + float target_error) { + if (sd.meshes.empty() || sd.vertices.empty() || sd.indices.empty()) return; + + const size_t vtx_stride_bytes = INSTANCED_VERTEX_STRIDE_BYTES; + const size_t vtx_stride_floats = INSTANCED_VERTEX_STRIDE_FLOATS; + const size_t total_vertex_count = sd.vertices.size() / vtx_stride_floats; + + // Env var knobs so we can tune without rebuilding. + // IFC_LOD_LOCK_BORDER=1 re-enable LockBorder (off by default: BIM + // geometry is often non-manifold so locking + // borders prevents any collapse). + // IFC_LOD_ERROR= override target_error (default 0.05 → 0.2). + // IFC_LOD_RATIO= override target_ratio. + // IFC_LOD_MIN_SAVINGS=<0..1> minimum fraction of tris saved to accept + // (default 0.25). + // IFC_LOD_DEBUG=1 print per-mesh diagnostics for the first + // few meshes of each call. + // IFC_LOD_SLOPPY=0 disable sloppy (clustering) decimator. + // Default ON: BIM brep output is usually + // non-manifold, so edge-collapse simplify + // returns the input unchanged. + const char* env_lock = std::getenv("IFC_LOD_LOCK_BORDER"); + const char* env_err = std::getenv("IFC_LOD_ERROR"); + const char* env_ratio = std::getenv("IFC_LOD_RATIO"); + const char* env_savings = std::getenv("IFC_LOD_MIN_SAVINGS"); + const char* env_debug = std::getenv("IFC_LOD_DEBUG"); + const char* env_sloppy = std::getenv("IFC_LOD_SLOPPY"); + + const bool lock_border = env_lock && env_lock[0] == '1'; + const bool use_sloppy = !(env_sloppy && env_sloppy[0] == '0'); + if (env_err) target_error = static_cast(std::atof(env_err)); + if (env_ratio) target_ratio = static_cast(std::atof(env_ratio)); + float min_savings = 0.25f; + if (env_savings) min_savings = static_cast(std::atof(env_savings)); + const bool debug = env_debug && env_debug[0] == '1'; + + // Loosened defaults: BIM meshes are non-manifold; LockBorder ≈ zero + // collapses. A 0.2 error budget still looks fine at sub-4px. + if (target_error < 0.2f) target_error = 0.2f; + + // Scratch buffers reused across meshes so we only allocate once. + std::vector simplified; + std::vector shadow; + simplified.reserve(1024); + shadow.reserve(1024); + + int dbg_printed = 0; + int dbg_rejected_savings = 0; + int dbg_rejected_noreduce = 0; + int dbg_accepted = 0; + + for (auto& mesh : sd.meshes) { + mesh.lod1_ebo_byte_offset = 0; + mesh.lod1_index_count = 0; + + const uint32_t tri_count = mesh.index_count / 3; + if (static_cast(tri_count) < min_triangles) continue; + if (mesh.vertex_count == 0) continue; + + // meshopt wants a pointer to the *first position* and a vertex_count + // equal to the number of referenced vertices (i.e. the absolute upper + // bound on indices we might see). Indices in `sd.indices` for this + // mesh are mesh-local (0..mesh.vertex_count). Pass the base-vertex + // as an offset into sd.vertices so meshopt reads positions at the + // right place. + const uint32_t base_vertex = mesh.vbo_byte_offset / vtx_stride_bytes; + if (base_vertex + mesh.vertex_count > total_vertex_count) continue; + + const uint32_t first_index = mesh.ebo_byte_offset / sizeof(uint32_t); + if (first_index + mesh.index_count > sd.indices.size()) continue; + + const float* positions = + sd.vertices.data() + base_vertex * vtx_stride_floats; + const uint32_t* indices = sd.indices.data() + first_index; + + const size_t target_index_count = std::max( + 3, static_cast(mesh.index_count * target_ratio) / 3 * 3); + + // The instanced VBO stores each triangle's vertices separately, so the + // mesh's index buffer is topologically disconnected — every edge is + // boundary, every vertex is unique, and meshopt_simplify can't collapse + // anything. Build a shadow index buffer that welds by position, so + // shared-position vertices share an ID; then simplify on that. Output + // indices are still valid mesh-local IDs (canonical representatives), + // usable directly as LOD1 indices against the same VBO. + shadow.resize(mesh.index_count); + meshopt_generateShadowIndexBuffer( + shadow.data(), + indices, mesh.index_count, + positions, mesh.vertex_count, + sizeof(float) * 3, // compare only xyz + vtx_stride_bytes); + + simplified.resize(mesh.index_count); + float result_error = 0.0f; + size_t new_index_count = 0; + + if (use_sloppy) { + // Cluster-based decimator. Ignores topology entirely; great for + // BIM brep output which is usually non-manifold / has T-junctions. + // Operates directly on the original indices — welding isn't + // needed since it quantises positions into voxel cells. + new_index_count = meshopt_simplifySloppy( + simplified.data(), + indices, mesh.index_count, + positions, mesh.vertex_count, vtx_stride_bytes, + target_index_count, target_error, + &result_error); + } else { + const unsigned int options = + lock_border ? static_cast(meshopt_SimplifyLockBorder) : 0u; + new_index_count = meshopt_simplify( + simplified.data(), + shadow.data(), mesh.index_count, + positions, mesh.vertex_count, vtx_stride_bytes, + target_index_count, target_error, + options, &result_error); + } + + if (debug && dbg_printed < 8) { + std::fprintf(stderr, + " [lod] mesh tris=%u target=%zu got=%zu err=%.4f\n", + tri_count, target_index_count / 3, + new_index_count / 3, result_error); + ++dbg_printed; + } + + // Accept only if we actually saved a meaningful chunk of tris. + if (new_index_count == 0 || new_index_count >= mesh.index_count) { + ++dbg_rejected_noreduce; + continue; + } + + const uint32_t saved = mesh.index_count - static_cast(new_index_count); + if (static_cast(saved) < min_savings * static_cast(mesh.index_count)) { + ++dbg_rejected_savings; + continue; + } + ++dbg_accepted; + + // Append the surviving indices to sd.indices; record the offset. + const size_t append_offset_bytes = sd.indices.size() * sizeof(uint32_t); + sd.indices.insert(sd.indices.end(), + simplified.begin(), + simplified.begin() + new_index_count); + mesh.lod1_ebo_byte_offset = static_cast(append_offset_bytes); + mesh.lod1_index_count = static_cast(new_index_count); + } + + if (debug) { + std::fprintf(stderr, + " [lod] summary: accepted=%d rejected_noreduce=%d rejected_savings=%d " + "(lock_border=%d target_error=%.3f target_ratio=%.3f min_savings=%.3f)\n", + dbg_accepted, dbg_rejected_noreduce, dbg_rejected_savings, + lock_border ? 1 : 0, target_error, target_ratio, min_savings); + } +} + +LodStats summariseLods(const SidecarData& sd) { + LodStats s; + s.meshes_total = static_cast(sd.meshes.size()); + for (const auto& m : sd.meshes) { + s.tris_lod0 += m.index_count / 3; + if (m.lod1_index_count > 0) { + ++s.meshes_with_lod1; + s.tris_lod1 += m.lod1_index_count / 3; + s.tris_lod0_for_lod1 += m.index_count / 3; + } + } + return s; +} diff --git a/src/ifcviewer/LodBuilder.h b/src/ifcviewer/LodBuilder.h new file mode 100644 index 0000000000..a937ae4987 --- /dev/null +++ b/src/ifcviewer/LodBuilder.h @@ -0,0 +1,56 @@ +/******************************************************************************** + * * + * This file is part of IfcOpenShell. * + * * + * IfcOpenShell is free software: you can redistribute it and/or modify * + * it under the terms of the Lesser GNU General Public License as published by * + * the Free Software Foundation, either version 3.0 of the License, or * + * (at your option) any later version. * + * * + * IfcOpenShell is distributed in the hope that it will be useful, * + * but WITHOUT ANY WARRANTY; without even the implied warranty of * + * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the * + * Lesser GNU General Public License for more details. * + * * + * You should have received a copy of the Lesser GNU General Public License * + * along with this program. If not, see . * + * * + ********************************************************************************/ + +#ifndef LODBUILDER_H +#define LODBUILDER_H + +#include "SidecarCache.h" + +// Build a LOD1 index slice for every mesh in `sd` whose triangle count is +// above `min_triangles`, using meshoptimizer's edge-collapse decimator. The +// LOD1 indices are appended to `sd.indices`; each MeshInfo's +// lod1_ebo_byte_offset + lod1_index_count are populated to point at the +// appended range. Meshes that don't qualify (too small) or where the +// decimator couldn't meet the target within the error budget have +// lod1_index_count left at 0 (renderer falls back to LOD0). +// +// Defaults match the Phase 3B first-iteration design: +// min_triangles = 500 — below this the overhead dominates +// target_ratio = 0.25 — aim for 25% of original tris +// target_error = 0.05 — stop if relative error exceeds 5% +// +// `sd.vertices` is read (position is the first 3 floats of each +// INSTANCED_VERTEX_STRIDE_FLOATS-wide vertex) but not modified — LOD1 +// reuses the same vertex buffer, just with a different index list. +void buildLods(SidecarData& sd, + int min_triangles = 500, + float target_ratio = 0.25f, + float target_error = 0.05f); + +// Cheap summary for logging. Safe to call before or after buildLods. +struct LodStats { + uint32_t meshes_total = 0; + uint32_t meshes_with_lod1 = 0; + uint32_t tris_lod0 = 0; // sum across all meshes + uint32_t tris_lod1 = 0; // only for meshes that got LOD1 + uint32_t tris_lod0_for_lod1 = 0; // LOD0 tris of the meshes that got LOD1 +}; +LodStats summariseLods(const SidecarData& sd); + +#endif // LODBUILDER_H diff --git a/src/ifcviewer/MainWindow.cpp b/src/ifcviewer/MainWindow.cpp index 8b63f3bdf6..7dc5454700 100644 --- a/src/ifcviewer/MainWindow.cpp +++ b/src/ifcviewer/MainWindow.cpp @@ -20,6 +20,7 @@ #include "MainWindow.h" #include "AppSettings.h" #include "SettingsWindow.h" +#include "LodBuilder.h" #include "SidecarCache.h" #include @@ -395,6 +396,19 @@ void MainWindow::onStreamingFinished() { sd.elements.push_back(pe); } + // Build LOD1 for eligible meshes (extends sd.indices and + // populates MeshInfo::lod1_*), push the extension onto the + // live GPU state so this session benefits too, then cache. + QElapsedTimer t_lod; t_lod.start(); + buildLods(sd); + LodStats ls = summariseLods(sd); + qDebug(" LOD build: %lld ms — %u/%u meshes got LOD1 " + "(%u tris → %u tris for those meshes)", + t_lod.elapsed(), + ls.meshes_with_lod1, ls.meshes_total, + ls.tris_lod0_for_lod1, ls.tris_lod1); + viewport_->applyLodExtension(loading_model_id_, sd); + std::string ifc_path = it->second.file_path.toStdString(); uint64_t file_size = static_cast( QFileInfo(it->second.file_path).size()); diff --git a/src/ifcviewer/README.md b/src/ifcviewer/README.md index 7bb6972b83..82bd89555c 100644 --- a/src/ifcviewer/README.md +++ b/src/ifcviewer/README.md @@ -92,7 +92,8 @@ engine with a Qt6 interface and OpenGL 4.5 rendering. | `GeometryStreamer.h/cpp` | Background iterator runner; emits `MeshChunk` + `InstanceChunk` | | `InstancedGeometry.h` | Shared structs: `MeshInfo`, `InstanceCpu`, `InstanceGpu`, chunk records | | `BvhAccel.h/cpp` | Median-split BVH builder; operates on instance world-AABBs | -| `SidecarCache.h/cpp` | Raw binary `.ifcview` (v4) sidecar read/write | +| `LodBuilder.h/cpp` | Post-stream decimation of unique meshes via meshoptimizer (`simplifySloppy`) | +| `SidecarCache.h/cpp` | Raw binary `.ifcview` (v5) sidecar read/write | | `AppSettings.h/cpp` | Persisted preferences (geometry library, stats overlay, backface culling) | | `SettingsWindow.h/cpp` | Settings dialog | | `CMakeLists.txt` | Build configuration | @@ -106,6 +107,9 @@ engine with a Qt6 interface and OpenGL 4.5 rendering. at GL 4.1). - **IfcOpenShell C++ libraries** (IfcParse, IfcGeom, and their dependencies: Open CASCADE, Boost, Eigen3, optionally CGAL). +- **[meshoptimizer](https://github.com/zeux/meshoptimizer)** — linked via + `find_package(meshoptimizer REQUIRED)`. Used at sidecar-build time for LOD + decimation; not needed at runtime once a sidecar exists. ## Building @@ -266,7 +270,7 @@ while stack not empty: Depth 64 is enough for billions of items on any balanced tree. The stack is on the C++ stack, zero per-frame allocation. -#### Sidecar format (`.ifcview`, v4) +#### Sidecar format (`.ifcview`, v5) Raw memory dump, Blender-`.blend`-style — no serialisation, no parsing. Stores everything needed to skip the `IfcGeom::Iterator` pass: @@ -276,7 +280,7 @@ SidecarHeader (magic "IFVW", version, endian, ...) uint64_t source_file_size uint32_t + float[] vertex data (7 floats × N_verts, local coords) uint32_t + uint32_t[] index data (mesh-local) -uint32_t + MeshInfo[] per-unique-mesh metadata (48 B each) +uint32_t + MeshInfo[] per-unique-mesh metadata (56 B each, incl. LOD1 slice) uint32_t + InstanceCpu[] per-placement records (transform + AABB + ids) uint32_t + PackedElementInfo[] element tree records uint32_t + char[] string table @@ -449,14 +453,117 @@ At 4 px, frame time breakdown matches: ~16 ms non-draw baseline (from throughput on the post-cull geometry — next steps (LOD, HiZ) attack that directly. -#### 3B. Distance / contribution LOD (medium-term) +#### 3B. Distance / contribution LOD — ✅ done -Pre-simplify unique representations at ingress time (store LOD 0 / 1 / -2 meshes in the VBO/EBO with offsets), select LOD per instance per -frame by the same projected-size metric as 3A. The visible-SSBO -plumbing and MDI structure don't change — only `firstIndex`/`count` in -the indirect command does. Ingress side needs a decimation pass -(`meshoptimizer` or similar); GPU side is nearly free. +Decimate each unique representation once (at sidecar-build time), store +the reduced index slice in the same EBO, and switch to it per-instance +per-frame whenever the projected sphere radius is small enough that the +reduced silhouette is indistinguishable from the original. + +##### Pipeline + +1. **After streaming finishes**, `MainWindow` calls `buildLods(sd)` on + the snapshotted `SidecarData`. Each eligible mesh's decimated index + list is appended to `sd.indices`; the per-mesh `MeshInfo` gains two + new fields: + + ```cpp + uint32_t lod1_ebo_byte_offset; // appended slice, same VBO + uint32_t lod1_index_count; // 0 = no LOD1 was built + ``` + + `MeshInfo` grew from 48 to 56 bytes, which also bumps the sidecar + format to v5. + +2. `viewport_->applyLodExtension(model_id, sd)` pushes the new index + suffix onto the live EBO via `glNamedBufferSubData` and replaces the + CPU-side `m.meshes` vector. The VBO and instance SSBO are untouched + — LOD1 reuses the same vertices, only the indices differ. + +3. The sidecar is then written with both LOD0 and LOD1 indices baked in, + so subsequent loads of the same file pick up LOD1 for free. + +##### Selection + +The contribution-cull pass already computes each instance's projected +pixel radius. LOD1 is selected when that radius falls below +`IFC_LOD1_PX` (default 30 px) and the mesh has a non-empty LOD1 slice. +Camera-inside-AABB short-circuits select LOD0 (treated as "infinite +radius") so you never accidentally see the reduced mesh up close. + +The visible-instance pipeline gains two more buckets (`fwd_lod1_`, +`rev_lod1_`), so the four-way split is now `{fwd, rev} × {LOD0, LOD1}`. +LOD0/LOD1 within a winding slice are contiguous — only winding requires +`glFrontFace` to flip between MDI calls, LOD does not. `firstIndex` / +`count` in the `DrawElementsIndirectCommand` pick which slice of the EBO +to walk; everything else (base vertex, base instance, SSBO bindings, +shader) is unchanged. + +##### Decimator choice: `meshopt_simplifySloppy` + +The first attempt used `meshopt_simplify`, which is an edge-collapse +decimator. It returned every input mesh unchanged (`err = 0.0`) for two +reasons, both inherent to BIM brep output: + +1. **Per-triangle vertex duplication.** The instanced VBO stores each + triangle's vertices separately so that hard-edge normals can differ + across triangles. Topologically there are no shared vertices, so no + edges exist for `meshopt_simplify` to collapse. A + `meshopt_generateShadowIndexBuffer` welding pass (hash xyz only, + ignore the interleaved normal/colour) fixes this half cheaply — the + VBO isn't touched, only a per-call shadow index buffer is built. +2. **Non-manifold topology even after welding.** BIM brep output has + T-junctions, coplanar slivers, separate solids meeting at a plane, + and multi-material cuts. `meshopt_simplify` needs valid 2-manifold + edge pairs to score collapses; it refuses the non-manifold ones, the + priority queue never fires, and it returns the input untouched. + +`meshopt_simplifySloppy` is a **voxel-clustering decimator** — it +quantises positions into cells and merges everything in a cell to a +single point. Topology is irrelevant, so it works directly on the +original indices (welding isn't even needed). The trade-off is that it +rounds off sharp corners and can produce slightly degenerate triangles, +so it doesn't look great at mid-screen size. For a LOD1 that only +activates below 30 px projected radius that's invisible in practice. If +you ever want LOD1 to remain active at larger sizes, the only robust +fix is to pre-process BIM meshes into manifold form (fuse coplanar +faces, split at T-junctions) — a significant project unto itself. + +##### Tuning knobs (env vars) + +| Var | Default | Effect | +|-----|---------|--------| +| `IFC_LOD1_PX` | `30` | Projected sphere radius (px) below which LOD1 kicks in. `0` disables LOD1 entirely. | +| `IFC_LOD_SLOPPY` | `1` | `0` falls back to edge-collapse (`meshopt_simplify`) on shadow-welded indices. Typically produces zero LOD1 output for BIM — useful only for A/B comparison. | +| `IFC_LOD_ERROR` | `0.2` | Target relative error passed to meshopt. | +| `IFC_LOD_RATIO` | `0.25` | Target triangle-count ratio (LOD1 aims for 25 % of LOD0 tris). | +| `IFC_LOD_MIN_SAVINGS` | `0.25` | Reject the LOD1 result if it doesn't shave at least this fraction of triangles. | +| `IFC_LOD_LOCK_BORDER` | `0` | `1` re-enables `meshopt_SimplifyLockBorder` (only meaningful with `IFC_LOD_SLOPPY=0`). | +| `IFC_LOD_DEBUG` | `0` | `1` prints per-mesh `tris / target / got / err` for the first 8 candidate meshes plus an accept/reject summary per model. | + +##### Measured results + +Same 10-model / 128 M-tri scene as Phase 3A (GTX 1650), 2 px contribution +threshold, overview camera, all models finalised with LOD1 built: + +| Build | FPS | Frame time | Visible tris | Visible objs | +|-------|-----|-----------|--------------|--------------| +| Phase 3A alone (2 px) | 20.2 | 49 ms | 40 M | 89 k | +| Phase 3A + 3B (LOD1 ≤ 30 px) | **43.2** | **23 ms** | 14 M | 81 k | + +Roughly half the remaining frame time, same object count (LOD is +lossless w.r.t. visibility — swapping index slice doesn't hide +anything). The triangle reduction on meshes that qualified for LOD1 is +~80 %: e.g. 4.17 M → 0.82 M tris for the 3618 eligible meshes of Model +1, 3.25 M → 0.65 M for Model 2, etc. Only about 20 % of unique meshes +qualify (the threshold is 500 tris — below that the indirect-command +overhead dominates), but those are the fat tail carrying most of the +rasterisation cost. + +LOD build itself runs on the main thread inside `onStreamingFinished`; +typical cost is 100–600 ms per model, folded into the already-visible +"finalizing" step. Cached into the sidecar afterwards, so subsequent +opens skip it entirely. #### 3C. Hierarchical-Z occlusion culling (longer-term) @@ -490,7 +597,7 @@ Scene size Bottleneck Fix < 100k instances CPU cull scan Phase 1 only 100k–500k CPU cull scan BVH (Phase 2) — done 500k+ tris / overview shot GPU vertex + raster Phase 3A contribution cull - (+ 3B LOD for close-ups) + + Phase 3B LOD (done) multi-million + occluders redundant rasterisation Phase 3C HiZ occlusion ``` @@ -508,10 +615,10 @@ multi-million + occluders redundant rasterisation Phase 3C HiZ occlusion - [x] Reflection-aware two-pass draw for mirrored placements - [x] Backface culling (user-toggleable, default on) - [x] `reorient-shells` enabled in iterator -- [x] Perf diagnostic env vars (`IFC_SKIP_MDI`, `IFC_MAX_SUBDRAWS`, `IFC_MIN_PX`) +- [x] Perf diagnostic env vars (`IFC_SKIP_MDI`, `IFC_MAX_SUBDRAWS`, `IFC_MIN_PX`, `IFC_LOD1_PX`) - [x] Phase 3A — screen-space contribution culling -- [ ] **Phase 3B — distance / contribution LOD** (next) -- [ ] Phase 3C — Hierarchical-Z occlusion culling +- [x] Phase 3B — distance / contribution LOD (meshoptimizer `simplifySloppy`) +- [ ] **Phase 3C — Hierarchical-Z occlusion culling** (next) - [ ] Phase 3D — GPU-side compute-shader culling - [ ] Vulkan/MoltenVK backend for macOS - [ ] Embedded Python scripting console diff --git a/src/ifcviewer/SidecarCache.cpp b/src/ifcviewer/SidecarCache.cpp index 3c5ca9cd8d..da3943988d 100644 --- a/src/ifcviewer/SidecarCache.cpp +++ b/src/ifcviewer/SidecarCache.cpp @@ -17,7 +17,11 @@ * * ********************************************************************************/ -// v4 layout (all multi-byte fields native-endian; endianness marker in header): +// v5 layout (all multi-byte fields native-endian; endianness marker in header). +// Same sequence as v4; the only change is that MeshInfo grew two uint32_ts +// (lod1_ebo_byte_offset + lod1_index_count) and `indices` may contain extra +// appended LOD1 slices pointed at by those offsets. +// // // SidecarHeader (16 bytes) // uint64_t source_file_size diff --git a/src/ifcviewer/SidecarCache.h b/src/ifcviewer/SidecarCache.h index e14eb9d256..332abdc802 100644 --- a/src/ifcviewer/SidecarCache.h +++ b/src/ifcviewer/SidecarCache.h @@ -34,7 +34,10 @@ #include static constexpr uint32_t SIDECAR_MAGIC = 0x49465657; // "IFVW" -static constexpr uint32_t SIDECAR_VERSION = 4; +// v5 = MeshInfo extended with lod1_ebo_byte_offset + lod1_index_count (56 B). +// sd.indices may contain an appended LOD1 index slice for each mesh +// where meshoptimizer decimation produced useful output. +static constexpr uint32_t SIDECAR_VERSION = 5; static constexpr uint32_t SIDECAR_ENDIAN = 0x01020304; // Fixed-size element record. Strings are stored as (offset, length) pairs diff --git a/src/ifcviewer/ViewportWindow.cpp b/src/ifcviewer/ViewportWindow.cpp index db002c1870..2606ffd3f3 100644 --- a/src/ifcviewer/ViewportWindow.cpp +++ b/src/ifcviewer/ViewportWindow.cpp @@ -751,6 +751,34 @@ void ViewportWindow::applyCachedModel(uint32_t model_id, SidecarData data) { ssbo_bytes / (1024.0*1024.0)); } +void ViewportWindow::applyLodExtension(uint32_t model_id, const SidecarData& sd) { + if (!gl_initialized_) return; + auto it = models_gpu_.find(model_id); + if (it == models_gpu_.end() || !it->second.finalized) return; + ModelGpuData& m = it->second; + + const size_t total_ib_bytes = sd.indices.size() * sizeof(uint32_t); + if (total_ib_bytes <= m.ebo_used) { + // buildLods didn't add anything; just refresh the meshes vector in + // case lod1_* fields were touched. + m.meshes = sd.meshes; + return; + } + + context_->makeCurrent(this); + if (total_ib_bytes > m.ebo_capacity) { + if (!growModelEbo(m, total_ib_bytes)) return; + } + const size_t append_bytes = total_ib_bytes - m.ebo_used; + const uint32_t* appended_src = + sd.indices.data() + (m.ebo_used / sizeof(uint32_t)); + gl_->glNamedBufferSubData(m.ebo, m.ebo_used, append_bytes, appended_src); + m.ebo_used = total_ib_bytes; + + // Replace mesh metadata so cullAndUploadVisible sees the new lod1_ fields. + m.meshes = sd.meshes; +} + void ViewportWindow::resetScene() { if (!gl_initialized_) return; context_->makeCurrent(this); @@ -826,17 +854,33 @@ uint32_t ViewportWindow::pickObjectAt(int x, int y) { void ViewportWindow::cullAndUploadVisible(ModelGpuData& m, const float planes[6][4], float focal_px, float min_pixel_radius) { - // Per-mesh scratch, split by winding: fwd = non-reflected (CCW in screen - // space), rev = reflected (CW in screen space). Splitting lets the draw + // Per-mesh scratch, split by winding × LOD. Winding split lets the draw // pass toggle glFrontFace once between two MDI calls so GL_CULL_FACE does - // the right thing for both. - if (visible_by_mesh_fwd_.size() < m.meshes.size()) visible_by_mesh_fwd_.resize(m.meshes.size()); - if (visible_by_mesh_rev_.size() < m.meshes.size()) visible_by_mesh_rev_.resize(m.meshes.size()); + // the right thing for both. LOD split means instances that want the + // decimated mesh go into a different bucket that emits against + // mesh.lod1_ebo_byte_offset / lod1_index_count. + auto resize_if = [&](std::vector>& v) { + if (v.size() < m.meshes.size()) v.resize(m.meshes.size()); + }; + resize_if(visible_by_mesh_fwd_lod0_); + resize_if(visible_by_mesh_fwd_lod1_); + resize_if(visible_by_mesh_rev_lod0_); + resize_if(visible_by_mesh_rev_lod1_); for (size_t i = 0; i < m.meshes.size(); ++i) { - visible_by_mesh_fwd_[i].clear(); - visible_by_mesh_rev_[i].clear(); + visible_by_mesh_fwd_lod0_[i].clear(); + visible_by_mesh_fwd_lod1_[i].clear(); + visible_by_mesh_rev_lod0_[i].clear(); + visible_by_mesh_rev_lod1_[i].clear(); } + // LOD1 switches in when projected sphere radius (in pixels) drops below + // this threshold. Overridable for tuning. Set to 0 to disable LOD1 + // entirely (always draw LOD0). + static const float lod1_px_threshold = []{ + const char* e = std::getenv("IFC_LOD1_PX"); + return (e && *e) ? static_cast(std::atof(e)) : 30.0f; + }(); + // Bounding-sphere contribution test: approximate an AABB by its enclosing // sphere (centre = midpoint, radius = half-diagonal). Project radius to // pixels as r_px = focal_px * r / distance (perspective). Reject if @@ -871,15 +915,44 @@ void ViewportWindow::cullAndUploadVisible(ModelGpuData& m, const float planes[6] return focal_px * radius >= min_pixel_radius * dist; }; + // Returns projected sphere radius in pixels (or +inf when camera is + // inside the AABB). Shares the geometry with contributionPasses; this + // version returns the value so we can also use it for LOD selection. + auto pixelRadius = [&](const float mn[3], const float mx[3]) -> float { + if (cx >= mn[0] && cx <= mx[0] && + cy >= mn[1] && cy <= mx[1] && + cz >= mn[2] && cz <= mx[2]) { + return std::numeric_limits::infinity(); + } + float ex = 0.5f * (mx[0] - mn[0]); + float ey = 0.5f * (mx[1] - mn[1]); + float ez = 0.5f * (mx[2] - mn[2]); + float radius = std::sqrt(ex*ex + ey*ey + ez*ez); + float dx = 0.5f * (mx[0] + mn[0]) - cx; + float dy = 0.5f * (mx[1] + mn[1]) - cy; + float dz = 0.5f * (mx[2] + mn[2]) - cz; + float dist = std::sqrt(dx*dx + dy*dy + dz*dz); + return dist > 0.0f ? focal_px * radius / dist + : std::numeric_limits::infinity(); + }; + auto test_and_push = [&](uint32_t inst_idx) { const InstanceCpu& inst = m.instances[inst_idx]; if (!aabbInFrustum(inst.world_aabb_min, inst.world_aabb_max, planes)) return; if (!contributionPasses(inst.world_aabb_min, inst.world_aabb_max)) return; if (inst.mesh_id >= m.meshes.size()) return; + const MeshInfo& mesh = m.meshes[inst.mesh_id]; + const bool want_lod1 = mesh.lod1_index_count > 0 && + lod1_px_threshold > 0.0f && + pixelRadius(inst.world_aabb_min, inst.world_aabb_max) < lod1_px_threshold; const bool reflected = inst_idx < m.instance_reflected.size() && m.instance_reflected[inst_idx] != 0; - if (reflected) visible_by_mesh_rev_[inst.mesh_id].push_back(inst_idx); - else visible_by_mesh_fwd_[inst.mesh_id].push_back(inst_idx); + auto& bucket = + reflected ? (want_lod1 ? visible_by_mesh_rev_lod1_ + : visible_by_mesh_rev_lod0_) + : (want_lod1 ? visible_by_mesh_fwd_lod1_ + : visible_by_mesh_fwd_lod0_); + bucket[inst.mesh_id].push_back(inst_idx); }; if (!m.bvh.nodes.empty()) { @@ -911,22 +984,28 @@ void ViewportWindow::cullAndUploadVisible(ModelGpuData& m, const float planes[6] for (uint32_t i = 0; i < m.instances.size(); ++i) test_and_push(i); } - // Flatten fwd-slice first, then rev-slice, into visible_flat_. Build - // matching DrawElementsIndirectCommands; commands for the fwd slice fill - // [0, indirect_forward_count), rev fills [indirect_forward_count, end). + // Flatten fwd-slice first (LOD0 then LOD1), then rev-slice (ditto), into + // visible_flat_. Commands for the fwd slice fill [0, indirect_forward_count), + // rev fills [indirect_forward_count, end). LOD0/LOD1 within a winding + // slice are contiguous — winding is what requires glFrontFace to flip + // between MDI calls, LOD is not. visible_flat_.clear(); indirect_scratch_.clear(); - auto emit_slice = [&](std::vector>& by_mesh) { + auto emit_slice = [&](std::vector>& by_mesh, int lod) { for (size_t mi = 0; mi < m.meshes.size(); ++mi) { const auto& mesh = m.meshes[mi]; const uint32_t vis_count = static_cast(by_mesh[mi].size()); - if (vis_count == 0 || mesh.index_count == 0) continue; + const uint32_t idx_count = + (lod == 1) ? mesh.lod1_index_count : mesh.index_count; + const uint32_t ebo_off = + (lod == 1) ? mesh.lod1_ebo_byte_offset : mesh.ebo_byte_offset; + if (vis_count == 0 || idx_count == 0) continue; DrawElementsIndirectCommand cmd; - cmd.count = mesh.index_count; + cmd.count = idx_count; cmd.instanceCount = vis_count; - cmd.firstIndex = mesh.ebo_byte_offset / sizeof(uint32_t); + cmd.firstIndex = ebo_off / sizeof(uint32_t); cmd.baseVertex = mesh.vbo_byte_offset / INSTANCED_VERTEX_STRIDE_BYTES; cmd.baseInstance = static_cast(visible_flat_.size()); indirect_scratch_.push_back(cmd); @@ -936,9 +1015,11 @@ void ViewportWindow::cullAndUploadVisible(ModelGpuData& m, const float planes[6] } }; - emit_slice(visible_by_mesh_fwd_); + emit_slice(visible_by_mesh_fwd_lod0_, 0); + emit_slice(visible_by_mesh_fwd_lod1_, 1); m.indirect_forward_count = static_cast(indirect_scratch_.size()); - emit_slice(visible_by_mesh_rev_); + emit_slice(visible_by_mesh_rev_lod0_, 0); + emit_slice(visible_by_mesh_rev_lod1_, 1); m.indirect_command_count = static_cast(indirect_scratch_.size()); // Upload visible list (keep binding alive even when empty). diff --git a/src/ifcviewer/ViewportWindow.h b/src/ifcviewer/ViewportWindow.h index a8d696121a..fe54cce921 100644 --- a/src/ifcviewer/ViewportWindow.h +++ b/src/ifcviewer/ViewportWindow.h @@ -127,6 +127,13 @@ public: // any existing state for model_id and marks it drawable. void applyCachedModel(uint32_t model_id, SidecarData data); + // After buildLods() has extended sd.indices + populated lod1_* fields, + // push just the appended index slice + the refreshed mesh metadata onto + // the live GPU state for model_id. VBO / SSBO / instance array are left + // alone; only the EBO grows and m.meshes is replaced. No-op if the + // model isn't finalised on the viewport. + void applyLodExtension(uint32_t model_id, const SidecarData& sd); + void hideModel(uint32_t model_id); void showModel(uint32_t model_id); void removeModel(uint32_t model_id); @@ -223,8 +230,13 @@ private: // per-frame allocation. indirect_scratch_ is the matching array of // DrawElementsIndirectCommand records — forward-declared as bytes so // the header doesn't need the struct definition. - std::vector> visible_by_mesh_fwd_; - std::vector> visible_by_mesh_rev_; + // Four buckets = {fwd, rev} × {LOD0, LOD1}. LOD1 buckets are only + // populated when the mesh has lod1_index_count > 0 and the projected + // pixel radius is below the LOD switch threshold. + std::vector> visible_by_mesh_fwd_lod0_; + std::vector> visible_by_mesh_fwd_lod1_; + std::vector> visible_by_mesh_rev_lod0_; + std::vector> visible_by_mesh_rev_lod1_; std::vector visible_flat_; std::vector indirect_scratch_;