mirror of
https://github.com/IfcOpenShell/IfcOpenShell.git
synced 2026-08-09 09:21:46 +00:00
dcc2bf1c01
The sync chunk-read on the render thread was causing 100-300 ms spikes during orbit whenever a new chunk needed to scatter-gather its mesh bytes from disk. p99 was 326 ms on the close-camera benchmark. New WgpuStreamingThread: one worker thread with a condvar-protected request/result queue. driveStreamingLoads becomes drain-then-enqueue: 1. Drain any results the worker pushed since last frame. For each, pool-allocate slices + queueWriteBuffer + build the chunk bind group (still main-thread because wgpu queue ops aren't thread-safe). 2. Walk visible non-resident chunks (sorted by distance), evict to make pool room, and enqueue the request. Chunk gains is_loading flag to prevent re-enqueueing while in flight. loadChunkBytesAndUploadGpu becomes the sync fallback path, used only when a screenshot is pending — the deferred-capture wait would otherwise let the window manager re-layout the window between frames and the test framework would capture at the wrong size. Normal streaming always goes through the worker. Bench warm-gate / requestUpdate gating updated to consider streaming_thread_.inFlightApprox() so we don't declare "converged" while a worker read is still in flight, and the render loop stays alive until the worker queue is empty. Refactored loadChunkBytesAndUploadGpu into two helpers: - makeChunkRequest: builds the worker request from chunk metadata - applyStreamedChunk: pool.alloc + queueWriteBuffer + bind group Both the sync and async paths share applyStreamedChunk. Benchmark (big federation, --streaming): close camera: avg 24 fps p99 47 ms (was 27/326) default camera: avg 24 fps p99 46 ms (was 31/186) stream time: ~2 ms (was 8-12) cull is now the bottleneck (20 ms median) — task #17 (GPU compute cull) is the next frontier. Pixel-identical to non-streaming on basic.ifc on both paths. Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
123 lines
4.4 KiB
C++
123 lines
4.4 KiB
C++
/********************************************************************************
|
|
* *
|
|
* This file is part of IfcOpenShell. *
|
|
* *
|
|
* IfcOpenShell is free software: you can redistribute it and/or modify *
|
|
* it under the terms of the Lesser GNU General Public License as published by *
|
|
* the Free Software Foundation, either version 3.0 of the License, or *
|
|
* (at your option) any later version. *
|
|
* *
|
|
* IfcOpenShell is distributed in the hope that it will be useful, *
|
|
* but WITHOUT ANY WARRANTY; without even the implied warranty of *
|
|
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the *
|
|
* Lesser GNU General Public License for more details. *
|
|
* *
|
|
* You should have received a copy of the Lesser GNU General Public License *
|
|
* along with this program. If not, see <http://www.gnu.org/licenses/>. *
|
|
* *
|
|
********************************************************************************/
|
|
|
|
#include "WgpuStreamingThread.h"
|
|
|
|
#include "WgpuStreamingLoader.h"
|
|
|
|
WgpuStreamingThread::~WgpuStreamingThread() {
|
|
stop();
|
|
}
|
|
|
|
void WgpuStreamingThread::start() {
|
|
std::unique_lock lk(mu_);
|
|
if (running_) return;
|
|
shutdown_ = false;
|
|
running_ = true;
|
|
lk.unlock();
|
|
worker_ = std::thread(&WgpuStreamingThread::workerLoop, this);
|
|
}
|
|
|
|
void WgpuStreamingThread::stop() {
|
|
{
|
|
std::unique_lock lk(mu_);
|
|
if (!running_) return;
|
|
shutdown_ = true;
|
|
}
|
|
cv_.notify_all();
|
|
if (worker_.joinable()) worker_.join();
|
|
std::unique_lock lk(mu_);
|
|
running_ = false;
|
|
requests_.clear();
|
|
results_.clear();
|
|
}
|
|
|
|
bool WgpuStreamingThread::enqueue(Request req) {
|
|
{
|
|
std::unique_lock lk(mu_);
|
|
if (!running_ || shutdown_) return false;
|
|
requests_.push_back(std::move(req));
|
|
}
|
|
cv_.notify_one();
|
|
return true;
|
|
}
|
|
|
|
std::vector<WgpuStreamingThread::Result> WgpuStreamingThread::drainResults() {
|
|
std::vector<Result> out;
|
|
{
|
|
std::unique_lock lk(mu_);
|
|
out.reserve(results_.size());
|
|
while (!results_.empty()) {
|
|
out.push_back(std::move(results_.front()));
|
|
results_.pop_front();
|
|
}
|
|
}
|
|
return out;
|
|
}
|
|
|
|
std::size_t WgpuStreamingThread::inFlightApprox() const {
|
|
std::unique_lock lk(mu_);
|
|
return requests_.size() + (in_progress_ ? 1u : 0u);
|
|
}
|
|
|
|
void WgpuStreamingThread::workerLoop() {
|
|
for (;;) {
|
|
Request req;
|
|
{
|
|
std::unique_lock lk(mu_);
|
|
cv_.wait(lk, [this]() { return shutdown_ || !requests_.empty(); });
|
|
if (shutdown_ && requests_.empty()) return;
|
|
req = std::move(requests_.front());
|
|
requests_.pop_front();
|
|
in_progress_ = true;
|
|
}
|
|
|
|
// Disk reads happen off-thread. Each Request carries everything
|
|
// the reader needs; the viewport keeps the corresponding chunk
|
|
// marked is_loading so eviction won't yank the slot underneath
|
|
// us. The vbytes / idx buffers are allocated here on the worker
|
|
// thread — they cross back to the main thread when the result
|
|
// is drained and applied (pool.alloc + queueWriteBuffer).
|
|
Result res;
|
|
res.model_id = req.model_id;
|
|
res.chunk_idx = req.chunk_idx;
|
|
res.success = true;
|
|
if (!req.v_ranges.empty()) {
|
|
if (!readSidecarVertexRanges(req.file_path,
|
|
req.vertex_section_offset,
|
|
req.v_ranges, res.vbytes)) {
|
|
res.success = false;
|
|
}
|
|
}
|
|
if (res.success && !req.i_ranges.empty()) {
|
|
if (!readSidecarIndexRanges(req.file_path,
|
|
req.index_section_offset,
|
|
req.i_ranges, res.idx)) {
|
|
res.success = false;
|
|
}
|
|
}
|
|
|
|
{
|
|
std::unique_lock lk(mu_);
|
|
results_.push_back(std::move(res));
|
|
in_progress_ = false;
|
|
}
|
|
}
|
|
}
|