mirror of
https://github.com/IfcOpenShell/IfcOpenShell.git
synced 2026-08-10 09:48:32 +00:00
wgpu: probed-size pool replaces per-chunk createBuffer
Drops the per-machine "guess the OOM ceiling" budget knob in favour of a single buffer pool whose capacity is *probed* at device-init time. The runtime answers the question: descend from min(maxBufferSize, 4 GB) through OOM error scopes, accept the largest size that allocates cleanly. On a desktop wgpu-native v29 box this lands at 2 GB; on browser-class platforms it'll land at 256 MB – 1 GB depending on the implementation. Same code path either way. Architecture: - WgpuBufferPool (new): single WGPUBuffer + free-list sub-allocator with adjacent-range coalescing and first-fit. 256 B alignment for storage-binding offsets. - Chunks now hold (pool_vertex_offset, pool_vertex_size) and (pool_index_offset, pool_index_size) instead of per-chunk WGPUBuffer handles. Load = pool.alloc + queueWriteBuffer. Unload = pool.free. - Bind groups bind pool_.buffer() at the chunk's specific (offset, size) for both the vertex and index storage bindings. - Eviction queries pool.largest_free_run_bytes() instead of a tracked budget; the two-phase LRU/distance evictor's policy is unchanged. What this fixes: - No more gpu-alloc-rs fragmentation OOM: one VkDeviceMemory block instead of N per-chunk blocks with rounding overhead. On the test dataset (~3 GB on disk, 562 k visible instances) the wgpu backend now runs through to render without OOM at any point. - No --streaming-vram-mb knob, no hardcoded budget constant, no per-machine calibration. The pool size adapts to whatever the runtime grants. Notes: - Error scope probing: wgpu-native v29 classifies "Not enough memory left" as WGPUErrorType_Validation, not OutOfMemory. We push both filters (nested) and treat either firing as probe failure. - The 4 GB probe cap is principled, not magic: above that, wgpu-native's advertised maxBufferSize is sometimes a sentinel (1 TB) that just forces wasteful halving steps. 4 GB is the largest buffer any realistic WebGPU implementation will grant a single allocation today. - Pool destroy()/release happens after model release in shutdown() so the underlying buffer outlives every bind group that references it. Follow-ups: spatial chunking (task #22) for finer eviction granularity; cull perf needs work at 100+ models / 1M+ instances (separate from streaming concerns). Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,130 @@
|
||||
/********************************************************************************
|
||||
* *
|
||||
* This file is part of IfcOpenShell. *
|
||||
* *
|
||||
* IfcOpenShell is free software: you can redistribute it and/or modify *
|
||||
* it under the terms of the Lesser GNU General Public License as published by *
|
||||
* the Free Software Foundation, either version 3.0 of the License, or *
|
||||
* (at your option) any later version. *
|
||||
* *
|
||||
* IfcOpenShell is distributed in the hope that it will be useful, *
|
||||
* but WITHOUT ANY WARRANTY; without even the implied warranty of *
|
||||
* MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the *
|
||||
* Lesser GNU General Public License for more details. *
|
||||
* *
|
||||
* You should have received a copy of the Lesser GNU General Public License *
|
||||
* along with this program. If not, see <http://www.gnu.org/licenses/>. *
|
||||
* *
|
||||
********************************************************************************/
|
||||
|
||||
#include "WgpuBufferPool.h"
|
||||
|
||||
#include <cassert>
|
||||
#include <cstring>
|
||||
|
||||
WgpuBufferPool::~WgpuBufferPool() {
|
||||
destroy();
|
||||
}
|
||||
|
||||
bool WgpuBufferPool::init(WGPUDevice device, uint64_t capacity_bytes,
|
||||
WGPUBufferUsage usage, const char* label) {
|
||||
destroy();
|
||||
if (capacity_bytes == 0) return false;
|
||||
|
||||
WGPUBufferDescriptor desc = {};
|
||||
desc.usage = usage;
|
||||
desc.size = capacity_bytes;
|
||||
if (label) {
|
||||
desc.label.data = label;
|
||||
desc.label.length = std::strlen(label);
|
||||
}
|
||||
buffer_ = wgpuDeviceCreateBuffer(device, &desc);
|
||||
if (!buffer_) return false;
|
||||
|
||||
capacity_ = capacity_bytes;
|
||||
used_ = 0;
|
||||
free_ranges_.clear();
|
||||
free_ranges_.push_back({0, capacity_bytes});
|
||||
return true;
|
||||
}
|
||||
|
||||
void WgpuBufferPool::destroy() {
|
||||
if (buffer_) {
|
||||
wgpuBufferRelease(buffer_);
|
||||
buffer_ = nullptr;
|
||||
}
|
||||
capacity_ = 0;
|
||||
used_ = 0;
|
||||
free_ranges_.clear();
|
||||
}
|
||||
|
||||
bool WgpuBufferPool::alloc(uint64_t size, uint64_t align, uint64_t* out_offset) {
|
||||
if (size == 0 || align == 0) return false;
|
||||
// First-fit: scan free ranges, pick the first that fits with alignment.
|
||||
for (size_t i = 0; i < free_ranges_.size(); ++i) {
|
||||
const FreeRange& r = free_ranges_[i];
|
||||
const uint64_t aligned = (r.offset + (align - 1)) & ~(align - 1);
|
||||
const uint64_t pad = aligned - r.offset;
|
||||
if (pad >= r.size) continue; // alignment alone won't fit
|
||||
if (size > r.size - pad) continue; // payload won't fit
|
||||
|
||||
// Split the range. Three resulting pieces:
|
||||
// [r.offset, aligned) -> pre-pad, returned to free list
|
||||
// [aligned, aligned + size) -> the allocation (claimed)
|
||||
// [aligned + size, r.offset + r.size) -> post-pad, returned to free list
|
||||
const uint64_t post_off = aligned + size;
|
||||
const uint64_t post_size = (r.offset + r.size) - post_off;
|
||||
|
||||
// Mutate in place: replace the matched range with the pre-pad
|
||||
// (or erase it if there's no pre-pad), then optionally insert
|
||||
// the post-pad immediately after.
|
||||
if (pad == 0 && post_size == 0) {
|
||||
free_ranges_.erase(free_ranges_.begin() + i);
|
||||
} else if (pad == 0) {
|
||||
free_ranges_[i] = {post_off, post_size};
|
||||
} else if (post_size == 0) {
|
||||
free_ranges_[i] = {r.offset, pad};
|
||||
} else {
|
||||
free_ranges_[i] = {r.offset, pad};
|
||||
free_ranges_.insert(free_ranges_.begin() + i + 1, {post_off, post_size});
|
||||
}
|
||||
|
||||
used_ += size; // pre-/post-pad remain in free_ranges_, not used_
|
||||
*out_offset = aligned;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void WgpuBufferPool::free(uint64_t offset, uint64_t size) {
|
||||
if (size == 0) return;
|
||||
assert(offset + size <= capacity_);
|
||||
|
||||
// Find insertion point: first range whose offset > released offset.
|
||||
size_t i = 0;
|
||||
while (i < free_ranges_.size() && free_ranges_[i].offset < offset) ++i;
|
||||
free_ranges_.insert(free_ranges_.begin() + i, {offset, size});
|
||||
used_ -= size;
|
||||
|
||||
// Coalesce with right neighbour first (so subsequent left-coalesce
|
||||
// sees the merged range).
|
||||
if (i + 1 < free_ranges_.size()
|
||||
&& free_ranges_[i].offset + free_ranges_[i].size == free_ranges_[i + 1].offset) {
|
||||
free_ranges_[i].size += free_ranges_[i + 1].size;
|
||||
free_ranges_.erase(free_ranges_.begin() + i + 1);
|
||||
}
|
||||
// Coalesce with left neighbour.
|
||||
if (i > 0
|
||||
&& free_ranges_[i - 1].offset + free_ranges_[i - 1].size == free_ranges_[i].offset) {
|
||||
free_ranges_[i - 1].size += free_ranges_[i].size;
|
||||
free_ranges_.erase(free_ranges_.begin() + i);
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t WgpuBufferPool::largest_free_run_bytes() const {
|
||||
uint64_t m = 0;
|
||||
for (const auto& r : free_ranges_) {
|
||||
if (r.size > m) m = r.size;
|
||||
}
|
||||
return m;
|
||||
}
|
||||
Reference in New Issue
Block a user