ifcviewer: stop the live budget from over-shrinking and from going stale

A 66-model session showed the pool reach a 1938 MB ceiling, the next
poll lower the budget to 1756, and the shrink drop 402 MB (73+73+256)
for a 182 MB excess, which the pool then spent seconds re-growing. Two
causes.

The release granularity is whole sub-buffers but the shrink ran "until
capacity ≤ target", so the last 36 MB of excess cost a 256 MB
sub-buffer. shrinkToCapacity now never undershoots — it releases only
while doing so keeps capacity ≥ target, leaving a sub-buffer's worth of
excess for the margin to absorb — and the pressure path uses a separate
releaseAtLeast(bytes), whose contract is the opposite: free at least
what the failed allocation needs, whatever the granularity. Resident
geometry is also only evicted once the pool is over budget by half the
margin (GpuBudget::shrinkTarget), so report jitter does not trigger a
shrink-and-reload.

The ceiling was a second old when the pool grew into it, and the upload
staging that rides on growth had pushed device free memory to ~74 MB —
below the driver's observed refusal point — before the next scheduled
poll. pollDeviceMemory now re-derives the budget immediately after any
sub-buffer is added, so the next growth decision sees the device as it
is.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Dion Moult
2026-08-23 19:19:44 +10:00
parent 6201c4052b
commit 24616ed655
7 changed files with 152 additions and 42 deletions
+53 -24
View File
@@ -166,36 +166,65 @@ bool BufferPool::addSubBuffer() {
#endif // __EMSCRIPTEN__ #endif // __EMSCRIPTEN__
} }
uint64_t BufferPool::releaseNewestSubBuffer(
const std::function<void(int sub_idx)>& evict_sub_buffer) {
if (sub_pools_.empty()) return 0;
const int idx = int(sub_pools_.size()) - 1;
if (sub_pools_[size_t(idx)].provisional) return 0;
evict_sub_buffer(idx);
SubPool& sub_pool = sub_pools_[size_t(idx)];
assert(sub_pool.used == 0 && "owner must free every slice before a sub-buffer is released");
if (sub_pool.buffer && sub_pool.owns_handle) {
// Destroy, not just release: the handle may still be
// referenced by in-flight work, and destroy tells the
// backend to reclaim the memory as soon as that completes
// instead of when the last reference goes away.
wgpuBufferDestroy(sub_pool.buffer);
wgpuBufferRelease(sub_pool.buffer);
}
const uint64_t released = sub_pool.capacity;
sub_pools_.pop_back();
return released;
}
namespace {
void logRelease(uint64_t released, uint64_t total, size_t count, uint64_t budget) {
if (released == 0) return;
std::fprintf(stderr,
"[wgpu pool] released %llu MB under memory pressure; pool now %llu MB "
"across %zu sub-buffer(s), budget %llu MB\n",
(unsigned long long)(released / (1024 * 1024)),
(unsigned long long)(total / (1024 * 1024)),
count,
(unsigned long long)(budget / (1024 * 1024)));
}
} // namespace
uint64_t BufferPool::shrinkToCapacity( uint64_t BufferPool::shrinkToCapacity(
uint64_t target_bytes, uint64_t target_bytes,
const std::function<void(int sub_idx)>& evict_sub_buffer) { const std::function<void(int sub_idx)>& evict_sub_buffer) {
uint64_t released = 0; uint64_t released = 0;
while (!sub_pools_.empty() && total_capacity_bytes() > target_bytes) { while (!sub_pools_.empty()) {
const int idx = int(sub_pools_.size()) - 1; const SubPool& newest = sub_pools_.back();
if (sub_pools_[size_t(idx)].provisional) break; if (newest.provisional) break;
evict_sub_buffer(idx); const uint64_t capacity = total_capacity_bytes();
SubPool& sub_pool = sub_pools_[size_t(idx)]; if (capacity < target_bytes + newest.capacity) break; // would undershoot
assert(sub_pool.used == 0 && "owner must free every slice before a sub-buffer is released"); released += releaseNewestSubBuffer(evict_sub_buffer);
if (sub_pool.buffer && sub_pool.owns_handle) {
// Destroy, not just release: the handle may still be
// referenced by in-flight work, and destroy tells the
// backend to reclaim the memory as soon as that completes
// instead of when the last reference goes away.
wgpuBufferDestroy(sub_pool.buffer);
wgpuBufferRelease(sub_pool.buffer);
}
released += sub_pool.capacity;
sub_pools_.pop_back();
} }
if (released > 0) { logRelease(released, total_capacity_bytes(), sub_pools_.size(), max_total_capacity_bytes_);
std::fprintf(stderr, return released;
"[wgpu pool] released %llu MB under memory pressure; pool now %llu MB " }
"across %zu sub-buffer(s), budget %llu MB\n",
(unsigned long long)(released / (1024 * 1024)), uint64_t BufferPool::releaseAtLeast(
(unsigned long long)(total_capacity_bytes() / (1024 * 1024)), uint64_t bytes,
sub_pools_.size(), const std::function<void(int sub_idx)>& evict_sub_buffer) {
(unsigned long long)(max_total_capacity_bytes_ / (1024 * 1024))); uint64_t released = 0;
while (released < bytes) {
const uint64_t got = releaseNewestSubBuffer(evict_sub_buffer);
if (got == 0) break;
released += got;
} }
logRelease(released, total_capacity_bytes(), sub_pools_.size(), max_total_capacity_bytes_);
return released; return released;
} }
+22 -10
View File
@@ -137,18 +137,26 @@ public:
void setMaxTotalCapacity(uint64_t max_bytes) { max_total_capacity_bytes_ = max_bytes; } void setMaxTotalCapacity(uint64_t max_bytes) { max_total_capacity_bytes_ = max_bytes; }
uint64_t max_total_capacity_bytes() const { return max_total_capacity_bytes_; } uint64_t max_total_capacity_bytes() const { return max_total_capacity_bytes_; }
// Release whole sub-buffers, newest first, until total capacity is // Release whole sub-buffers, newest first. Before each is dropped,
// ≤ target_bytes (or nothing is left). Before each sub-buffer is // `evict_sub_buffer(sub_idx)` is invoked so the owner can free every
// dropped, `evict_sub_buffer(sub_idx)` is invoked so the owner can // slice that lives in it — the pool does not know what a slice holds,
// free every slice that lives in it — the pool does not know what a // and a sub-buffer is only released once it is empty. Releasing from
// slice holds, and a sub-buffer is only released once it is empty. // the back keeps every surviving Slice::sub_idx valid. Both return the
// Releasing from the back keeps every surviving Slice::sub_idx valid. // bytes released. This is how the cache yields memory to the required
// Returns the number of bytes released. This is how the cache yields // tier (see GpuBudget); on web a provisional sub-buffer that is still
// memory to the required tier (see GpuBudget); on web a provisional // validating is left alone and the caller retries once it resolves.
// sub-buffer that is still validating is left alone and the shrink is //
// retried once that resolves. // shrinkToCapacity never goes *below* target_bytes: a sub-buffer is
// released only while doing so keeps capacity ≥ target, so an excess
// smaller than the newest sub-buffer releases nothing (the budget's
// margin absorbs it) instead of dropping 256 MB for the last 36.
uint64_t shrinkToCapacity(uint64_t target_bytes, uint64_t shrinkToCapacity(uint64_t target_bytes,
const std::function<void(int sub_idx)>& evict_sub_buffer); const std::function<void(int sub_idx)>& evict_sub_buffer);
// releaseAtLeast frees sub-buffers until at least `bytes` have gone
// (or nothing is left) — for a failed required allocation that needs
// that much back no matter the granularity.
uint64_t releaseAtLeast(uint64_t bytes,
const std::function<void(int sub_idx)>& evict_sub_buffer);
// Proactively add a sub-buffer (no allocation). On web this kicks off the // Proactively add a sub-buffer (no allocation). On web this kicks off the
// async provisional-validation cycle so validated free space appears a // async provisional-validation cycle so validated free space appears a
@@ -201,6 +209,10 @@ private:
// ≥ MIN_SUB_BUFFER_BYTES; false only when even the minimum size is // ≥ MIN_SUB_BUFFER_BYTES; false only when even the minimum size is
// refused, at which point growth_disabled_ latches. // refused, at which point growth_disabled_ latches.
bool addSubBuffer(); bool addSubBuffer();
// Drop the newest sub-buffer after `evict_sub_buffer` empties it.
// Returns its capacity; 0 when the pool is empty or the newest
// sub-buffer is still provisional (web).
uint64_t releaseNewestSubBuffer(const std::function<void(int sub_idx)>& evict_sub_buffer);
#if defined(__EMSCRIPTEN__) #if defined(__EMSCRIPTEN__)
// Web-only async-growth resolver. Called from the AllowSpontaneous // Web-only async-growth resolver. Called from the AllowSpontaneous
+14
View File
@@ -74,6 +74,13 @@ public:
// budget, so the very next small required allocation does not fail // budget, so the very next small required allocation does not fail
// again and trigger another shrink cycle. // again and trigger another shrink cycle.
static constexpr std::uint64_t kPressureSlackBytes = 32ull * 1024 * 1024; static constexpr std::uint64_t kPressureSlackBytes = 32ull * 1024 * 1024;
// The live budget moves with every driver report, and reports jitter
// (upload staging, other processes). The pool's ceiling follows the
// budget exactly, but geometry already resident is only evicted once
// the pool is over budget by this much — i.e. once the device's free
// memory has dropped below half the margin — so a transient dip does
// not cost a shrink-and-reload.
static constexpr std::uint64_t kShrinkHysteresisBytes = kFixedMarginBytes / 2;
// Absolute ceiling regardless of device memory (the wasm heap on web). // Absolute ceiling regardless of device memory (the wasm heap on web).
// 0 = none. // 0 = none.
@@ -95,6 +102,13 @@ public:
std::uint64_t bytes_needed, std::uint64_t bytes_needed,
std::uint64_t device_free_bytes); std::uint64_t device_free_bytes);
// Capacity the pool should shrink to right now, or 0 when it is within
// the hysteresis band (or the budget is unbounded).
std::uint64_t shrinkTarget(std::uint64_t cache_capacity_bytes) const {
if (!bounded_ || cache_capacity_bytes < budget_ + kShrinkHysteresisBytes) return 0;
return budget_;
}
// False until something bounds the cache (a device report, a cap, or // False until something bounds the cache (a device report, a cap, or
// a pressure event). The pool then grows until the driver refuses, // a pressure event). The pool then grows until the driver refuses,
// exactly as before; the first of those bounds it. // exactly as before; the first of those bounds it.
+14 -8
View File
@@ -1657,9 +1657,15 @@ ifcviewer::GpuMemoryInfo ViewportCore::queryDeviceMemory() const {
} }
void ViewportCore::pollDeviceMemory() { void ViewportCore::pollDeviceMemory() {
if (device_vram_poll_timer_.isValid() // Re-derive immediately after the pool has grown: a ceiling computed
// from a second-old report can be reached by growth plus the upload
// staging that rides on it, leaving the device far below the margin
// before the next scheduled poll notices.
const bool pool_grew = pool_.sub_buffer_count() != polled_sub_buffer_count_;
if (!pool_grew && device_vram_poll_timer_.isValid()
&& device_vram_poll_timer_.elapsed() < 1000) return; && device_vram_poll_timer_.elapsed() < 1000) return;
device_vram_poll_timer_.start(); device_vram_poll_timer_.start();
polled_sub_buffer_count_ = pool_.sub_buffer_count();
const ifcviewer::GpuMemoryInfo mem = queryDeviceMemory(); const ifcviewer::GpuMemoryInfo mem = queryDeviceMemory();
if (!mem.valid) return; if (!mem.valid) return;
device_vram_used_bytes_ = mem.used_bytes; device_vram_used_bytes_ = mem.used_bytes;
@@ -1672,12 +1678,10 @@ void ViewportCore::applyBudgetToPool() {
if (!budget_.bounded()) return; if (!budget_.bounded()) return;
const std::uint64_t budget = budget_.cache_budget_bytes(); const std::uint64_t budget = budget_.cache_budget_bytes();
pool_.setMaxTotalCapacity(budget); pool_.setMaxTotalCapacity(budget);
// Whole sub-buffers are the release granularity, so only act once the const std::uint64_t target = budget_.shrinkTarget(pool_.total_capacity_bytes());
// excess is worth one; below that the fixed margin covers it. if (target == 0) return;
const std::uint64_t capacity = pool_.total_capacity_bytes();
if (capacity < budget + BufferPool::MIN_SUB_BUFFER_BYTES) return;
const std::uint64_t released = pool_.shrinkToCapacity( const std::uint64_t released = pool_.shrinkToCapacity(
budget, [this](int sub_idx) { evictChunksInSubBuffer(sub_idx); }); target, [this](int sub_idx) { evictChunksInSubBuffer(sub_idx); });
if (released > 0) { if (released > 0) {
const double mb = 1.0 / (1024.0 * 1024.0); const double mb = 1.0 / (1024.0 * 1024.0);
Log::info() << "[wgpu] device has less to give: geometry cache budget now " Log::info() << "[wgpu] device has less to give: geometry cache budget now "
@@ -1712,8 +1716,10 @@ bool ViewportCore::onRequiredAllocationFailed(const char* what, std::uint64_t by
return false; return false;
} }
pool_.setMaxTotalCapacity(budget_.cache_budget_bytes()); pool_.setMaxTotalCapacity(budget_.cache_budget_bytes());
const std::uint64_t released = pool_.shrinkToCapacity( // The budget was just lowered by what the allocation needs; free at
budget_.cache_budget_bytes(), // least that much, whatever the sub-buffer granularity.
const std::uint64_t released = pool_.releaseAtLeast(
capacity_before - budget_.cache_budget_bytes(),
[this](int sub_idx) { evictChunksInSubBuffer(sub_idx); }); [this](int sub_idx) { evictChunksInSubBuffer(sub_idx); });
Log::warn() << "[wgpu] out of memory allocating " << what << " (" Log::warn() << "[wgpu] out of memory allocating " << what << " ("
<< double(bytes) * mb << " MB); geometry cache budget lowered to " << double(bytes) * mb << " MB); geometry cache budget lowered to "
+1
View File
@@ -1578,6 +1578,7 @@ private:
std::uint64_t device_vram_used_bytes_ = 0; std::uint64_t device_vram_used_bytes_ = 0;
std::uint64_t device_vram_total_bytes_ = 0; std::uint64_t device_vram_total_bytes_ = 0;
Stopwatch device_vram_poll_timer_; Stopwatch device_vram_poll_timer_;
std::size_t polled_sub_buffer_count_ = 0;
double last_cull_ms_ = 0.0; double last_cull_ms_ = 0.0;
double last_cull_compute_ms_ = 0.0; double last_cull_compute_ms_ = 0.0;
double last_cull_upload_ms_ = 0.0; double last_cull_upload_ms_ = 0.0;
+38
View File
@@ -312,3 +312,41 @@ TEST_CASE("shrinkToCapacity releases sub-buffers newest-first after the owner em
REQUIRE(pool.sub_buffer_count() == 0); REQUIRE(pool.sub_buffer_count() == 0);
REQUIRE(evicted == std::vector<int>{0}); REQUIRE(evicted == std::vector<int>{0});
} }
TEST_CASE("shrinkToCapacity never undershoots the target", "[buffer_pool]") {
BufferPool pool;
FakePoolGuard guard{pool};
pool.addSubBufferForTesting(fake_handle(1), 256);
pool.addSubBufferForTesting(fake_handle(2), 256);
pool.addSubBufferForTesting(fake_handle(3), 73);
pool.addSubBufferForTesting(fake_handle(4), 73);
auto evict = [](int) {};
// 658 held, target 476: 182 over. The two 73s go (36 still over);
// the 256 would undershoot, so it stays — the margin absorbs 36.
REQUIRE(pool.shrinkToCapacity(476, evict) == 146);
REQUIRE(pool.total_capacity_bytes() == 512);
// An excess smaller than the newest sub-buffer releases nothing.
REQUIRE(pool.shrinkToCapacity(500, evict) == 0);
REQUIRE(pool.total_capacity_bytes() == 512);
}
TEST_CASE("releaseAtLeast frees whole sub-buffers until the requested bytes are gone", "[buffer_pool]") {
BufferPool pool;
FakePoolGuard guard{pool};
pool.addSubBufferForTesting(fake_handle(1), 256);
pool.addSubBufferForTesting(fake_handle(2), 73);
pool.addSubBufferForTesting(fake_handle(3), 73);
std::vector<int> evicted;
auto evict = [&](int sub_idx) { evicted.push_back(sub_idx); };
// Needs 100: 73 is not enough, 73+73 is. Overshoot by a sub-buffer is
// the point — the allocation must fit.
REQUIRE(pool.releaseAtLeast(100, evict) == 146);
REQUIRE(evicted == std::vector<int>{2, 1});
REQUIRE(pool.total_capacity_bytes() == 256);
// More than the pool holds: everything goes, no crash.
REQUIRE(pool.releaseAtLeast(1000, evict) == 256);
REQUIRE(pool.sub_buffer_count() == 0);
REQUIRE(pool.releaseAtLeast(1, evict) == 0);
}
+10
View File
@@ -150,3 +150,13 @@ TEST_CASE("a refusal with memory still reported free teaches the margin", "[gpu_
REQUIRE(b.margin_bytes() REQUIRE(b.margin_bytes()
== GpuBudget::kFixedMarginBytes + 162 * MB + GpuBudget::kPressureSlackBytes); == GpuBudget::kFixedMarginBytes + 162 * MB + GpuBudget::kPressureSlackBytes);
} }
TEST_CASE("resident geometry is only shrunk once over budget by the hysteresis", "[gpu_budget]") {
GpuBudget b;
REQUIRE(b.shrinkTarget(4096 * MB) == 0); // unbounded: never
b.update(2800 * MB, 0);
const std::uint64_t budget = b.cache_budget_bytes();
REQUIRE(b.shrinkTarget(budget) == 0);
REQUIRE(b.shrinkTarget(budget + GpuBudget::kShrinkHysteresisBytes - 1) == 0);
REQUIRE(b.shrinkTarget(budget + GpuBudget::kShrinkHysteresisBytes) == budget);
}