mirror of
https://github.com/IfcOpenShell/IfcOpenShell.git
synced 2026-09-26 18:21:59 +00:00
ifcviewer: move driveStreamingLoads into ViewportCore (#84-o)
The per-frame streaming residency driver — LRU/priority eviction,
worker-result drain, candidate selection, click-and-track diagnostic,
sync-fallback for screenshot capture — now lives in ViewportCore.
ViewportWindow::driveStreamingLoads is a one-line forwarder.
Streaming-related state moves to core with reference aliases on VW:
streaming_{loads,more_pending,candidates,evictions_{lru,pri},drained,
blocked_oom}_this_frame_, streaming_debug_, tracked_{object_id,
chunk_mid,chunk_idx,was_resident}_, and pending_screenshot_path_. The
pick handler and bench-warm gate (still VW) read/write through the
aliases unchanged.
Qt-isms in the body were replaced en route:
- QFileInfo(...).completeBaseName() → std::filesystem::path::stem()
- requestUpdate() → host_->requestFrame()
- QString::number(x, 'f', N) in numeric logs → raw double / int (we lose
fixed-precision in a couple of diag lines; acceptable tradeoff).
host_->requestFrame() means the streaming loop is now host-agnostic:
the WebViewportHost will provide its own requestAnimationFrame
equivalent when it lands.
This commit is contained in:
@@ -1730,3 +1730,421 @@ void ViewportCore::unloadChunk(ModelGpuData& m, std::size_t chunk_idx) {
|
|||||||
c.total_visible_vertices = 0;
|
c.total_visible_vertices = 0;
|
||||||
c.is_resident = false;
|
c.is_resident = false;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// Streaming driver (#84-o): driveStreamingLoads
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
#include <filesystem>
|
||||||
|
#include <set>
|
||||||
|
|
||||||
|
namespace {
|
||||||
|
// Extract the file's base name (no extension, no parent dirs) for log
|
||||||
|
// readability — replaces the previous QFileInfo(...).completeBaseName().
|
||||||
|
std::string pathStem(const std::string& path) {
|
||||||
|
if (path.empty()) return {};
|
||||||
|
return std::filesystem::path(path).stem().string();
|
||||||
|
}
|
||||||
|
} // namespace
|
||||||
|
|
||||||
|
void ViewportCore::driveStreamingLoads() {
|
||||||
|
// Bump LRU clock once per call. Resident-and-visible chunks get
|
||||||
|
// stamped with this value below; the evictor uses it to find the
|
||||||
|
// least-recently-visible non-visible resident chunk.
|
||||||
|
++streaming_frame_idx_;
|
||||||
|
|
||||||
|
// Refresh per-chunk frame state. (a) LRU stamp on frustum-visible
|
||||||
|
// residents (HiZ flicker can't un-stamp them; cull-with-HiZ would
|
||||||
|
// thrash the LRU). (b) EMA-smoothed visibility_history: how often
|
||||||
|
// the chunk has *actually* contributed pixels (post-HiZ) over the
|
||||||
|
// last ~30 frames.
|
||||||
|
constexpr float HISTORY_ALPHA = 1.0f / 30.0f;
|
||||||
|
for (auto& [mid, m] : models_gpu_) {
|
||||||
|
if (m.hidden) continue;
|
||||||
|
for (auto& c : m.chunks) {
|
||||||
|
if (c.is_resident && c.frustum_visible_count > 0) {
|
||||||
|
c.last_visible_frame_idx = streaming_frame_idx_;
|
||||||
|
}
|
||||||
|
const float current = (c.total_visible_draws > 0) ? 1.0f : 0.0f;
|
||||||
|
c.visibility_history =
|
||||||
|
c.visibility_history * (1.0f - HISTORY_ALPHA)
|
||||||
|
+ current * HISTORY_ALPHA;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Build the camera's view-projection for the diagnostic dump below.
|
||||||
|
Eigen::Matrix4f v_mat, p_mat;
|
||||||
|
buildViewProj(v_mat, p_mat);
|
||||||
|
const Eigen::Matrix4f vp_mat = p_mat * v_mat;
|
||||||
|
|
||||||
|
// chunk.current_priority was accumulated during cullModelCpuCompute
|
||||||
|
// (one add per frustum-passing instance). No standalone walk needed
|
||||||
|
// here; the candidate/resident priority lambdas just read it.
|
||||||
|
auto chunk_screen_area_px = [&](const ModelGpuData::Chunk& c) -> float {
|
||||||
|
return c.current_priority;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Resident chunks: contribution × visibility_history (floored), so
|
||||||
|
// chunks that don't actually render lose priority over time and
|
||||||
|
// become evictable. Candidates: pure contribution — best-case
|
||||||
|
// estimate. Newly-loaded chunks get a GRACE_FRAMES grace period at
|
||||||
|
// full max-history factor to stop equal-priority swap loops.
|
||||||
|
constexpr float HISTORY_FLOOR = 0.05f;
|
||||||
|
constexpr std::uint64_t GRACE_FRAMES = 30;
|
||||||
|
auto resident_priority = [&](const ModelGpuData::Chunk& c) -> float {
|
||||||
|
const std::uint64_t age = streaming_frame_idx_ - c.loaded_frame_idx;
|
||||||
|
const float vis = (age < GRACE_FRAMES)
|
||||||
|
? 1.0f
|
||||||
|
: std::max(c.visibility_history, HISTORY_FLOOR);
|
||||||
|
return chunk_screen_area_px(c) * vis;
|
||||||
|
};
|
||||||
|
auto candidate_priority = [&](const ModelGpuData::Chunk& c) -> float {
|
||||||
|
return chunk_screen_area_px(c);
|
||||||
|
};
|
||||||
|
|
||||||
|
// Per-frame load budget. 4 chunks/frame × 60fps ingests 240/sec —
|
||||||
|
// a 100-model scene fully resides in ~1s.
|
||||||
|
constexpr int MAX_STREAMING_LOADS_PER_FRAME = 4;
|
||||||
|
int loads = 0;
|
||||||
|
bool more_pending = false;
|
||||||
|
|
||||||
|
// Reset per-frame counters used by WGPU_STREAM_DEBUG output.
|
||||||
|
streaming_candidates_this_frame_ = 0;
|
||||||
|
streaming_evictions_lru_this_frame_ = 0;
|
||||||
|
streaming_evictions_pri_this_frame_ = 0;
|
||||||
|
streaming_drained_this_frame_ = 0;
|
||||||
|
streaming_blocked_oom_this_frame_ = 0;
|
||||||
|
|
||||||
|
auto pool_can_fit = [&](std::uint64_t bytes) -> bool {
|
||||||
|
if (pool_.largest_free_run_bytes() >= bytes) return true;
|
||||||
|
if (pool_.can_grow() && pool_.next_growth_size_bytes() >= bytes) return true;
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Phase-1 evictor: drop the LRU non-visible resident chunk. Skips
|
||||||
|
// chunks stamped on streaming_frame_idx_ to avoid yanking what cull
|
||||||
|
// just marked visible.
|
||||||
|
auto evict_one_lru = [&]() -> bool {
|
||||||
|
ModelGpuData* victim_m = nullptr;
|
||||||
|
std::size_t victim_ci = 0;
|
||||||
|
std::uint64_t victim_lru = std::numeric_limits<std::uint64_t>::max();
|
||||||
|
for (auto& [mid, m] : models_gpu_) {
|
||||||
|
for (std::size_t ci = 0; ci < m.chunks.size(); ++ci) {
|
||||||
|
auto& c = m.chunks[ci];
|
||||||
|
if (!c.is_resident) continue;
|
||||||
|
if (c.last_visible_frame_idx == streaming_frame_idx_) continue;
|
||||||
|
if (c.last_visible_frame_idx < victim_lru) {
|
||||||
|
victim_lru = c.last_visible_frame_idx;
|
||||||
|
victim_m = &m;
|
||||||
|
victim_ci = ci;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!victim_m) return false;
|
||||||
|
unloadChunk(*victim_m, victim_ci);
|
||||||
|
++streaming_evictions_lru_this_frame_;
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Phase-2 evictor: when every resident is visible-this-frame but a
|
||||||
|
// higher-priority candidate needs room, drop the lowest-priority
|
||||||
|
// resident provided the candidate's contribution is meaningfully
|
||||||
|
// bigger (2× area hysteresis stops oscillation).
|
||||||
|
constexpr float EVICT_PRIORITY_RATIO = 2.0f;
|
||||||
|
// WGPU_STREAM_EVICT_LOG=1 — log every priority-eviction with the
|
||||||
|
// (candidate, victim) pair and detect direct A→B→A 2-cycles.
|
||||||
|
static const bool evict_log =
|
||||||
|
std::getenv("WGPU_STREAM_EVICT_LOG") != nullptr;
|
||||||
|
auto evict_lowest_priority_than = [&](std::uint32_t cand_mid,
|
||||||
|
std::uint32_t cand_ci,
|
||||||
|
float cand_priority) -> bool {
|
||||||
|
const float threshold = cand_priority / EVICT_PRIORITY_RATIO;
|
||||||
|
ModelGpuData* victim_m = nullptr;
|
||||||
|
std::size_t victim_ci = 0;
|
||||||
|
float victim_priority = threshold;
|
||||||
|
for (auto& [mid, m] : models_gpu_) {
|
||||||
|
for (std::size_t ci = 0; ci < m.chunks.size(); ++ci) {
|
||||||
|
auto& c = m.chunks[ci];
|
||||||
|
if (!c.is_resident) continue;
|
||||||
|
const float p = resident_priority(c);
|
||||||
|
if (p < victim_priority) {
|
||||||
|
victim_priority = p;
|
||||||
|
victim_m = &m;
|
||||||
|
victim_ci = ci;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!victim_m) return false;
|
||||||
|
|
||||||
|
auto& victim = victim_m->chunks[victim_ci];
|
||||||
|
|
||||||
|
if (evict_log) {
|
||||||
|
const std::string cand_stem = pathStem(
|
||||||
|
models_gpu_.at(cand_mid).streaming_file_path);
|
||||||
|
const std::string vic_stem = pathStem(victim_m->streaming_file_path);
|
||||||
|
// 2-cycle detection: this victim was previously evicted by
|
||||||
|
// THIS exact candidate — the smoking gun for a swap loop.
|
||||||
|
const bool is_2_cycle =
|
||||||
|
victim.last_evicted_by_model_id == cand_mid
|
||||||
|
&& victim.last_evicted_by_chunk_idx == cand_ci
|
||||||
|
&& victim.load_count > 1;
|
||||||
|
Log::info()
|
||||||
|
<< (is_2_cycle ? "[evict 2-cycle] " : "[evict] ")
|
||||||
|
<< "kicked chunk " << victim_ci
|
||||||
|
<< " of " << vic_stem
|
||||||
|
<< " (eff=" << int(victim_priority)
|
||||||
|
<< ", load_count=" << victim.load_count
|
||||||
|
<< ") for chunk " << cand_ci
|
||||||
|
<< " of " << cand_stem
|
||||||
|
<< " (pri=" << int(cand_priority)
|
||||||
|
<< ", threshold=" << int(threshold) << ")";
|
||||||
|
}
|
||||||
|
|
||||||
|
victim.last_evicted_by_model_id = cand_mid;
|
||||||
|
victim.last_evicted_by_chunk_idx = cand_ci;
|
||||||
|
victim.last_evicted_by_priority = cand_priority;
|
||||||
|
victim.last_evicted_frame_idx = streaming_frame_idx_;
|
||||||
|
|
||||||
|
unloadChunk(*victim_m, victim_ci);
|
||||||
|
++streaming_evictions_pri_this_frame_;
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
|
||||||
|
constexpr std::uint64_t BLOCKED_COOLDOWN_FRAMES = 180;
|
||||||
|
|
||||||
|
// ---- Drain worker results -------------------------------------------
|
||||||
|
{
|
||||||
|
auto results = streaming_thread_.drainResults();
|
||||||
|
for (auto& res : results) {
|
||||||
|
auto it = models_gpu_.find(res.model_id);
|
||||||
|
if (it == models_gpu_.end()) continue; // model unloaded
|
||||||
|
auto& m = it->second;
|
||||||
|
if (res.chunk_idx >= m.chunks.size()) continue;
|
||||||
|
auto& c = m.chunks[res.chunk_idx];
|
||||||
|
c.is_loading = false;
|
||||||
|
if (!res.success) {
|
||||||
|
Log::warn() << "[wgpu stream] worker read failed for model "
|
||||||
|
<< res.model_id << " chunk " << res.chunk_idx;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if (!applyStreamedChunk(m, res.chunk_idx, res.vbytes, res.idx)) {
|
||||||
|
c.blocked_cooldown_until_frame_idx =
|
||||||
|
streaming_frame_idx_ + BLOCKED_COOLDOWN_FRAMES;
|
||||||
|
if (evict_log) {
|
||||||
|
Log::info()
|
||||||
|
<< "[blocked-apply] chunk " << res.chunk_idx
|
||||||
|
<< " of " << pathStem(m.streaming_file_path)
|
||||||
|
<< " — pool OOM at apply, fetched bytes discarded"
|
||||||
|
<< " — cooldown " << BLOCKED_COOLDOWN_FRAMES << "f";
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
++loads;
|
||||||
|
++streaming_drained_this_frame_;
|
||||||
|
++c.load_count;
|
||||||
|
c.last_visible_frame_idx = streaming_frame_idx_;
|
||||||
|
// Thrash watch — fire once per power-of-≈3 threshold.
|
||||||
|
const std::uint32_t lc = c.load_count;
|
||||||
|
if (lc == 3 || lc == 10 || lc == 30 || lc == 100
|
||||||
|
|| (lc > 100 && (lc % 100) == 0)) {
|
||||||
|
Log::info()
|
||||||
|
<< "[stream thrash] chunk " << res.chunk_idx
|
||||||
|
<< " of " << pathStem(m.streaming_file_path)
|
||||||
|
<< " loaded " << lc << "x -- pool saturated?";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- Enqueue new requests -------------------------------------------
|
||||||
|
struct Candidate {
|
||||||
|
ModelGpuData* m;
|
||||||
|
std::size_t ci;
|
||||||
|
std::uint32_t mid;
|
||||||
|
float priority;
|
||||||
|
};
|
||||||
|
std::vector<Candidate> candidates;
|
||||||
|
candidates.reserve(64);
|
||||||
|
for (auto& [mid, m] : models_gpu_) {
|
||||||
|
if (m.streaming_file_path.empty() || m.hidden) continue;
|
||||||
|
for (std::size_t ci = 0; ci < m.chunks.size(); ++ci) {
|
||||||
|
auto& c = m.chunks[ci];
|
||||||
|
if (c.is_resident) continue;
|
||||||
|
if (c.is_loading) continue;
|
||||||
|
if (c.frustum_visible_count == 0) continue;
|
||||||
|
if (c.blocked_cooldown_until_frame_idx > streaming_frame_idx_) continue;
|
||||||
|
candidates.push_back({&m, ci, mid, candidate_priority(c)});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
streaming_candidates_this_frame_ = int(candidates.size());
|
||||||
|
std::sort(candidates.begin(), candidates.end(),
|
||||||
|
[](const Candidate& a, const Candidate& b) {
|
||||||
|
return a.priority > b.priority;
|
||||||
|
});
|
||||||
|
|
||||||
|
int enqueued = 0;
|
||||||
|
for (const Candidate& cand : candidates) {
|
||||||
|
if (enqueued >= MAX_STREAMING_LOADS_PER_FRAME) {
|
||||||
|
more_pending = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
auto& c = cand.m->chunks[cand.ci];
|
||||||
|
|
||||||
|
const std::uint64_t need = c.vertex_byte_size
|
||||||
|
+ c.index_count * sizeof(std::uint32_t);
|
||||||
|
while (!pool_can_fit(c.vertex_byte_size)
|
||||||
|
|| (c.index_count > 0
|
||||||
|
&& !pool_can_fit(c.index_count * sizeof(std::uint32_t)))
|
||||||
|
|| pool_.total_free_bytes() < need) {
|
||||||
|
if (evict_one_lru()) continue;
|
||||||
|
if (evict_lowest_priority_than(cand.mid, std::uint32_t(cand.ci),
|
||||||
|
cand.priority)) continue;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (!pool_can_fit(c.vertex_byte_size)
|
||||||
|
|| (c.index_count > 0
|
||||||
|
&& !pool_can_fit(c.index_count * sizeof(std::uint32_t)))) {
|
||||||
|
++streaming_blocked_oom_this_frame_;
|
||||||
|
c.blocked_cooldown_until_frame_idx =
|
||||||
|
streaming_frame_idx_ + BLOCKED_COOLDOWN_FRAMES;
|
||||||
|
if (evict_log) {
|
||||||
|
const std::uint64_t v_bytes = c.vertex_byte_size;
|
||||||
|
const std::uint64_t i_bytes = c.index_count * sizeof(std::uint32_t);
|
||||||
|
const double mb = 1.0 / (1024.0 * 1024.0);
|
||||||
|
Log::info()
|
||||||
|
<< "[blocked] chunk " << cand.ci
|
||||||
|
<< " of " << pathStem(cand.m->streaming_file_path)
|
||||||
|
<< " (pri=" << int(cand.priority)
|
||||||
|
<< ") -- needs v=" << double(v_bytes) * mb
|
||||||
|
<< " MB + i=" << double(i_bytes) * mb
|
||||||
|
<< " MB; pool largest_free="
|
||||||
|
<< double(pool_.largest_free_run_bytes()) * mb
|
||||||
|
<< " MB total_free="
|
||||||
|
<< double(pool_.total_free_bytes()) * mb
|
||||||
|
<< " MB can_grow=" << (pool_.can_grow() ? "Y" : "N")
|
||||||
|
<< " -- cooldown " << BLOCKED_COOLDOWN_FRAMES << "f";
|
||||||
|
}
|
||||||
|
more_pending = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Sync fallback when a screenshot is pending: the deferred-
|
||||||
|
// capture wait would let the window manager re-layout the
|
||||||
|
// window while we wait, capturing at the wrong size. With sync
|
||||||
|
// loads the chunk appears in the same frame we enqueue.
|
||||||
|
if (!pending_screenshot_path_.empty()) {
|
||||||
|
if (loadChunkBytesAndUploadGpu(*cand.m, cand.ci)) {
|
||||||
|
++enqueued;
|
||||||
|
c.last_visible_frame_idx = streaming_frame_idx_;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (streaming_thread_.enqueue(makeChunkRequest(*cand.m, cand.ci, cand.mid))) {
|
||||||
|
c.is_loading = true;
|
||||||
|
++enqueued;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
loads += enqueued;
|
||||||
|
if (loads > 0 || streaming_thread_.inFlightApprox() > 0) host_->requestFrame();
|
||||||
|
|
||||||
|
streaming_loads_this_frame_ = loads;
|
||||||
|
streaming_more_pending_ = more_pending;
|
||||||
|
|
||||||
|
// Click-and-track diagnostic. When the user picked an object, we
|
||||||
|
// noted which chunk holds it. If that chunk has just transitioned
|
||||||
|
// resident→evicted, dump the priority + pool state at the moment
|
||||||
|
// of loss.
|
||||||
|
if (tracked_chunk_idx_ != SIZE_MAX) {
|
||||||
|
auto it = models_gpu_.find(tracked_chunk_mid_);
|
||||||
|
if (it != models_gpu_.end()
|
||||||
|
&& tracked_chunk_idx_ < it->second.chunks.size()) {
|
||||||
|
const auto& m = it->second;
|
||||||
|
const auto& c = m.chunks[tracked_chunk_idx_];
|
||||||
|
if (tracked_was_resident_ && !c.is_resident) {
|
||||||
|
const double mb = 1.0 / (1024.0 * 1024.0);
|
||||||
|
const float my_area = chunkScreenAreaPx(c, vp_mat);
|
||||||
|
const std::uint64_t my_bytes = c.vertex_byte_size
|
||||||
|
+ c.index_count * sizeof(std::uint32_t);
|
||||||
|
Log::info()
|
||||||
|
<< "[track] chunk " << tracked_chunk_idx_
|
||||||
|
<< " (object " << tracked_object_id_
|
||||||
|
<< ", model " << tracked_chunk_mid_
|
||||||
|
<< ") EVICTED this frame";
|
||||||
|
Log::info()
|
||||||
|
<< " area=" << int(my_area) << "px2"
|
||||||
|
<< " frustum_vis=" << c.frustum_visible_count
|
||||||
|
<< " hist=" << c.visibility_history
|
||||||
|
<< " load_count=" << c.load_count
|
||||||
|
<< " size=" << double(my_bytes) * mb << "MB";
|
||||||
|
Log::info()
|
||||||
|
<< " chunk aabb "
|
||||||
|
<< (c.aabb_max[0] - c.aabb_min[0]) << "x"
|
||||||
|
<< (c.aabb_max[1] - c.aabb_min[1]) << "x"
|
||||||
|
<< (c.aabb_max[2] - c.aabb_min[2]) << "m"
|
||||||
|
<< " centre=("
|
||||||
|
<< 0.5f * (c.aabb_min[0] + c.aabb_max[0]) << ","
|
||||||
|
<< 0.5f * (c.aabb_min[1] + c.aabb_max[1]) << ","
|
||||||
|
<< 0.5f * (c.aabb_min[2] + c.aabb_max[2]) << ")";
|
||||||
|
Log::info()
|
||||||
|
<< " pool used="
|
||||||
|
<< int(double(pool_.total_used_bytes()) * mb)
|
||||||
|
<< "/"
|
||||||
|
<< int(double(pool_.total_capacity_bytes()) * mb)
|
||||||
|
<< "MB largest_free="
|
||||||
|
<< double(pool_.largest_free_run_bytes()) * mb << "MB";
|
||||||
|
Log::info()
|
||||||
|
<< " this-frame: cands=" << streaming_candidates_this_frame_
|
||||||
|
<< " enq=" << enqueued
|
||||||
|
<< " ev_lru=" << streaming_evictions_lru_this_frame_
|
||||||
|
<< " ev_pri=" << streaming_evictions_pri_this_frame_
|
||||||
|
<< " blocked=" << streaming_blocked_oom_this_frame_;
|
||||||
|
|
||||||
|
struct Stat { std::uint32_t mid; std::size_t ci; float area; };
|
||||||
|
std::vector<Stat> all;
|
||||||
|
all.reserve(64);
|
||||||
|
for (const auto& [mid2, m2] : models_gpu_) {
|
||||||
|
for (std::size_t ci2 = 0; ci2 < m2.chunks.size(); ++ci2) {
|
||||||
|
const auto& cc = m2.chunks[ci2];
|
||||||
|
if (cc.is_resident) continue;
|
||||||
|
if (cc.frustum_visible_count == 0) continue;
|
||||||
|
all.push_back({mid2, ci2, chunkScreenAreaPx(cc, vp_mat)});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
std::sort(all.begin(), all.end(),
|
||||||
|
[](const Stat& a, const Stat& b){ return a.area > b.area; });
|
||||||
|
const std::size_t n = std::min<std::size_t>(5, all.size());
|
||||||
|
for (std::size_t i = 0; i < n; ++i) {
|
||||||
|
Log::info()
|
||||||
|
<< " top cand #" << i << ": model " << all[i].mid
|
||||||
|
<< " chunk " << all[i].ci
|
||||||
|
<< " area=" << int(all[i].area) << "px2";
|
||||||
|
}
|
||||||
|
}
|
||||||
|
tracked_was_resident_ = c.is_resident;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (streaming_debug_) {
|
||||||
|
std::size_t resident = 0;
|
||||||
|
std::uint32_t max_load_count = 0;
|
||||||
|
std::size_t cycled = 0;
|
||||||
|
for (const auto& [mid, m] : models_gpu_) {
|
||||||
|
for (const auto& c : m.chunks) {
|
||||||
|
if (c.is_resident) ++resident;
|
||||||
|
if (c.load_count > max_load_count) max_load_count = c.load_count;
|
||||||
|
if (c.load_count > 1) ++cycled;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Log::info()
|
||||||
|
<< "[stream-debug] f" << streaming_frame_idx_
|
||||||
|
<< " cands=" << streaming_candidates_this_frame_
|
||||||
|
<< " enq=" << enqueued
|
||||||
|
<< " drained=" << streaming_drained_this_frame_
|
||||||
|
<< " ev_lru=" << streaming_evictions_lru_this_frame_
|
||||||
|
<< " ev_pri=" << streaming_evictions_pri_this_frame_
|
||||||
|
<< " blocked=" << streaming_blocked_oom_this_frame_
|
||||||
|
<< " resident=" << resident
|
||||||
|
<< " cycled=" << cycled
|
||||||
|
<< " max_load=" << max_load_count;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -288,11 +288,18 @@ public:
|
|||||||
// Build the worker request for a chunk. Walks the chunk's mesh_ids
|
// Build the worker request for a chunk. Walks the chunk's mesh_ids
|
||||||
// and derives scatter-gather byte/index ranges from each mesh's
|
// and derives scatter-gather byte/index ranges from each mesh's
|
||||||
// sidecar offsets. Pure function of model + chunk metadata; safe to
|
// sidecar offsets. Pure function of model + chunk metadata; safe to
|
||||||
// call from the main thread. Static — also used by the still-in-VW
|
// call from the main thread.
|
||||||
// driveStreamingLoads to enqueue against streaming_thread_.
|
|
||||||
static StreamingThread::Request makeChunkRequest(
|
static StreamingThread::Request makeChunkRequest(
|
||||||
const ModelGpuData& m, std::size_t chunk_idx, std::uint32_t model_id);
|
const ModelGpuData& m, std::size_t chunk_idx, std::uint32_t model_id);
|
||||||
|
|
||||||
|
// Per-frame streaming driver. Called from render() after cull. Walks
|
||||||
|
// every model's chunks once for residency bookkeeping, drains the
|
||||||
|
// worker's completed results into the pool, then enqueues new
|
||||||
|
// requests for visible non-resident chunks (LRU + priority eviction
|
||||||
|
// when the pool can't fit). Triggers host_->requestFrame() while
|
||||||
|
// residency is still settling so the render loop keeps ticking.
|
||||||
|
void driveStreamingLoads();
|
||||||
|
|
||||||
private:
|
private:
|
||||||
bool probeAndCreatePool();
|
bool probeAndCreatePool();
|
||||||
|
|
||||||
@@ -418,6 +425,37 @@ private:
|
|||||||
// (resets only at shutdown).
|
// (resets only at shutdown).
|
||||||
std::uint64_t streaming_frame_idx_ = 0;
|
std::uint64_t streaming_frame_idx_ = 0;
|
||||||
|
|
||||||
|
// Per-frame streaming activity. Written by driveStreamingLoads,
|
||||||
|
// consumed by the benchmark warm-gate (`loads_this_frame == 0 AND
|
||||||
|
// worker idle == settled`). `more_pending` is a soft hint — true
|
||||||
|
// means residency hasn't converged and the loop should keep ticking.
|
||||||
|
int streaming_loads_this_frame_ = 0;
|
||||||
|
bool streaming_more_pending_ = false;
|
||||||
|
|
||||||
|
// Per-frame breakdown counters consumed by the WGPU_STREAM_DEBUG
|
||||||
|
// log. All reset at the top of driveStreamingLoads.
|
||||||
|
int streaming_candidates_this_frame_ = 0;
|
||||||
|
int streaming_evictions_lru_this_frame_ = 0;
|
||||||
|
int streaming_evictions_pri_this_frame_ = 0;
|
||||||
|
int streaming_drained_this_frame_ = 0;
|
||||||
|
int streaming_blocked_oom_this_frame_ = 0;
|
||||||
|
bool streaming_debug_ = false; // WGPU_STREAM_DEBUG=1
|
||||||
|
|
||||||
|
// Click-and-track diagnostic. Set by the pick handler when an
|
||||||
|
// object is selected; driveStreamingLoads dumps priority + pool
|
||||||
|
// state every time that chunk transitions resident→evicted so
|
||||||
|
// we can pinpoint WHY a piece of geometry disappeared.
|
||||||
|
std::uint32_t tracked_object_id_ = 0;
|
||||||
|
std::uint32_t tracked_chunk_mid_ = 0;
|
||||||
|
std::size_t tracked_chunk_idx_ = SIZE_MAX;
|
||||||
|
bool tracked_was_resident_ = false;
|
||||||
|
|
||||||
|
// When non-empty, a screenshot capture is pending and the streaming
|
||||||
|
// loader switches to the synchronous-fetch fallback so the
|
||||||
|
// first-frame capture isn't an empty buffer. Cleared after capture
|
||||||
|
// completes.
|
||||||
|
std::string pending_screenshot_path_;
|
||||||
|
|
||||||
// Tool-refresh callback: fired by applyStreamedChunk when a newly-
|
// Tool-refresh callback: fired by applyStreamedChunk when a newly-
|
||||||
// arrived chunk filled in a mesh-local volume. ViewportWindow wires
|
// arrived chunk filled in a mesh-local volume. ViewportWindow wires
|
||||||
// this to its Volume-tool HUD refresh in the ctor. Null by default
|
// this to its Volume-tool HUD refresh in the ctor. Null by default
|
||||||
|
|||||||
@@ -267,7 +267,20 @@ ViewportWindow::ViewportWindow(QWindow* parent)
|
|||||||
section_planes_ (core_.section_planes_),
|
section_planes_ (core_.section_planes_),
|
||||||
xray_alpha_cap_ (core_.xray_alpha_cap_),
|
xray_alpha_cap_ (core_.xray_alpha_cap_),
|
||||||
selection_ (core_.selection_),
|
selection_ (core_.selection_),
|
||||||
visibility_ (core_.visibility_) {
|
visibility_ (core_.visibility_),
|
||||||
|
tracked_object_id_ (core_.tracked_object_id_),
|
||||||
|
tracked_chunk_mid_ (core_.tracked_chunk_mid_),
|
||||||
|
tracked_chunk_idx_ (core_.tracked_chunk_idx_),
|
||||||
|
tracked_was_resident_ (core_.tracked_was_resident_),
|
||||||
|
streaming_loads_this_frame_ (core_.streaming_loads_this_frame_),
|
||||||
|
streaming_more_pending_ (core_.streaming_more_pending_),
|
||||||
|
streaming_candidates_this_frame_ (core_.streaming_candidates_this_frame_),
|
||||||
|
streaming_evictions_lru_this_frame_(core_.streaming_evictions_lru_this_frame_),
|
||||||
|
streaming_evictions_pri_this_frame_(core_.streaming_evictions_pri_this_frame_),
|
||||||
|
streaming_drained_this_frame_ (core_.streaming_drained_this_frame_),
|
||||||
|
streaming_blocked_oom_this_frame_(core_.streaming_blocked_oom_this_frame_),
|
||||||
|
streaming_debug_ (core_.streaming_debug_),
|
||||||
|
pending_screenshot_path_(core_.pending_screenshot_path_) {
|
||||||
// wgpu doesn't need a GL context; we just need a real native window
|
// wgpu doesn't need a GL context; we just need a real native window
|
||||||
// whose backing layer matches the GPU API wgpu will drive.
|
// whose backing layer matches the GPU API wgpu will drive.
|
||||||
//
|
//
|
||||||
@@ -4902,502 +4915,7 @@ void ViewportWindow::buildModelBindGroup(ModelGpuData& m) {
|
|||||||
|
|
||||||
// unloadChunk moved to ViewportCore (#84-n).
|
// unloadChunk moved to ViewportCore (#84-n).
|
||||||
|
|
||||||
void ViewportWindow::driveStreamingLoads() {
|
void ViewportWindow::driveStreamingLoads() { core_.driveStreamingLoads(); }
|
||||||
// Bump LRU clock once per call. Resident-and-visible chunks get
|
|
||||||
// stamped with this value below; the evictor uses it to find the
|
|
||||||
// least-recently-visible non-visible resident chunk.
|
|
||||||
++streaming_frame_idx_;
|
|
||||||
|
|
||||||
// Refresh per-chunk frame state. (a) LRU stamp on frustum-visible
|
|
||||||
// residents (HiZ flicker can't un-stamp them; cull-with-HiZ would
|
|
||||||
// thrash the LRU). (b) EMA-smoothed visibility_history: how often
|
|
||||||
// the chunk has *actually* contributed pixels (post-HiZ) over the
|
|
||||||
// last ~30 frames. The two metrics serve different jobs — LRU
|
|
||||||
// distinguishes "out of view" from "in view", history distinguishes
|
|
||||||
// "in view AND not occluded" from "in view BUT mostly occluded".
|
|
||||||
constexpr float HISTORY_ALPHA = 1.0f / 30.0f;
|
|
||||||
for (auto& [mid, m] : models_gpu_) {
|
|
||||||
if (m.hidden) continue;
|
|
||||||
for (auto& c : m.chunks) {
|
|
||||||
if (c.is_resident && c.frustum_visible_count > 0) {
|
|
||||||
c.last_visible_frame_idx = streaming_frame_idx_;
|
|
||||||
}
|
|
||||||
const float current = (c.total_visible_draws > 0) ? 1.0f : 0.0f;
|
|
||||||
c.visibility_history =
|
|
||||||
c.visibility_history * (1.0f - HISTORY_ALPHA)
|
|
||||||
+ current * HISTORY_ALPHA;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Build the camera's view-projection (still needed for the AABB-based
|
|
||||||
// diagnostic dump in the tracking output below). Cull/render use the
|
|
||||||
// same helper.
|
|
||||||
Eigen::Matrix4f v_mat, p_mat;
|
|
||||||
core_.buildViewProj(v_mat, p_mat);
|
|
||||||
const Eigen::Matrix4f vp_mat = p_mat * v_mat;
|
|
||||||
|
|
||||||
// chunk.current_priority was accumulated during cullModelCpuCompute
|
|
||||||
// (one add per frustum-passing instance). No standalone walk needed
|
|
||||||
// here; the candidate/resident priority lambdas just read it.
|
|
||||||
auto chunk_screen_area_px = [&](const ModelGpuData::Chunk& c) -> float {
|
|
||||||
return c.current_priority;
|
|
||||||
};
|
|
||||||
|
|
||||||
// Resident chunks: contribution × visibility_history (floored), so
|
|
||||||
// chunks that don't actually render lose priority over time and
|
|
||||||
// become evictable. Candidates: pure contribution — best-case
|
|
||||||
// estimate. Asymmetry lets new high-contribution chunks displace
|
|
||||||
// long-resident-but-occluded ones.
|
|
||||||
//
|
|
||||||
// CRITICAL: newly-loaded chunks get a "grace period" of GRACE_FRAMES
|
|
||||||
// at the full max-history factor. Without it, a freshly-loaded
|
|
||||||
// chunk's effective priority crashes to contribution × 0.05 next
|
|
||||||
// frame (history hasn't had time to develop), and the chunk it
|
|
||||||
// displaced — back as a candidate at full priority — re-displaces
|
|
||||||
// it. Infinite reverse-swap between equal-priority chunks. The
|
|
||||||
// cycle starves the per-frame load budget (MAX_STREAMING_LOADS = 4)
|
|
||||||
// so candidates ranked below the cyclers (e.g. brace chunks at
|
|
||||||
// priority position 20) never get attempted. Grace period gives
|
|
||||||
// visibility_history time to settle and breaks the cycle.
|
|
||||||
constexpr float HISTORY_FLOOR = 0.05f;
|
|
||||||
constexpr uint64_t GRACE_FRAMES = 30;
|
|
||||||
auto resident_priority = [&](const ModelGpuData::Chunk& c) -> float {
|
|
||||||
const uint64_t age = streaming_frame_idx_ - c.loaded_frame_idx;
|
|
||||||
const float vis = (age < GRACE_FRAMES)
|
|
||||||
? 1.0f
|
|
||||||
: std::max(c.visibility_history, HISTORY_FLOOR);
|
|
||||||
return chunk_screen_area_px(c) * vis;
|
|
||||||
};
|
|
||||||
auto candidate_priority = [&](const ModelGpuData::Chunk& c) -> float {
|
|
||||||
return chunk_screen_area_px(c);
|
|
||||||
};
|
|
||||||
|
|
||||||
// Per-frame load budget. Caps first-frame stall on a fresh load — at
|
|
||||||
// 4 chunks/frame × 60fps we ingest 240 chunks/sec, fast enough that
|
|
||||||
// a 100-model scene fully resides in ~1s. The hard ceiling on total
|
|
||||||
// residency is the pool capacity (probed at startup); when the pool
|
|
||||||
// can't fit a candidate, the evictors below free closer-fitting
|
|
||||||
// ranges until it does.
|
|
||||||
constexpr int MAX_STREAMING_LOADS_PER_FRAME = 4;
|
|
||||||
int loads = 0;
|
|
||||||
bool more_pending = false;
|
|
||||||
|
|
||||||
// Reset per-frame counters used by WGPU_STREAM_DEBUG output.
|
|
||||||
streaming_candidates_this_frame_ = 0;
|
|
||||||
streaming_evictions_lru_this_frame_ = 0;
|
|
||||||
streaming_evictions_pri_this_frame_ = 0;
|
|
||||||
streaming_drained_this_frame_ = 0;
|
|
||||||
streaming_blocked_oom_this_frame_ = 0;
|
|
||||||
|
|
||||||
// The pool needs `need` contiguous bytes free for both the vertex and
|
|
||||||
// index allocations a load requires. Fragmentation matters: a chunk
|
|
||||||
// may fit total-free-bytes but not largest_free_run_bytes(). With
|
|
||||||
// multi-sub-buffer pools, an alloc can also succeed by growing the
|
|
||||||
// pool (adding a new sub-buffer at per_sub_buffer_capacity_bytes()),
|
|
||||||
// so a chunk also "fits" if it's smaller than one fresh sub-buffer.
|
|
||||||
// The actual alloc handles the growth attempt; this predicate only
|
|
||||||
// avoids wasted evict-then-fail loops.
|
|
||||||
auto pool_can_fit = [&](uint64_t bytes) -> bool {
|
|
||||||
if (pool_.largest_free_run_bytes() >= bytes) return true;
|
|
||||||
// Growth might still rescue us. Use next_growth_size_bytes()
|
|
||||||
// rather than per_sub_buffer_capacity_bytes() — after a refusal
|
|
||||||
// at e.g. 2 GB, halve-on-failure pushes the next achievable
|
|
||||||
// sub-buffer down to 1 GB; saying "fits if ≤2 GB" would lie.
|
|
||||||
if (pool_.can_grow() && pool_.next_growth_size_bytes() >= bytes) return true;
|
|
||||||
return false;
|
|
||||||
};
|
|
||||||
|
|
||||||
// Phase-1 evictor: drop the LRU non-visible resident chunk. Skips
|
|
||||||
// chunks stamped on streaming_frame_idx_ to avoid yanking what cull
|
|
||||||
// just marked visible. Returns true iff a chunk was evicted.
|
|
||||||
auto evict_one_lru = [&]() -> bool {
|
|
||||||
ModelGpuData* victim_m = nullptr;
|
|
||||||
size_t victim_ci = 0;
|
|
||||||
uint64_t victim_lru = std::numeric_limits<uint64_t>::max();
|
|
||||||
for (auto& [mid, m] : models_gpu_) {
|
|
||||||
for (size_t ci = 0; ci < m.chunks.size(); ++ci) {
|
|
||||||
auto& c = m.chunks[ci];
|
|
||||||
if (!c.is_resident) continue;
|
|
||||||
if (c.last_visible_frame_idx == streaming_frame_idx_) continue;
|
|
||||||
if (c.last_visible_frame_idx < victim_lru) {
|
|
||||||
victim_lru = c.last_visible_frame_idx;
|
|
||||||
victim_m = &m;
|
|
||||||
victim_ci = ci;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!victim_m) return false;
|
|
||||||
core_.unloadChunk(*victim_m, victim_ci);
|
|
||||||
++streaming_evictions_lru_this_frame_;
|
|
||||||
return true;
|
|
||||||
};
|
|
||||||
|
|
||||||
// Phase-2 evictor: when every resident chunk is visible-this-frame
|
|
||||||
// but we still need room for a higher-priority candidate, drop the
|
|
||||||
// resident with the lowest priority (contribution × history) —
|
|
||||||
// provided the candidate's contribution is meaningfully bigger.
|
|
||||||
// 2.0× hysteresis: candidate must have 2× more pixel area than the
|
|
||||||
// victim's effective priority. In linear-radius terms that's a
|
|
||||||
// ~41% gap, which is what stops 5 m vs 7 m chunks from oscillating.
|
|
||||||
// Area metric is much more discriminating than radius, so we can
|
|
||||||
// afford a bigger gap and still leave room for genuine swaps.
|
|
||||||
constexpr float EVICT_PRIORITY_RATIO = 2.0f;
|
|
||||||
// WGPU_STREAM_EVICT_LOG=1 — log every priority-eviction with the
|
|
||||||
// (candidate, victim) pair and detect direct A→B→A 2-cycles. Noisy
|
|
||||||
// when working-set > pool; gated separately from WGPU_STREAM_DEEP_DEBUG
|
|
||||||
// so you can run one without the other.
|
|
||||||
static const bool evict_log =
|
|
||||||
std::getenv("WGPU_STREAM_EVICT_LOG") != nullptr;
|
|
||||||
auto evict_lowest_priority_than = [&](uint32_t cand_mid,
|
|
||||||
uint32_t cand_ci,
|
|
||||||
float cand_priority) -> bool {
|
|
||||||
const float threshold = cand_priority / EVICT_PRIORITY_RATIO;
|
|
||||||
ModelGpuData* victim_m = nullptr;
|
|
||||||
size_t victim_ci = 0;
|
|
||||||
float victim_priority = threshold;
|
|
||||||
for (auto& [mid, m] : models_gpu_) {
|
|
||||||
for (size_t ci = 0; ci < m.chunks.size(); ++ci) {
|
|
||||||
auto& c = m.chunks[ci];
|
|
||||||
if (!c.is_resident) continue;
|
|
||||||
const float p = resident_priority(c);
|
|
||||||
if (p < victim_priority) {
|
|
||||||
victim_priority = p;
|
|
||||||
victim_m = &m;
|
|
||||||
victim_ci = ci;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (!victim_m) return false;
|
|
||||||
|
|
||||||
auto& victim = victim_m->chunks[victim_ci];
|
|
||||||
|
|
||||||
if (evict_log) {
|
|
||||||
QFileInfo cand_fi(QString::fromStdString(
|
|
||||||
models_gpu_.at(cand_mid).streaming_file_path));
|
|
||||||
QFileInfo vic_fi(QString::fromStdString(victim_m->streaming_file_path));
|
|
||||||
// 2-cycle detection: this victim was previously evicted by
|
|
||||||
// THIS exact candidate. That's the smoking gun for a swap-
|
|
||||||
// loop — A pushes B out, B comes back as candidate, B
|
|
||||||
// pushes A out, A comes back as candidate, …
|
|
||||||
const bool is_2_cycle =
|
|
||||||
victim.last_evicted_by_model_id == cand_mid
|
|
||||||
&& victim.last_evicted_by_chunk_idx == cand_ci
|
|
||||||
&& victim.load_count > 1;
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< (is_2_cycle ? "[evict 2-cycle] " : "[evict] ")
|
|
||||||
<< "kicked chunk " << victim_ci
|
|
||||||
<< " of " << vic_fi.completeBaseName()
|
|
||||||
<< " (eff=" << QString::number(victim_priority, 'f', 0)
|
|
||||||
<< ", load_count=" << victim.load_count
|
|
||||||
<< ") for chunk " << cand_ci
|
|
||||||
<< " of " << cand_fi.completeBaseName()
|
|
||||||
<< " (pri=" << QString::number(cand_priority, 'f', 0)
|
|
||||||
<< ", threshold=" << QString::number(threshold, 'f', 0) << ")";
|
|
||||||
}
|
|
||||||
|
|
||||||
victim.last_evicted_by_model_id = cand_mid;
|
|
||||||
victim.last_evicted_by_chunk_idx = cand_ci;
|
|
||||||
victim.last_evicted_by_priority = cand_priority;
|
|
||||||
victim.last_evicted_frame_idx = streaming_frame_idx_;
|
|
||||||
|
|
||||||
core_.unloadChunk(*victim_m, victim_ci);
|
|
||||||
++streaming_evictions_pri_this_frame_;
|
|
||||||
return true;
|
|
||||||
};
|
|
||||||
|
|
||||||
// Cooldown duration for chunks that hit OOM (at can-fit time or at
|
|
||||||
// apply time). 180 frames ≈ 3s at 60 fps. Web-friendly: caps re-
|
|
||||||
// fetches of a chronically-unfittable chunk's byte range at one
|
|
||||||
// every ~3 seconds, instead of every frame. If the pool layout
|
|
||||||
// changes within the cooldown (other chunks evicted, fragmentation
|
|
||||||
// resolved) the chunk re-attempts once the cooldown expires.
|
|
||||||
constexpr uint64_t BLOCKED_COOLDOWN_FRAMES = 180;
|
|
||||||
|
|
||||||
// ---- Drain worker results -------------------------------------------
|
|
||||||
// Apply any chunk reads that the streaming thread finished since
|
|
||||||
// last frame. Each apply does pool.alloc + queueWriteBuffer + bind
|
|
||||||
// group build — strictly main-thread work because wgpu queue ops
|
|
||||||
// are not thread-safe. Counts toward loads_this_frame for the
|
|
||||||
// bench warm gate's "settled" check.
|
|
||||||
{
|
|
||||||
auto results = streaming_thread_.drainResults();
|
|
||||||
for (auto& res : results) {
|
|
||||||
auto it = models_gpu_.find(res.model_id);
|
|
||||||
if (it == models_gpu_.end()) continue; // model unloaded
|
|
||||||
auto& m = it->second;
|
|
||||||
if (res.chunk_idx >= m.chunks.size()) continue;
|
|
||||||
auto& c = m.chunks[res.chunk_idx];
|
|
||||||
// The chunk may have been "unloaded" mid-flight (it wasn't
|
|
||||||
// resident yet — eviction only acts on residents — but the
|
|
||||||
// loader could have re-enqueued or the model could have
|
|
||||||
// been hidden). Clear the loading flag regardless.
|
|
||||||
c.is_loading = false;
|
|
||||||
if (!res.success) {
|
|
||||||
Log::warn().noquote().nospace()
|
|
||||||
<< "[wgpu stream] worker read failed for model "
|
|
||||||
<< res.model_id << " chunk " << res.chunk_idx;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (!core_.applyStreamedChunk(m, res.chunk_idx, res.vbytes, res.idx)) {
|
|
||||||
// Pool OOM at apply time — pool fragmented further
|
|
||||||
// between enqueue and worker-result. Set the same
|
|
||||||
// cooldown as the enqueue-time block: we just paid for
|
|
||||||
// a disk read / web fetch and discarded it; without
|
|
||||||
// the cooldown the same byte range would be re-fetched
|
|
||||||
// every frame until pool layout changes.
|
|
||||||
c.blocked_cooldown_until_frame_idx =
|
|
||||||
streaming_frame_idx_ + BLOCKED_COOLDOWN_FRAMES;
|
|
||||||
if (evict_log) {
|
|
||||||
QFileInfo fi(QString::fromStdString(m.streaming_file_path));
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< "[blocked-apply] chunk " << res.chunk_idx
|
|
||||||
<< " of " << fi.completeBaseName()
|
|
||||||
<< " — pool OOM at apply, fetched bytes discarded"
|
|
||||||
<< " — cooldown " << BLOCKED_COOLDOWN_FRAMES << "f";
|
|
||||||
}
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
++loads;
|
|
||||||
++streaming_drained_this_frame_;
|
|
||||||
++c.load_count;
|
|
||||||
c.last_visible_frame_idx = streaming_frame_idx_;
|
|
||||||
// Thrash watch — fire once per power-of-≈3 threshold (3, 10,
|
|
||||||
// 30, 100). A chunk that crosses 10 has been re-loaded 10×
|
|
||||||
// this session; that points at either pool saturation or a
|
|
||||||
// hysteresis boundary keeping it on the evict/load edge.
|
|
||||||
// One line per crossing per chunk — bounded in noise.
|
|
||||||
const uint32_t lc = c.load_count;
|
|
||||||
if (lc == 3 || lc == 10 || lc == 30 || lc == 100
|
|
||||||
|| (lc > 100 && (lc % 100) == 0)) {
|
|
||||||
QFileInfo fi(QString::fromStdString(m.streaming_file_path));
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< "[stream thrash] chunk " << res.chunk_idx
|
|
||||||
<< " of " << fi.completeBaseName()
|
|
||||||
<< " loaded " << lc << "× — pool saturated?";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---- Enqueue new requests -------------------------------------------
|
|
||||||
// Gather non-resident, !is_loading, frustum-visible chunks; sort by
|
|
||||||
// candidate priority (contribution_px) DESCENDING so the biggest
|
|
||||||
// screen-coverage chunks load first. Each enqueue makes room in
|
|
||||||
// the pool by evicting low-priority residents (contribution ×
|
|
||||||
// visibility_history); apply's alloc is best-effort.
|
|
||||||
struct Candidate { ModelGpuData* m; size_t ci; uint32_t mid; float priority; };
|
|
||||||
std::vector<Candidate> candidates;
|
|
||||||
candidates.reserve(64);
|
|
||||||
for (auto& [mid, m] : models_gpu_) {
|
|
||||||
if (m.streaming_file_path.empty() || m.hidden) continue;
|
|
||||||
for (size_t ci = 0; ci < m.chunks.size(); ++ci) {
|
|
||||||
auto& c = m.chunks[ci];
|
|
||||||
if (c.is_resident) continue;
|
|
||||||
if (c.is_loading) continue;
|
|
||||||
if (c.frustum_visible_count == 0) continue;
|
|
||||||
// Cooldown after a previous OOM. Skip — don't waste a fetch
|
|
||||||
// or an enqueue slot on a chunk we just learned doesn't fit.
|
|
||||||
if (c.blocked_cooldown_until_frame_idx > streaming_frame_idx_) continue;
|
|
||||||
candidates.push_back({&m, ci, mid, candidate_priority(c)});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
streaming_candidates_this_frame_ = int(candidates.size());
|
|
||||||
std::sort(candidates.begin(), candidates.end(),
|
|
||||||
[](const Candidate& a, const Candidate& b) {
|
|
||||||
return a.priority > b.priority; // biggest first
|
|
||||||
});
|
|
||||||
|
|
||||||
int enqueued = 0;
|
|
||||||
for (const Candidate& cand : candidates) {
|
|
||||||
if (enqueued >= MAX_STREAMING_LOADS_PER_FRAME) {
|
|
||||||
more_pending = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
auto& c = cand.m->chunks[cand.ci];
|
|
||||||
|
|
||||||
const uint64_t need = c.vertex_byte_size
|
|
||||||
+ c.index_count * sizeof(uint32_t);
|
|
||||||
while (!pool_can_fit(c.vertex_byte_size)
|
|
||||||
|| (c.index_count > 0
|
|
||||||
&& !pool_can_fit(c.index_count * sizeof(uint32_t)))
|
|
||||||
|| pool_.total_free_bytes() < need) {
|
|
||||||
if (evict_one_lru()) continue;
|
|
||||||
if (evict_lowest_priority_than(cand.mid, uint32_t(cand.ci),
|
|
||||||
cand.priority)) continue;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (!pool_can_fit(c.vertex_byte_size)
|
|
||||||
|| (c.index_count > 0
|
|
||||||
&& !pool_can_fit(c.index_count * sizeof(uint32_t)))) {
|
|
||||||
// Block + cooldown. The break-here-on-block logic was
|
|
||||||
// wrong: it assumed lower-priority candidates can't beat
|
|
||||||
// this one, which is true for *priority* eviction but not
|
|
||||||
// for *size-based fitting*. A smaller candidate may slot
|
|
||||||
// happily into a 24 MB hole even when the 31 MB candidate
|
|
||||||
// can't. Continue to the next candidate; the cooldown stops
|
|
||||||
// the chronically-blocked chunk from re-entering candidacy
|
|
||||||
// every frame (would otherwise burn web bandwidth on the
|
|
||||||
// same wasted fetches).
|
|
||||||
++streaming_blocked_oom_this_frame_;
|
|
||||||
c.blocked_cooldown_until_frame_idx =
|
|
||||||
streaming_frame_idx_ + BLOCKED_COOLDOWN_FRAMES;
|
|
||||||
if (evict_log) {
|
|
||||||
const uint64_t v_bytes = c.vertex_byte_size;
|
|
||||||
const uint64_t i_bytes = c.index_count * sizeof(uint32_t);
|
|
||||||
const double mb = 1.0 / (1024.0 * 1024.0);
|
|
||||||
QFileInfo fi(QString::fromStdString(cand.m->streaming_file_path));
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< "[blocked] chunk " << cand.ci
|
|
||||||
<< " of " << fi.completeBaseName()
|
|
||||||
<< " (pri=" << QString::number(cand.priority, 'f', 0)
|
|
||||||
<< ") — needs v=" << QString::number(double(v_bytes) * mb, 'f', 1)
|
|
||||||
<< " MB + i=" << QString::number(double(i_bytes) * mb, 'f', 1)
|
|
||||||
<< " MB; pool largest_free="
|
|
||||||
<< QString::number(double(pool_.largest_free_run_bytes()) * mb, 'f', 1)
|
|
||||||
<< " MB total_free="
|
|
||||||
<< QString::number(double(pool_.total_free_bytes()) * mb, 'f', 1)
|
|
||||||
<< " MB can_grow=" << (pool_.can_grow() ? "Y" : "N")
|
|
||||||
<< " — cooldown " << BLOCKED_COOLDOWN_FRAMES << "f";
|
|
||||||
}
|
|
||||||
more_pending = true;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Sync fallback when a screenshot is pending: the deferred-capture
|
|
||||||
// wait would let the window manager re-layout the window while we
|
|
||||||
// wait, capturing at the wrong size. With sync loads the chunk
|
|
||||||
// appears in the same frame we enqueue, no deferred-state to manage.
|
|
||||||
if (!pending_screenshot_path_.empty()) {
|
|
||||||
if (core_.loadChunkBytesAndUploadGpu(*cand.m, cand.ci)) {
|
|
||||||
++enqueued;
|
|
||||||
c.last_visible_frame_idx = streaming_frame_idx_;
|
|
||||||
}
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (streaming_thread_.enqueue(ViewportCore::makeChunkRequest(*cand.m, cand.ci, cand.mid))) {
|
|
||||||
c.is_loading = true;
|
|
||||||
++enqueued;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
loads += enqueued;
|
|
||||||
// Keep the frame loop running while we're making progress or there
|
|
||||||
// are worker reads still in flight. When everything's quiet
|
|
||||||
// (no main-thread work this frame AND worker queue empty) we let
|
|
||||||
// the renderer idle until the camera moves or a model loads.
|
|
||||||
// Spinning otherwise would burn CPU forever on visible-set >
|
|
||||||
// pool-capacity scenes.
|
|
||||||
if (loads > 0 || streaming_thread_.inFlightApprox() > 0) requestUpdate();
|
|
||||||
|
|
||||||
// Surface per-frame activity for the bench harness to gate the
|
|
||||||
// orbit sweep against cold-load. We only export loads — more_pending
|
|
||||||
// can stay true forever in the can't-fit case and is not a "done"
|
|
||||||
// signal.
|
|
||||||
streaming_loads_this_frame_ = loads;
|
|
||||||
streaming_more_pending_ = more_pending;
|
|
||||||
|
|
||||||
// Click-and-track diagnostic. When the user picked an object, we noted
|
|
||||||
// which chunk holds it. If that chunk has just transitioned resident
|
|
||||||
// → evicted, dump the priority + pool state at the moment of loss so
|
|
||||||
// we can see WHY it lost (was the new candidate higher priority? did
|
|
||||||
// the pool fail to fit anyone? did frustum visibility just go to 0?).
|
|
||||||
if (tracked_chunk_idx_ != SIZE_MAX) {
|
|
||||||
auto it = models_gpu_.find(tracked_chunk_mid_);
|
|
||||||
if (it != models_gpu_.end()
|
|
||||||
&& tracked_chunk_idx_ < it->second.chunks.size()) {
|
|
||||||
const auto& m = it->second;
|
|
||||||
const auto& c = m.chunks[tracked_chunk_idx_];
|
|
||||||
if (tracked_was_resident_ && !c.is_resident) {
|
|
||||||
const double mb = 1.0 / (1024.0 * 1024.0);
|
|
||||||
const float my_area = core_.chunkScreenAreaPx(c, vp_mat);
|
|
||||||
const uint64_t my_bytes = c.vertex_byte_size
|
|
||||||
+ c.index_count * sizeof(uint32_t);
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< "[track] chunk " << tracked_chunk_idx_
|
|
||||||
<< " (object " << tracked_object_id_
|
|
||||||
<< ", model " << tracked_chunk_mid_
|
|
||||||
<< ") EVICTED this frame";
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< " area=" << QString::number(my_area, 'f', 0) << "px²"
|
|
||||||
<< " frustum_vis=" << c.frustum_visible_count
|
|
||||||
<< " hist=" << QString::number(c.visibility_history, 'f', 2)
|
|
||||||
<< " load_count=" << c.load_count
|
|
||||||
<< " size=" << QString::number(double(my_bytes) * mb, 'f', 1) << "MB";
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< " chunk aabb "
|
|
||||||
<< QString::number(c.aabb_max[0] - c.aabb_min[0], 'f', 1) << "×"
|
|
||||||
<< QString::number(c.aabb_max[1] - c.aabb_min[1], 'f', 1) << "×"
|
|
||||||
<< QString::number(c.aabb_max[2] - c.aabb_min[2], 'f', 1) << "m"
|
|
||||||
<< " centre=("
|
|
||||||
<< QString::number(0.5f * (c.aabb_min[0] + c.aabb_max[0]), 'f', 1) << ","
|
|
||||||
<< QString::number(0.5f * (c.aabb_min[1] + c.aabb_max[1]), 'f', 1) << ","
|
|
||||||
<< QString::number(0.5f * (c.aabb_min[2] + c.aabb_max[2]), 'f', 1) << ")";
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< " pool used="
|
|
||||||
<< QString::number(double(pool_.total_used_bytes()) * mb, 'f', 0)
|
|
||||||
<< "/"
|
|
||||||
<< QString::number(double(pool_.total_capacity_bytes()) * mb, 'f', 0)
|
|
||||||
<< "MB largest_free="
|
|
||||||
<< QString::number(double(pool_.largest_free_run_bytes()) * mb, 'f', 1) << "MB";
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< " this-frame: cands=" << streaming_candidates_this_frame_
|
|
||||||
<< " enq=" << enqueued
|
|
||||||
<< " ev_lru=" << streaming_evictions_lru_this_frame_
|
|
||||||
<< " ev_pri=" << streaming_evictions_pri_this_frame_
|
|
||||||
<< " blocked=" << streaming_blocked_oom_this_frame_;
|
|
||||||
|
|
||||||
// Top 5 candidates by priority — see which chunk(s) outscored ours.
|
|
||||||
struct Stat { uint32_t mid; size_t ci; float area; };
|
|
||||||
std::vector<Stat> all;
|
|
||||||
all.reserve(64);
|
|
||||||
for (const auto& [mid2, m2] : models_gpu_) {
|
|
||||||
for (size_t ci2 = 0; ci2 < m2.chunks.size(); ++ci2) {
|
|
||||||
const auto& cc = m2.chunks[ci2];
|
|
||||||
if (cc.is_resident) continue;
|
|
||||||
if (cc.frustum_visible_count == 0) continue;
|
|
||||||
all.push_back({mid2, ci2, core_.chunkScreenAreaPx(cc, vp_mat)});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
std::sort(all.begin(), all.end(),
|
|
||||||
[](const Stat& a, const Stat& b){ return a.area > b.area; });
|
|
||||||
const size_t n = std::min<size_t>(5, all.size());
|
|
||||||
for (size_t i = 0; i < n; ++i) {
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< " top cand #" << i << ": model " << all[i].mid
|
|
||||||
<< " chunk " << all[i].ci
|
|
||||||
<< " area=" << QString::number(all[i].area, 'f', 0) << "px²";
|
|
||||||
}
|
|
||||||
}
|
|
||||||
tracked_was_resident_ = c.is_resident;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (streaming_debug_) {
|
|
||||||
// Cheap per-frame breakdown so a thrash cycle's shape becomes
|
|
||||||
// visible — high candidates + high evictions + low net loads is
|
|
||||||
// the smoking gun for "working set > pool".
|
|
||||||
size_t resident = 0;
|
|
||||||
uint32_t max_load_count = 0;
|
|
||||||
size_t cycled = 0; // chunks loaded > 1 time this session
|
|
||||||
for (const auto& [mid, m] : models_gpu_) {
|
|
||||||
for (const auto& c : m.chunks) {
|
|
||||||
if (c.is_resident) ++resident;
|
|
||||||
if (c.load_count > max_load_count) max_load_count = c.load_count;
|
|
||||||
if (c.load_count > 1) ++cycled;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
Log::info().noquote().nospace()
|
|
||||||
<< "[stream-debug] f" << streaming_frame_idx_
|
|
||||||
<< " cands=" << streaming_candidates_this_frame_
|
|
||||||
<< " enq=" << enqueued
|
|
||||||
<< " drained=" << streaming_drained_this_frame_
|
|
||||||
<< " ev_lru=" << streaming_evictions_lru_this_frame_
|
|
||||||
<< " ev_pri=" << streaming_evictions_pri_this_frame_
|
|
||||||
<< " blocked=" << streaming_blocked_oom_this_frame_
|
|
||||||
<< " resident=" << resident
|
|
||||||
<< " cycled=" << cycled
|
|
||||||
<< " max_load=" << max_load_count;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// -----------------------------------------------------------------------------
|
// -----------------------------------------------------------------------------
|
||||||
// Depth attachment
|
// Depth attachment
|
||||||
|
|||||||
@@ -860,14 +860,13 @@ private:
|
|||||||
bool fly_debug_ = false;
|
bool fly_debug_ = false;
|
||||||
Stopwatch fly_render_clock_;
|
Stopwatch fly_render_clock_;
|
||||||
|
|
||||||
// Click-and-track diagnostic: when a pick lands, stash the chunk
|
// Click-and-track aliases (storage in core_). The pick handler still
|
||||||
// that holds the picked object. driveStreamingLoads watches for that
|
// lives in VW so it touches these by name; driveStreamingLoads in
|
||||||
// chunk's `is_resident` flipping true→false and dumps the priority
|
// core dumps the priority / pool stats at eviction.
|
||||||
// / pool stats at the moment of eviction so we can see why it lost.
|
uint32_t& tracked_object_id_;
|
||||||
uint32_t tracked_object_id_ = 0;
|
uint32_t& tracked_chunk_mid_;
|
||||||
uint32_t tracked_chunk_mid_ = 0;
|
size_t& tracked_chunk_idx_;
|
||||||
size_t tracked_chunk_idx_ = SIZE_MAX;
|
bool& tracked_was_resident_;
|
||||||
bool tracked_was_resident_ = false;
|
|
||||||
|
|
||||||
// Mouse-navigation bindings — mirrors GL's NavBindings + currentNavBindings().
|
// Mouse-navigation bindings — mirrors GL's NavBindings + currentNavBindings().
|
||||||
// Selection stays on LMB for every preset (none of the presets steal it),
|
// Selection stays on LMB for every preset (none of the presets steal it),
|
||||||
@@ -944,22 +943,20 @@ public:
|
|||||||
BufferPool& pool_;
|
BufferPool& pool_;
|
||||||
StreamingThread& streaming_thread_;
|
StreamingThread& streaming_thread_;
|
||||||
|
|
||||||
// Per-frame streaming activity, written by driveStreamingLoads,
|
// Per-frame streaming activity aliases (storage in core_). Written
|
||||||
// consumed by the benchmark harness to delay the orbit sweep until
|
// by driveStreamingLoads; consumed by the still-in-VW benchmark
|
||||||
// the initial cold-load settles. `loads` = chunks brought resident
|
// harness.
|
||||||
// this frame; `more_pending` = the loader wants to keep going.
|
int& streaming_loads_this_frame_;
|
||||||
int streaming_loads_this_frame_ = 0;
|
bool& streaming_more_pending_;
|
||||||
bool streaming_more_pending_ = false;
|
|
||||||
|
|
||||||
// Per-frame streaming counters for WGPU_STREAM_DEBUG. Mutated inside
|
// Per-frame streaming counters (storage in core_). Consumed by the
|
||||||
// driveStreamingLoads, consumed by the per-frame debug print and the
|
// bench-warm timeout dump in render() (still VW-side).
|
||||||
// bench-warm timeout dump.
|
int& streaming_candidates_this_frame_;
|
||||||
int streaming_candidates_this_frame_ = 0;
|
int& streaming_evictions_lru_this_frame_;
|
||||||
int streaming_evictions_lru_this_frame_ = 0;
|
int& streaming_evictions_pri_this_frame_;
|
||||||
int streaming_evictions_pri_this_frame_ = 0;
|
int& streaming_drained_this_frame_;
|
||||||
int streaming_drained_this_frame_ = 0;
|
int& streaming_blocked_oom_this_frame_;
|
||||||
int streaming_blocked_oom_this_frame_ = 0;
|
bool& streaming_debug_;
|
||||||
bool streaming_debug_ = false; // WGPU_STREAM_DEBUG=1
|
|
||||||
|
|
||||||
// Bench warm-phase counters. We wait until N consecutive frames with
|
// Bench warm-phase counters. We wait until N consecutive frames with
|
||||||
// 0 loads (convergence) before starting the orbit sweep, capped by
|
// 0 loads (convergence) before starting the orbit sweep, capped by
|
||||||
@@ -1015,8 +1012,11 @@ private:
|
|||||||
// mouse. Matches GL's last_cull_was_motion_ behaviour.
|
// mouse. Matches GL's last_cull_was_motion_ behaviour.
|
||||||
bool last_cull_was_motion_ = false;
|
bool last_cull_was_motion_ = false;
|
||||||
|
|
||||||
// Pending one-shot screenshot, captured at the end of the next render().
|
// Pending one-shot screenshot path alias (storage in core_).
|
||||||
std::string pending_screenshot_path_;
|
// Captured at the end of the next render(); driveStreamingLoads
|
||||||
|
// observes the non-empty value to switch into the sync chunk-load
|
||||||
|
// fallback so the first-frame capture isn't an empty buffer.
|
||||||
|
std::string& pending_screenshot_path_;
|
||||||
bool pending_screenshot_quit_ = false;
|
bool pending_screenshot_quit_ = false;
|
||||||
|
|
||||||
// Mouse navigation state. LMB drag orbits, MMB drag pans, wheel zooms.
|
// Mouse navigation state. LMB drag orbits, MMB drag pans, wheel zooms.
|
||||||
|
|||||||
Reference in New Issue
Block a user