From e5072460cddc3a5a1b37088b06a1614a26040cbe Mon Sep 17 00:00:00 2001 From: Dion Moult Date: Mon, 24 Aug 2026 10:23:47 +1000 Subject: [PATCH] ifcviewer-web: OPFS model cache behind addUrl(url, {cache: true}) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Streaming a federation over the network re-downloads everything on every visit: browsers do not populate their HTTP cache from ranged fetches (measured at 0 of 78 range requests served from cache even with a strong ETag). Host pages have started hand-rolling OPFS caches against the library's own source seam — this is the second app to port the same ~350 lines — so the capability moves into the library. The design keeps what those pages got right: the cache fills FROM THE VIEWER'S OWN RANGED READS (no second download, and only bytes the camera actually needed), entries are keyed by a hash of the URL and validated by ETag (falling back to Last-Modified + size), and a byte-span ledger guarantees a partial copy is never mistaken for a whole one. What it fixes: writes go through a FileSystemSyncAccessHandle in an inline worker — positional writes with no copy-on-open, where the pages' createWritable({keepExistingData}) paid a whole-file copy per flush (quadratic as the cache fills) and buffered up to 48 MB per model in JS to compensate — the handle's exclusive lock makes a second tab fall back to plain network instead of corrupting the entry; a complete copy now opens when the server is unreachable (offline was dead before despite the bytes being local); and entry names are hashes, where prefix-matched sanitised names could delete a sibling model's cache. viewer.cacheInfo() reports entries and the storage estimate; viewer.clearCache(url?) drops one or all. Browsers without OPFS or sync handles, servers without validators, and second tabs all degrade to exactly today's network streaming. Verified: the sample round-trips to zero range requests on reload, and a 42 MB model goes from 108 range requests to 5 on the second visit — the 85% the camera had viewed comes off disk, coverage honestly reports incomplete for the bytes streaming never needed. The test server now sends a content-hash ETag so the specs exercise real validation. Co-Authored-By: Claude Fable 5 --- src/ifcviewer-web/tests/opfs-cache.spec.mjs | 107 ++++++ src/ifcviewer-web/tests/serve.mjs | 8 +- src/ifcviewer-web/web/ifcviewer.js | 368 +++++++++++++++++++- 3 files changed, 481 insertions(+), 2 deletions(-) create mode 100644 src/ifcviewer-web/tests/opfs-cache.spec.mjs diff --git a/src/ifcviewer-web/tests/opfs-cache.spec.mjs b/src/ifcviewer-web/tests/opfs-cache.spec.mjs new file mode 100644 index 0000000000..af7ea1fd02 --- /dev/null +++ b/src/ifcviewer-web/tests/opfs-cache.spec.mjs @@ -0,0 +1,107 @@ +import { test, expect } from '@playwright/test'; + +// The OPFS model cache behind addUrl(url, {cache: true}): the first load +// streams over HTTP Range and fills a local copy from those same reads; a +// reload of the page then loads the model with zero geometry traffic. One +// browser context spans both loads — OPFS is origin storage, so it survives +// page reloads within the context. + +function watchRequests(page, counters) { + page.on('request', (req) => { + if (!req.url().includes('sample.ifcview')) return; + if (req.method() === 'HEAD') counters.head++; + else if (req.headers()['range']) counters.range++; + else counters.other++; + }); +} + +async function openScripting(page, errors) { + page.on('console', (msg) => { + const t = msg.text(); + if (/Uncaptured WebGPU error|is invalid|Not enough memory left/i.test(t)) errors.push(t); + }); + page.on('pageerror', (e) => errors.push('pageerror: ' + e.message)); + await page.goto('/scripting.html'); + await page.waitForFunction(() => !!(window.viewer && window.viewer.isLive()), null, + { timeout: 30_000 }); +} + +async function addCachedSampleAndSettle(page) { + await page.evaluate(async () => { + const v = window.viewer; + const loaded = new Promise((resolve) => { + const off = v.onModelLoaded((d) => { off(); resolve(d); }); + }); + await v.addUrl('/sample.ifcview', { replace: true, cache: true, name: 'cached' }); + await loaded; + }); + await page.waitForFunction(() => { + const v = window.viewer; + if (!v.modelCount()) return false; + const p = v.modelProgress(0); + return p.total > 0 && p.resident === p.total; + }, null, { timeout: 30_000 }); + // The element table is read through the same source — pull it so its + // ranges land in the cache too, then let the write chain drain. + await page.evaluate(() => window.viewer.getObjects()); + await page.waitForFunction(async () => { + const info = await window.viewer.cacheInfo(); + const e = info.entries.find((x) => x.url.endsWith('/sample.ifcview')); + return !!(e && e.complete); + }, null, { timeout: 30_000 }); +} + +test('first load fills the cache from its own reads; a reload streams nothing', async ({ page }) => { + const errors = []; + const first = { head: 0, range: 0, other: 0 }; + watchRequests(page, first); + await openScripting(page, errors); + await page.evaluate(() => window.viewer.clearCache()); + + await addCachedSampleAndSettle(page); + expect(first.range, 'first visit must stream over HTTP Range').toBeGreaterThan(0); + + const info = await page.evaluate(() => window.viewer.cacheInfo()); + const entry = info.entries.find((x) => x.url.endsWith('/sample.ifcview')); + expect(entry.complete).toBe(true); + expect(entry.cachedBytes).toBe(entry.size); + + // Second visit: same context, fresh page. Only the HEAD validation may + // touch the network — every byte of geometry and metadata comes from OPFS. + await page.reload(); + const second = { head: 0, range: 0, other: 0 }; + watchRequests(page, second); + await page.waitForFunction(() => !!(window.viewer && window.viewer.isLive()), null, + { timeout: 30_000 }); + await addCachedSampleAndSettle(page); + expect(second.range, 'a complete validated copy must stream zero ranges').toBe(0); + expect(second.other, 'and never download the file whole').toBe(0); + expect(second.head).toBeGreaterThan(0); + + // The cached model is actually usable: objects enumerate with GUIDs. + const objects = await page.evaluate(() => window.viewer.getObjects()); + expect(objects.length).toBeGreaterThan(0); + expect(objects.some((o) => o.guid)).toBe(true); + + expect(errors).toEqual([]); +}); + +test('clearCache drops the entry and the next load streams again', async ({ page }) => { + const errors = []; + await openScripting(page, errors); + await addCachedSampleAndSettle(page); + + const cleared = await page.evaluate(() => window.viewer.clearCache('/sample.ifcview')); + expect(cleared).toBe(1); + const info = await page.evaluate(() => window.viewer.cacheInfo()); + expect(info.entries.find((x) => x.url.endsWith('/sample.ifcview'))).toBeUndefined(); + + await page.reload(); + const counters = { head: 0, range: 0, other: 0 }; + watchRequests(page, counters); + await page.waitForFunction(() => !!(window.viewer && window.viewer.isLive()), null, + { timeout: 30_000 }); + await addCachedSampleAndSettle(page); + expect(counters.range).toBeGreaterThan(0); + expect(errors).toEqual([]); +}); diff --git a/src/ifcviewer-web/tests/serve.mjs b/src/ifcviewer-web/tests/serve.mjs index 124868dd77..89fc403d16 100644 --- a/src/ifcviewer-web/tests/serve.mjs +++ b/src/ifcviewer-web/tests/serve.mjs @@ -6,6 +6,7 @@ // Serve dir resolution: $WEB_BUILD_DIR if set, else the repo's build-web. import http from 'node:http'; import { readFile } from 'node:fs/promises'; +import { createHash } from 'node:crypto'; import { fileURLToPath } from 'node:url'; import path from 'node:path'; @@ -46,6 +47,9 @@ http.createServer(async (req, res) => { try { body = await readFile(inRoot); } catch { body = await readFile(inSrc); } // fall back to the source dir const ctype = MIME[path.extname(p)] || 'application/octet-stream'; + // A strong ETag from the content, so the OPFS cache spec can exercise + // validation exactly the way a real Accept-Ranges host would offer it. + const etag = '"' + createHash('sha1').update(body).digest('hex').slice(0, 16) + '"'; // HEAD: headers only — lets the remote backend resolve total size. if (req.method === 'HEAD') { @@ -53,6 +57,7 @@ http.createServer(async (req, res) => { 'Content-Type': ctype, 'Content-Length': body.length, 'Accept-Ranges': 'bytes', + 'ETag': etag, }); res.end(); return; @@ -74,12 +79,13 @@ http.createServer(async (req, res) => { 'Content-Range': `bytes ${start}-${end}/${body.length}`, 'Accept-Ranges': 'bytes', 'Content-Length': slice.length, + 'ETag': etag, }); res.end(slice); return; } - res.writeHead(200, { 'Content-Type': ctype, 'Accept-Ranges': 'bytes' }); + res.writeHead(200, { 'Content-Type': ctype, 'Accept-Ranges': 'bytes', 'ETag': etag }); res.end(body); } catch { res.writeHead(404).end('not found'); diff --git a/src/ifcviewer-web/web/ifcviewer.js b/src/ifcviewer-web/web/ifcviewer.js index b4cb80c28a..9def85ffc4 100644 --- a/src/ifcviewer-web/web/ifcviewer.js +++ b/src/ifcviewer-web/web/ifcviewer.js @@ -51,6 +51,307 @@ return cr ? parseInt(cr.split('/')[1] || '0', 10) : 0; } + // ---- OPFS model cache ------------------------------------------------------ + // + // addUrl(url, {cache: true}) keeps a local copy of the sidecar in the + // Origin Private File System, filled FROM THE VIEWER'S OWN RANGED READS — + // no second download, and only the bytes the camera actually needed. + // (Browsers do not populate their HTTP cache from ranged fetches: measured + // 0 of 78 range requests served from cache even with a strong ETag.) + // + // Entries are keyed by a hash of the URL and validated by ETag (falling + // back to Last-Modified + size); a byte-span ledger records which ranges + // are really on disk, so a partial copy is never mistaken for a whole one — + // a read is served locally only when its span is fully covered. On the next + // visit a complete validated copy loads with zero geometry traffic; if the + // server is unreachable, the newest complete copy is used as-is (offline). + // + // Writes go through a dedicated worker holding a FileSystemSyncAccessHandle: + // positional writes with no copy-on-open (createWritable({keepExistingData}) + // copies the whole existing file into a swap file per open — quadratic as + // the cache fills), and the handle's exclusive lock makes a second tab fall + // back to plain network instead of corrupting the entry. Browsers without + // sync access handles just never cache — behaviour is exactly as without + // the flag. + const CACHE_DIR = 'ifcviewer-cache'; + + const cacheWorkerSource = ` + const handles = new Map(); + onmessage = async (e) => { + const { id, op, name, pos, data, len } = e.data; + const reply = (msg, transfer) => postMessage(Object.assign({ id }, msg), transfer || []); + try { + if (op === 'open') { + const root = await navigator.storage.getDirectory(); + const dir = await root.getDirectoryHandle('${CACHE_DIR}', { create: true }); + const fh = await dir.getFileHandle(name, { create: true }); + handles.set(name, await fh.createSyncAccessHandle()); + reply({ ok: true, size: handles.get(name).getSize() }); + } else if (op === 'write') { + handles.get(name).write(new Uint8Array(data), { at: pos }); + reply({ ok: true }); + } else if (op === 'read') { + const buf = new Uint8Array(len); + const n = handles.get(name).read(buf, { at: pos }); + reply({ ok: n === len, data: buf.buffer }, [buf.buffer]); + } else if (op === 'close') { + const h = handles.get(name); + if (h) { h.flush(); h.close(); handles.delete(name); } + reply({ ok: true }); + } else { + reply({ ok: false, error: 'unknown op ' + op }); + } + } catch (err) { + reply({ ok: false, error: String((err && err.message) || err) }); + } + }; + `; + + let cacheWorker = null; // lazily created; false once known unusable + let cacheMsgId = 0; + const cachePending = new Map(); + function cacheCall(op, name, extra, transfer) { + if (cacheWorker === false) return Promise.reject(new Error('no cache worker')); + if (!cacheWorker) { + try { + cacheWorker = new Worker(URL.createObjectURL( + new Blob([cacheWorkerSource], { type: 'text/javascript' }))); + cacheWorker.onmessage = (e) => { + const pending = cachePending.get(e.data.id); + if (!pending) return; + cachePending.delete(e.data.id); + if (e.data.ok) pending.resolve(e.data); + else pending.reject(new Error(e.data.error || (op + ' failed'))); + }; + } catch (err) { + cacheWorker = false; + return Promise.reject(err); + } + } + const id = ++cacheMsgId; + return new Promise((resolve, reject) => { + cachePending.set(id, { resolve: resolve, reject: reject }); + cacheWorker.postMessage(Object.assign({ id: id, op: op, name: name }, extra || {}), + transfer || []); + }); + } + + async function cacheDirHandle(create) { + const root = await navigator.storage.getDirectory(); + return root.getDirectoryHandle(CACHE_DIR, { create: !!create }); + } + + let persistAsked = false; + async function openCacheDir() { + try { + const dir = await cacheDirHandle(true); + // Without this a large cache is "best effort" and the browser may drop + // it under disk pressure — invisibly, looking like the site being slow + // again on the next visit. + if (!persistAsked && navigator.storage.persist) { + persistAsked = true; + navigator.storage.persist().catch(() => {}); + } + return dir; + } catch (err) { + return null; + } + } + + async function cacheEntryName(url) { + const bytes = new TextEncoder().encode(url); + const digest = await crypto.subtle.digest('SHA-256', bytes); + return Array.from(new Uint8Array(digest).slice(0, 16)) + .map((b) => b.toString(16).padStart(2, '0')).join(''); + } + + // Sorted, merged, half-open [start, end) byte spans. + function spansAdd(spans, start, end) { + const out = []; + let s0 = start, e0 = end; + for (const [a, b] of spans) { + if (b < s0 || a > e0) out.push([a, b]); + else { s0 = Math.min(s0, a); e0 = Math.max(e0, b); } + } + out.push([s0, e0]); + out.sort((x, y) => x[0] - y[0]); + return out; + } + const spansCover = (spans, start, end) => + spans.some(([a, b]) => a <= start && b >= end); + const spansBytes = (spans) => spans.reduce((sum, [a, b]) => sum + (b - a), 0); + + async function readCacheMeta(dir, name) { + try { + const fh = await dir.getFileHandle(name + '.meta'); + return JSON.parse(await (await fh.getFile()).text()); + } catch (err) { + return null; + } + } + // The meta file is tiny, so main-thread createWritable is fine here; only + // the tab holding the data file's exclusive lock ever writes it. + async function writeCacheMeta(dir, name, meta) { + const fh = await dir.getFileHandle(name + '.meta', { create: true }); + const w = await fh.createWritable(); + await w.write(JSON.stringify(meta)); + await w.close(); + } + async function removeCacheEntry(dir, name) { + await dir.removeEntry(name).catch(() => {}); + await dir.removeEntry(name + '.meta').catch(() => {}); + } + + async function headValidators(url) { + try { + const res = await fetch(url, { method: 'HEAD' }); + if (!res.ok) return null; + return { + etag: res.headers.get('ETag') || null, + lastModified: res.headers.get('Last-Modified') || null, + size: parseInt(res.headers.get('Content-Length') || '0', 10) || 0, + }; + } catch (err) { + return null; // offline, or CORS refused HEAD + } + } + function validatorsMatch(meta, head) { + if (meta.etag && head.etag) return meta.etag === head.etag; + if (meta.lastModified && head.lastModified) { + return meta.lastModified === head.lastModified && meta.size === head.size; + } + return false; + } + + // A Blob-shaped source (`size` + `slice(a, b).arrayBuffer()`) backed by the + // OPFS entry, filling from every ranged read that goes through it. The + // wasm's read shim only ever calls those two members, so this passes for + // the File it would get from a picked file. + function fillingCacheSource(dir, name, url, meta) { + let spans = meta.spans.slice(); + let dirtySince = 0; // bytes written since the ledger was persisted + let broken = false; // a write failed (quota?): serve network, stop filling + let complete = spansCover(spans, 0, meta.size); + let writeChain = Promise.resolve(); + + const persistLedger = () => { + dirtySince = 0; + return writeCacheMeta(dir, name, Object.assign({}, meta, { spans: spans })) + .catch(() => {}); + }; + const finishIfComplete = () => { + if (complete || !spansCover(spans, 0, meta.size)) return; + complete = true; + // Flush + release the lock; from here reads come off the closed file. + writeChain = writeChain + .then(() => cacheCall('close', name)) + .then(persistLedger) + .catch(() => {}); + }; + + return { + size: meta.size, + slice(start, end) { + const stop = Math.min(end, meta.size); + return { + arrayBuffer: async () => { + if (spansCover(spans, start, stop)) { + if (complete) { + const fh = await dir.getFileHandle(name); + return (await fh.getFile()).slice(start, stop).arrayBuffer(); + } + // Serialise behind the writes so a just-written range is + // readable (sync-handle writes are visible to the same handle + // immediately; ordering through the chain keeps it simple). + return (writeChain = writeChain.then(() => + cacheCall('read', name, { pos: start, len: stop - start }) + )).then((r) => r.data); + } + const res = await fetch(url, { + headers: { Range: 'bytes=' + start + '-' + (stop - 1) }, + }); + if (res.status !== 206 && res.status !== 200) { + throw new Error('range fetch failed: ' + res.status); + } + let buf = await res.arrayBuffer(); + if (res.status === 200 && buf.byteLength > stop - start) { + buf = buf.slice(start, stop); + } + if (!broken && !complete && buf.byteLength === stop - start) { + const copy = buf.slice(0); + writeChain = writeChain + .then(() => cacheCall('write', name, { pos: start, data: copy }, [copy])) + .then(() => { + spans = spansAdd(spans, start, stop); + dirtySince += stop - start; + // Persist the ledger periodically — bytes on disk that the + // ledger does not record are merely re-fetched next visit. + if (dirtySince >= (4 << 20)) return persistLedger(); + }) + .then(finishIfComplete) + .catch(() => { broken = true; }); + } + return buf; + }, + }; + }, + }; + } + + // Decide what to hand the loader for a cached URL: + // {file} — a complete validated local copy (zero network) + // {source} — a Blob-shaped self-filling source + // null — cache unusable here (no OPFS / no validators / second tab): + // caller falls back to the plain URL path. + async function cachedUrlSource(url) { + if (!(crypto && crypto.subtle) || !(navigator.storage && navigator.storage.getDirectory)) { + return null; + } + const head = await headValidators(url); + const dir = await openCacheDir(); + if (!dir) return null; + const name = await cacheEntryName(url); + const meta = await readCacheMeta(dir, name); + + const completeLocal = async (m) => { + const fh = await dir.getFileHandle(name); + const file = await fh.getFile(); + return file.size === m.size ? { file: file } : null; + }; + + if (!head) { + // Offline (or HEAD refused): a complete copy is better than nothing — + // this is the offline story. Anything less falls back to the URL path, + // which will fail the same way it always did. + if (meta && spansCover(meta.spans, 0, meta.size)) return completeLocal(meta); + return null; + } + if (!head.etag && !head.lastModified) return null; // nothing to validate by + if (!head.size) return null; + + if (meta && validatorsMatch(meta, head)) { + if (spansCover(meta.spans, 0, meta.size)) { + const local = await completeLocal(meta); + if (local) return local; + } + // Partial copy of the still-current file: resume filling it. + } else if (meta) { + await removeCacheEntry(dir, name); // server has a different file now + } + + const fresh = (!meta || !validatorsMatch(meta, head)) + ? { url: url, etag: head.etag, lastModified: head.lastModified, + size: head.size, spans: [] } + : meta; + try { + await cacheCall('open', name); // exclusive: a second tab lands in catch + } catch (err) { + return null; + } + await writeCacheMeta(dir, name, fresh).catch(() => {}); + return { source: fillingCacheSource(dir, name, url, fresh) }; + } + // Mouse navigation schemes the wasm's classifyPress understands. Named so a // typo is an error here rather than a silent fall-back to blender in the core. const NAV_PRESETS = ['blender', 'rhino', 'revit', 'web']; @@ -593,14 +894,79 @@ Module._load_sidecar_from_source_c(sid); return sid; }, + // `cache: true` keeps a local OPFS copy filled from the viewer's own + // ranged reads (see the OPFS model cache section above): the next visit + // loads it with zero geometry traffic, and a complete copy still opens + // when the server is unreachable. Falls back to plain URL streaming + // wherever the cache cannot help (no OPFS, no validators from the + // server, another tab already filling this entry). addUrl: async function (url, o) { if (o && o.replace) this.clearScene(); - const sid = await registerUrl(url); + let sid = null; + if (o && o.cache) { + const cached = await cachedUrlSource(url).catch(() => null); + if (cached) { + const src = cached.file || cached.source; + sid = Module.__ifcvSources.length; + Module.__ifcvSources.push({ file: src, url: null, size: src.size }); + } + } + if (sid === null) sid = await registerUrl(url); if (o && o.name) this.setModelName(sid, o.name); Module._load_sidecar_from_source_c(sid); return sid; }, + // What the OPFS cache holds: [{url, size, cachedBytes, complete}] plus + // the browser's storage estimate. Entries whose ledger has not caught + // up with the last few reads under-report slightly; nothing over-reports. + cacheInfo: async function () { + const out = { entries: [], estimate: null }; + try { + const dir = await cacheDirHandle(false); + for await (const key of dir.keys()) { + if (!key.endsWith('.meta')) continue; + try { + const meta = JSON.parse(await (await + (await dir.getFileHandle(key)).getFile()).text()); + out.entries.push({ + url: meta.url, + size: meta.size, + cachedBytes: spansBytes(meta.spans), + complete: spansCover(meta.spans, 0, meta.size), + }); + } catch (err) { /* torn meta: skip */ } + } + } catch (err) { /* no cache dir yet */ } + if (navigator.storage && navigator.storage.estimate) { + out.estimate = await navigator.storage.estimate().catch(() => null); + } + return out; + }, + + // Drop cached models — one URL, or everything. Sources already handed + // to the viewer keep working (open handles and Files stay readable); + // the next visit simply streams from the network again. + clearCache: async function (url) { + try { + const dir = await cacheDirHandle(false); + const only = url ? await cacheEntryName(url) : null; + const names = []; + for await (const key of dir.keys()) { + const base = key.endsWith('.meta') ? key.slice(0, -5) : key; + if (!only || base === only) names.push(key); + } + let n = 0; + for (const key of names) { + await dir.removeEntry(key).catch(() => {}); + if (!key.endsWith('.meta')) n++; + } + return n; + } catch (err) { + return 0; + } + }, + // ---- Federation ------------------------------------------------------ // // The concepts an .ifcfed file carries, without the file format: a