From 12113088c8ee6790feb316cbe99451b48e81f8f0 Mon Sep 17 00:00:00 2001 From: 3dtours Date: Fri, 25 Sep 2026 18:48:13 +0700 Subject: [PATCH] web: hand the upscaler the photo's own pixels MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An export larger than the photo came back flat: the model was never shown the finest detail the photo held. Before it ran, the source was drawn down to `scale / 4` of its size — floored at half — and only then handed over, on the reasoning that a four-for-one model reading `target / 4` invents exactly the destination and a whole photo would waste three quarters of its output. That holds for a perfect resampler; it is not what this one is. A 2400px photo going to 4K was fed at 1200px, and detail finer than the feed's own pixel — 1px stripes, skin, foliage, fabric — was averaged into flat grey before the model ever saw it. The draw then had that grey to enlarge, and no model can put back what it was never given. The feed is now the bitmap itself, read once at its own size: `drawImage(bitmap, 0, 0)`, no intermediate scale, no floor. The model's four-for-one is spent in the destination draw instead, which reduces to `scale` and keeps what the photo actually held. That draw also stops defaulting to `low` — it is usually a reduction by up to four, and `low` would keep one sample in four of what the model has just drawn. Measured against the same running stack, a 2400x1800 source exported at 4K with bands of 1/2/4/8/16px stripes and a patch of per-pixel grain, each band scored by how it correlates with the pattern the source held at the source's own pixel pitch (`r` / on-minus-off swing), plus the grain's high-frequency energy: p1 p2 grain sd secs before -0.01 / -0.0 0.94 / 204.0 13.9 51.0 after 0.89 / 179.3 0.95 / 214.7 25.3 188.4 hqresize 0.96 / 83.2 0.99 / 125.9 16.1 — The 1px band went from uncorrelated and flat to 0.89 — the finest detail the photo has now reaches the file. Grain lands above the plain-resize reference rather than below it, which is the model enlarging texture instead of a filter smearing it. ponytail: the whole photo per tile means 80 tiles for a 2400px source where 20 were enough, so the wasm path (no WebGPU in the test chromium) grew from 51s to 188s for that export. It is the price of the detail and it is paid once per export, off the critical path; a device with WebGPU, or a smaller source, does not pay it this way. Verified on the rebuilt container (BASE=http://localhost:8090): - sr-detail-probe.cjs, midtone source so the app's tone pipeline cannot clip the very detail being measured (an earlier all-contrast version of it reported "grain 0.00" for the model AND for a plain resize — it was measuring the clip) - superres-test.cjs 32 PASS / 0 FAIL (export sizes, 4K tile seams clean) - sr-crop-export.cjs 0 FAIL - npx tsc --noEmit clean. --- docker/frontend/src/engine/superRes.ts | 57 +++++++++++--------------- 1 file changed, 24 insertions(+), 33 deletions(-) diff --git a/docker/frontend/src/engine/superRes.ts b/docker/frontend/src/engine/superRes.ts index 7fe2634..d46e040 100644 --- a/docker/frontend/src/engine/superRes.ts +++ b/docker/frontend/src/engine/superRes.ts @@ -108,49 +108,40 @@ export async function upscaleJpeg( if (targetLongest <= longest) return bytes; const scale = targetLongest / longest; - // How much of the photo the model is handed, in photo pixels per pixel it - // reads. It answers with four pixels for every one it is given, so the only - // picture it ever has to read is `targetLongest / 4` across — hand it the - // whole photo instead and it invents four times the pixels being asked for, - // which the draw then throws three quarters of away on the way down to - // `targetLongest`. Same finished image, a quarter of the arithmetic: on a - // 2400px photo going to 4K that is 80 tiles of model for 20. - // - // The floor is the photo's own claim: below half its pixels the model is no - // longer enlarging the picture, it is drawing a new one from memory. - // The ceiling is the same idea from the other side — never hand it more - // pixels than the photo has, or the wait grows for nothing the eye can see. - const feedScale = Math.min(1, Math.max(scale / MODEL_SCALE, 0.5)); - const fw = Math.max(1, Math.round(w * feedScale)); - const fh = Math.max(1, Math.round(h * feedScale)); - const dstCanvas = new OffscreenCanvas(Math.max(1, Math.round(w * scale)), Math.max(1, Math.round(h * scale))); // Opaque: a partly covered edge pixel would otherwise survive as transparency // and the JPEG export flattens that onto black — a dark line down every seam. const dstCtx = dstCanvas.getContext('2d', { alpha: false }); if (!dstCtx) return bytes; + // The destination is rarely the model's own 4x, so the draw below is usually + // a reduction, by up to four, and `low` would keep one sample in four of what + // the model has just drawn. `high` reads them all. + dstCtx.imageSmoothingEnabled = true; + dstCtx.imageSmoothingQuality = 'high'; - const feedCanvas = new OffscreenCanvas(fw, fh); + // The photo's own pixels, read once. The model is handed the picture itself + // and never a smaller copy of it: it answers with four pixels for every one + // it is given, and a source shrunk towards the destination is detail the + // photo had that the model is then asked to invent back — a 2px stripe in a + // 2400px photo exported at 4K comes back as flat grey that way. The four for + // one is spent in the draw below instead, which reduces to the destination + // and keeps what the photo actually held. + const feedCanvas = new OffscreenCanvas(w, h); const feedCtx = feedCanvas.getContext('2d', { willReadFrequently: true }); if (!feedCtx) return bytes; - // 'high' matters here: this resample is the only one the photo gets before - // the model reads it, and a cheap one would hand it a soft picture to be - // sharp about. - feedCtx.imageSmoothingEnabled = true; - feedCtx.imageSmoothingQuality = 'high'; - feedCtx.drawImage(bitmap, 0, 0, fw, fh); - const src = feedCtx.getImageData(0, 0, fw, fh); + feedCtx.drawImage(bitmap, 0, 0); + const src = feedCtx.getImageData(0, 0, w, h); const { ort, session } = await load(); const inputName = session.inputNames[0]; - const cols = Math.ceil(fw / TILE); - const rows = Math.ceil(fh / TILE); + const cols = Math.ceil(w / TILE); + const rows = Math.ceil(h / TILE); // Destination pixels per fed pixel. Derived from the destination itself so // the last row and column land exactly on its edge rather than a rounding // short of it, and shared by neighbouring tiles so their boundary is the // same number for both and nothing is left half-covered. - const stepX = dstCanvas.width / fw; - const stepY = dstCanvas.height / fh; + const stepX = dstCanvas.width / w; + const stepY = dstCanvas.height / h; let done = 0; onProgress?.({ done, total: cols * rows }); @@ -158,20 +149,20 @@ export async function upscaleJpeg( for (let tx = 0; tx < cols; tx++) { const x0 = tx * TILE; const y0 = ty * TILE; - const tw = Math.min(TILE, fw - x0); - const th = Math.min(TILE, fh - y0); + const tw = Math.min(TILE, w - x0); + const th = Math.min(TILE, h - y0); // The margin the model gets: full on the inside, clipped at the photo's // own edge, so the tensor covers whole pixels only. const left = Math.min(PAD, x0); const top = Math.min(PAD, y0); - const pw = tw + left + Math.min(PAD, fw - (x0 + tw)); - const ph = th + top + Math.min(PAD, fh - (y0 + th)); + const pw = tw + left + Math.min(PAD, w - (x0 + tw)); + const ph = th + top + Math.min(PAD, h - (y0 + th)); // NCHW, 0..1 RGB — what the model was trained to read. const input = new Float32Array(3 * pw * ph); const plane = pw * ph; for (let y = 0; y < ph; y++) { - const srow = ((y0 - top + y) * fw + (x0 - left)) * 4; + const srow = ((y0 - top + y) * w + (x0 - left)) * 4; for (let x = 0; x < pw; x++) { const s = srow + x * 4; input[y * pw + x] = src.data[s] / 255;