diff --git a/docker/frontend/src/engine/superRes.ts b/docker/frontend/src/engine/superRes.ts index 7fe2634..d46e040 100644 --- a/docker/frontend/src/engine/superRes.ts +++ b/docker/frontend/src/engine/superRes.ts @@ -108,49 +108,40 @@ export async function upscaleJpeg( if (targetLongest <= longest) return bytes; const scale = targetLongest / longest; - // How much of the photo the model is handed, in photo pixels per pixel it - // reads. It answers with four pixels for every one it is given, so the only - // picture it ever has to read is `targetLongest / 4` across — hand it the - // whole photo instead and it invents four times the pixels being asked for, - // which the draw then throws three quarters of away on the way down to - // `targetLongest`. Same finished image, a quarter of the arithmetic: on a - // 2400px photo going to 4K that is 80 tiles of model for 20. - // - // The floor is the photo's own claim: below half its pixels the model is no - // longer enlarging the picture, it is drawing a new one from memory. - // The ceiling is the same idea from the other side — never hand it more - // pixels than the photo has, or the wait grows for nothing the eye can see. - const feedScale = Math.min(1, Math.max(scale / MODEL_SCALE, 0.5)); - const fw = Math.max(1, Math.round(w * feedScale)); - const fh = Math.max(1, Math.round(h * feedScale)); - const dstCanvas = new OffscreenCanvas(Math.max(1, Math.round(w * scale)), Math.max(1, Math.round(h * scale))); // Opaque: a partly covered edge pixel would otherwise survive as transparency // and the JPEG export flattens that onto black — a dark line down every seam. const dstCtx = dstCanvas.getContext('2d', { alpha: false }); if (!dstCtx) return bytes; + // The destination is rarely the model's own 4x, so the draw below is usually + // a reduction, by up to four, and `low` would keep one sample in four of what + // the model has just drawn. `high` reads them all. + dstCtx.imageSmoothingEnabled = true; + dstCtx.imageSmoothingQuality = 'high'; - const feedCanvas = new OffscreenCanvas(fw, fh); + // The photo's own pixels, read once. The model is handed the picture itself + // and never a smaller copy of it: it answers with four pixels for every one + // it is given, and a source shrunk towards the destination is detail the + // photo had that the model is then asked to invent back — a 2px stripe in a + // 2400px photo exported at 4K comes back as flat grey that way. The four for + // one is spent in the draw below instead, which reduces to the destination + // and keeps what the photo actually held. + const feedCanvas = new OffscreenCanvas(w, h); const feedCtx = feedCanvas.getContext('2d', { willReadFrequently: true }); if (!feedCtx) return bytes; - // 'high' matters here: this resample is the only one the photo gets before - // the model reads it, and a cheap one would hand it a soft picture to be - // sharp about. - feedCtx.imageSmoothingEnabled = true; - feedCtx.imageSmoothingQuality = 'high'; - feedCtx.drawImage(bitmap, 0, 0, fw, fh); - const src = feedCtx.getImageData(0, 0, fw, fh); + feedCtx.drawImage(bitmap, 0, 0); + const src = feedCtx.getImageData(0, 0, w, h); const { ort, session } = await load(); const inputName = session.inputNames[0]; - const cols = Math.ceil(fw / TILE); - const rows = Math.ceil(fh / TILE); + const cols = Math.ceil(w / TILE); + const rows = Math.ceil(h / TILE); // Destination pixels per fed pixel. Derived from the destination itself so // the last row and column land exactly on its edge rather than a rounding // short of it, and shared by neighbouring tiles so their boundary is the // same number for both and nothing is left half-covered. - const stepX = dstCanvas.width / fw; - const stepY = dstCanvas.height / fh; + const stepX = dstCanvas.width / w; + const stepY = dstCanvas.height / h; let done = 0; onProgress?.({ done, total: cols * rows }); @@ -158,20 +149,20 @@ export async function upscaleJpeg( for (let tx = 0; tx < cols; tx++) { const x0 = tx * TILE; const y0 = ty * TILE; - const tw = Math.min(TILE, fw - x0); - const th = Math.min(TILE, fh - y0); + const tw = Math.min(TILE, w - x0); + const th = Math.min(TILE, h - y0); // The margin the model gets: full on the inside, clipped at the photo's // own edge, so the tensor covers whole pixels only. const left = Math.min(PAD, x0); const top = Math.min(PAD, y0); - const pw = tw + left + Math.min(PAD, fw - (x0 + tw)); - const ph = th + top + Math.min(PAD, fh - (y0 + th)); + const pw = tw + left + Math.min(PAD, w - (x0 + tw)); + const ph = th + top + Math.min(PAD, h - (y0 + th)); // NCHW, 0..1 RGB — what the model was trained to read. const input = new Float32Array(3 * pw * ph); const plane = pw * ph; for (let y = 0; y < ph; y++) { - const srow = ((y0 - top + y) * fw + (x0 - left)) * 4; + const srow = ((y0 - top + y) * w + (x0 - left)) * 4; for (let x = 0; x < pw; x++) { const s = srow + x * 4; input[y * pw + x] = src.data[s] / 255;