diff --git a/changes/dilate-parallel.md b/changes/dilate-parallel.md new file mode 100644 index 00000000..345109c2 --- /dev/null +++ b/changes/dilate-parallel.md @@ -0,0 +1,8 @@ +bump: patch +type: performance +**Cut-out edge padding runs on every core.** `tex_dilate`'s passes hand their rows, sixteen at a time, to +`Job.parallel_for`: within a pass a row writes only its own still-masked texels and reads only +neighbours the mask already let go, so the bytes are the ones the single-threaded loop made. The worker +is handed plain buffers in a `DilateJob` and makes nothing. `tex_dilate_bytes` is the slice-taking +form (safe_api.ludic), and `examples/rendering/dilate.ludic` holds the result against the old loop +(prints DILATE OK). It was 206 ms of the main thread in a Maroon Lake boot. diff --git a/examples/rendering/dilate.ludic b/examples/rendering/dilate.ludic new file mode 100644 index 00000000..637ebd6e --- /dev/null +++ b/examples/rendering/dilate.ludic @@ -0,0 +1,77 @@ +# dilate.ludic - render3d's cut-out padding, now a pass's rows on every core, gives exactly the +# bytes the one-thread version gave: the old loop is kept here as the oracle, over a patterned atlas +# +# bin/ludic build examples/rendering/dilate.ludic --headless && ./build/dilate_headless (prints DILATE OK) +program Dilate { + numbers float + import "ludic.render3d/r3d.ludic" + + # the dilate as it was, one row after another + function oracle(px: []byte, w: int, h: int, c: int, thresh: int, passes: int) -> void { + let mask = buffer(w * h) + for i in 0 .. w * h { let o = i * c; if px[o] + px[o + 1] + px[o + 2] < thresh { mask[i] = 1 } else { mask[i] = 0 } } + let next = buffer(w * h) + for pass in 0 .. passes { + for q in 0 .. w * h { next[q] = mask[q] } + for y in 0 .. h { + for x in 0 .. w { + let k = y * w + x + if mask[k] == 1 { + var r = 0; var g = 0; var b = 0; var n = 0 + if x > 0 and mask[k - 1] == 0 { let o = (k - 1) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } + if x < w - 1 and mask[k + 1] == 0 { let o = (k + 1) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } + if y > 0 and mask[k - w] == 0 { let o = (k - w) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } + if y < h - 1 and mask[k + w] == 0 { let o = (k + w) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } + if n > 0 { let o = k * c; px[o] = r / n; px[o + 1] = g / n; px[o + 2] = b / n; next[k] = 0 } + } + } + } + for q in 0 .. w * h { mask[q] = next[q] } + } + } + + # blades and blobs of colour on black, different in every row and column + function atlas(w: int, h: int, c: int) -> []byte { + let px = buffer(w * h * c) + for y in 0 .. h { + for x in 0 .. w { + let o = (y * w + x) * c + let lit = ((x * 7 + y * 13) % 29 < 6) or ((x / 9 + y / 11) % 5 == 0) + for k in 0 .. c { px[o + k] = 0 } + if lit { + px[o] = (x * 3 + y) % 256 + px[o + 1] = (x + y * 5) % 256 + px[o + 2] = (x * y) % 256 + } + if c == 4 { px[o + 3] = 255 } + } + } + return px + } + + handler Boot(render3d_st: mut Render3dState) phase Start { + var ok = true + for c in 3 .. 5 { + let w = 173 + let h = 101 + let a = atlas(w, h, c) + let b = atlas(w, h, c) + render3d_st.tex_w = w + render3d_st.tex_h = h + render3d_st.tex_channels = c + render3d_st.tex_depth = 8 + tex_dilate_bytes(render3d_st, a, 60, 24) + oracle(b, w, h, c, 60, 24) + let orig = atlas(w, h, c) + var same = 0 + var moved = 0 + for i in 0 .. w * h * c { + if a[i] == b[i] { same += 1 } + if a[i] != orig[i] { moved += 1 } + } + if same != w * h * c or moved == 0 { ok = false } # the same bytes, and the padding did pad + } + if ok { print("DILATE OK") } else { print("DILATE FAILED") } + quit() + } +} diff --git a/packages/ludic.render3d/safe_api.ludic b/packages/ludic.render3d/safe_api.ludic index 3db24a92..e74d583a 100644 --- a/packages/ludic.render3d/safe_api.ludic +++ b/packages/ludic.render3d/safe_api.ludic @@ -39,3 +39,8 @@ function png_decode_bytes(render3d_st: mut Render3dState, path: string, reuse: [ function tex_upload_bytes(render3d_st: mut Render3dState, px: []byte, srgb: bool, mips: bool) -> int { return tex_upload(render3d_st, data_of(px), srgb, mips) } +# cut-out edge padding over samples laid out as the last decode left them (tex_dilate) +function tex_dilate_bytes(render3d_st: Render3dState, px: []byte, thresh: int, passes: int) -> void { + if px == null or len(px) < render3d_st.tex_w * render3d_st.tex_h * render3d_st.tex_channels { return } + tex_dilate(render3d_st, data_of(px), thresh, passes) +} diff --git a/packages/ludic.render3d/texture.ludic b/packages/ludic.render3d/texture.ludic index 48ab06f6..3ec299fd 100644 --- a/packages/ludic.render3d/texture.ludic +++ b/packages/ludic.render3d/texture.ludic @@ -198,30 +198,51 @@ function tex_dilate(render3d_st: Render3dState, px: pointer, thresh: int, passes var i = 0 while i < w * h { let o = i * c; if px[o] + px[o + 1] + px[o + 2] < thresh { mask[i] = 1 } else { mask[i] = 0 }; i += 1 } let next = bytes(w * h) + let j = new DilateJob + j.px = px; j.mask = mask; j.next = next; j.w = w; j.h = h; j.c = c + let blocks = (h + TEX_DILATE_ROWS - 1) / TEX_DILATE_ROWS for pass in 0 .. passes { mem_copy(next, mask, w * h) - var y = 0 - while y < h { - var x = 0 - while x < w { - let k = y * w + x - if mask[k] == 1 { - var r = 0; var g = 0; var b = 0; var n = 0 - if x > 0 and mask[k - 1] == 0 { let o = (k - 1) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } - if x < w - 1 and mask[k + 1] == 0 { let o = (k + 1) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } - if y > 0 and mask[k - w] == 0 { let o = (k - w) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } - if y < h - 1 and mask[k + w] == 0 { let o = (k + w) * c; r += px[o]; g += px[o + 1]; b += px[o + 2]; n += 1 } - if n > 0 { let o = k * c; px[o] = r / n; px[o + 1] = g / n; px[o + 2] = b / n; next[k] = 0 } - } - x += 1 - } - y += 1 - } + Job.parallel_for(blocks, fn tex_dilate_rows, j) mem_copy(mask, next, w * h) } free(mask); free(next) } +# A pass's rows on every core. Within a pass a row writes only its own still-masked texels and reads +# only neighbours the mask already let go, which no row writes this pass - so rows are independent and +# the result is the one row-by-row order gave. The worker is handed plain buffers and makes nothing. +const TEX_DILATE_ROWS: int = 16 +property DilateJob { + px: pointer = null + mask: pointer = null + next: pointer = null + w: int = 0 + h: int = 0 + c: int = 0 +} +function tex_dilate_rows(b: int, j: DilateJob) -> void { + let w = j.w; let c = j.c; let px = j.px; let mask = j.mask; let next = j.next + var y = b * TEX_DILATE_ROWS + let y1 = min(j.h, y + TEX_DILATE_ROWS) + while y < y1 { + var x = 0 + while x < w { + let k = y * w + x + if mask[k] == 1 { + var r = 0; var g = 0; var bl = 0; var n = 0 + if x > 0 and mask[k - 1] == 0 { let o = (k - 1) * c; r += px[o]; g += px[o + 1]; bl += px[o + 2]; n += 1 } + if x < w - 1 and mask[k + 1] == 0 { let o = (k + 1) * c; r += px[o]; g += px[o + 1]; bl += px[o + 2]; n += 1 } + if y > 0 and mask[k - w] == 0 { let o = (k - w) * c; r += px[o]; g += px[o + 1]; bl += px[o + 2]; n += 1 } + if y < j.h - 1 and mask[k + w] == 0 { let o = (k + w) * c; r += px[o]; g += px[o + 1]; bl += px[o + 2]; n += 1 } + if n > 0 { let o = k * c; px[o] = r / n; px[o + 1] = g / n; px[o + 2] = bl / n; next[k] = 0 } + } + x += 1 + } + y += 1 + } +} + # Upload the last-decoded samples as a 2D texture. srgb: colour data (8-bit only). function tex_upload(render3d_st: mut Render3dState, px: pointer, srgb: bool, mips: bool) -> int { let id = gpu_tex_new(render3d_st)