feat(render3d): 23.3 - a .dds beside a .png is uploaded BC-compressed with its whole mip chain

The device's textureCompressionBC is asked for and remembered (gvk_has_bc). tex_load_ex prefers a
DX10 .dds with the full chain beside the .png (not for an edge-padded cut-out atlas):
texture_dds.ludic reads BC7 / BC5 / BC4, gpu_tex_compressed makes the image with every level and
no colour-attachment use (a compressed image is only sampled and copied into), and
gvk_tex_upload_blocks copies each level's blocks from one staging buffer. A colour map is BC7
sampled as sRGB, a data map BC7 read as it is. examples/rendering/bc.ludic holds it, in the suite;
ludic.lab's plate carries its .dds (its three shots render at 56-60 dB against the .png's).

ludic-dev test 307/307, no Vulkan SDK in the environment.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Orkun ÇAKILKAYA 2026-09-28 01:23:44 +03:00
parent 6c663cc87b
commit 8cc4bdf66b
16 changed files with 217 additions and 1 deletions

View file

@ -0,0 +1,11 @@
bump: minor
type: feature
**A texture with a `.dds` beside its `.png` goes to the GPU BC-compressed, its mips included.**
`tex_load` (and everything built on it: glTF materials, the terrain's materials, ludic.ui's
pictures) looks for `<name>.dds` when the device has `textureCompressionBC` - every desktop GPU,
Apple silicon through MoltenVK too - and uploads its blocks as they are: BC7 (colour in sRGB,
or data read as it is), BC5 or BC4, a quarter of RGBA8 or less in video memory, no decoding and
no mipmapping on the load. Only a DX10 `.dds` carrying the whole chain to 1x1 is taken; anything
else is reported and the `.png` is used, as it is for a cut-out atlas (padded from the `.png`).
`gpu_tex_compressed` and `R3D_BC7` / `R3D_BC7_SRGB` / `R3D_BC5` / `R3D_BC4` are the backend's
entry; `examples/rendering/bc.ludic` is the check, and ludic.lab's plate carries its `.dds`.

View file

@ -0,0 +1,59 @@
# bc.ludic - a texture whose .png has a .dds beside it arrives BC7-compressed with its whole mip
# chain (plan 23.3 of maroon-lake), and one without a .dds as the .png it always was. Prints BC OK.
#
# bin/ludic build examples/rendering/bc.ludic --headless && ./build/bc_headless
program Bc {
numbers float
import "ludic.render3d/r3d.ludic"
property Marker { on: int = 1 }
model Anchor { Marker }
function scene_draw(render3d_st: mut Render3dState) -> void { }
function scene_draw_casters(render3d_st: mut Render3dState, light_vp: floats) -> void { }
function stream_fill(s: Stream, cx: int, cz: int, band: int) -> void { }
function check(what: string, cond: bool) -> bool {
if not cond { print(`bc: FAILED - {what}`) }
return cond
}
# the internal format render3d recorded for a texture
function fmt_of(render3d_st: mut Render3dState, tex: int) -> int {
let o = gpu_tx_at(render3d_st, tex)
if o < 0 { return 0 }
return render3d_st.gpu_tx[o + 4]
}
handler Boot(render3d_st: mut Render3dState) phase Start {
spawn Anchor {}
r3d_on_draw(render3d_st, fn scene_draw)
r3d_on_casters(render3d_st, fn scene_draw_casters)
r3d_on_stream_fill(render3d_st, fn stream_fill)
r3d_plate_mode(render3d_st, true)
let dir = "packages/ludic.lab/plate"
render3d_st.r3d_sky_path = `{dir}/sky.hdr`
if not r3d_init(render3d_st, 320, 180, "Bc") {
quit()
return
}
var ok = true
ok = check("this device reads BC textures", render3d_st.gvk_has_bc) and ok
let t = tex_load(render3d_st, `{dir}/plate_diff.png`, true)
ok = check("the plate's colour loads", t != 0) and ok
ok = check("from its .dds, as BC7 in sRGB", fmt_of(render3d_st, t) == R3D_BC7_SRGB) and ok
ok = check("with every level down to 1x1", render3d_st.gvk_tex_levels[t] == gvk_mip_levels(tex_width(render3d_st, t), tex_height(render3d_st, t))) and ok
let a = tex_load(render3d_st, `{dir}/plate_arm.png`, false)
ok = check("a data map's .dds is BC7 read as it is (not sRGB)", fmt_of(render3d_st, a) == R3D_BC7) and ok
let s = tex_load(render3d_st, `{dir}/probe_arm.png`, false)
ok = check("the second one loads too", s != 0) and ok
# a picture with no .dds: the .png, uncompressed, as before
let p = tex_load_ex(render3d_st, `{dir}/plate_diff.png`, true, 1)
ok = check("a cut-out atlas (edge-padded from the .png) is not the .dds", p != 0 and fmt_of(render3d_st, p) != R3D_BC7_SRGB) and ok
if ok { print("BC OK") } else { print("BC FAILED") }
quit()
}
handler Present(render3d_st: mut Render3dState) phase Render {
r3d_present(render3d_st)
}
}

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

View file

@ -200,6 +200,7 @@ export state Render3dState {
gvk_labels: bool = false # R3D_VK_LABELS: each profiled pass is a debug label (Metal System Trace)
gvk_inflight: int = -1
gvk_nopool: bool = false
gvk_has_bc: bool = false # textureCompressionBC: .dds textures are uploaded compressed
# R3D_VK_PROF: Vulkan objects made and destroyed, reported every 120 frames
gvk_mk_img: int = 0
gvk_mk_view: int = 0

View file

@ -571,6 +571,18 @@ function gpu_tex_paramf(render3d_st: mut Render3dState, kind: int, pname: int, v
}
# the border colour clamp-to-border reads (four fixed values in `rgba`)
function gpu_tex_border(render3d_st: Render3dState, kind: int, rgba: pointer) -> void { }
# a block-compressed texture with its whole mip chain, into the bound 2D texture: levels as they
# are in the file (level l at offs[l]), no decoding on the CPU and no mips made here
function gpu_tex_compressed(render3d_st: mut Render3dState, ifmt: int, w: int, h: int, levels: int, data: pointer, offs: words) -> bool {
gvk_flush(render3d_st)
let tex = render3d_st.gpu_bound_2d
if not gvk_tex_storage(render3d_st, tex, false, ifmt, w, h, 1, true) { return false }
if render3d_st.gvk_tex_levels[tex] != levels { print(`r3d: a {w}x{h} compressed texture brought {levels} levels, not {render3d_st.gvk_tex_levels[tex]}`); return false }
if not gvk_tex_upload_blocks(render3d_st, tex, w, h, levels, data, offs) { return false }
let o = gpu_tx_at(render3d_st, tex)
if o >= 0 { render3d_st.gpu_tx[o] = GPU_TEX2D; render3d_st.gpu_tx[o + 1] = w; render3d_st.gpu_tx[o + 2] = h; render3d_st.gpu_tx[o + 3] = 1; render3d_st.gpu_tx[o + 4] = ifmt; render3d_st.gpu_tx[o + 10] = 1 }
return true
}
function gpu_tex_mips(render3d_st: mut Render3dState, kind: int) -> void {
let mt = gpu_bound(render3d_st, kind)
let mo = gpu_tx_at(render3d_st, mt)

View file

@ -174,6 +174,10 @@ function gvk_init(render3d_st: mut Render3dState) -> bool {
# how many of them there are. Asked for where the device has them; the renderer checks
# gvk_has_mdi / gvk_has_dic before it takes that path.
render3d_st.gvk_has_mdi = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect) == 1 and Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance) == 1
# BC-compressed textures (BC4/5/7): every desktop GPU, Apple silicon through MoltenVK too; a
# texture with a .dds beside its .png is uploaded compressed when this is on (texture.ludic)
render3d_st.gvk_has_bc = Vk.get_i32(f2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_textureCompressionBC) == 1
if render3d_st.gvk_has_bc { Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_textureCompressionBC, 1) }
if render3d_st.gvk_has_mdi {
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_multiDrawIndirect, 1)
Vk.put_i32(want2, VkPhysicalDeviceFeatures2_features + VkPhysicalDeviceFeatures_drawIndirectFirstInstance, 1)

View file

@ -0,0 +1,31 @@
# gpu_vk_bc.ludic - a block-compressed texture's levels uploaded as they are: the GPU decodes BC
# blocks itself, so the bytes in the file are the bytes in video memory, a quarter of RGBA8's.
# every level of `tex` (made by gvk_tex_storage with its whole chain) from `data`, level l's
# blocks at offs[l] .. offs[l + 1]
function gvk_tex_upload_blocks(render3d_st: mut Render3dState, tex: int, w: int, h: int, levels: int, data: pointer, offs: words) -> bool {
let n = offs[levels]
let dst = gvk_staging(render3d_st, n, VK_BUFFER_USAGE_TRANSFER_SRC_BIT)
if dst == null { print("r3d: vulkan: no staging buffer for a compressed upload"); return false }
mem_copy(dst, data, n)
let image = render3d_st.gvk_tex_image[tex]
let cb = gvk_once_begin(render3d_st)
gvk_barrier(render3d_st, cb, image, false, 0, levels, 1, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL)
for l in 0 .. levels {
let bic = gvk_tmp(render3d_st, VkBufferImageCopy_sizeof)
Vk.zero(bic, VkBufferImageCopy_sizeof)
let off: long = offs[l]
Vk.put_i64(bic, VkBufferImageCopy_bufferOffset, off)
Vk.put_i32(bic, VkBufferImageCopy_imageSubresource + VkImageSubresourceLayers_aspectMask, VK_IMAGE_ASPECT_COLOR_BIT)
Vk.put_i32(bic, VkBufferImageCopy_imageSubresource + VkImageSubresourceLayers_mipLevel, l)
Vk.put_i32(bic, VkBufferImageCopy_imageSubresource + VkImageSubresourceLayers_layerCount, 1)
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_width, max(w >> l, 1))
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_height, max(h >> l, 1))
Vk.put_i32(bic, VkBufferImageCopy_imageExtent + VkExtent3D_depth, 1)
Vk.cmd_copy_buffer_to_image(cb, render3d_st.gvk_st_buf, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, bic)
}
gvk_barrier(render3d_st, cb, image, false, 0, levels, 1, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL)
let ok = gvk_once_end(render3d_st, cb)
gvk_staging_free(render3d_st)
return ok
}

View file

@ -9,7 +9,18 @@
# The renderer names formats the way OpenGL does (they are gpu.ludic's vocabulary); these turn
# one into the Vulkan image format that holds it. Three-channel formats have no widely
# supported Vulkan image format, so they are stored with four and expanded on upload.
# block-compressed formats, named here (the GL header generated for render3d carries no BPTC):
# the values are GL's, so they sit in the same vocabulary as every other internal format
const R3D_BC7: int = 0x8E8C
const R3D_BC7_SRGB: int = 0x8E8D
const R3D_BC5: int = 0x8DBD
const R3D_BC4: int = 0x8DBB
function gvk_is_compressed(ifmt: int) -> bool { return ifmt == R3D_BC7 or ifmt == R3D_BC7_SRGB or ifmt == R3D_BC5 or ifmt == R3D_BC4 }
function gvk_format(ifmt: int) -> int {
if ifmt == R3D_BC7 { return VK_FORMAT_BC7_UNORM_BLOCK }
if ifmt == R3D_BC7_SRGB { return VK_FORMAT_BC7_SRGB_BLOCK }
if ifmt == R3D_BC5 { return VK_FORMAT_BC5_UNORM_BLOCK }
if ifmt == R3D_BC4 { return VK_FORMAT_BC4_UNORM_BLOCK }
if ifmt == GL_RGBA8 or ifmt == GL_RGB8 { return VK_FORMAT_R8G8B8A8_UNORM }
if ifmt == GL_RGB10_A2 { return VK_FORMAT_A2B10G10R10_UNORM_PACK32 } # the HDR10 screen and LDR image
if ifmt == GL_SRGB8_ALPHA8 or ifmt == GL_SRGB8 { return VK_FORMAT_R8G8B8A8_SRGB }
@ -158,7 +169,8 @@ function gvk_tex_storage(render3d_st: mut Render3dState, tex: int, array: bool,
if samples < 1 { samples = 1 }
if samples > 1 { levels = 1 }
var usage = VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT
if depth { usage = usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT } else { usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT }
# a block-compressed image is only ever sampled and copied into: it cannot be drawn to
if depth { usage = usage | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT } else if not gvk_is_compressed(ifmt) { usage = usage | VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT }
# DLSS reads and writes the frame from compute: a float colour target is storage too while
# Streamline is running (8-bit sRGB formats cannot be, so only the float ones)
if render3d_st.gsl_on and samples == 1 and (vkfmt == VK_FORMAT_R16G16B16A16_SFLOAT or vkfmt == VK_FORMAT_R32_SFLOAT) { usage = usage | VK_IMAGE_USAGE_STORAGE_BIT }

View file

@ -11,6 +11,8 @@ import "gpu.ludic"
import "gpu_manifest.ludic"
import "gpu_vk.ludic"
import "gpu_vk_res.ludic"
import "gpu_vk_bc.ludic"
import "texture_dds.ludic"
import "prof.ludic"
import "drawstats.ludic"
import "programs.ludic"

View file

@ -263,6 +263,18 @@ function tex_upload(render3d_st: mut Render3dState, px: pointer, srgb: bool, mip
function tex_load(render3d_st: mut Render3dState, path: pointer, srgb: bool) -> int { return tex_load_ex(render3d_st, path, srgb, 0) }
# ... with `dilate` passes of edge padding for a cut-out atlas (0 = none)
function tex_load_ex(render3d_st: mut Render3dState, path: pointer, srgb: bool, dilate: int) -> int {
# a .dds encoded beside the .png at build time goes to the GPU compressed, its mips included
# (not for a cut-out atlas: its edge padding is made here, from the .png)
if render3d_st.gvk_has_bc and dilate == 0 {
let dds = dds_path_of(path)
if len(dds) > 0 and Fs.exists(dds) {
let t = tex_load_dds(render3d_st, dds, srgb)
if t != 0 {
tex_note_size(render3d_st, t)
return t
}
}
}
let px = png_decode(render3d_st, path)
if px == null { return 0 }
if dilate > 0 { tex_dilate(render3d_st, px, 60, dilate) }

View file

@ -0,0 +1,71 @@
# texture_dds.ludic - a .dds beside a .png: BC7 (colour, and anything packed into three channels),
# BC5 (a normal's x and y) or BC4 (one channel), every mip level encoded at build time. Only the
# DX10 form with its whole chain is taken; anything else is 0 and the caller loads the .png.
# DXGI's numbers for the formats read here
const DDS_BC4: int = 80
const DDS_BC5: int = 83
const DDS_BC7: int = 98
const DDS_BC7_SRGB: int = 99
function dds_u32(b: []byte, o: int) -> int { return b[o] | (b[o + 1] << 8) | (b[o + 2] << 16) | (b[o + 3] << 24) }
# the .dds a .png may have beside it, or "" when the path is not a .png
function dds_path_of(path: string) -> string {
let n = len(path)
if n < 5 or path[n - 4 .. n] != ".png" { return "" }
return path[0 .. n - 4] + ".dds"
}
# Load the .dds at `path` as a texture (srgb: colour, sampled as sRGB), or 0
function tex_load_dds(render3d_st: mut Render3dState, path: string, srgb: bool) -> int {
let b = Fs.read_bytes(path)
if b == null { return 0 }
let id = tex_load_dds_bytes(render3d_st, b, srgb)
free(b)
if id == 0 { print(`r3d: {path} is not a DX10 .dds with its whole mip chain; the .png is used`) }
return id
}
function tex_load_dds_bytes(render3d_st: mut Render3dState, b: []byte, srgb: bool) -> int {
if len(b) < 148 or b[0] != 68 or b[1] != 68 or b[2] != 83 or b[3] != 32 { return 0 }
if b[84] != 68 or b[85] != 88 or b[86] != 49 or b[87] != 48 { return 0 } # "DX10"
let h = dds_u32(b, 12)
let w = dds_u32(b, 16)
var levels = dds_u32(b, 28)
if levels < 1 { levels = 1 }
let dxgi = dds_u32(b, 128)
var ifmt = 0
var block = 16
if dxgi == DDS_BC7 or dxgi == DDS_BC7_SRGB { ifmt = R3D_BC7; if srgb { ifmt = R3D_BC7_SRGB } }
if dxgi == DDS_BC5 { ifmt = R3D_BC5 }
if dxgi == DDS_BC4 { ifmt = R3D_BC4; block = 8 }
if ifmt == 0 or w < 1 or h < 1 or levels != gvk_mip_levels(w, h) { return 0 }
let offs = words(levels + 1)
var at = 0
for l in 0 .. levels {
offs[l] = at
at += max((max(w >> l, 1) + 3) / 4, 1) * max((max(h >> l, 1) + 3) / 4, 1) * block
}
offs[levels] = at
if 148 + at > len(b) {
free(offs)
return 0
}
let id = gpu_tex_new(render3d_st)
gpu_tex_bind(render3d_st, GPU_TEX2D, id)
let ok = gpu_tex_compressed(render3d_st, ifmt, w, h, levels, mem_off(data_of(b), 148), offs)
free(offs)
if not ok {
gpu_tex_free(render3d_st, id)
return 0
}
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_S, GL_REPEAT)
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_WRAP_T, GL_REPEAT)
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MAG_FILTER, GL_LINEAR)
gpu_tex_param(render3d_st, GPU_TEX2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR)
gpu_tex_paramf(render3d_st, GPU_TEX2D, GL_TEXTURE_MAX_ANISOTROPY_EXT, render3d_st.tex_anisotropy)
render3d_st.tex_w = w
render3d_st.tex_h = h
return id
}

View file

@ -1053,6 +1053,7 @@ function cmd_dev_test() -> int {
controller_case("state/component", "", " #10 big 10 1", "component.ludic (0.S: a component's header names its states; its getters, functions and events take them and the template never sees them)")
headless_case("rendering/ui_render3d", "ok", "ui_render3d.ludic (ludic.ui's render3d backend builds against the renderer)")
headless_line_case("rendering/release", "RELEASE OK", "release.ludic (model_release: meshes go, a shared texture stays until its last user goes)")
headless_line_case("rendering/bc", "BC OK", "bc.ludic (a .dds beside a .png arrives BC7-compressed with every mip level)")
migrate_component_case()
migrate_foreign_case()
reject_case("rejected/runtime_type_clash", "PadButton is the runtime's enum", "a program's type named like one of the runtime's is refused")